LLVM 24.0.0git
AMDGPUISelLowering.cpp
Go to the documentation of this file.
1//===-- AMDGPUISelLowering.cpp - AMDGPU Common DAG lowering functions -----===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This is the parent TargetLowering class for hardware code gen
11/// targets.
12//
13//===----------------------------------------------------------------------===//
14
15#include "AMDGPUISelLowering.h"
16#include "AMDGPU.h"
17#include "AMDGPUInstrInfo.h"
19#include "AMDGPUMemoryUtils.h"
26#include "llvm/IR/IntrinsicsAMDGPU.h"
31
32using namespace llvm;
33
34#define GET_CALLING_CONV_IMPL
35#include "AMDGPUGenCallingConv.inc"
36
38 "amdgpu-bypass-slow-div",
39 cl::desc("Skip 64-bit divide for dynamic 32-bit values"),
40 cl::init(true));
41
42// Find a larger type to do a load / store of a vector with.
44 unsigned StoreSize = VT.getStoreSizeInBits();
45 if (StoreSize <= 32)
46 return EVT::getIntegerVT(Ctx, StoreSize);
47
48 if (StoreSize % 32 == 0)
49 return EVT::getVectorVT(Ctx, MVT::i32, StoreSize / 32);
50
51 return VT;
52}
53
57
59 // In order for this to be a signed 24-bit value, bit 23, must
60 // be a sign bit.
61 return DAG.ComputeMaxSignificantBits(Op);
62}
63
65 const TargetSubtargetInfo &STI,
66 const AMDGPUSubtarget &AMDGPUSTI)
67 : TargetLowering(TM, STI), Subtarget(&AMDGPUSTI) {
68 // Always lower memset, memcpy, and memmove intrinsics to load/store
69 // instructions, rather then generating calls to memset, mempcy or memmove.
73
74 // Enable ganging up loads and stores in the memcpy DAG lowering.
76
77 // Lower floating point store/load to integer store/load to reduce the number
78 // of patterns in tablegen.
80 AddPromotedToType(ISD::LOAD, MVT::f32, MVT::i32);
81
83 AddPromotedToType(ISD::LOAD, MVT::v2f32, MVT::v2i32);
84
86 AddPromotedToType(ISD::LOAD, MVT::v3f32, MVT::v3i32);
87
89 AddPromotedToType(ISD::LOAD, MVT::v4f32, MVT::v4i32);
90
92 AddPromotedToType(ISD::LOAD, MVT::v5f32, MVT::v5i32);
93
95 AddPromotedToType(ISD::LOAD, MVT::v6f32, MVT::v6i32);
96
98 AddPromotedToType(ISD::LOAD, MVT::v7f32, MVT::v7i32);
99
101 AddPromotedToType(ISD::LOAD, MVT::v8f32, MVT::v8i32);
102
104 AddPromotedToType(ISD::LOAD, MVT::v9f32, MVT::v9i32);
105
106 setOperationAction(ISD::LOAD, MVT::v10f32, Promote);
107 AddPromotedToType(ISD::LOAD, MVT::v10f32, MVT::v10i32);
108
109 setOperationAction(ISD::LOAD, MVT::v11f32, Promote);
110 AddPromotedToType(ISD::LOAD, MVT::v11f32, MVT::v11i32);
111
112 setOperationAction(ISD::LOAD, MVT::v12f32, Promote);
113 AddPromotedToType(ISD::LOAD, MVT::v12f32, MVT::v12i32);
114
115 setOperationAction(ISD::LOAD, MVT::v16f32, Promote);
116 AddPromotedToType(ISD::LOAD, MVT::v16f32, MVT::v16i32);
117
118 setOperationAction(ISD::LOAD, MVT::v32f32, Promote);
119 AddPromotedToType(ISD::LOAD, MVT::v32f32, MVT::v32i32);
120
122 AddPromotedToType(ISD::LOAD, MVT::i64, MVT::v2i32);
123
125 AddPromotedToType(ISD::LOAD, MVT::v2i64, MVT::v4i32);
126
128 AddPromotedToType(ISD::LOAD, MVT::f64, MVT::v2i32);
129
131 AddPromotedToType(ISD::LOAD, MVT::v2f64, MVT::v4i32);
132
134 AddPromotedToType(ISD::LOAD, MVT::v3i64, MVT::v6i32);
135
137 AddPromotedToType(ISD::LOAD, MVT::v4i64, MVT::v8i32);
138
140 AddPromotedToType(ISD::LOAD, MVT::v3f64, MVT::v6i32);
141
143 AddPromotedToType(ISD::LOAD, MVT::v4f64, MVT::v8i32);
144
146 AddPromotedToType(ISD::LOAD, MVT::v8i64, MVT::v16i32);
147
149 AddPromotedToType(ISD::LOAD, MVT::v8f64, MVT::v16i32);
150
151 setOperationAction(ISD::LOAD, MVT::v16i64, Promote);
152 AddPromotedToType(ISD::LOAD, MVT::v16i64, MVT::v32i32);
153
154 setOperationAction(ISD::LOAD, MVT::v16f64, Promote);
155 AddPromotedToType(ISD::LOAD, MVT::v16f64, MVT::v32i32);
156
158 AddPromotedToType(ISD::LOAD, MVT::i128, MVT::v4i32);
159
160 // TODO: Would be better to consume as directly legal
162 AddPromotedToType(ISD::ATOMIC_LOAD, MVT::f32, MVT::i32);
163
165 AddPromotedToType(ISD::ATOMIC_LOAD, MVT::f64, MVT::i64);
166
168 AddPromotedToType(ISD::ATOMIC_LOAD, MVT::f16, MVT::i16);
169
171 AddPromotedToType(ISD::ATOMIC_LOAD, MVT::bf16, MVT::i16);
172
174 AddPromotedToType(ISD::ATOMIC_LOAD, MVT::v2f32, MVT::i64);
175
177 AddPromotedToType(ISD::ATOMIC_STORE, MVT::f32, MVT::i32);
178
180 AddPromotedToType(ISD::ATOMIC_STORE, MVT::f64, MVT::i64);
181
183 AddPromotedToType(ISD::ATOMIC_STORE, MVT::f16, MVT::i16);
184
186 AddPromotedToType(ISD::ATOMIC_STORE, MVT::bf16, MVT::i16);
187
189 AddPromotedToType(ISD::ATOMIC_STORE, MVT::v2f32, MVT::i64);
190
191 // There are no 64-bit extloads. These should be done as a 32-bit extload and
192 // an extension to 64-bit.
193 for (MVT VT : MVT::integer_valuetypes())
195 Expand);
196
197 for (MVT VT : MVT::integer_valuetypes()) {
198 if (VT == MVT::i64)
199 continue;
200
201 for (auto Op : {ISD::SEXTLOAD, ISD::ZEXTLOAD, ISD::EXTLOAD}) {
202 setLoadExtAction(Op, VT, MVT::i1, Promote);
203 setLoadExtAction(Op, VT, MVT::i8, Legal);
204 setLoadExtAction(Op, VT, MVT::i16, Legal);
205 setLoadExtAction(Op, VT, MVT::i32, Expand);
206 }
207 }
208
210 for (auto MemVT :
211 {MVT::v2i8, MVT::v4i8, MVT::v2i16, MVT::v3i16, MVT::v4i16})
213 Expand);
214
215 setLoadExtAction(ISD::EXTLOAD, MVT::f32, MVT::f16, Expand);
216 setLoadExtAction(ISD::EXTLOAD, MVT::f32, MVT::bf16, Expand);
217 setLoadExtAction(ISD::EXTLOAD, MVT::v2f32, MVT::v2f16, Expand);
218 setLoadExtAction(ISD::EXTLOAD, MVT::v2f32, MVT::v2bf16, Expand);
219 setLoadExtAction(ISD::EXTLOAD, MVT::v3f32, MVT::v3f16, Expand);
220 setLoadExtAction(ISD::EXTLOAD, MVT::v3f32, MVT::v3bf16, Expand);
221 setLoadExtAction(ISD::EXTLOAD, MVT::v4f32, MVT::v4f16, Expand);
222 setLoadExtAction(ISD::EXTLOAD, MVT::v4f32, MVT::v4bf16, Expand);
223 setLoadExtAction(ISD::EXTLOAD, MVT::v8f32, MVT::v8f16, Expand);
224 setLoadExtAction(ISD::EXTLOAD, MVT::v8f32, MVT::v8bf16, Expand);
225 setLoadExtAction(ISD::EXTLOAD, MVT::v16f32, MVT::v16f16, Expand);
226 setLoadExtAction(ISD::EXTLOAD, MVT::v16f32, MVT::v16bf16, Expand);
227 setLoadExtAction(ISD::EXTLOAD, MVT::v32f32, MVT::v32f16, Expand);
228 setLoadExtAction(ISD::EXTLOAD, MVT::v32f32, MVT::v32bf16, Expand);
229
230 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::f32, Expand);
231 setLoadExtAction(ISD::EXTLOAD, MVT::v2f64, MVT::v2f32, Expand);
232 setLoadExtAction(ISD::EXTLOAD, MVT::v3f64, MVT::v3f32, Expand);
233 setLoadExtAction(ISD::EXTLOAD, MVT::v4f64, MVT::v4f32, Expand);
234 setLoadExtAction(ISD::EXTLOAD, MVT::v8f64, MVT::v8f32, Expand);
235 setLoadExtAction(ISD::EXTLOAD, MVT::v16f64, MVT::v16f32, Expand);
236
237 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::f16, Expand);
238 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::bf16, Expand);
239 setLoadExtAction(ISD::EXTLOAD, MVT::v2f64, MVT::v2f16, Expand);
240 setLoadExtAction(ISD::EXTLOAD, MVT::v2f64, MVT::v2bf16, Expand);
241 setLoadExtAction(ISD::EXTLOAD, MVT::v3f64, MVT::v3f16, Expand);
242 setLoadExtAction(ISD::EXTLOAD, MVT::v3f64, MVT::v3bf16, Expand);
243 setLoadExtAction(ISD::EXTLOAD, MVT::v4f64, MVT::v4f16, Expand);
244 setLoadExtAction(ISD::EXTLOAD, MVT::v4f64, MVT::v4bf16, Expand);
245 setLoadExtAction(ISD::EXTLOAD, MVT::v8f64, MVT::v8f16, Expand);
246 setLoadExtAction(ISD::EXTLOAD, MVT::v8f64, MVT::v8bf16, Expand);
247 setLoadExtAction(ISD::EXTLOAD, MVT::v16f64, MVT::v16f16, Expand);
248 setLoadExtAction(ISD::EXTLOAD, MVT::v16f64, MVT::v16bf16, Expand);
249
251 AddPromotedToType(ISD::STORE, MVT::f32, MVT::i32);
252
254 AddPromotedToType(ISD::STORE, MVT::v2f32, MVT::v2i32);
255
257 AddPromotedToType(ISD::STORE, MVT::v3f32, MVT::v3i32);
258
260 AddPromotedToType(ISD::STORE, MVT::v4f32, MVT::v4i32);
261
263 AddPromotedToType(ISD::STORE, MVT::v5f32, MVT::v5i32);
264
266 AddPromotedToType(ISD::STORE, MVT::v6f32, MVT::v6i32);
267
269 AddPromotedToType(ISD::STORE, MVT::v7f32, MVT::v7i32);
270
272 AddPromotedToType(ISD::STORE, MVT::v8f32, MVT::v8i32);
273
275 AddPromotedToType(ISD::STORE, MVT::v9f32, MVT::v9i32);
276
278 AddPromotedToType(ISD::STORE, MVT::v10f32, MVT::v10i32);
279
281 AddPromotedToType(ISD::STORE, MVT::v11f32, MVT::v11i32);
282
284 AddPromotedToType(ISD::STORE, MVT::v12f32, MVT::v12i32);
285
287 AddPromotedToType(ISD::STORE, MVT::v16f32, MVT::v16i32);
288
290 AddPromotedToType(ISD::STORE, MVT::v32f32, MVT::v32i32);
291
293 AddPromotedToType(ISD::STORE, MVT::i64, MVT::v2i32);
294
296 AddPromotedToType(ISD::STORE, MVT::v2i64, MVT::v4i32);
297
299 AddPromotedToType(ISD::STORE, MVT::f64, MVT::v2i32);
300
302 AddPromotedToType(ISD::STORE, MVT::v2f64, MVT::v4i32);
303
305 AddPromotedToType(ISD::STORE, MVT::v3i64, MVT::v6i32);
306
308 AddPromotedToType(ISD::STORE, MVT::v3f64, MVT::v6i32);
309
311 AddPromotedToType(ISD::STORE, MVT::v4i64, MVT::v8i32);
312
314 AddPromotedToType(ISD::STORE, MVT::v4f64, MVT::v8i32);
315
317 AddPromotedToType(ISD::STORE, MVT::v8i64, MVT::v16i32);
318
320 AddPromotedToType(ISD::STORE, MVT::v8f64, MVT::v16i32);
321
323 AddPromotedToType(ISD::STORE, MVT::v16i64, MVT::v32i32);
324
326 AddPromotedToType(ISD::STORE, MVT::v16f64, MVT::v32i32);
327
329 AddPromotedToType(ISD::STORE, MVT::i128, MVT::v4i32);
330
331 setTruncStoreAction(MVT::i64, MVT::i1, Expand);
332 setTruncStoreAction(MVT::i64, MVT::i8, Expand);
333 setTruncStoreAction(MVT::i64, MVT::i16, Expand);
334 setTruncStoreAction(MVT::i64, MVT::i32, Expand);
335
336 setTruncStoreAction(MVT::v2i64, MVT::v2i1, Expand);
337 setTruncStoreAction(MVT::v2i64, MVT::v2i8, Expand);
338 setTruncStoreAction(MVT::v2i64, MVT::v2i16, Expand);
339 setTruncStoreAction(MVT::v2i64, MVT::v2i32, Expand);
340
341 setTruncStoreAction(MVT::f32, MVT::bf16, Expand);
342 setTruncStoreAction(MVT::f32, MVT::f16, Expand);
343 setTruncStoreAction(MVT::v2f32, MVT::v2bf16, Expand);
344 setTruncStoreAction(MVT::v2f32, MVT::v2f16, Expand);
345 setTruncStoreAction(MVT::v3f32, MVT::v3bf16, Expand);
346 setTruncStoreAction(MVT::v3f32, MVT::v3f16, Expand);
347 setTruncStoreAction(MVT::v4f32, MVT::v4bf16, Expand);
348 setTruncStoreAction(MVT::v4f32, MVT::v4f16, Expand);
349 setTruncStoreAction(MVT::v6f32, MVT::v6f16, Expand);
350 setTruncStoreAction(MVT::v8f32, MVT::v8bf16, Expand);
351 setTruncStoreAction(MVT::v8f32, MVT::v8f16, Expand);
352 setTruncStoreAction(MVT::v16f32, MVT::v16bf16, Expand);
353 setTruncStoreAction(MVT::v16f32, MVT::v16f16, Expand);
354 setTruncStoreAction(MVT::v32f32, MVT::v32bf16, Expand);
355 setTruncStoreAction(MVT::v32f32, MVT::v32f16, Expand);
356
357 setTruncStoreAction(MVT::f64, MVT::bf16, Expand);
358 setTruncStoreAction(MVT::f64, MVT::f16, Expand);
359 setTruncStoreAction(MVT::f64, MVT::f32, Expand);
360
361 setTruncStoreAction(MVT::v2f64, MVT::v2f32, Expand);
362 setTruncStoreAction(MVT::v2f64, MVT::v2bf16, Expand);
363 setTruncStoreAction(MVT::v2f64, MVT::v2f16, Expand);
364
365 setTruncStoreAction(MVT::v3i32, MVT::v3i8, Expand);
366
367 setTruncStoreAction(MVT::v3i64, MVT::v3i32, Expand);
368 setTruncStoreAction(MVT::v3i64, MVT::v3i16, Expand);
369 setTruncStoreAction(MVT::v3i64, MVT::v3i8, Expand);
370 setTruncStoreAction(MVT::v3i64, MVT::v3i1, Expand);
371 setTruncStoreAction(MVT::v3f64, MVT::v3f32, Expand);
372 setTruncStoreAction(MVT::v3f64, MVT::v3bf16, Expand);
373 setTruncStoreAction(MVT::v3f64, MVT::v3f16, Expand);
374
375 setTruncStoreAction(MVT::v4i64, MVT::v4i32, Expand);
376 setTruncStoreAction(MVT::v4i64, MVT::v4i16, Expand);
377 setTruncStoreAction(MVT::v4f64, MVT::v4f32, Expand);
378 setTruncStoreAction(MVT::v4f64, MVT::v4bf16, Expand);
379 setTruncStoreAction(MVT::v4f64, MVT::v4f16, Expand);
380
381 setTruncStoreAction(MVT::v5i32, MVT::v5i1, Expand);
382 setTruncStoreAction(MVT::v5i32, MVT::v5i8, Expand);
383 setTruncStoreAction(MVT::v5i32, MVT::v5i16, Expand);
384
385 setTruncStoreAction(MVT::v6i32, MVT::v6i1, Expand);
386 setTruncStoreAction(MVT::v6i32, MVT::v6i8, Expand);
387 setTruncStoreAction(MVT::v6i32, MVT::v6i16, Expand);
388
389 setTruncStoreAction(MVT::v7i32, MVT::v7i1, Expand);
390 setTruncStoreAction(MVT::v7i32, MVT::v7i8, Expand);
391 setTruncStoreAction(MVT::v7i32, MVT::v7i16, Expand);
392
393 setTruncStoreAction(MVT::v8f64, MVT::v8f32, Expand);
394 setTruncStoreAction(MVT::v8f64, MVT::v8bf16, Expand);
395 setTruncStoreAction(MVT::v8f64, MVT::v8f16, Expand);
396
397 setTruncStoreAction(MVT::v16f64, MVT::v16f32, Expand);
398 setTruncStoreAction(MVT::v16f64, MVT::v16bf16, Expand);
399 setTruncStoreAction(MVT::v16f64, MVT::v16f16, Expand);
400 setTruncStoreAction(MVT::v16i64, MVT::v16i16, Expand);
401 setTruncStoreAction(MVT::v16i64, MVT::v16i8, Expand);
402 setTruncStoreAction(MVT::v16i64, MVT::v16i8, Expand);
403 setTruncStoreAction(MVT::v16i64, MVT::v16i1, Expand);
404
405 setOperationAction(ISD::Constant, {MVT::i32, MVT::i64}, Legal);
406 setOperationAction(ISD::ConstantFP, {MVT::f32, MVT::f64}, Legal);
407
409
410 // For R600, this is totally unsupported, just custom lower to produce an
411 // error.
413
414 // Library functions. These default to Expand, but we have instructions
415 // for them.
418 {MVT::f16, MVT::f32}, Legal);
420
422 setOperationAction(ISD::FROUND, {MVT::f32, MVT::f64}, Custom);
424 {MVT::f16, MVT::f32, MVT::f64}, Expand);
425
428 MVT::f32, Custom);
430
431 setOperationAction(ISD::FNEARBYINT, {MVT::f16, MVT::f32, MVT::f64}, Custom);
432
433 setOperationAction(ISD::FRINT, {MVT::f16, MVT::f32, MVT::f64}, Custom);
434
435 setOperationAction({ISD::LRINT, ISD::LLRINT}, {MVT::f16, MVT::f32, MVT::f64},
436 Expand);
437
438 setOperationAction(ISD::FREM, {MVT::f16, MVT::f32, MVT::f64}, Expand);
439 setOperationAction(ISD::IS_FPCLASS, {MVT::f32, MVT::f64}, Legal);
441
443 Custom);
444
445 setOperationAction(ISD::FCANONICALIZE, {MVT::f32, MVT::f64}, Legal);
446
447 // FIXME: These IS_FPCLASS vector fp types are marked custom so it reaches
448 // scalarization code. Can be removed when IS_FPCLASS expand isn't called by
449 // default unless marked custom/legal.
451 {MVT::v2f32, MVT::v3f32, MVT::v4f32, MVT::v5f32,
452 MVT::v6f32, MVT::v7f32, MVT::v8f32, MVT::v16f32,
453 MVT::v2f64, MVT::v3f64, MVT::v4f64, MVT::v8f64,
454 MVT::v16f64},
455 Custom);
456
457 // Expand to fneg + fadd.
459
461 {MVT::v3i32, MVT::v3f32, MVT::v4i32, MVT::v4f32,
462 MVT::v5i32, MVT::v5f32, MVT::v6i32, MVT::v6f32,
463 MVT::v7i32, MVT::v7f32, MVT::v8i32, MVT::v8f32,
464 MVT::v9i32, MVT::v9f32, MVT::v10i32, MVT::v10f32,
465 MVT::v11i32, MVT::v11f32, MVT::v12i32, MVT::v12f32},
466 Custom);
467
470 {MVT::v2f32, MVT::v2i32, MVT::v3f32, MVT::v3i32, MVT::v4f32,
471 MVT::v4i32, MVT::v5f32, MVT::v5i32, MVT::v6f32, MVT::v6i32,
472 MVT::v7f32, MVT::v7i32, MVT::v8f32, MVT::v8i32, MVT::v9f32,
473 MVT::v9i32, MVT::v10i32, MVT::v10f32, MVT::v11i32, MVT::v11f32,
474 MVT::v12i32, MVT::v12f32, MVT::v16i32, MVT::v32f32, MVT::v32i32,
475 MVT::v2f64, MVT::v2i64, MVT::v3f64, MVT::v3i64, MVT::v4f64,
476 MVT::v4i64, MVT::v8f64, MVT::v8i64, MVT::v16f64, MVT::v16i64},
477 Custom);
478
480 Expand);
481 setOperationAction(ISD::FP_TO_FP16, {MVT::f64, MVT::f32}, Custom);
482
483 const MVT ScalarIntVTs[] = { MVT::i32, MVT::i64 };
484 for (MVT VT : ScalarIntVTs) {
485 // These should use [SU]DIVREM, so set them to expand
487 Expand);
488
489 // GPU does not have divrem function for signed or unsigned.
491
492 // GPU does not have [S|U]MUL_LOHI functions as a single instruction.
494
496
498 Expand);
499 }
500
501 // The hardware supports 32-bit FSHR, but not FSHL.
503
504 setOperationAction({ISD::ROTL, ISD::ROTR}, {MVT::i32, MVT::i64}, Expand);
505
507
512 MVT::i64, Custom);
514
516 Legal);
517
520 MVT::i64, Custom);
521
522 for (auto VT : {MVT::i8, MVT::i16})
524
525 static const MVT::SimpleValueType VectorIntTypes[] = {
526 MVT::v2i32, MVT::v3i32, MVT::v4i32, MVT::v5i32, MVT::v6i32, MVT::v7i32,
527 MVT::v9i32, MVT::v10i32, MVT::v11i32, MVT::v12i32};
528
529 for (MVT VT : VectorIntTypes) {
530 // Expand the following operations for the current type by default.
531 // clang-format off
551 VT, Expand);
552 // clang-format on
553 }
554
555 static const MVT::SimpleValueType FloatVectorTypes[] = {
556 MVT::v2f32, MVT::v3f32, MVT::v4f32, MVT::v5f32, MVT::v6f32, MVT::v7f32,
557 MVT::v9f32, MVT::v10f32, MVT::v11f32, MVT::v12f32};
558
559 for (MVT VT : FloatVectorTypes) {
572 VT, Expand);
573 }
574
575 // This causes using an unrolled select operation rather than expansion with
576 // bit operations. This is in general better, but the alternative using BFI
577 // instructions may be better if the select sources are SGPRs.
579 AddPromotedToType(ISD::SELECT, MVT::v2f32, MVT::v2i32);
580
582 AddPromotedToType(ISD::SELECT, MVT::v3f32, MVT::v3i32);
583
585 AddPromotedToType(ISD::SELECT, MVT::v4f32, MVT::v4i32);
586
588 AddPromotedToType(ISD::SELECT, MVT::v5f32, MVT::v5i32);
589
591 AddPromotedToType(ISD::SELECT, MVT::v6f32, MVT::v6i32);
592
594 AddPromotedToType(ISD::SELECT, MVT::v7f32, MVT::v7i32);
595
597 AddPromotedToType(ISD::SELECT, MVT::v9f32, MVT::v9i32);
598
600 AddPromotedToType(ISD::SELECT, MVT::v10f32, MVT::v10i32);
601
603 AddPromotedToType(ISD::SELECT, MVT::v11f32, MVT::v11i32);
604
606 AddPromotedToType(ISD::SELECT, MVT::v12f32, MVT::v12i32);
607
609 setJumpIsExpensive(true);
610
613
615
616 // We want to find all load dependencies for long chains of stores to enable
617 // merging into very wide vectors. The problem is with vectors with > 4
618 // elements. MergeConsecutiveStores will attempt to merge these because x8/x16
619 // vectors are a legal type, even though we have to split the loads
620 // usually. When we can more precisely specify load legality per address
621 // space, we should be able to make FindBetterChain/MergeConsecutiveStores
622 // smarter so that they can figure out what to do in 2 iterations without all
623 // N > 4 stores on the same chain.
625
626 // memcpy/memmove/memset are expanded in the IR, so we shouldn't need to worry
627 // about these during lowering.
628 MaxStoresPerMemcpy = 0xffffffff;
629 MaxStoresPerMemmove = 0xffffffff;
630 MaxStoresPerMemset = 0xffffffff;
631
632 // The expansion for 64-bit division is enormous.
634 addBypassSlowDiv(64, 32);
635
646
650}
651
652//===----------------------------------------------------------------------===//
653// Target Information
654//===----------------------------------------------------------------------===//
655
657static bool fnegFoldsIntoOpcode(unsigned Opc) {
658 switch (Opc) {
659 case ISD::FADD:
660 case ISD::FSUB:
661 case ISD::FMUL:
662 case ISD::FMA:
663 case ISD::FMAD:
664 case ISD::FMINNUM:
665 case ISD::FMAXNUM:
668 case ISD::FMINIMUM:
669 case ISD::FMAXIMUM:
670 case ISD::FMINIMUMNUM:
671 case ISD::FMAXIMUMNUM:
672 case ISD::SELECT:
673 case ISD::FSIN:
674 case ISD::FTRUNC:
675 case ISD::FRINT:
676 case ISD::FNEARBYINT:
677 case ISD::FROUNDEVEN:
679 case AMDGPUISD::RCP:
680 case AMDGPUISD::RCP_LEGACY:
681 case AMDGPUISD::RCP_IFLAG:
682 case AMDGPUISD::SIN_HW:
683 case AMDGPUISD::FMUL_LEGACY:
684 case AMDGPUISD::FMIN_LEGACY:
685 case AMDGPUISD::FMAX_LEGACY:
686 case AMDGPUISD::FMED3:
687 // TODO: handle llvm.amdgcn.fma.legacy
688 return true;
689 case ISD::BITCAST:
690 llvm_unreachable("bitcast is special cased");
691 default:
692 return false;
693 }
694}
695
696static bool fnegFoldsIntoOp(const SDNode *N) {
697 unsigned Opc = N->getOpcode();
698 if (Opc == ISD::BITCAST) {
699 // TODO: Is there a benefit to checking the conditions performFNegCombine
700 // does? We don't for the other cases.
701 SDValue BCSrc = N->getOperand(0);
702 if (BCSrc.getOpcode() == ISD::BUILD_VECTOR) {
703 return BCSrc.getNumOperands() == 2 &&
704 BCSrc.getOperand(1).getValueSizeInBits() == 32;
705 }
706
707 return BCSrc.getOpcode() == ISD::SELECT && BCSrc.getValueType() == MVT::f32;
708 }
709
710 return fnegFoldsIntoOpcode(Opc);
711}
712
713/// \p returns true if the operation will definitely need to use a 64-bit
714/// encoding, and thus will use a VOP3 encoding regardless of the source
715/// modifiers.
717static bool opMustUseVOP3Encoding(const SDNode *N, MVT VT) {
718 return (N->getNumOperands() > 2 && N->getOpcode() != ISD::SELECT) ||
719 VT == MVT::f64;
720}
721
722/// Return true if v_cndmask_b32 will support fabs/fneg source modifiers for the
723/// type for ISD::SELECT.
725static bool selectSupportsSourceMods(const SDNode *N) {
726 // TODO: Only applies if select will be vector
727 return N->getValueType(0) == MVT::f32;
728}
729
730// Most FP instructions support source modifiers, but this could be refined
731// slightly.
733static bool hasSourceMods(const SDNode *N) {
734 if (isa<MemSDNode>(N))
735 return false;
736
737 switch (N->getOpcode()) {
738 case ISD::CopyToReg:
739 case ISD::FDIV:
740 case ISD::FREM:
741 case ISD::INLINEASM:
743 case AMDGPUISD::DIV_SCALE:
745
746 // TODO: Should really be looking at the users of the bitcast. These are
747 // problematic because bitcasts are used to legalize all stores to integer
748 // types.
749 case ISD::BITCAST:
750 return false;
752 switch (N->getConstantOperandVal(0)) {
753 case Intrinsic::amdgcn_interp_p1:
754 case Intrinsic::amdgcn_interp_p2:
755 case Intrinsic::amdgcn_interp_mov:
756 case Intrinsic::amdgcn_interp_p1_f16:
757 case Intrinsic::amdgcn_interp_p2_f16:
758 return false;
759 default:
760 return true;
761 }
762 }
763 case ISD::SELECT:
765 default:
766 return true;
767 }
768}
769
771 unsigned CostThreshold) {
772 // Some users (such as 3-operand FMA/MAD) must use a VOP3 encoding, and thus
773 // it is truly free to use a source modifier in all cases. If there are
774 // multiple users but for each one will necessitate using VOP3, there will be
775 // a code size increase. Try to avoid increasing code size unless we know it
776 // will save on the instruction count.
777 unsigned NumMayIncreaseSize = 0;
778 MVT VT = N->getValueType(0).getScalarType().getSimpleVT();
779
780 assert(!N->use_empty());
781
782 // XXX - Should this limit number of uses to check?
783 for (const SDNode *U : N->users()) {
784 if (!hasSourceMods(U))
785 return false;
786
787 if (!opMustUseVOP3Encoding(U, VT)) {
788 if (++NumMayIncreaseSize > CostThreshold)
789 return false;
790 }
791 }
792
793 return true;
794}
795
797 ISD::NodeType ExtendKind) const {
798 assert(!VT.isVector() && "only scalar expected");
799
800 // Round to the next multiple of 32-bits.
801 unsigned Size = VT.getSizeInBits();
802 if (Size <= 32)
803 return MVT::i32;
804 return EVT::getIntegerVT(Context, 32 * ((Size + 31) / 32));
805}
806
808 return 32;
809}
810
812 return true;
813}
814
815// The backend supports 32 and 64 bit floating point immediates.
816// FIXME: Why are we reporting vectors of FP immediates as legal?
818 bool ForCodeSize) const {
819 return isTypeLegal(VT.getScalarType());
820}
821
822// We don't want to shrink f64 / f32 constants.
824 EVT ScalarVT = VT.getScalarType();
825 return (ScalarVT != MVT::f32 && ScalarVT != MVT::f64);
826}
827
829 SDNode *N, ISD::LoadExtType ExtTy, EVT NewVT,
830 std::optional<unsigned> ByteOffset) const {
831 // TODO: This may be worth removing. Check regression tests for diffs.
832 if (!TargetLoweringBase::shouldReduceLoadWidth(N, ExtTy, NewVT, ByteOffset))
833 return false;
834
835 unsigned NewSize = NewVT.getStoreSizeInBits();
836
837 // If we are reducing to a 32-bit load or a smaller multi-dword load,
838 // this is always better.
839 if (NewSize >= 32)
840 return true;
841
842 EVT OldVT = N->getValueType(0);
843 unsigned OldSize = OldVT.getStoreSizeInBits();
844
846 unsigned AS = MN->getAddressSpace();
847 // Do not shrink an aligned scalar load to sub-dword.
848 // Scalar engine cannot do sub-dword loads.
849 // Do not enable for gfx1250+ even though it has sub-dword loads because
850 // this will convert:
851 // i16 = trunc (zextload i16->i32)
852 // to:
853 // i16 = (load i16)
854 // This transformation will be reversed by LowerLOAD resulting in an infinite
855 // loop. Also, tablegen already has a pattern to match zextload i16->i32, but
856 // load i16 will not be matched since there is no instruction that does it.
857 if (OldSize >= 32 && NewSize < 32 && MN->getAlign() >= Align(4) &&
861 MN->isInvariant())) &&
863 return false;
864
865 // Don't produce extloads from sub 32-bit types. SI doesn't have scalar
866 // extloads, so doing one requires using a buffer_load. In cases where we
867 // still couldn't use a scalar load, using the wider load shouldn't really
868 // hurt anything.
869
870 // If the old size already had to be an extload, there's no harm in continuing
871 // to reduce the width.
872 return (OldSize < 32);
873}
874
876 const SelectionDAG &DAG,
877 const MachineMemOperand &MMO) const {
878
879 assert(LoadTy.getSizeInBits() == CastTy.getSizeInBits());
880
881 if (LoadTy.getScalarType() == MVT::i32)
882 return false;
883
884 unsigned LScalarSize = LoadTy.getScalarSizeInBits();
885 unsigned CastScalarSize = CastTy.getScalarSizeInBits();
886
887 if ((LScalarSize >= CastScalarSize) && (CastScalarSize < 32))
888 return false;
889
890 unsigned Fast = 0;
892 CastTy, MMO, &Fast) &&
893 Fast;
894}
895
896// SI+ has instructions for cttz / ctlz for 32-bit values. This is probably also
897// profitable with the expansion for 64-bit since it's generally good to
898// speculate things.
900 return true;
901}
902
904 return true;
905}
906
908 switch (N->getOpcode()) {
909 case ISD::EntryToken:
910 case ISD::TokenFactor:
911 return true;
913 unsigned IntrID = N->getConstantOperandVal(0);
915 }
917 unsigned IntrID = N->getConstantOperandVal(1);
919 }
920 case ISD::LOAD:
921 if (cast<LoadSDNode>(N)->getMemOperand()->getAddrSpace() ==
923 return true;
924 return false;
925 case AMDGPUISD::SETCC: // ballot-style instruction
926 return true;
927 }
928 return false;
929}
930
932 SDValue Op, SelectionDAG &DAG, bool LegalOperations, bool ForCodeSize,
933 NegatibleCost &Cost, unsigned Depth) const {
934
935 switch (Op.getOpcode()) {
936 case ISD::FMA:
937 case ISD::FMAD: {
938 // Negating a fma is not free if it has users without source mods.
939 if (!allUsesHaveSourceMods(Op.getNode()))
940 return SDValue();
941 break;
942 }
943 case AMDGPUISD::RCP: {
944 SDValue Src = Op.getOperand(0);
945 EVT VT = Op.getValueType();
946 SDLoc SL(Op);
947
948 SDValue NegSrc = getNegatedExpression(Src, DAG, LegalOperations,
949 ForCodeSize, Cost, Depth + 1);
950 if (NegSrc)
951 return DAG.getNode(AMDGPUISD::RCP, SL, VT, NegSrc, Op->getFlags());
952 return SDValue();
953 }
954 default:
955 break;
956 }
957
958 return TargetLowering::getNegatedExpression(Op, DAG, LegalOperations,
959 ForCodeSize, Cost, Depth);
960}
961
962//===---------------------------------------------------------------------===//
963// Target Properties
964//===---------------------------------------------------------------------===//
965
968
969 // Packed operations do not have a fabs modifier.
970 // Report this based on the end legalized type.
971 return VT == MVT::f32 || VT == MVT::f64 || VT == MVT::f16 || VT == MVT::bf16;
972}
973
976 // Report this based on the end legalized type.
977 VT = VT.getScalarType();
978 return VT == MVT::f32 || VT == MVT::f64 || VT == MVT::f16 || VT == MVT::bf16;
979}
980
982 unsigned NumElem,
983 unsigned AS) const {
984 return true;
985}
986
988 // There are few operations which truly have vector input operands. Any vector
989 // operation is going to involve operations on each component, and a
990 // build_vector will be a copy per element, so it always makes sense to use a
991 // build_vector input in place of the extracted element to avoid a copy into a
992 // super register.
993 //
994 // We should probably only do this if all users are extracts only, but this
995 // should be the common case.
996 return true;
997}
998
1000 // Truncate is just accessing a subregister.
1001
1002 unsigned SrcSize = Source.getSizeInBits();
1003 unsigned DestSize = Dest.getSizeInBits();
1004
1005 return DestSize < SrcSize && DestSize % 32 == 0 ;
1006}
1007
1009 // Truncate is just accessing a subregister.
1010
1011 unsigned SrcSize = Source->getScalarSizeInBits();
1012 unsigned DestSize = Dest->getScalarSizeInBits();
1013
1014 if (DestSize== 16 && Subtarget->has16BitInsts())
1015 return SrcSize >= 32;
1016
1017 return DestSize < SrcSize && DestSize % 32 == 0;
1018}
1019
1021 unsigned SrcSize = Src->getScalarSizeInBits();
1022 unsigned DestSize = Dest->getScalarSizeInBits();
1023
1024 if (SrcSize == 16 && Subtarget->has16BitInsts())
1025 return DestSize >= 32;
1026
1027 return SrcSize == 32 && DestSize == 64;
1028}
1029
1031 // Any register load of a 64-bit value really requires 2 32-bit moves. For all
1032 // practical purposes, the extra mov 0 to load a 64-bit is free. As used,
1033 // this will enable reducing 64-bit operations the 32-bit, which is always
1034 // good.
1035
1036 if (Src == MVT::i16)
1037 return Dest == MVT::i32 ||Dest == MVT::i64 ;
1038
1039 return Src == MVT::i32 && Dest == MVT::i64;
1040}
1041
1043 EVT DestVT) const {
1044 switch (N->getOpcode()) {
1045 case ISD::ABS:
1046 case ISD::ADD:
1047 case ISD::SUB:
1048 case ISD::SHL:
1049 case ISD::SRL:
1050 case ISD::SRA:
1051 case ISD::AND:
1052 case ISD::OR:
1053 case ISD::XOR:
1054 case ISD::MUL:
1055 case ISD::SETCC:
1056 case ISD::SELECT:
1057 case ISD::SMIN:
1058 case ISD::SMAX:
1059 case ISD::UMIN:
1060 case ISD::UMAX:
1061 case ISD::USUBSAT:
1062 case ISD::UADDSAT:
1063 if (isTypeLegal(MVT::i16) &&
1064 (!DestVT.isVector() ||
1065 !isOperationLegal(ISD::ADD, MVT::v2i16))) { // Check if VOP3P
1066 // Don't narrow back down to i16 if promoted to i32 already.
1067 if (!N->isDivergent() && DestVT.isInteger() &&
1068 DestVT.getScalarSizeInBits() > 1 &&
1069 DestVT.getScalarSizeInBits() <= 16 &&
1070 SrcVT.getScalarSizeInBits() > 16) {
1071 return false;
1072 }
1073 }
1074 return true;
1075 default:
1076 break;
1077 }
1078
1079 // There aren't really 64-bit registers, but pairs of 32-bit ones and only a
1080 // limited number of native 64-bit operations. Shrinking an operation to fit
1081 // in a single 32-bit register should always be helpful. As currently used,
1082 // this is much less general than the name suggests, and is only used in
1083 // places trying to reduce the sizes of loads. Shrinking loads to < 32-bits is
1084 // not profitable, and may actually be harmful.
1085 if (isa<LoadSDNode>(N))
1086 return SrcVT.getSizeInBits() > 32 && DestVT.getSizeInBits() == 32;
1087
1088 return true;
1089}
1090
1092 const SDNode* N, CombineLevel Level) const {
1093 assert((N->getOpcode() == ISD::SHL || N->getOpcode() == ISD::SRA ||
1094 N->getOpcode() == ISD::SRL) &&
1095 "Expected shift op");
1096
1097 SDValue ShiftLHS = N->getOperand(0);
1098 if (!ShiftLHS->hasOneUse())
1099 return false;
1100
1101 if (ShiftLHS.getOpcode() == ISD::SIGN_EXTEND &&
1102 !ShiftLHS.getOperand(0)->hasOneUse())
1103 return false;
1104
1105 // Always commute pre-type legalization and right shifts.
1106 // We're looking for shl(or(x,y),z) patterns.
1108 N->getOpcode() != ISD::SHL || N->getOperand(0).getOpcode() != ISD::OR)
1109 return true;
1110
1111 // If only user is a i32 right-shift, then don't destroy a BFE pattern.
1112 if (N->getValueType(0) == MVT::i32 && N->hasOneUse() &&
1113 (N->user_begin()->getOpcode() == ISD::SRA ||
1114 N->user_begin()->getOpcode() == ISD::SRL))
1115 return false;
1116
1117 // Don't destroy or(shl(load_zext(),c), load_zext()) patterns.
1118 auto IsShiftAndLoad = [](SDValue LHS, SDValue RHS) {
1119 if (LHS.getOpcode() != ISD::SHL)
1120 return false;
1121 auto *RHSLd = dyn_cast<LoadSDNode>(RHS);
1122 auto *LHS0 = dyn_cast<LoadSDNode>(LHS.getOperand(0));
1123 auto *LHS1 = dyn_cast<ConstantSDNode>(LHS.getOperand(1));
1124 return LHS0 && LHS1 && RHSLd && LHS0->getExtensionType() == ISD::ZEXTLOAD &&
1125 LHS1->getAPIntValue() == LHS0->getMemoryVT().getScalarSizeInBits() &&
1126 RHSLd->getExtensionType() == ISD::ZEXTLOAD;
1127 };
1128 SDValue LHS = N->getOperand(0).getOperand(0);
1129 SDValue RHS = N->getOperand(0).getOperand(1);
1130 return !(IsShiftAndLoad(LHS, RHS) || IsShiftAndLoad(RHS, LHS));
1131}
1132
1133//===---------------------------------------------------------------------===//
1134// TargetLowering Callbacks
1135//===---------------------------------------------------------------------===//
1136
1138 bool IsVarArg) {
1139 switch (CC) {
1147 return CC_AMDGPU;
1150 return CC_AMDGPU_CS_CHAIN;
1151 case CallingConv::C:
1152 case CallingConv::Fast:
1153 case CallingConv::Cold:
1154 return CC_AMDGPU_Func;
1157 return CC_SI_Gfx;
1160 default:
1161 reportFatalUsageError("unsupported calling convention for call");
1162 }
1163}
1164
1166 bool IsVarArg) {
1167 switch (CC) {
1170 llvm_unreachable("kernels should not be handled here");
1180 return RetCC_SI_Shader;
1183 return RetCC_SI_Gfx;
1184 case CallingConv::C:
1185 case CallingConv::Fast:
1186 case CallingConv::Cold:
1187 return RetCC_AMDGPU_Func;
1188 default:
1189 reportFatalUsageError("unsupported calling convention");
1190 }
1191}
1192
1193/// The SelectionDAGBuilder will automatically promote function arguments
1194/// with illegal types. However, this does not work for the AMDGPU targets
1195/// since the function arguments are stored in memory as these illegal types.
1196/// In order to handle this properly we need to get the original types sizes
1197/// from the LLVM IR Function and fixup the ISD:InputArg values before
1198/// passing them to AnalyzeFormalArguments()
1199
1200/// When the SelectionDAGBuilder computes the Ins, it takes care of splitting
1201/// input values across multiple registers. Each item in the Ins array
1202/// represents a single value that will be stored in registers. Ins[x].VT is
1203/// the value type of the value that will be stored in the register, so
1204/// whatever SDNode we lower the argument to needs to be this type.
1205///
1206/// In order to correctly lower the arguments we need to know the size of each
1207/// argument. Since Ins[x].VT gives us the size of the register that will
1208/// hold the value, we need to look at Ins[x].ArgVT to see the 'real' type
1209/// for the original function argument so that we can deduce the correct memory
1210/// type to use for Ins[x]. In most cases the correct memory type will be
1211/// Ins[x].ArgVT. However, this will not always be the case. If, for example,
1212/// we have a kernel argument of type v8i8, this argument will be split into
1213/// 8 parts and each part will be represented by its own item in the Ins array.
1214/// For each part the Ins[x].ArgVT will be the v8i8, which is the full type of
1215/// the argument before it was split. From this, we deduce that the memory type
1216/// for each individual part is i8. We pass the memory type as LocVT to the
1217/// calling convention analysis function and the register type (Ins[x].VT) as
1218/// the ValVT.
1220 CCState &State,
1221 const SmallVectorImpl<ISD::InputArg> &Ins) const {
1222 const MachineFunction &MF = State.getMachineFunction();
1223 const Function &Fn = MF.getFunction();
1224 LLVMContext &Ctx = Fn.getContext();
1225 const unsigned ExplicitOffset = Subtarget->getExplicitKernelArgOffset();
1227
1228 Align MaxAlign = Align(1);
1229 uint64_t ExplicitArgOffset = 0;
1230 const DataLayout &DL = Fn.getDataLayout();
1231
1232 unsigned InIndex = 0;
1233
1234 for (const Argument &Arg : Fn.args()) {
1235 const bool IsByRef = Arg.hasByRefAttr();
1236 Type *BaseArgTy = Arg.getType();
1237 Type *MemArgTy = IsByRef ? Arg.getParamByRefType() : BaseArgTy;
1238 Align Alignment = DL.getValueOrABITypeAlignment(
1239 IsByRef ? Arg.getParamAlign() : std::nullopt, MemArgTy);
1240 MaxAlign = std::max(Alignment, MaxAlign);
1241 uint64_t AllocSize = DL.getTypeAllocSize(MemArgTy);
1242
1243 uint64_t ArgOffset = alignTo(ExplicitArgOffset, Alignment) + ExplicitOffset;
1244 ExplicitArgOffset = alignTo(ExplicitArgOffset, Alignment) + AllocSize;
1245
1246 // We're basically throwing away everything passed into us and starting over
1247 // to get accurate in-memory offsets. The "PartOffset" is completely useless
1248 // to us as computed in Ins.
1249 //
1250 // We also need to figure out what type legalization is trying to do to get
1251 // the correct memory offsets.
1252
1253 SmallVector<EVT, 16> ValueVTs;
1255 ComputeValueVTs(*this, DL, BaseArgTy, ValueVTs, /*MemVTs=*/nullptr,
1256 &Offsets, ArgOffset);
1257
1258 for (unsigned Value = 0, NumValues = ValueVTs.size();
1259 Value != NumValues; ++Value) {
1260 uint64_t BasePartOffset = Offsets[Value];
1261
1262 EVT ArgVT = ValueVTs[Value];
1263 EVT MemVT = ArgVT;
1264 MVT RegisterVT = getRegisterTypeForCallingConv(Ctx, CC, ArgVT);
1265 unsigned NumRegs = getNumRegistersForCallingConv(Ctx, CC, ArgVT);
1266
1267 if (NumRegs == 1) {
1268 // This argument is not split, so the IR type is the memory type.
1269 if (ArgVT.isExtended()) {
1270 // We have an extended type, like i24, so we should just use the
1271 // register type.
1272 MemVT = RegisterVT;
1273 } else {
1274 MemVT = ArgVT;
1275 }
1276 } else if (ArgVT.isVector() && RegisterVT.isVector() &&
1277 ArgVT.getScalarType() == RegisterVT.getScalarType()) {
1278 assert(ArgVT.getVectorNumElements() > RegisterVT.getVectorNumElements());
1279 // We have a vector value which has been split into a vector with
1280 // the same scalar type, but fewer elements. This should handle
1281 // all the floating-point vector types.
1282 MemVT = RegisterVT;
1283 } else if (ArgVT.isVector() &&
1284 ArgVT.getVectorNumElements() == NumRegs) {
1285 // This arg has been split so that each element is stored in a separate
1286 // register.
1287 MemVT = ArgVT.getScalarType();
1288 } else if (ArgVT.isExtended()) {
1289 // We have an extended type, like i65.
1290 MemVT = RegisterVT;
1291 } else {
1292 unsigned MemoryBits = ArgVT.getStoreSizeInBits() / NumRegs;
1293 assert(ArgVT.getStoreSizeInBits() % NumRegs == 0);
1294 if (RegisterVT.isInteger()) {
1295 MemVT = EVT::getIntegerVT(State.getContext(), MemoryBits);
1296 } else if (RegisterVT.isVector()) {
1297 assert(!RegisterVT.getScalarType().isFloatingPoint());
1298 unsigned NumElements = RegisterVT.getVectorNumElements();
1299 assert(MemoryBits % NumElements == 0);
1300 // This vector type has been split into another vector type with
1301 // a different elements size.
1302 EVT ScalarVT = EVT::getIntegerVT(State.getContext(),
1303 MemoryBits / NumElements);
1304 MemVT = EVT::getVectorVT(State.getContext(), ScalarVT, NumElements);
1305 } else {
1306 llvm_unreachable("cannot deduce memory type.");
1307 }
1308 }
1309
1310 // Convert one element vectors to scalar.
1311 if (MemVT.isVector() && MemVT.getVectorNumElements() == 1)
1312 MemVT = MemVT.getScalarType();
1313
1314 // Round up vec3/vec5 argument.
1315 if (MemVT.isVector() && !MemVT.isPow2VectorType()) {
1316 MemVT = MemVT.getPow2VectorType(State.getContext());
1317 } else if (!MemVT.isSimple() && !MemVT.isVector()) {
1318 MemVT = MemVT.getRoundIntegerType(State.getContext());
1319 }
1320
1321 unsigned PartOffset = 0;
1322 for (unsigned i = 0; i != NumRegs; ++i) {
1323 State.addLoc(CCValAssign::getCustomMem(InIndex++, RegisterVT,
1324 BasePartOffset + PartOffset,
1325 MemVT.getSimpleVT(),
1327 PartOffset += MemVT.getStoreSize();
1328 }
1329 }
1330 }
1331}
1332
1334 SDValue Chain, CallingConv::ID CallConv,
1335 bool isVarArg,
1337 const SmallVectorImpl<SDValue> &OutVals,
1338 const SDLoc &DL, SelectionDAG &DAG) const {
1339 // FIXME: Fails for r600 tests
1340 //assert(!isVarArg && Outs.empty() && OutVals.empty() &&
1341 // "wave terminate should not have return values");
1342 return DAG.getNode(AMDGPUISD::ENDPGM, DL, MVT::Other, Chain);
1343}
1344
1345//===---------------------------------------------------------------------===//
1346// Target specific lowering
1347//===---------------------------------------------------------------------===//
1348
1349/// Selects the correct CCAssignFn for a given CallingConvention value.
1354
1359
1361 SelectionDAG &DAG,
1362 MachineFrameInfo &MFI,
1363 int ClobberedFI) const {
1364 SmallVector<SDValue, 8> ArgChains;
1365 int64_t FirstByte = MFI.getObjectOffset(ClobberedFI);
1366 int64_t LastByte = FirstByte + MFI.getObjectSize(ClobberedFI) - 1;
1367
1368 // Include the original chain at the beginning of the list. When this is
1369 // used by target LowerCall hooks, this helps legalize find the
1370 // CALLSEQ_BEGIN node.
1371 ArgChains.push_back(Chain);
1372
1373 // Add a chain value for each stack argument corresponding
1374 for (SDNode *U : DAG.getEntryNode().getNode()->users()) {
1375 if (LoadSDNode *L = dyn_cast<LoadSDNode>(U)) {
1376 if (FrameIndexSDNode *FI = dyn_cast<FrameIndexSDNode>(L->getBasePtr())) {
1377 if (FI->getIndex() < 0) {
1378 int64_t InFirstByte = MFI.getObjectOffset(FI->getIndex());
1379 int64_t InLastByte = InFirstByte;
1380 InLastByte += MFI.getObjectSize(FI->getIndex()) - 1;
1381
1382 if ((InFirstByte <= FirstByte && FirstByte <= InLastByte) ||
1383 (FirstByte <= InFirstByte && InFirstByte <= LastByte))
1384 ArgChains.push_back(SDValue(L, 1));
1385 }
1386 }
1387 }
1388 }
1389
1390 // Build a tokenfactor for all the chains.
1391 return DAG.getNode(ISD::TokenFactor, SDLoc(Chain), MVT::Other, ArgChains);
1392}
1393
1396 StringRef Reason) const {
1397 SDValue Callee = CLI.Callee;
1398 SelectionDAG &DAG = CLI.DAG;
1399
1400 const Function &Fn = DAG.getMachineFunction().getFunction();
1401
1402 StringRef FuncName("<unknown>");
1403
1405 FuncName = G->getSymbol();
1406 else if (const GlobalAddressSDNode *G = dyn_cast<GlobalAddressSDNode>(Callee))
1407 FuncName = G->getGlobal()->getName();
1408
1409 DAG.getContext()->diagnose(
1410 DiagnosticInfoUnsupported(Fn, Reason + FuncName, CLI.DL.getDebugLoc()));
1411
1412 if (!CLI.IsTailCall) {
1413 for (ISD::InputArg &Arg : CLI.Ins)
1414 InVals.push_back(DAG.getPOISON(Arg.VT));
1415 }
1416
1417 // FIXME: Hack because R600 doesn't handle callseq pseudos yet.
1418 if (getTargetMachine().getTargetTriple().getArch() == Triple::r600)
1419 return CLI.Chain;
1420
1421 SDValue Chain = DAG.getCALLSEQ_START(CLI.Chain, 0, 0, CLI.DL);
1422 return DAG.getCALLSEQ_END(Chain, 0, 0, /*InGlue=*/SDValue(), CLI.DL);
1423}
1424
1426 SmallVectorImpl<SDValue> &InVals) const {
1427 return lowerUnhandledCall(CLI, InVals, "unsupported call to function ");
1428}
1429
1431 SelectionDAG &DAG) const {
1432 const Function &Fn = DAG.getMachineFunction().getFunction();
1433
1435 Fn, "unsupported dynamic alloca", SDLoc(Op).getDebugLoc()));
1436 auto Ops = {DAG.getConstant(0, SDLoc(), Op.getValueType()), Op.getOperand(0)};
1437 return DAG.getMergeValues(Ops, SDLoc());
1438}
1439
1441 SelectionDAG &DAG) const {
1442 switch (Op.getOpcode()) {
1443 default:
1444 Op->print(errs(), &DAG);
1445 llvm_unreachable("Custom lowering code for this "
1446 "instruction is not implemented yet!");
1447 break;
1449 case ISD::CONCAT_VECTORS: return LowerCONCAT_VECTORS(Op, DAG);
1451 case ISD::UDIVREM: return LowerUDIVREM(Op, DAG);
1452 case ISD::SDIVREM:
1453 return LowerSDIVREM(Op, DAG);
1454 case ISD::FCEIL: return LowerFCEIL(Op, DAG);
1455 case ISD::FTRUNC: return LowerFTRUNC(Op, DAG);
1456 case ISD::FRINT: return LowerFRINT(Op, DAG);
1457 case ISD::FNEARBYINT: return LowerFNEARBYINT(Op, DAG);
1458 case ISD::FROUNDEVEN:
1459 return LowerFROUNDEVEN(Op, DAG);
1460 case ISD::FROUND: return LowerFROUND(Op, DAG);
1461 case ISD::FFLOOR: return LowerFFLOOR(Op, DAG);
1462 case ISD::FLOG2:
1463 return LowerFLOG2(Op, DAG);
1464 case ISD::FLOG:
1465 case ISD::FLOG10:
1466 return LowerFLOGCommon(Op, DAG);
1467 case ISD::FEXP:
1468 case ISD::FEXP10:
1469 return lowerFEXP(Op, DAG);
1470 case ISD::FEXP2:
1471 return lowerFEXP2(Op, DAG);
1472 case ISD::FPOW:
1473 return lowerFPOW(Op, DAG);
1474 case ISD::SINT_TO_FP: return LowerSINT_TO_FP(Op, DAG);
1475 case ISD::UINT_TO_FP: return LowerUINT_TO_FP(Op, DAG);
1476 case ISD::FP_TO_FP16: return LowerFP_TO_FP16(Op, DAG);
1477 case ISD::FP_TO_SINT:
1478 case ISD::FP_TO_UINT:
1479 return LowerFP_TO_INT(Op, DAG);
1482 return LowerFP_TO_INT_SAT(Op, DAG);
1483 case ISD::CTTZ:
1485 case ISD::CTLZ:
1487 return LowerCTLZ_CTTZ(Op, DAG);
1488 case ISD::CTLS:
1489 return LowerCTLS(Op, DAG);
1491 }
1492 return Op;
1493}
1494
1497 SelectionDAG &DAG) const {
1498 switch (N->getOpcode()) {
1500 // Different parts of legalization seem to interpret which type of
1501 // sign_extend_inreg is the one to check for custom lowering. The extended
1502 // from type is what really matters, but some places check for custom
1503 // lowering of the result type. This results in trying to use
1504 // ReplaceNodeResults to sext_in_reg to an illegal type, so we'll just do
1505 // nothing here and let the illegal result integer be handled normally.
1506 return;
1507 case ISD::FLOG2:
1508 if (SDValue Lowered = LowerFLOG2(SDValue(N, 0), DAG))
1509 Results.push_back(Lowered);
1510 return;
1511 case ISD::FLOG:
1512 case ISD::FLOG10:
1513 if (SDValue Lowered = LowerFLOGCommon(SDValue(N, 0), DAG))
1514 Results.push_back(Lowered);
1515 return;
1516 case ISD::FEXP2:
1517 if (SDValue Lowered = lowerFEXP2(SDValue(N, 0), DAG))
1518 Results.push_back(Lowered);
1519 return;
1520 case ISD::FEXP:
1521 case ISD::FEXP10:
1522 if (SDValue Lowered = lowerFEXP(SDValue(N, 0), DAG))
1523 Results.push_back(Lowered);
1524 return;
1525 case ISD::CTLZ:
1527 if (auto Lowered = lowerCTLZResults(SDValue(N, 0u), DAG))
1528 Results.push_back(Lowered);
1529 return;
1530 default:
1531 return;
1532 }
1533}
1534
1536 SelectionDAG &DAG) const {
1538 SDLoc SL(Op);
1539 EVT VT = Op.getValueType();
1540 return DAG.getTargetBlockAddress(BA->getBlockAddress(), VT, BA->getOffset(),
1541 BA->getTargetFlags());
1542}
1543
1545 SDValue Op,
1546 SelectionDAG &DAG) const {
1547
1548 const DataLayout &DL = DAG.getDataLayout();
1550 const GlobalValue *GV = G->getGlobal();
1551
1552 if (G->getAddressSpace() == AMDGPUAS::BARRIER) {
1553 const GlobalVariable *GVar = cast<GlobalVariable>(GV);
1554
1555 if (!AMDGPU::isNamedBarrier(*GVar)) {
1556 const Function &Fn = DAG.getMachineFunction().getFunction();
1558 Fn, "unsupported use of BARRIER address space",
1560 return DAG.getPOISON(Op.getValueType());
1561 }
1562
1563 unsigned Offset = MFI->allocateBarrierGlobal(DL, *cast<GlobalVariable>(GV));
1564 return DAG.getConstant(Offset, SDLoc(Op), Op.getValueType());
1565 }
1566
1567 if (!MFI->isModuleEntryFunction()) {
1568 if (std::optional<uint32_t> Address =
1571 return DAG.getConstant(*Address, SDLoc(Op), Op.getValueType());
1572 }
1573 }
1574
1575 if (G->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS ||
1576 G->getAddressSpace() == AMDGPUAS::REGION_ADDRESS) {
1577 if (!MFI->isModuleEntryFunction() &&
1578 GV->getName() != "llvm.amdgcn.module.lds") {
1579 SDLoc DL(Op);
1580 const Function &Fn = DAG.getMachineFunction().getFunction();
1582 Fn, "local memory global used by non-kernel function",
1583 DL.getDebugLoc(), DS_Warning));
1584
1585 // We currently don't have a way to correctly allocate LDS objects that
1586 // aren't directly associated with a kernel. We do force inlining of
1587 // functions that use local objects. However, if these dead functions are
1588 // not eliminated, we don't want a compile time error. Just emit a warning
1589 // and a trap, since there should be no callable path here.
1590 SDValue Trap = DAG.getNode(ISD::TRAP, DL, MVT::Other, DAG.getEntryNode());
1591 SDValue OutputChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other,
1592 Trap, DAG.getRoot());
1593 DAG.setRoot(OutputChain);
1594 return DAG.getPOISON(Op.getValueType());
1595 }
1596
1597 // TODO: We could emit code to handle the initialization somewhere.
1598 // We ignore the initializer for now and legalize it to allow selection.
1599 // The initializer will anyway get errored out during assembly emission.
1600 unsigned Offset = MFI->allocateLDSGlobal(DL, *cast<GlobalVariable>(GV));
1601 // A constant byte offset (e.g. from a GEP into an array of named barriers)
1602 // folds directly into the allocated LDS address.
1603 return DAG.getConstant(Offset + G->getOffset(), SDLoc(Op),
1604 Op.getValueType());
1605 }
1606 return SDValue();
1607}
1608
1610 SelectionDAG &DAG) const {
1612 SDLoc SL(Op);
1613
1614 EVT VT = Op.getValueType();
1615 if (VT.getVectorElementType().getSizeInBits() < 32) {
1616 unsigned OpBitSize = Op.getOperand(0).getValueType().getSizeInBits();
1617 if (OpBitSize >= 32 && OpBitSize % 32 == 0) {
1618 unsigned NewNumElt = OpBitSize / 32;
1619 EVT NewEltVT = (NewNumElt == 1) ? MVT::i32
1621 MVT::i32, NewNumElt);
1622 for (const SDUse &U : Op->ops()) {
1623 SDValue In = U.get();
1624 SDValue NewIn = DAG.getNode(ISD::BITCAST, SL, NewEltVT, In);
1625 if (NewNumElt > 1)
1626 DAG.ExtractVectorElements(NewIn, Args);
1627 else
1628 Args.push_back(NewIn);
1629 }
1630
1631 EVT NewVT = EVT::getVectorVT(*DAG.getContext(), MVT::i32,
1632 NewNumElt * Op.getNumOperands());
1633 SDValue BV = DAG.getBuildVector(NewVT, SL, Args);
1634 return DAG.getNode(ISD::BITCAST, SL, VT, BV);
1635 }
1636 }
1637
1638 for (const SDUse &U : Op->ops())
1639 DAG.ExtractVectorElements(U.get(), Args);
1640
1641 return DAG.getBuildVector(Op.getValueType(), SL, Args);
1642}
1643
1645 SelectionDAG &DAG) const {
1646 SDLoc SL(Op);
1648 unsigned Start = Op.getConstantOperandVal(1);
1649 EVT VT = Op.getValueType();
1650 EVT SrcVT = Op.getOperand(0).getValueType();
1651
1652 if (VT.getScalarSizeInBits() == 16 && Start % 2 == 0) {
1653 unsigned NumElt = VT.getVectorNumElements();
1654 unsigned NumSrcElt = SrcVT.getVectorNumElements();
1655 assert(NumElt % 2 == 0 && NumSrcElt % 2 == 0 && "expect legal types");
1656
1657 // Extract 32-bit registers at a time.
1658 EVT NewSrcVT = EVT::getVectorVT(*DAG.getContext(), MVT::i32, NumSrcElt / 2);
1659 EVT NewVT = NumElt == 2
1660 ? MVT::i32
1661 : EVT::getVectorVT(*DAG.getContext(), MVT::i32, NumElt / 2);
1662 SDValue Tmp = DAG.getNode(ISD::BITCAST, SL, NewSrcVT, Op.getOperand(0));
1663
1664 DAG.ExtractVectorElements(Tmp, Args, Start / 2, NumElt / 2);
1665 if (NumElt == 2)
1666 Tmp = Args[0];
1667 else
1668 Tmp = DAG.getBuildVector(NewVT, SL, Args);
1669
1670 return DAG.getNode(ISD::BITCAST, SL, VT, Tmp);
1671 }
1672
1673 DAG.ExtractVectorElements(Op.getOperand(0), Args, Start,
1675
1676 return DAG.getBuildVector(Op.getValueType(), SL, Args);
1677}
1678
1679// TODO: Handle fabs too
1681 if (Val.getOpcode() == ISD::FNEG)
1682 return Val.getOperand(0);
1683
1684 return Val;
1685}
1686
1687// SelectionDAG twin of AMDGPUCombinerHelper::canIgnoreLegacyMinMaxTies.
1689 SDNodeFlags Flags, SDValue LHS,
1690 SDValue RHS) {
1691 return Flags.hasNoSignedZeros() || DAG.isKnownNeverLogicalZero(LHS) ||
1693}
1694
1696 const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, SDValue True,
1697 SDValue False, SDValue CC, SDNodeFlags Flags, DAGCombinerInfo &DCI) const {
1698 SelectionDAG &DAG = DCI.DAG;
1699 ISD::CondCode CCOpcode = cast<CondCodeSDNode>(CC)->get();
1700 assert(CCOpcode != ISD::SETCC_INVALID && "Invalid setcc condcode!");
1701
1702 switch (CCOpcode) {
1703 case ISD::SETOLE:
1704 case ISD::SETOLT:
1705 case ISD::SETLE:
1706 case ISD::SETLT:
1707 case ISD::SETOGE:
1708 case ISD::SETOGT:
1709 case ISD::SETGE:
1710 case ISD::SETGT:
1711 // Only do this after legalization to avoid interfering with other combines
1712 // which might occur.
1714 !DCI.isCalledByLegalizer())
1715 return SDValue();
1716 break;
1717 default:
1718 break;
1719 }
1720
1721 // Canonicalize so the select returns the compare's LHS on a true predicate.
1722 if (LHS != True)
1723 CCOpcode = ISD::getSetCCInverse(CCOpcode, VT);
1724
1725 unsigned Opc;
1726 bool Swap; // Emit (rhs, lhs) instead of (lhs, rhs).
1727 switch (CCOpcode) {
1728 case ISD::SETOLT:
1729 case ISD::SETLT:
1730 case ISD::SETOLE:
1731 Opc = AMDGPUISD::FMIN_LEGACY;
1732 Swap = false;
1733 break;
1734 case ISD::SETULE:
1735 case ISD::SETLE:
1736 case ISD::SETULT:
1737 Opc = AMDGPUISD::FMIN_LEGACY;
1738 Swap = true;
1739 break;
1740 case ISD::SETOGE:
1741 case ISD::SETGE:
1742 case ISD::SETOGT:
1743 Opc = AMDGPUISD::FMAX_LEGACY;
1744 Swap = false;
1745 break;
1746 case ISD::SETUGT:
1747 case ISD::SETGT:
1748 case ISD::SETUGE:
1749 Opc = AMDGPUISD::FMAX_LEGACY;
1750 Swap = true;
1751 break;
1752 default:
1753 return SDValue();
1754 }
1755
1756 // For these predicates the NaN-correct operand order is the signed zero
1757 // tie-incorrect one, so the fold needs the tie to be unobservable.
1758 if ((CCOpcode == ISD::SETOLE || CCOpcode == ISD::SETULT ||
1759 CCOpcode == ISD::SETOGT || CCOpcode == ISD::SETUGE) &&
1760 !canIgnoreLegacyMinMaxTies(DAG, Flags, LHS, RHS))
1761 return SDValue();
1762
1763 if (Swap)
1764 std::swap(LHS, RHS);
1765 return DAG.getNode(Opc, DL, VT, LHS, RHS, Flags);
1766}
1767
1768/// Generate Min/Max node
1770 const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, SDValue True,
1771 SDValue False, SDValue CC, SDNodeFlags Flags, DAGCombinerInfo &DCI) const {
1772 if ((LHS == True && RHS == False) || (LHS == False && RHS == True))
1773 return combineFMinMaxLegacyImpl(DL, VT, LHS, RHS, True, False, CC, Flags,
1774 DCI);
1775
1776 SelectionDAG &DAG = DCI.DAG;
1777
1778 // If we can't directly match this, try to see if we can fold an fneg to
1779 // match.
1780
1783 SDValue NegTrue = peekFNeg(True);
1784
1785 // Undo the combine foldFreeOpFromSelect does if it helps us match the
1786 // fmin/fmax.
1787 //
1788 // select (fcmp olt (lhs, K)), (fneg lhs), -K
1789 // -> fneg (fmin_legacy lhs, K)
1790 //
1791 // TODO: Use getNegatedExpression
1792 if (LHS == NegTrue && CFalse && CRHS) {
1793 APFloat NegRHS = neg(CRHS->getValueAPF());
1794 if (NegRHS == CFalse->getValueAPF()) {
1795 SDValue Combined = combineFMinMaxLegacyImpl(DL, VT, LHS, RHS, NegTrue,
1796 False, CC, Flags, DCI);
1797 if (Combined)
1798 return DAG.getNode(ISD::FNEG, DL, VT, Combined);
1799 return SDValue();
1800 }
1801 }
1802
1803 return SDValue();
1804}
1805
1806std::pair<SDValue, SDValue>
1808 SDLoc SL(Op);
1809
1810 SDValue Vec = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, Op);
1811
1812 const SDValue Zero = DAG.getConstant(0, SL, MVT::i32);
1813 const SDValue One = DAG.getConstant(1, SL, MVT::i32);
1814
1815 SDValue Lo = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Vec, Zero);
1816 SDValue Hi = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Vec, One);
1817
1818 return std::pair(Lo, Hi);
1819}
1820
1822 SDLoc SL(Op);
1823
1824 SDValue Vec = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, Op);
1825 const SDValue Zero = DAG.getConstant(0, SL, MVT::i32);
1826 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Vec, Zero);
1827}
1828
1830 SDLoc SL(Op);
1831
1832 SDValue Vec = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, Op);
1833 const SDValue One = DAG.getConstant(1, SL, MVT::i32);
1834 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Vec, One);
1835}
1836
1837// Split a vector type into two parts. The first part is a power of two vector.
1838// The second part is whatever is left over, and is a scalar if it would
1839// otherwise be a 1-vector.
1840std::pair<EVT, EVT>
1842 EVT LoVT, HiVT;
1843 EVT EltVT = VT.getVectorElementType();
1844 unsigned NumElts = VT.getVectorNumElements();
1845 unsigned LoNumElts = PowerOf2Ceil((NumElts + 1) / 2);
1846 LoVT = EVT::getVectorVT(*DAG.getContext(), EltVT, LoNumElts);
1847 HiVT = NumElts - LoNumElts == 1
1848 ? EltVT
1849 : EVT::getVectorVT(*DAG.getContext(), EltVT, NumElts - LoNumElts);
1850 return std::pair(LoVT, HiVT);
1851}
1852
1853// Split a vector value into two parts of types LoVT and HiVT. HiVT could be
1854// scalar.
1855std::pair<SDValue, SDValue>
1857 const EVT &LoVT, const EVT &HiVT,
1858 SelectionDAG &DAG) const {
1859 EVT VT = N.getValueType();
1861 (HiVT.isVector() ? HiVT.getVectorNumElements() : 1) <=
1862 VT.getVectorNumElements() &&
1863 "More vector elements requested than available!");
1865 DAG.getVectorIdxConstant(0, DL));
1866
1867 unsigned LoNumElts = LoVT.getVectorNumElements();
1868
1869 if (HiVT.isVector()) {
1870 unsigned HiNumElts = HiVT.getVectorNumElements();
1871 if ((VT.getVectorNumElements() % HiNumElts) == 0) {
1872 // Avoid creating an extract_subvector with an index that isn't a multiple
1873 // of the result type.
1875 DAG.getConstant(LoNumElts, DL, MVT::i32));
1876 return {Lo, Hi};
1877 }
1878
1880 DAG.ExtractVectorElements(N, Elts, /*Start=*/LoNumElts,
1881 /*Count=*/HiNumElts);
1882 SDValue Hi = DAG.getBuildVector(HiVT, DL, Elts);
1883 return {Lo, Hi};
1884 }
1885
1887 DAG.getVectorIdxConstant(LoNumElts, DL));
1888 return {Lo, Hi};
1889}
1890
1892 SelectionDAG &DAG) const {
1894 EVT VT = Op.getValueType();
1895 SDLoc SL(Op);
1896
1897
1898 // If this is a 2 element vector, we really want to scalarize and not create
1899 // weird 1 element vectors.
1900 if (VT.getVectorNumElements() == 2) {
1901 SDValue Ops[2];
1902 std::tie(Ops[0], Ops[1]) = scalarizeVectorLoad(Load, DAG);
1903 return DAG.getMergeValues(Ops, SL);
1904 }
1905
1906 SDValue BasePtr = Load->getBasePtr();
1907 EVT MemVT = Load->getMemoryVT();
1908
1909 const MachinePointerInfo &SrcValue = Load->getMemOperand()->getPointerInfo();
1910
1911 EVT LoVT, HiVT;
1912 EVT LoMemVT, HiMemVT;
1913 SDValue Lo, Hi;
1914
1915 std::tie(LoVT, HiVT) = getSplitDestVTs(VT, DAG);
1916 std::tie(LoMemVT, HiMemVT) = getSplitDestVTs(MemVT, DAG);
1917 std::tie(Lo, Hi) = splitVector(Op, SL, LoVT, HiVT, DAG);
1918
1919 unsigned Size = LoMemVT.getStoreSize();
1920 Align BaseAlign = Load->getAlign();
1921 Align HiAlign = commonAlignment(BaseAlign, Size);
1922
1923 SDValue LoLoad = DAG.getExtLoad(
1924 Load->getExtensionType(), SL, LoVT, Load->getChain(), BasePtr, SrcValue,
1925 LoMemVT, BaseAlign, Load->getMemOperand()->getFlags(), Load->getAAInfo());
1926 SDValue HiPtr = DAG.getObjectPtrOffset(SL, BasePtr, TypeSize::getFixed(Size));
1927 SDValue HiLoad = DAG.getExtLoad(
1928 Load->getExtensionType(), SL, HiVT, Load->getChain(), HiPtr,
1929 SrcValue.getWithOffset(LoMemVT.getStoreSize()), HiMemVT, HiAlign,
1930 Load->getMemOperand()->getFlags(), Load->getAAInfo());
1931
1932 SDValue Join;
1933 if (LoVT == HiVT) {
1934 // This is the case that the vector is power of two so was evenly split.
1935 Join = DAG.getNode(ISD::CONCAT_VECTORS, SL, VT, LoLoad, HiLoad);
1936 } else {
1937 Join = DAG.getNode(ISD::INSERT_SUBVECTOR, SL, VT, DAG.getPOISON(VT), LoLoad,
1938 DAG.getVectorIdxConstant(0, SL));
1939 Join = DAG.getNode(
1941 VT, Join, HiLoad,
1943 }
1944
1945 SDValue Ops[] = {Join, DAG.getNode(ISD::TokenFactor, SL, MVT::Other,
1946 LoLoad.getValue(1), HiLoad.getValue(1))};
1947
1948 return DAG.getMergeValues(Ops, SL);
1949}
1950
1952 SelectionDAG &DAG) const {
1954 EVT VT = Op.getValueType();
1955 SDValue BasePtr = Load->getBasePtr();
1956 EVT MemVT = Load->getMemoryVT();
1957 SDLoc SL(Op);
1958 const MachinePointerInfo &SrcValue = Load->getMemOperand()->getPointerInfo();
1959 Align BaseAlign = Load->getAlign();
1960 unsigned NumElements = MemVT.getVectorNumElements();
1961
1962 // Widen from vec3 to vec4 when the load is at least 8-byte aligned
1963 // or 16-byte fully dereferenceable. Otherwise, split the vector load.
1964 if (NumElements != 3 ||
1965 (BaseAlign < Align(8) &&
1966 !SrcValue.isDereferenceable(16, *DAG.getContext(), DAG.getDataLayout())))
1967 return SplitVectorLoad(Op, DAG);
1968
1969 assert(NumElements == 3);
1970
1971 EVT WideVT =
1973 EVT WideMemVT =
1975 SDValue WideLoad = DAG.getExtLoad(
1976 Load->getExtensionType(), SL, WideVT, Load->getChain(), BasePtr, SrcValue,
1977 WideMemVT, BaseAlign, Load->getMemOperand()->getFlags());
1978 return DAG.getMergeValues(
1979 {DAG.getNode(ISD::EXTRACT_SUBVECTOR, SL, VT, WideLoad,
1980 DAG.getVectorIdxConstant(0, SL)),
1981 WideLoad.getValue(1)},
1982 SL);
1983}
1984
1986 SelectionDAG &DAG) const {
1988 SDValue Val = Store->getValue();
1989 EVT VT = Val.getValueType();
1990
1991 // If this is a 2 element vector, we really want to scalarize and not create
1992 // weird 1 element vectors.
1993 if (VT.getVectorNumElements() == 2)
1994 return scalarizeVectorStore(Store, DAG);
1995
1996 EVT MemVT = Store->getMemoryVT();
1997 SDValue Chain = Store->getChain();
1998 SDValue BasePtr = Store->getBasePtr();
1999 SDLoc SL(Op);
2000
2001 EVT LoVT, HiVT;
2002 EVT LoMemVT, HiMemVT;
2003 SDValue Lo, Hi;
2004
2005 std::tie(LoVT, HiVT) = getSplitDestVTs(VT, DAG);
2006 std::tie(LoMemVT, HiMemVT) = getSplitDestVTs(MemVT, DAG);
2007 std::tie(Lo, Hi) = splitVector(Val, SL, LoVT, HiVT, DAG);
2008
2009 SDValue HiPtr = DAG.getObjectPtrOffset(SL, BasePtr, LoMemVT.getStoreSize());
2010
2011 const MachinePointerInfo &SrcValue = Store->getMemOperand()->getPointerInfo();
2012 Align BaseAlign = Store->getAlign();
2013 unsigned Size = LoMemVT.getStoreSize();
2014 Align HiAlign = commonAlignment(BaseAlign, Size);
2015
2016 SDValue LoStore =
2017 DAG.getTruncStore(Chain, SL, Lo, BasePtr, SrcValue, LoMemVT, BaseAlign,
2018 Store->getMemOperand()->getFlags(), Store->getAAInfo());
2019 SDValue HiStore = DAG.getTruncStore(
2020 Chain, SL, Hi, HiPtr, SrcValue.getWithOffset(Size), HiMemVT, HiAlign,
2021 Store->getMemOperand()->getFlags(), Store->getAAInfo());
2022
2023 return DAG.getNode(ISD::TokenFactor, SL, MVT::Other, LoStore, HiStore);
2024}
2025
2026// This is a shortcut for integer division because we have fast i32<->f32
2027// conversions, and fast f32 reciprocal instructions.
2029 bool Sign) const {
2030 SDLoc DL(Op);
2031 EVT VT = Op.getValueType();
2032 assert(VT == MVT::i32 && "LowerDIVREMToFloat expects an i32");
2033
2034 SDValue LHS = Op.getOperand(0);
2035 SDValue RHS = Op.getOperand(1);
2036 MVT IntVT = MVT::i32;
2037 MVT FltVT = MVT::f32;
2038
2039 unsigned LHSSignBits;
2040 unsigned RHSSignBits;
2041 if (Sign) {
2042 LHSSignBits = DAG.ComputeNumSignBits(LHS);
2043 RHSSignBits = DAG.ComputeNumSignBits(RHS);
2044 if (LHSSignBits < 9 || RHSSignBits < 9)
2045 return SDValue();
2046 } else {
2047 KnownBits LHSKnown = DAG.computeKnownBits(LHS);
2048 KnownBits RHSKnown = DAG.computeKnownBits(RHS);
2049
2050 LHSSignBits = LHSKnown.countMinLeadingZeros();
2051 RHSSignBits = RHSKnown.countMinLeadingZeros();
2052 }
2053
2054 unsigned BitSize = VT.getSizeInBits();
2055 unsigned SignBits = std::min(LHSSignBits, RHSSignBits);
2056 unsigned DivBits = BitSize - SignBits;
2057 if (Sign)
2058 ++DivBits;
2059
2060 // In order to avoid problems due to 1 ulp accuracy issues with v_rcp_f32,
2061 // limit LowerDIVREMToFloat to:
2062 // [-0x400000,0x3FFFFF] for Sign
2063 // [ 0x000000,0x3FFFFF] for !Sign
2064 // This matches what is done in expandDivRemToFloatImpl.
2065 if (DivBits > (Sign ? 23 : 22))
2066 return SDValue();
2067
2070
2071 // int ia = (int)LHS;
2072 SDValue ia = LHS;
2073
2074 // int ib, (int)RHS;
2075 SDValue ib = RHS;
2076
2077 // The calculation:
2078 // fq = fa*recip(fb)
2079 // may be too small due to the 1ulp accuracy in the recip
2080 // operation and rounding issues. Since fq is truncated to produce
2081 // an integer value it may be too small by one. This is
2082 // dealt with by incrementing fa by 1ulp:
2083 // fq = (fa+1ulp)*recip(fb)
2084 // This will increase fa's magnitude by at most 0.5
2085 // (i.e. when fabs(fa)==0x400000 the LSB of the mantissa represents 0.5).
2086 // Thus, this method is safe since fa must be incremented by at least 1.0
2087 // for the quotient to increase by one.
2088 SDValue fa = DAG.getNode(ToFp, DL, FltVT, ia);
2089 SDValue faAsInt = DAG.getNode(ISD::BITCAST, DL, MVT::i32, fa);
2090 SDValue faIncremented = DAG.getNode(ISD::ADD, DL, MVT::i32, faAsInt,
2091 DAG.getConstant(1, DL, MVT::i32));
2092 fa = DAG.getNode(ISD::BITCAST, DL, FltVT, faIncremented);
2093
2094 // float fb = (float)ib;
2095 SDValue fb = DAG.getNode(ToFp, DL, FltVT, ib);
2096
2097 SDValue fq = DAG.getNode(ISD::FMUL, DL, FltVT,
2098 fa, DAG.getNode(AMDGPUISD::RCP, DL, FltVT, fb));
2099
2100 // fq = trunc(fq);
2101 fq = DAG.getNode(ISD::FTRUNC, DL, FltVT, fq);
2102
2103 // int iq = (int)fq;
2104 SDValue Div = DAG.getNode(ToInt, DL, IntVT, fq);
2105
2106 // Rem needs compensation, it's easier to recompute it
2107 SDValue Rem = DAG.getNode(ISD::MUL, DL, VT, Div, RHS);
2108 Rem = DAG.getNode(ISD::SUB, DL, VT, LHS, Rem);
2109
2110 return DAG.getMergeValues({ Div, Rem }, DL);
2111}
2112
2114 SelectionDAG &DAG,
2116 SDLoc DL(Op);
2117 EVT VT = Op.getValueType();
2118
2119 assert(VT == MVT::i64 && "LowerUDIVREM64 expects an i64");
2120
2121 EVT HalfVT = VT.getHalfSizedIntegerVT(*DAG.getContext());
2122
2123 SDValue One = DAG.getConstant(1, DL, HalfVT);
2124 SDValue Zero = DAG.getConstant(0, DL, HalfVT);
2125
2126 //HiLo split
2127 SDValue LHS_Lo, LHS_Hi;
2128 SDValue LHS = Op.getOperand(0);
2129 std::tie(LHS_Lo, LHS_Hi) = DAG.SplitScalar(LHS, DL, HalfVT, HalfVT);
2130
2131 SDValue RHS_Lo, RHS_Hi;
2132 SDValue RHS = Op.getOperand(1);
2133 std::tie(RHS_Lo, RHS_Hi) = DAG.SplitScalar(RHS, DL, HalfVT, HalfVT);
2134
2135 if (DAG.MaskedValueIsZero(RHS, APInt::getHighBitsSet(64, 32)) &&
2136 DAG.MaskedValueIsZero(LHS, APInt::getHighBitsSet(64, 32))) {
2137
2138 SDValue Res = DAG.getNode(ISD::UDIVREM, DL, DAG.getVTList(HalfVT, HalfVT),
2139 LHS_Lo, RHS_Lo);
2140
2141 SDValue DIV = DAG.getBuildVector(MVT::v2i32, DL, {Res.getValue(0), Zero});
2142 SDValue REM = DAG.getBuildVector(MVT::v2i32, DL, {Res.getValue(1), Zero});
2143
2144 Results.push_back(DAG.getNode(ISD::BITCAST, DL, MVT::i64, DIV));
2145 Results.push_back(DAG.getNode(ISD::BITCAST, DL, MVT::i64, REM));
2146 return;
2147 }
2148
2149 if (isTypeLegal(MVT::i64)) {
2150 // The algorithm here is based on ideas from "Software Integer Division",
2151 // Tom Rodeheffer, August 2008.
2152
2155
2156 // Compute denominator reciprocal.
2157 unsigned FMAD =
2158 !Subtarget->hasMadMacF32Insts() ? (unsigned)ISD::FMA
2161 : (unsigned)AMDGPUISD::FMAD_FTZ;
2162
2163 SDValue Cvt_Lo = DAG.getNode(ISD::UINT_TO_FP, DL, MVT::f32, RHS_Lo);
2164 SDValue Cvt_Hi = DAG.getNode(ISD::UINT_TO_FP, DL, MVT::f32, RHS_Hi);
2165 SDValue Mad1 = DAG.getNode(FMAD, DL, MVT::f32, Cvt_Hi,
2166 DAG.getConstantFP(APInt(32, 0x4f800000).bitsToFloat(), DL, MVT::f32),
2167 Cvt_Lo);
2168 SDValue Rcp = DAG.getNode(AMDGPUISD::RCP, DL, MVT::f32, Mad1);
2169 SDValue Mul1 = DAG.getNode(ISD::FMUL, DL, MVT::f32, Rcp,
2170 DAG.getConstantFP(APInt(32, 0x5f7ffffc).bitsToFloat(), DL, MVT::f32));
2171 SDValue Mul2 = DAG.getNode(ISD::FMUL, DL, MVT::f32, Mul1,
2172 DAG.getConstantFP(APInt(32, 0x2f800000).bitsToFloat(), DL, MVT::f32));
2173 SDValue Trunc = DAG.getNode(ISD::FTRUNC, DL, MVT::f32, Mul2);
2174 SDValue Mad2 = DAG.getNode(FMAD, DL, MVT::f32, Trunc,
2175 DAG.getConstantFP(APInt(32, 0xcf800000).bitsToFloat(), DL, MVT::f32),
2176 Mul1);
2177 SDValue Rcp_Lo = DAG.getNode(ISD::FP_TO_UINT, DL, HalfVT, Mad2);
2178 SDValue Rcp_Hi = DAG.getNode(ISD::FP_TO_UINT, DL, HalfVT, Trunc);
2179 SDValue Rcp64 = DAG.getBitcast(VT,
2180 DAG.getBuildVector(MVT::v2i32, DL, {Rcp_Lo, Rcp_Hi}));
2181
2182 SDValue Zero64 = DAG.getConstant(0, DL, VT);
2183 SDValue One64 = DAG.getConstant(1, DL, VT);
2184 SDValue Zero1 = DAG.getConstant(0, DL, MVT::i1);
2185 SDVTList HalfCarryVT = DAG.getVTList(HalfVT, MVT::i1);
2186
2187 // First round of UNR (Unsigned integer Newton-Raphson).
2188 SDValue Neg_RHS = DAG.getNode(ISD::SUB, DL, VT, Zero64, RHS);
2189 SDValue Mullo1 = DAG.getNode(ISD::MUL, DL, VT, Neg_RHS, Rcp64);
2190 SDValue Mulhi1 = DAG.getNode(ISD::MULHU, DL, VT, Rcp64, Mullo1);
2191 SDValue Mulhi1_Lo, Mulhi1_Hi;
2192 std::tie(Mulhi1_Lo, Mulhi1_Hi) =
2193 DAG.SplitScalar(Mulhi1, DL, HalfVT, HalfVT);
2194 SDValue Add1_Lo = DAG.getNode(ISD::UADDO_CARRY, DL, HalfCarryVT, Rcp_Lo,
2195 Mulhi1_Lo, Zero1);
2196 SDValue Add1_Hi = DAG.getNode(ISD::UADDO_CARRY, DL, HalfCarryVT, Rcp_Hi,
2197 Mulhi1_Hi, Add1_Lo.getValue(1));
2198 SDValue Add1 = DAG.getBitcast(VT,
2199 DAG.getBuildVector(MVT::v2i32, DL, {Add1_Lo, Add1_Hi}));
2200
2201 // Second round of UNR.
2202 SDValue Mullo2 = DAG.getNode(ISD::MUL, DL, VT, Neg_RHS, Add1);
2203 SDValue Mulhi2 = DAG.getNode(ISD::MULHU, DL, VT, Add1, Mullo2);
2204 SDValue Mulhi2_Lo, Mulhi2_Hi;
2205 std::tie(Mulhi2_Lo, Mulhi2_Hi) =
2206 DAG.SplitScalar(Mulhi2, DL, HalfVT, HalfVT);
2207 SDValue Add2_Lo = DAG.getNode(ISD::UADDO_CARRY, DL, HalfCarryVT, Add1_Lo,
2208 Mulhi2_Lo, Zero1);
2209 SDValue Add2_Hi = DAG.getNode(ISD::UADDO_CARRY, DL, HalfCarryVT, Add1_Hi,
2210 Mulhi2_Hi, Add2_Lo.getValue(1));
2211 SDValue Add2 = DAG.getBitcast(VT,
2212 DAG.getBuildVector(MVT::v2i32, DL, {Add2_Lo, Add2_Hi}));
2213
2214 SDValue Mulhi3 = DAG.getNode(ISD::MULHU, DL, VT, LHS, Add2);
2215
2216 SDValue Mul3 = DAG.getNode(ISD::MUL, DL, VT, RHS, Mulhi3);
2217
2218 SDValue Mul3_Lo, Mul3_Hi;
2219 std::tie(Mul3_Lo, Mul3_Hi) = DAG.SplitScalar(Mul3, DL, HalfVT, HalfVT);
2220 SDValue Sub1_Lo = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, LHS_Lo,
2221 Mul3_Lo, Zero1);
2222 SDValue Sub1_Hi = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, LHS_Hi,
2223 Mul3_Hi, Sub1_Lo.getValue(1));
2224 SDValue Sub1_Mi = DAG.getNode(ISD::SUB, DL, HalfVT, LHS_Hi, Mul3_Hi);
2225 SDValue Sub1 = DAG.getBitcast(VT,
2226 DAG.getBuildVector(MVT::v2i32, DL, {Sub1_Lo, Sub1_Hi}));
2227
2228 SDValue MinusOne = DAG.getConstant(0xffffffffu, DL, HalfVT);
2229 SDValue C1 = DAG.getSelectCC(DL, Sub1_Hi, RHS_Hi, MinusOne, Zero,
2230 ISD::SETUGE);
2231 SDValue C2 = DAG.getSelectCC(DL, Sub1_Lo, RHS_Lo, MinusOne, Zero,
2232 ISD::SETUGE);
2233 SDValue C3 = DAG.getSelectCC(DL, Sub1_Hi, RHS_Hi, C2, C1, ISD::SETEQ);
2234
2235 // TODO: Here and below portions of the code can be enclosed into if/endif.
2236 // Currently control flow is unconditional and we have 4 selects after
2237 // potential endif to substitute PHIs.
2238
2239 // if C3 != 0 ...
2240 SDValue Sub2_Lo = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub1_Lo,
2241 RHS_Lo, Zero1);
2242 SDValue Sub2_Mi = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub1_Mi,
2243 RHS_Hi, Sub1_Lo.getValue(1));
2244 SDValue Sub2_Hi = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub2_Mi,
2245 Zero, Sub2_Lo.getValue(1));
2246 SDValue Sub2 = DAG.getBitcast(VT,
2247 DAG.getBuildVector(MVT::v2i32, DL, {Sub2_Lo, Sub2_Hi}));
2248
2249 SDValue Add3 = DAG.getNode(ISD::ADD, DL, VT, Mulhi3, One64);
2250
2251 SDValue C4 = DAG.getSelectCC(DL, Sub2_Hi, RHS_Hi, MinusOne, Zero,
2252 ISD::SETUGE);
2253 SDValue C5 = DAG.getSelectCC(DL, Sub2_Lo, RHS_Lo, MinusOne, Zero,
2254 ISD::SETUGE);
2255 SDValue C6 = DAG.getSelectCC(DL, Sub2_Hi, RHS_Hi, C5, C4, ISD::SETEQ);
2256
2257 // if (C6 != 0)
2258 SDValue Add4 = DAG.getNode(ISD::ADD, DL, VT, Add3, One64);
2259
2260 SDValue Sub3_Lo = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub2_Lo,
2261 RHS_Lo, Zero1);
2262 SDValue Sub3_Mi = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub2_Mi,
2263 RHS_Hi, Sub2_Lo.getValue(1));
2264 SDValue Sub3_Hi = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub3_Mi,
2265 Zero, Sub3_Lo.getValue(1));
2266 SDValue Sub3 = DAG.getBitcast(VT,
2267 DAG.getBuildVector(MVT::v2i32, DL, {Sub3_Lo, Sub3_Hi}));
2268
2269 // endif C6
2270 // endif C3
2271
2272 SDValue Sel1 = DAG.getSelectCC(DL, C6, Zero, Add4, Add3, ISD::SETNE);
2273 SDValue Div = DAG.getSelectCC(DL, C3, Zero, Sel1, Mulhi3, ISD::SETNE);
2274
2275 SDValue Sel2 = DAG.getSelectCC(DL, C6, Zero, Sub3, Sub2, ISD::SETNE);
2276 SDValue Rem = DAG.getSelectCC(DL, C3, Zero, Sel2, Sub1, ISD::SETNE);
2277
2278 Results.push_back(Div);
2279 Results.push_back(Rem);
2280
2281 return;
2282 }
2283
2284 // r600 expandion.
2285 // Get Speculative values
2286 SDValue DIV_Part = DAG.getNode(ISD::UDIV, DL, HalfVT, LHS_Hi, RHS_Lo);
2287 SDValue REM_Part = DAG.getNode(ISD::UREM, DL, HalfVT, LHS_Hi, RHS_Lo);
2288
2289 SDValue REM_Lo = DAG.getSelectCC(DL, RHS_Hi, Zero, REM_Part, LHS_Hi, ISD::SETEQ);
2290 SDValue REM = DAG.getBuildVector(MVT::v2i32, DL, {REM_Lo, Zero});
2291 REM = DAG.getNode(ISD::BITCAST, DL, MVT::i64, REM);
2292
2293 SDValue DIV_Hi = DAG.getSelectCC(DL, RHS_Hi, Zero, DIV_Part, Zero, ISD::SETEQ);
2294 SDValue DIV_Lo = Zero;
2295
2296 const unsigned halfBitWidth = HalfVT.getSizeInBits();
2297
2298 for (unsigned i = 0; i < halfBitWidth; ++i) {
2299 const unsigned bitPos = halfBitWidth - i - 1;
2300 SDValue POS = DAG.getConstant(bitPos, DL, HalfVT);
2301 // Get value of high bit
2302 SDValue HBit = DAG.getNode(ISD::SRL, DL, HalfVT, LHS_Lo, POS);
2303 HBit = DAG.getNode(ISD::AND, DL, HalfVT, HBit, One);
2304 HBit = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, HBit);
2305
2306 // Shift
2307 REM = DAG.getNode(ISD::SHL, DL, VT, REM, DAG.getConstant(1, DL, VT));
2308 // Add LHS high bit
2309 REM = DAG.getNode(ISD::OR, DL, VT, REM, HBit);
2310
2311 SDValue BIT = DAG.getConstant(1ULL << bitPos, DL, HalfVT);
2312 SDValue realBIT = DAG.getSelectCC(DL, REM, RHS, BIT, Zero, ISD::SETUGE);
2313
2314 DIV_Lo = DAG.getNode(ISD::OR, DL, HalfVT, DIV_Lo, realBIT);
2315
2316 // Update REM
2317 SDValue REM_sub = DAG.getNode(ISD::SUB, DL, VT, REM, RHS);
2318 REM = DAG.getSelectCC(DL, REM, RHS, REM_sub, REM, ISD::SETUGE);
2319 }
2320
2321 SDValue DIV = DAG.getBuildVector(MVT::v2i32, DL, {DIV_Lo, DIV_Hi});
2322 DIV = DAG.getNode(ISD::BITCAST, DL, MVT::i64, DIV);
2323 Results.push_back(DIV);
2324 Results.push_back(REM);
2325}
2326
2328 SelectionDAG &DAG) const {
2329 SDLoc DL(Op);
2330 EVT VT = Op.getValueType();
2331
2332 if (VT == MVT::i64) {
2334 LowerUDIVREM64(Op, DAG, Results);
2335 return DAG.getMergeValues(Results, DL);
2336 }
2337
2338 if (VT == MVT::i32) {
2339 if (SDValue Res = LowerDIVREMToFloat(Op, DAG, false))
2340 return Res;
2341 }
2342
2343 SDValue X = Op.getOperand(0);
2344 SDValue Y = Op.getOperand(1);
2345
2346 // See AMDGPUCodeGenPrepare::expandDivRem32 for a description of the
2347 // algorithm used here.
2348
2349 // Initial estimate of inv(y).
2350 SDValue Z = DAG.getNode(AMDGPUISD::URECIP, DL, VT, Y);
2351
2352 // One round of UNR.
2353 SDValue NegY = DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(0, DL, VT), Y);
2354 SDValue NegYZ = DAG.getNode(ISD::MUL, DL, VT, NegY, Z);
2355 Z = DAG.getNode(ISD::ADD, DL, VT, Z,
2356 DAG.getNode(ISD::MULHU, DL, VT, Z, NegYZ));
2357
2358 // Quotient/remainder estimate.
2359 SDValue Q = DAG.getNode(ISD::MULHU, DL, VT, X, Z);
2360 SDValue R =
2361 DAG.getNode(ISD::SUB, DL, VT, X, DAG.getNode(ISD::MUL, DL, VT, Q, Y));
2362
2363 // First quotient/remainder refinement.
2364 EVT CCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
2365 SDValue One = DAG.getConstant(1, DL, VT);
2366 SDValue Cond = DAG.getSetCC(DL, CCVT, R, Y, ISD::SETUGE);
2367 Q = DAG.getNode(ISD::SELECT, DL, VT, Cond,
2368 DAG.getNode(ISD::ADD, DL, VT, Q, One), Q);
2369 R = DAG.getNode(ISD::SELECT, DL, VT, Cond,
2370 DAG.getNode(ISD::SUB, DL, VT, R, Y), R);
2371
2372 // Second quotient/remainder refinement.
2373 Cond = DAG.getSetCC(DL, CCVT, R, Y, ISD::SETUGE);
2374 Q = DAG.getNode(ISD::SELECT, DL, VT, Cond,
2375 DAG.getNode(ISD::ADD, DL, VT, Q, One), Q);
2376 R = DAG.getNode(ISD::SELECT, DL, VT, Cond,
2377 DAG.getNode(ISD::SUB, DL, VT, R, Y), R);
2378
2379 return DAG.getMergeValues({Q, R}, DL);
2380}
2381
2383 SelectionDAG &DAG) const {
2384 SDLoc DL(Op);
2385 EVT VT = Op.getValueType();
2386
2387 SDValue LHS = Op.getOperand(0);
2388 SDValue RHS = Op.getOperand(1);
2389
2390 SDValue Zero = DAG.getConstant(0, DL, VT);
2391 SDValue NegOne = DAG.getAllOnesConstant(DL, VT);
2392
2393 if (VT == MVT::i32) {
2394 if (SDValue Res = LowerDIVREMToFloat(Op, DAG, true))
2395 return Res;
2396 }
2397
2398 // LHS must have > 33 sign-bits to ensure that LHS != -2147483648
2399 // Otherwise 32-bit division cannot be used safely.
2400 // -2147483648/1 and -2147483648/-1 are not equal,
2401 // but they produce the same lower 32-bit result.
2402 if (VT == MVT::i64 && DAG.ComputeNumSignBits(LHS) > 33 &&
2403 DAG.ComputeNumSignBits(RHS) > 32) {
2404 EVT HalfVT = VT.getHalfSizedIntegerVT(*DAG.getContext());
2405
2406 //HiLo split
2407 SDValue LHS_Lo = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, HalfVT, LHS, Zero);
2408 SDValue RHS_Lo = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, HalfVT, RHS, Zero);
2409 SDValue DIVREM = DAG.getNode(ISD::SDIVREM, DL, DAG.getVTList(HalfVT, HalfVT),
2410 LHS_Lo, RHS_Lo);
2411 SDValue Res[2] = {
2412 DAG.getNode(ISD::SIGN_EXTEND, DL, VT, DIVREM.getValue(0)),
2413 DAG.getNode(ISD::SIGN_EXTEND, DL, VT, DIVREM.getValue(1))
2414 };
2415 return DAG.getMergeValues(Res, DL);
2416 }
2417
2418 SDValue LHSign = DAG.getSelectCC(DL, LHS, Zero, NegOne, Zero, ISD::SETLT);
2419 SDValue RHSign = DAG.getSelectCC(DL, RHS, Zero, NegOne, Zero, ISD::SETLT);
2420 SDValue DSign = DAG.getNode(ISD::XOR, DL, VT, LHSign, RHSign);
2421 SDValue RSign = LHSign; // Remainder sign is the same as LHS
2422
2423 LHS = DAG.getNode(ISD::ADD, DL, VT, LHS, LHSign);
2424 RHS = DAG.getNode(ISD::ADD, DL, VT, RHS, RHSign);
2425
2426 LHS = DAG.getNode(ISD::XOR, DL, VT, LHS, LHSign);
2427 RHS = DAG.getNode(ISD::XOR, DL, VT, RHS, RHSign);
2428
2429 SDValue Div = DAG.getNode(ISD::UDIVREM, DL, DAG.getVTList(VT, VT), LHS, RHS);
2430 SDValue Rem = Div.getValue(1);
2431
2432 Div = DAG.getNode(ISD::XOR, DL, VT, Div, DSign);
2433 Rem = DAG.getNode(ISD::XOR, DL, VT, Rem, RSign);
2434
2435 Div = DAG.getNode(ISD::SUB, DL, VT, Div, DSign);
2436 Rem = DAG.getNode(ISD::SUB, DL, VT, Rem, RSign);
2437
2438 SDValue Res[2] = {
2439 Div,
2440 Rem
2441 };
2442 return DAG.getMergeValues(Res, DL);
2443}
2444
2446 SDLoc SL(Op);
2447 SDValue Src = Op.getOperand(0);
2448
2449 // result = trunc(src)
2450 // if (src > 0.0 && src != result)
2451 // result += 1.0
2452
2453 SDValue Trunc = DAG.getNode(ISD::FTRUNC, SL, MVT::f64, Src);
2454
2455 const SDValue Zero = DAG.getConstantFP(0.0, SL, MVT::f64);
2456 const SDValue One = DAG.getConstantFP(1.0, SL, MVT::f64);
2457
2458 EVT SetCCVT =
2459 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), MVT::f64);
2460
2461 SDValue Lt0 = DAG.getSetCC(SL, SetCCVT, Src, Zero, ISD::SETOGT);
2462 SDValue NeTrunc = DAG.getSetCC(SL, SetCCVT, Src, Trunc, ISD::SETONE);
2463 SDValue And = DAG.getNode(ISD::AND, SL, SetCCVT, Lt0, NeTrunc);
2464
2465 SDValue Add = DAG.getNode(ISD::SELECT, SL, MVT::f64, And, One, Zero);
2466 // TODO: Should this propagate fast-math-flags?
2467 return DAG.getNode(ISD::FADD, SL, MVT::f64, Trunc, Add);
2468}
2469
2471 SelectionDAG &DAG) {
2472 const unsigned FractBits = 52;
2473 const unsigned ExpBits = 11;
2474
2475 SDValue ExpPart = DAG.getNode(AMDGPUISD::BFE_U32, SL, MVT::i32,
2476 Hi,
2477 DAG.getConstant(FractBits - 32, SL, MVT::i32),
2478 DAG.getConstant(ExpBits, SL, MVT::i32));
2479 SDValue Exp = DAG.getNode(ISD::SUB, SL, MVT::i32, ExpPart,
2480 DAG.getConstant(1023, SL, MVT::i32));
2481
2482 return Exp;
2483}
2484
2486 SDLoc SL(Op);
2487 SDValue Src = Op.getOperand(0);
2488
2489 assert(Op.getValueType() == MVT::f64);
2490
2491 const SDValue Zero = DAG.getConstant(0, SL, MVT::i32);
2492
2493 // Extract the upper half, since this is where we will find the sign and
2494 // exponent.
2495 SDValue Hi = getHiHalf64(Src, DAG);
2496
2497 SDValue Exp = extractF64Exponent(Hi, SL, DAG);
2498
2499 const unsigned FractBits = 52;
2500
2501 // Extract the sign bit.
2502 const SDValue SignBitMask = DAG.getConstant(UINT32_C(1) << 31, SL, MVT::i32);
2503 SDValue SignBit = DAG.getNode(ISD::AND, SL, MVT::i32, Hi, SignBitMask);
2504
2505 // Extend back to 64-bits.
2506 SDValue SignBit64 = DAG.getBuildVector(MVT::v2i32, SL, {Zero, SignBit});
2507 SignBit64 = DAG.getNode(ISD::BITCAST, SL, MVT::i64, SignBit64);
2508
2509 SDValue BcInt = DAG.getNode(ISD::BITCAST, SL, MVT::i64, Src);
2510 const SDValue FractMask
2511 = DAG.getConstant((UINT64_C(1) << FractBits) - 1, SL, MVT::i64);
2512
2513 SDValue Shr = DAG.getNode(ISD::SRA, SL, MVT::i64, FractMask, Exp);
2514 SDValue Not = DAG.getNOT(SL, Shr, MVT::i64);
2515 SDValue Tmp0 = DAG.getNode(ISD::AND, SL, MVT::i64, BcInt, Not);
2516
2517 EVT SetCCVT =
2518 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), MVT::i32);
2519
2520 const SDValue FiftyOne = DAG.getConstant(FractBits - 1, SL, MVT::i32);
2521
2522 SDValue ExpLt0 = DAG.getSetCC(SL, SetCCVT, Exp, Zero, ISD::SETLT);
2523 SDValue ExpGt51 = DAG.getSetCC(SL, SetCCVT, Exp, FiftyOne, ISD::SETGT);
2524
2525 SDValue Tmp1 = DAG.getNode(ISD::SELECT, SL, MVT::i64, ExpLt0, SignBit64, Tmp0);
2526 SDValue Tmp2 = DAG.getNode(ISD::SELECT, SL, MVT::i64, ExpGt51, BcInt, Tmp1);
2527
2528 return DAG.getNode(ISD::BITCAST, SL, MVT::f64, Tmp2);
2529}
2530
2532 SelectionDAG &DAG) const {
2533 SDLoc SL(Op);
2534 SDValue Src = Op.getOperand(0);
2535
2536 assert(Op.getValueType() == MVT::f64);
2537
2538 APFloat C1Val(APFloat::IEEEdouble(), "0x1.0p+52");
2539 SDValue C1 = DAG.getConstantFP(C1Val, SL, MVT::f64);
2540 SDValue CopySign = DAG.getNode(ISD::FCOPYSIGN, SL, MVT::f64, C1, Src);
2541
2542 // TODO: Should this propagate fast-math-flags?
2543
2544 SDValue Tmp1 = DAG.getNode(ISD::FADD, SL, MVT::f64, Src, CopySign);
2545 SDValue Tmp2 = DAG.getNode(ISD::FSUB, SL, MVT::f64, Tmp1, CopySign);
2546
2547 SDValue Fabs = DAG.getNode(ISD::FABS, SL, MVT::f64, Src);
2548
2549 APFloat C2Val(APFloat::IEEEdouble(), "0x1.fffffffffffffp+51");
2550 SDValue C2 = DAG.getConstantFP(C2Val, SL, MVT::f64);
2551
2552 EVT SetCCVT =
2553 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), MVT::f64);
2554 SDValue Cond = DAG.getSetCC(SL, SetCCVT, Fabs, C2, ISD::SETOGT);
2555
2556 return DAG.getSelect(SL, MVT::f64, Cond, Src, Tmp2);
2557}
2558
2560 SelectionDAG &DAG) const {
2561 // FNEARBYINT and FRINT are the same, except in their handling of FP
2562 // exceptions. Those aren't really meaningful for us, and OpenCL only has
2563 // rint, so just treat them as equivalent.
2564 return DAG.getNode(ISD::FROUNDEVEN, SDLoc(Op), Op.getValueType(),
2565 Op.getOperand(0));
2566}
2567
2569 auto VT = Op.getValueType();
2570 auto Arg = Op.getOperand(0u);
2571 return DAG.getNode(ISD::FROUNDEVEN, SDLoc(Op), VT, Arg);
2572}
2573
2574// XXX - May require not supporting f32 denormals?
2575
2576// Don't handle v2f16. The extra instructions to scalarize and repack around the
2577// compare and vselect end up producing worse code than scalarizing the whole
2578// operation.
2580 SDLoc SL(Op);
2581 SDValue X = Op.getOperand(0);
2582 EVT VT = Op.getValueType();
2583
2584 SDValue T = DAG.getNode(ISD::FTRUNC, SL, VT, X);
2585
2586 // TODO: Should this propagate fast-math-flags?
2587
2588 SDValue Diff = DAG.getNode(ISD::FSUB, SL, VT, X, T);
2589
2590 SDValue AbsDiff = DAG.getNode(ISD::FABS, SL, VT, Diff);
2591
2592 const SDValue Zero = DAG.getConstantFP(0.0, SL, VT);
2593 const SDValue One = DAG.getConstantFP(1.0, SL, VT);
2594
2595 EVT SetCCVT =
2596 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
2597
2598 const SDValue Half = DAG.getConstantFP(0.5, SL, VT);
2599 SDValue Cmp = DAG.getSetCC(SL, SetCCVT, AbsDiff, Half, ISD::SETOGE);
2600 SDValue OneOrZeroFP = DAG.getNode(ISD::SELECT, SL, VT, Cmp, One, Zero);
2601
2602 SDValue SignedOffset = DAG.getNode(ISD::FCOPYSIGN, SL, VT, OneOrZeroFP, X);
2603 return DAG.getNode(ISD::FADD, SL, VT, T, SignedOffset);
2604}
2605
2607 SDLoc SL(Op);
2608 SDValue Src = Op.getOperand(0);
2609
2610 // result = trunc(src);
2611 // if (src < 0.0 && src != result)
2612 // result += -1.0.
2613
2614 SDValue Trunc = DAG.getNode(ISD::FTRUNC, SL, MVT::f64, Src);
2615
2616 const SDValue Zero = DAG.getConstantFP(0.0, SL, MVT::f64);
2617 const SDValue NegOne = DAG.getConstantFP(-1.0, SL, MVT::f64);
2618
2619 EVT SetCCVT =
2620 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), MVT::f64);
2621
2622 SDValue Lt0 = DAG.getSetCC(SL, SetCCVT, Src, Zero, ISD::SETOLT);
2623 SDValue NeTrunc = DAG.getSetCC(SL, SetCCVT, Src, Trunc, ISD::SETONE);
2624 SDValue And = DAG.getNode(ISD::AND, SL, SetCCVT, Lt0, NeTrunc);
2625
2626 SDValue Add = DAG.getNode(ISD::SELECT, SL, MVT::f64, And, NegOne, Zero);
2627 // TODO: Should this propagate fast-math-flags?
2628 return DAG.getNode(ISD::FADD, SL, MVT::f64, Trunc, Add);
2629}
2630
2631/// Return true if it's known that \p Src can never be an f32 denormal value.
2633 switch (Src.getOpcode()) {
2634 case ISD::FP_EXTEND:
2635 return Src.getOperand(0).getValueType() == MVT::f16;
2636 case ISD::FP16_TO_FP:
2637 case ISD::FFREXP:
2638 case ISD::FSQRT:
2639 case AMDGPUISD::LOG:
2640 case AMDGPUISD::EXP:
2641 return true;
2643 unsigned IntrinsicID = Src.getConstantOperandVal(0);
2644 switch (IntrinsicID) {
2645 case Intrinsic::amdgcn_frexp_mant:
2646 case Intrinsic::amdgcn_log:
2647 case Intrinsic::amdgcn_log_clamp:
2648 case Intrinsic::amdgcn_exp2:
2649 case Intrinsic::amdgcn_sqrt:
2650 return true;
2651 default:
2652 return false;
2653 }
2654 }
2655 default:
2656 return false;
2657 }
2658
2659 llvm_unreachable("covered opcode switch");
2660}
2661
2663 SDNodeFlags Flags) {
2664 return Flags.hasApproximateFuncs();
2665}
2666
2675
2677 SDValue Src,
2678 SDNodeFlags Flags) const {
2679 SDLoc SL(Src);
2680 EVT VT = Src.getValueType();
2681 const fltSemantics &Semantics = VT.getFltSemantics();
2682 SDValue SmallestNormal =
2683 DAG.getConstantFP(APFloat::getSmallestNormalized(Semantics), SL, VT);
2684
2685 // Want to scale denormals up, but negatives and 0 work just as well on the
2686 // scaled path.
2687 SDValue IsLtSmallestNormal = DAG.getSetCC(
2688 SL, getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT), Src,
2689 SmallestNormal, ISD::SETOLT);
2690
2691 return IsLtSmallestNormal;
2692}
2693
2695 SDNodeFlags Flags) const {
2696 SDLoc SL(Src);
2697 EVT VT = Src.getValueType();
2698 const fltSemantics &Semantics = VT.getFltSemantics();
2699 SDValue Inf = DAG.getConstantFP(APFloat::getInf(Semantics), SL, VT);
2700
2701 SDValue Fabs = DAG.getNode(ISD::FABS, SL, VT, Src, Flags);
2702 SDValue IsFinite = DAG.getSetCC(
2703 SL, getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT), Fabs,
2704 Inf, ISD::SETOLT);
2705 return IsFinite;
2706}
2707
2708/// If denormal handling is required return the scaled input to FLOG2, and the
2709/// check for denormal range. Otherwise, return null values.
2710std::pair<SDValue, SDValue>
2712 SDValue Src, SDNodeFlags Flags) const {
2713 if (!needsDenormHandlingF32(DAG, Src, Flags))
2714 return {};
2715
2716 MVT VT = MVT::f32;
2717 const fltSemantics &Semantics = APFloat::IEEEsingle();
2718 SDValue SmallestNormal =
2719 DAG.getConstantFP(APFloat::getSmallestNormalized(Semantics), SL, VT);
2720
2721 SDValue IsLtSmallestNormal = DAG.getSetCC(
2722 SL, getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT), Src,
2723 SmallestNormal, ISD::SETOLT);
2724
2725 SDValue Scale32 = DAG.getConstantFP(0x1.0p+32, SL, VT);
2726 SDValue One = DAG.getConstantFP(1.0, SL, VT);
2727 SDValue ScaleFactor =
2728 DAG.getNode(ISD::SELECT, SL, VT, IsLtSmallestNormal, Scale32, One, Flags);
2729
2730 SDValue ScaledInput = DAG.getNode(ISD::FMUL, SL, VT, Src, ScaleFactor, Flags);
2731 return {ScaledInput, IsLtSmallestNormal};
2732}
2733
2735 // v_log_f32 is good enough for OpenCL, except it doesn't handle denormals.
2736 // If we have to handle denormals, scale up the input and adjust the result.
2737
2738 // scaled = x * (is_denormal ? 0x1.0p+32 : 1.0)
2739 // log2 = amdgpu_log2 - (is_denormal ? 32.0 : 0.0)
2740
2741 SDLoc SL(Op);
2742 EVT VT = Op.getValueType();
2743 SDValue Src = Op.getOperand(0);
2744 SDNodeFlags Flags = Op->getFlags();
2745
2746 if (VT == MVT::f16) {
2747 // Nothing in half is a denormal when promoted to f32.
2748 assert(!isTypeLegal(VT));
2749 SDValue Ext = DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, Src, Flags);
2750 SDValue Log = DAG.getNode(AMDGPUISD::LOG, SL, MVT::f32, Ext, Flags);
2751 return DAG.getNode(ISD::FP_ROUND, SL, VT, Log,
2752 DAG.getTargetConstant(0, SL, MVT::i32), Flags);
2753 }
2754
2755 auto [ScaledInput, IsLtSmallestNormal] =
2756 getScaledLogInput(DAG, SL, Src, Flags);
2757 if (!ScaledInput)
2758 return DAG.getNode(AMDGPUISD::LOG, SL, VT, Src, Flags);
2759
2760 SDValue Log2 = DAG.getNode(AMDGPUISD::LOG, SL, VT, ScaledInput, Flags);
2761
2762 SDValue ThirtyTwo = DAG.getConstantFP(32.0, SL, VT);
2763 SDValue Zero = DAG.getConstantFP(0.0, SL, VT);
2764 SDValue ResultOffset =
2765 DAG.getNode(ISD::SELECT, SL, VT, IsLtSmallestNormal, ThirtyTwo, Zero);
2766 return DAG.getNode(ISD::FSUB, SL, VT, Log2, ResultOffset, Flags);
2767}
2768
2769static SDValue getMad(SelectionDAG &DAG, const SDLoc &SL, EVT VT, SDValue X,
2770 SDValue Y, SDValue C, SDNodeFlags Flags = SDNodeFlags()) {
2771 SDValue Mul = DAG.getNode(ISD::FMUL, SL, VT, X, Y, Flags);
2772 return DAG.getNode(ISD::FADD, SL, VT, Mul, C, Flags);
2773}
2774
2776 SelectionDAG &DAG) const {
2777 SDValue X = Op.getOperand(0);
2778 EVT VT = Op.getValueType();
2779 SDNodeFlags Flags = Op->getFlags();
2780 SDLoc DL(Op);
2781 const bool IsLog10 = Op.getOpcode() == ISD::FLOG10;
2782 assert(IsLog10 || Op.getOpcode() == ISD::FLOG);
2783
2784 if (VT == MVT::f16 || Flags.hasApproximateFuncs()) {
2785 // TODO: The direct f16 path is 1.79 ulp for f16. This should be used
2786 // depending on !fpmath metadata.
2787
2788 bool PromoteToF32 = VT == MVT::f16 && (!Flags.hasApproximateFuncs() ||
2789 !isTypeLegal(MVT::f16));
2790
2791 if (PromoteToF32) {
2792 // Log and multiply in f32 is always good enough for f16.
2793 X = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, X, Flags);
2794 }
2795
2796 SDValue Lowered = LowerFLOGUnsafe(X, DL, DAG, IsLog10, Flags);
2797 if (PromoteToF32) {
2798 return DAG.getNode(ISD::FP_ROUND, DL, VT, Lowered,
2799 DAG.getTargetConstant(0, DL, MVT::i32), Flags);
2800 }
2801
2802 return Lowered;
2803 }
2804
2805 SDValue ScaledInput, IsScaled;
2806 if (VT == MVT::f16)
2807 X = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, X, Flags);
2808 else {
2809 std::tie(ScaledInput, IsScaled) = getScaledLogInput(DAG, DL, X, Flags);
2810 if (ScaledInput)
2811 X = ScaledInput;
2812 }
2813
2814 SDValue Y = DAG.getNode(AMDGPUISD::LOG, DL, VT, X, Flags);
2815
2816 SDValue R;
2817 if (Subtarget->hasFastFMAF32()) {
2818 // c+cc are ln(2)/ln(10) to more than 49 bits
2819 const float c_log10 = 0x1.344134p-2f;
2820 const float cc_log10 = 0x1.09f79ep-26f;
2821
2822 // c + cc is ln(2) to more than 49 bits
2823 const float c_log = 0x1.62e42ep-1f;
2824 const float cc_log = 0x1.efa39ep-25f;
2825
2826 SDValue C = DAG.getConstantFP(IsLog10 ? c_log10 : c_log, DL, VT);
2827 SDValue CC = DAG.getConstantFP(IsLog10 ? cc_log10 : cc_log, DL, VT);
2828 // This adds correction terms for which contraction may lead to an increase
2829 // in the error of the approximation, so disable it.
2830 Flags.setAllowContract(false);
2831 R = DAG.getNode(ISD::FMUL, DL, VT, Y, C, Flags);
2832 SDValue NegR = DAG.getNode(ISD::FNEG, DL, VT, R, Flags);
2833 SDValue FMA0 = DAG.getNode(ISD::FMA, DL, VT, Y, C, NegR, Flags);
2834 SDValue FMA1 = DAG.getNode(ISD::FMA, DL, VT, Y, CC, FMA0, Flags);
2835 R = DAG.getNode(ISD::FADD, DL, VT, R, FMA1, Flags);
2836 } else {
2837 // ch+ct is ln(2)/ln(10) to more than 36 bits
2838 const float ch_log10 = 0x1.344000p-2f;
2839 const float ct_log10 = 0x1.3509f6p-18f;
2840
2841 // ch + ct is ln(2) to more than 36 bits
2842 const float ch_log = 0x1.62e000p-1f;
2843 const float ct_log = 0x1.0bfbe8p-15f;
2844
2845 SDValue CH = DAG.getConstantFP(IsLog10 ? ch_log10 : ch_log, DL, VT);
2846 SDValue CT = DAG.getConstantFP(IsLog10 ? ct_log10 : ct_log, DL, VT);
2847
2848 SDValue YAsInt = DAG.getNode(ISD::BITCAST, DL, MVT::i32, Y);
2849 SDValue MaskConst = DAG.getConstant(0xfffff000, DL, MVT::i32);
2850 SDValue YHInt = DAG.getNode(ISD::AND, DL, MVT::i32, YAsInt, MaskConst);
2851 SDValue YH = DAG.getNode(ISD::BITCAST, DL, MVT::f32, YHInt);
2852 SDValue YT = DAG.getNode(ISD::FSUB, DL, VT, Y, YH, Flags);
2853 // This adds correction terms for which contraction may lead to an increase
2854 // in the error of the approximation, so disable it.
2855 Flags.setAllowContract(false);
2856 SDValue YTCT = DAG.getNode(ISD::FMUL, DL, VT, YT, CT, Flags);
2857 SDValue Mad0 = getMad(DAG, DL, VT, YH, CT, YTCT, Flags);
2858 SDValue Mad1 = getMad(DAG, DL, VT, YT, CH, Mad0, Flags);
2859 R = getMad(DAG, DL, VT, YH, CH, Mad1);
2860 }
2861
2862 const bool IsFiniteOnly = Flags.hasNoNaNs() && Flags.hasNoInfs();
2863
2864 // TODO: Check if known finite from source value.
2865 if (!IsFiniteOnly) {
2866 SDValue IsFinite = getIsFinite(DAG, Y, Flags);
2867 R = DAG.getNode(ISD::SELECT, DL, VT, IsFinite, R, Y, Flags);
2868 }
2869
2870 if (IsScaled) {
2871 SDValue Zero = DAG.getConstantFP(0.0f, DL, VT);
2872 SDValue ShiftK =
2873 DAG.getConstantFP(IsLog10 ? 0x1.344136p+3f : 0x1.62e430p+4f, DL, VT);
2874 SDValue Shift =
2875 DAG.getNode(ISD::SELECT, DL, VT, IsScaled, ShiftK, Zero, Flags);
2876 R = DAG.getNode(ISD::FSUB, DL, VT, R, Shift, Flags);
2877 }
2878
2879 return R;
2880}
2881
2885
2886// Do f32 fast math expansion for flog2 or flog10. This is accurate enough for a
2887// promote f16 operation.
2889 SelectionDAG &DAG, bool IsLog10,
2890 SDNodeFlags Flags) const {
2891 EVT VT = Src.getValueType();
2892 unsigned LogOp =
2893 VT == MVT::f32 ? (unsigned)AMDGPUISD::LOG : (unsigned)ISD::FLOG2;
2894
2895 double Log2BaseInverted =
2897
2898 if (VT == MVT::f32) {
2899 auto [ScaledInput, IsScaled] = getScaledLogInput(DAG, SL, Src, Flags);
2900 if (ScaledInput) {
2901 SDValue LogSrc = DAG.getNode(AMDGPUISD::LOG, SL, VT, ScaledInput, Flags);
2902 SDValue ScaledResultOffset =
2903 DAG.getConstantFP(-32.0 * Log2BaseInverted, SL, VT);
2904
2905 SDValue Zero = DAG.getConstantFP(0.0f, SL, VT);
2906
2907 SDValue ResultOffset = DAG.getNode(ISD::SELECT, SL, VT, IsScaled,
2908 ScaledResultOffset, Zero, Flags);
2909
2910 SDValue Log2Inv = DAG.getConstantFP(Log2BaseInverted, SL, VT);
2911
2912 if (Subtarget->hasFastFMAF32())
2913 return DAG.getNode(ISD::FMA, SL, VT, LogSrc, Log2Inv, ResultOffset,
2914 Flags);
2915 SDValue Mul = DAG.getNode(ISD::FMUL, SL, VT, LogSrc, Log2Inv, Flags);
2916 return DAG.getNode(ISD::FADD, SL, VT, Mul, ResultOffset);
2917 }
2918 }
2919
2920 SDValue Log2Operand = DAG.getNode(LogOp, SL, VT, Src, Flags);
2921 SDValue Log2BaseInvertedOperand = DAG.getConstantFP(Log2BaseInverted, SL, VT);
2922
2923 return DAG.getNode(ISD::FMUL, SL, VT, Log2Operand, Log2BaseInvertedOperand,
2924 Flags);
2925}
2926
2927// This expansion gives a result slightly better than 1ulp.
2929 SelectionDAG &DAG) const {
2930 SDLoc DL(Op);
2931 SDValue X = Op.getOperand(0);
2932
2933 // TODO: Check if reassoc is safe. There is an output change in exp2 and
2934 // exp10, which slightly increases ulp.
2935 SDNodeFlags Flags = Op->getFlags() & ~SDNodeFlags::AllowReassociation;
2936
2937 SDValue DN, F, T;
2938
2939 if (Op.getOpcode() == ISD::FEXP2) {
2940 // dn = rint(x)
2941 DN = DAG.getNode(ISD::FRINT, DL, MVT::f64, X, Flags);
2942 // f = x - dn
2943 F = DAG.getNode(ISD::FSUB, DL, MVT::f64, X, DN, Flags);
2944 // t = f*C1 + f*C2
2945 SDValue C1 = DAG.getConstantFP(0x1.62e42fefa39efp-1, DL, MVT::f64);
2946 SDValue C2 = DAG.getConstantFP(0x1.abc9e3b39803fp-56, DL, MVT::f64);
2947 SDValue Mul2 = DAG.getNode(ISD::FMUL, DL, MVT::f64, F, C2, Flags);
2948 T = DAG.getNode(ISD::FMA, DL, MVT::f64, F, C1, Mul2, Flags);
2949 } else if (Op.getOpcode() == ISD::FEXP10) {
2950 // dn = rint(x * C1)
2951 SDValue C1 = DAG.getConstantFP(0x1.a934f0979a371p+1, DL, MVT::f64);
2952 SDValue Mul = DAG.getNode(ISD::FMUL, DL, MVT::f64, X, C1, Flags);
2953 DN = DAG.getNode(ISD::FRINT, DL, MVT::f64, Mul, Flags);
2954
2955 // f = FMA(-dn, C2, FMA(-dn, C3, x))
2956 SDValue NegDN = DAG.getNode(ISD::FNEG, DL, MVT::f64, DN, Flags);
2957 SDValue C2 = DAG.getConstantFP(-0x1.9dc1da994fd21p-59, DL, MVT::f64);
2958 SDValue C3 = DAG.getConstantFP(0x1.34413509f79ffp-2, DL, MVT::f64);
2959 SDValue Inner = DAG.getNode(ISD::FMA, DL, MVT::f64, NegDN, C3, X, Flags);
2960 F = DAG.getNode(ISD::FMA, DL, MVT::f64, NegDN, C2, Inner, Flags);
2961
2962 // t = FMA(f, C4, f*C5)
2963 SDValue C4 = DAG.getConstantFP(0x1.26bb1bbb55516p+1, DL, MVT::f64);
2964 SDValue C5 = DAG.getConstantFP(-0x1.f48ad494ea3e9p-53, DL, MVT::f64);
2965 SDValue MulF = DAG.getNode(ISD::FMUL, DL, MVT::f64, F, C5, Flags);
2966 T = DAG.getNode(ISD::FMA, DL, MVT::f64, F, C4, MulF, Flags);
2967 } else { // ISD::FEXP
2968 // dn = rint(x * C1)
2969 SDValue C1 = DAG.getConstantFP(0x1.71547652b82fep+0, DL, MVT::f64);
2970 SDValue Mul = DAG.getNode(ISD::FMUL, DL, MVT::f64, X, C1, Flags);
2971 DN = DAG.getNode(ISD::FRINT, DL, MVT::f64, Mul, Flags);
2972
2973 // t = FMA(-dn, C2, FMA(-dn, C3, x))
2974 SDValue NegDN = DAG.getNode(ISD::FNEG, DL, MVT::f64, DN, Flags);
2975 SDValue C2 = DAG.getConstantFP(0x1.abc9e3b39803fp-56, DL, MVT::f64);
2976 SDValue C3 = DAG.getConstantFP(0x1.62e42fefa39efp-1, DL, MVT::f64);
2977 SDValue Inner = DAG.getNode(ISD::FMA, DL, MVT::f64, NegDN, C3, X, Flags);
2978 T = DAG.getNode(ISD::FMA, DL, MVT::f64, NegDN, C2, Inner, Flags);
2979 }
2980
2981 // Polynomial expansion for p
2982 SDValue P = DAG.getConstantFP(0x1.ade156a5dcb37p-26, DL, MVT::f64);
2983 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2984 DAG.getConstantFP(0x1.28af3fca7ab0cp-22, DL, MVT::f64),
2985 Flags);
2986 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2987 DAG.getConstantFP(0x1.71dee623fde64p-19, DL, MVT::f64),
2988 Flags);
2989 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2990 DAG.getConstantFP(0x1.a01997c89e6b0p-16, DL, MVT::f64),
2991 Flags);
2992 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2993 DAG.getConstantFP(0x1.a01a014761f6ep-13, DL, MVT::f64),
2994 Flags);
2995 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2996 DAG.getConstantFP(0x1.6c16c1852b7b0p-10, DL, MVT::f64),
2997 Flags);
2998 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2999 DAG.getConstantFP(0x1.1111111122322p-7, DL, MVT::f64), Flags);
3000 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
3001 DAG.getConstantFP(0x1.55555555502a1p-5, DL, MVT::f64), Flags);
3002 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
3003 DAG.getConstantFP(0x1.5555555555511p-3, DL, MVT::f64), Flags);
3004 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
3005 DAG.getConstantFP(0x1.000000000000bp-1, DL, MVT::f64), Flags);
3006
3007 SDValue One = DAG.getConstantFP(1.0, DL, MVT::f64);
3008
3009 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P, One, Flags);
3010 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P, One, Flags);
3011
3012 // z = ldexp(p, (int)dn)
3013 SDValue DNInt = DAG.getNode(ISD::FP_TO_SINT, DL, MVT::i32, DN);
3014 SDValue Z = DAG.getNode(ISD::FLDEXP, DL, MVT::f64, P, DNInt, Flags);
3015
3016 // Overflow/underflow guards
3017 SDValue CondHi = DAG.getSetCC(
3018 DL, MVT::i1, X, DAG.getConstantFP(1024.0, DL, MVT::f64), ISD::SETULE);
3019
3020 if (!Flags.hasNoInfs()) {
3021 SDValue PInf = DAG.getConstantFP(std::numeric_limits<double>::infinity(),
3022 DL, MVT::f64);
3023 Z = DAG.getSelect(DL, MVT::f64, CondHi, Z, PInf, Flags);
3024 }
3025
3026 SDValue CondLo = DAG.getSetCC(
3027 DL, MVT::i1, X, DAG.getConstantFP(-1075.0, DL, MVT::f64), ISD::SETUGE);
3028 SDValue Zero = DAG.getConstantFP(0.0, DL, MVT::f64);
3029 Z = DAG.getSelect(DL, MVT::f64, CondLo, Z, Zero, Flags);
3030
3031 return Z;
3032}
3033
3035 // v_exp_f32 is good enough for OpenCL, except it doesn't handle denormals.
3036 // If we have to handle denormals, scale up the input and adjust the result.
3037
3038 EVT VT = Op.getValueType();
3039 if (VT == MVT::f64)
3040 return lowerFEXPF64(Op, DAG);
3041
3042 SDLoc SL(Op);
3043 SDValue Src = Op.getOperand(0);
3044 SDNodeFlags Flags = Op->getFlags();
3045
3046 if (VT == MVT::f16) {
3047 // Nothing in half is a denormal when promoted to f32.
3048 assert(!isTypeLegal(MVT::f16));
3049 SDValue Ext = DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, Src, Flags);
3050 SDValue Log = DAG.getNode(AMDGPUISD::EXP, SL, MVT::f32, Ext, Flags);
3051 return DAG.getNode(ISD::FP_ROUND, SL, VT, Log,
3052 DAG.getTargetConstant(0, SL, MVT::i32), Flags);
3053 }
3054
3055 assert(VT == MVT::f32);
3056
3057 if (!needsDenormHandlingF32(DAG, Src, Flags))
3058 return DAG.getNode(AMDGPUISD::EXP, SL, MVT::f32, Src, Flags);
3059
3060 // bool needs_scaling = x < -0x1.f80000p+6f;
3061 // v_exp_f32(x + (s ? 0x1.0p+6f : 0.0f)) * (s ? 0x1.0p-64f : 1.0f);
3062
3063 // -nextafter(128.0, -1)
3064 SDValue RangeCheckConst = DAG.getConstantFP(-0x1.f80000p+6f, SL, VT);
3065
3066 EVT SetCCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
3067
3068 SDValue NeedsScaling =
3069 DAG.getSetCC(SL, SetCCVT, Src, RangeCheckConst, ISD::SETOLT);
3070
3071 SDValue SixtyFour = DAG.getConstantFP(0x1.0p+6f, SL, VT);
3072 SDValue Zero = DAG.getConstantFP(0.0, SL, VT);
3073
3074 SDValue AddOffset =
3075 DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, SixtyFour, Zero);
3076
3077 SDValue AddInput = DAG.getNode(ISD::FADD, SL, VT, Src, AddOffset, Flags);
3078 SDValue Exp2 = DAG.getNode(AMDGPUISD::EXP, SL, VT, AddInput, Flags);
3079
3080 SDValue TwoExpNeg64 = DAG.getConstantFP(0x1.0p-64f, SL, VT);
3081 SDValue One = DAG.getConstantFP(1.0, SL, VT);
3082 SDValue ResultScale =
3083 DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, TwoExpNeg64, One);
3084
3085 return DAG.getNode(ISD::FMUL, SL, VT, Exp2, ResultScale, Flags);
3086}
3087
3089 SelectionDAG &DAG,
3090 SDNodeFlags Flags,
3091 bool IsExp10) const {
3092 // exp(x) -> exp2(M_LOG2E_F * x);
3093 // exp10(x) -> exp2(log2(10) * x);
3094 EVT VT = X.getValueType();
3095 SDValue Const =
3096 DAG.getConstantFP(IsExp10 ? 0x1.a934f0p+1f : numbers::log2e, SL, VT);
3097
3098 SDValue Mul = DAG.getNode(ISD::FMUL, SL, VT, X, Const, Flags);
3099 return DAG.getNode(VT == MVT::f32 ? (unsigned)AMDGPUISD::EXP
3100 : (unsigned)ISD::FEXP2,
3101 SL, VT, Mul, Flags);
3102}
3103
3105 SelectionDAG &DAG,
3106 SDNodeFlags Flags) const {
3107 EVT VT = X.getValueType();
3108 if (VT != MVT::f32 || !needsDenormHandlingF32(DAG, X, Flags))
3109 return lowerFEXPUnsafeImpl(X, SL, DAG, Flags, /*IsExp10=*/false);
3110
3111 EVT SetCCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
3112
3113 SDValue Threshold = DAG.getConstantFP(-0x1.5d58a0p+6f, SL, VT);
3114 SDValue NeedsScaling = DAG.getSetCC(SL, SetCCVT, X, Threshold, ISD::SETOLT);
3115
3116 SDValue ScaleOffset = DAG.getConstantFP(0x1.0p+6f, SL, VT);
3117
3118 SDValue ScaledX = DAG.getNode(ISD::FADD, SL, VT, X, ScaleOffset, Flags);
3119
3120 SDValue AdjustedX =
3121 DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, ScaledX, X);
3122
3123 const SDValue Log2E = DAG.getConstantFP(numbers::log2e, SL, VT);
3124 SDValue ExpInput = DAG.getNode(ISD::FMUL, SL, VT, AdjustedX, Log2E, Flags);
3125
3126 SDValue Exp2 = DAG.getNode(AMDGPUISD::EXP, SL, VT, ExpInput, Flags);
3127
3128 SDValue ResultScaleFactor = DAG.getConstantFP(0x1.969d48p-93f, SL, VT);
3129 SDValue AdjustedResult =
3130 DAG.getNode(ISD::FMUL, SL, VT, Exp2, ResultScaleFactor, Flags);
3131
3132 return DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, AdjustedResult, Exp2,
3133 Flags);
3134}
3135
3136/// Emit approx-funcs appropriate lowering for exp10. inf/nan should still be
3137/// handled correctly.
3139 SelectionDAG &DAG,
3140 SDNodeFlags Flags) const {
3141 const EVT VT = X.getValueType();
3142
3143 const unsigned Exp2Op = VT == MVT::f32 ? static_cast<unsigned>(AMDGPUISD::EXP)
3144 : static_cast<unsigned>(ISD::FEXP2);
3145
3146 if (VT != MVT::f32 || !needsDenormHandlingF32(DAG, X, Flags)) {
3147 // exp2(x * 0x1.a92000p+1f) * exp2(x * 0x1.4f0978p-11f);
3148 SDValue K0 = DAG.getConstantFP(0x1.a92000p+1f, SL, VT);
3149 SDValue K1 = DAG.getConstantFP(0x1.4f0978p-11f, SL, VT);
3150
3151 SDValue Mul0 = DAG.getNode(ISD::FMUL, SL, VT, X, K0, Flags);
3152 SDValue Exp2_0 = DAG.getNode(Exp2Op, SL, VT, Mul0, Flags);
3153 SDValue Mul1 = DAG.getNode(ISD::FMUL, SL, VT, X, K1, Flags);
3154 SDValue Exp2_1 = DAG.getNode(Exp2Op, SL, VT, Mul1, Flags);
3155 return DAG.getNode(ISD::FMUL, SL, VT, Exp2_0, Exp2_1);
3156 }
3157
3158 // bool s = x < -0x1.2f7030p+5f;
3159 // x += s ? 0x1.0p+5f : 0.0f;
3160 // exp10 = exp2(x * 0x1.a92000p+1f) *
3161 // exp2(x * 0x1.4f0978p-11f) *
3162 // (s ? 0x1.9f623ep-107f : 1.0f);
3163
3164 EVT SetCCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
3165
3166 SDValue Threshold = DAG.getConstantFP(-0x1.2f7030p+5f, SL, VT);
3167 SDValue NeedsScaling = DAG.getSetCC(SL, SetCCVT, X, Threshold, ISD::SETOLT);
3168
3169 SDValue ScaleOffset = DAG.getConstantFP(0x1.0p+5f, SL, VT);
3170 SDValue ScaledX = DAG.getNode(ISD::FADD, SL, VT, X, ScaleOffset, Flags);
3171 SDValue AdjustedX =
3172 DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, ScaledX, X);
3173
3174 SDValue K0 = DAG.getConstantFP(0x1.a92000p+1f, SL, VT);
3175 SDValue K1 = DAG.getConstantFP(0x1.4f0978p-11f, SL, VT);
3176
3177 SDValue Mul0 = DAG.getNode(ISD::FMUL, SL, VT, AdjustedX, K0, Flags);
3178 SDValue Exp2_0 = DAG.getNode(Exp2Op, SL, VT, Mul0, Flags);
3179 SDValue Mul1 = DAG.getNode(ISD::FMUL, SL, VT, AdjustedX, K1, Flags);
3180 SDValue Exp2_1 = DAG.getNode(Exp2Op, SL, VT, Mul1, Flags);
3181
3182 SDValue MulExps = DAG.getNode(ISD::FMUL, SL, VT, Exp2_0, Exp2_1, Flags);
3183
3184 SDValue ResultScaleFactor = DAG.getConstantFP(0x1.9f623ep-107f, SL, VT);
3185 SDValue AdjustedResult =
3186 DAG.getNode(ISD::FMUL, SL, VT, MulExps, ResultScaleFactor, Flags);
3187
3188 return DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, AdjustedResult, MulExps,
3189 Flags);
3190}
3191
3193 EVT VT = Op.getValueType();
3194
3195 if (VT == MVT::f64)
3196 return lowerFEXPF64(Op, DAG);
3197
3198 SDLoc SL(Op);
3199 SDValue X = Op.getOperand(0);
3200 SDNodeFlags Flags = Op->getFlags();
3201 const bool IsExp10 = Op.getOpcode() == ISD::FEXP10;
3202
3203 // TODO: Interpret allowApproxFunc as ignoring DAZ. This is currently copying
3204 // library behavior. Also, is known-not-daz source sufficient?
3205 if (allowApproxFunc(DAG, Flags)) { // TODO: Does this really require fast?
3206 return IsExp10 ? lowerFEXP10Unsafe(X, SL, DAG, Flags)
3207 : lowerFEXPUnsafe(X, SL, DAG, Flags);
3208 }
3209
3210 if (VT.getScalarType() == MVT::f16) {
3211 if (VT.isVector())
3212 return SDValue();
3213
3214 // Nothing in half is a denormal when promoted to f32.
3215 //
3216 // exp(f16 x) ->
3217 // fptrunc (v_exp_f32 (fmul (fpext x), log2e))
3218 //
3219 // exp10(f16 x) ->
3220 // fptrunc (v_exp_f32 (fmul (fpext x), log2(10)))
3221 SDValue Ext = DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, X, Flags);
3222 SDValue Lowered = lowerFEXPUnsafeImpl(Ext, SL, DAG, Flags, IsExp10);
3223 return DAG.getNode(ISD::FP_ROUND, SL, VT, Lowered,
3224 DAG.getTargetConstant(0, SL, MVT::i32), Flags);
3225 }
3226
3227 assert(VT == MVT::f32);
3228
3229 // Algorithm:
3230 //
3231 // e^x = 2^(x/ln(2)) = 2^(x*(64/ln(2))/64)
3232 //
3233 // x*(64/ln(2)) = n + f, |f| <= 0.5, n is integer
3234 // n = 64*m + j, 0 <= j < 64
3235 //
3236 // e^x = 2^((64*m + j + f)/64)
3237 // = (2^m) * (2^(j/64)) * 2^(f/64)
3238 // = (2^m) * (2^(j/64)) * e^(f*(ln(2)/64))
3239 //
3240 // f = x*(64/ln(2)) - n
3241 // r = f*(ln(2)/64) = x - n*(ln(2)/64)
3242 //
3243 // e^x = (2^m) * (2^(j/64)) * e^r
3244 //
3245 // (2^(j/64)) is precomputed
3246 //
3247 // e^r = 1 + r + (r^2)/2! + (r^3)/3! + (r^4)/4! + (r^5)/5!
3248 // e^r = 1 + q
3249 //
3250 // q = r + (r^2)/2! + (r^3)/3! + (r^4)/4! + (r^5)/5!
3251 //
3252 // e^x = (2^m) * ( (2^(j/64)) + q*(2^(j/64)) )
3253 SDNodeFlags FlagsNoContract = Flags;
3254 FlagsNoContract.setAllowContract(false);
3255
3256 SDValue PH, PL;
3257 if (Subtarget->hasFastFMAF32()) {
3258 const float c_exp = numbers::log2ef;
3259 const float cc_exp = 0x1.4ae0bep-26f; // c+cc are 49 bits
3260 const float c_exp10 = 0x1.a934f0p+1f;
3261 const float cc_exp10 = 0x1.2f346ep-24f;
3262
3263 SDValue C = DAG.getConstantFP(IsExp10 ? c_exp10 : c_exp, SL, VT);
3264 SDValue CC = DAG.getConstantFP(IsExp10 ? cc_exp10 : cc_exp, SL, VT);
3265
3266 PH = DAG.getNode(ISD::FMUL, SL, VT, X, C, Flags);
3267 SDValue NegPH = DAG.getNode(ISD::FNEG, SL, VT, PH, Flags);
3268 SDValue FMA0 = DAG.getNode(ISD::FMA, SL, VT, X, C, NegPH, Flags);
3269 PL = DAG.getNode(ISD::FMA, SL, VT, X, CC, FMA0, Flags);
3270 } else {
3271 const float ch_exp = 0x1.714000p+0f;
3272 const float cl_exp = 0x1.47652ap-12f; // ch + cl are 36 bits
3273
3274 const float ch_exp10 = 0x1.a92000p+1f;
3275 const float cl_exp10 = 0x1.4f0978p-11f;
3276
3277 SDValue CH = DAG.getConstantFP(IsExp10 ? ch_exp10 : ch_exp, SL, VT);
3278 SDValue CL = DAG.getConstantFP(IsExp10 ? cl_exp10 : cl_exp, SL, VT);
3279
3280 SDValue XAsInt = DAG.getNode(ISD::BITCAST, SL, MVT::i32, X);
3281 SDValue MaskConst = DAG.getConstant(0xfffff000, SL, MVT::i32);
3282 SDValue XHAsInt = DAG.getNode(ISD::AND, SL, MVT::i32, XAsInt, MaskConst);
3283 SDValue XH = DAG.getNode(ISD::BITCAST, SL, VT, XHAsInt);
3284 SDValue XL = DAG.getNode(ISD::FSUB, SL, VT, X, XH, Flags);
3285
3286 PH = DAG.getNode(ISD::FMUL, SL, VT, XH, CH, Flags);
3287
3288 SDValue XLCL = DAG.getNode(ISD::FMUL, SL, VT, XL, CL, Flags);
3289 SDValue Mad0 = getMad(DAG, SL, VT, XL, CH, XLCL, Flags);
3290 PL = getMad(DAG, SL, VT, XH, CL, Mad0, Flags);
3291 }
3292
3293 SDValue E = DAG.getNode(ISD::FROUNDEVEN, SL, VT, PH, Flags);
3294
3295 // It is unsafe to contract this fsub into the PH multiply.
3296 SDValue PHSubE = DAG.getNode(ISD::FSUB, SL, VT, PH, E, FlagsNoContract);
3297
3298 SDValue A = DAG.getNode(ISD::FADD, SL, VT, PHSubE, PL, Flags);
3299 SDValue IntE = DAG.getNode(ISD::FP_TO_SINT, SL, MVT::i32, E);
3300 SDValue Exp2 = DAG.getNode(AMDGPUISD::EXP, SL, VT, A, Flags);
3301
3302 SDValue R = DAG.getNode(ISD::FLDEXP, SL, VT, Exp2, IntE, Flags);
3303
3304 SDValue UnderflowCheckConst =
3305 DAG.getConstantFP(IsExp10 ? -0x1.66d3e8p+5f : -0x1.9d1da0p+6f, SL, VT);
3306
3307 EVT SetCCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
3308 SDValue Zero = DAG.getConstantFP(0.0, SL, VT);
3309 SDValue Underflow =
3310 DAG.getSetCC(SL, SetCCVT, X, UnderflowCheckConst, ISD::SETOLT);
3311
3312 R = DAG.getNode(ISD::SELECT, SL, VT, Underflow, Zero, R);
3313
3314 if (!Flags.hasNoInfs()) {
3315 SDValue OverflowCheckConst =
3316 DAG.getConstantFP(IsExp10 ? 0x1.344136p+5f : 0x1.62e430p+6f, SL, VT);
3317 SDValue Overflow =
3318 DAG.getSetCC(SL, SetCCVT, X, OverflowCheckConst, ISD::SETOGT);
3319 SDValue Inf =
3321 R = DAG.getNode(ISD::SELECT, SL, VT, Overflow, Inf, R);
3322 }
3323
3324 return R;
3325}
3326
3327// No pow instruction or libcall to fall back on. fmul_legacy returns 0 for a
3328// zero operand even against an infinity or a NaN, so pow(x, 0) and pow(1, y)
3329// fall out as exp2(0) = 1.
3331 EVT VT = Op.getValueType();
3332 assert(VT == MVT::f32);
3333
3334 SDLoc SL(Op);
3335 SDValue X = Op.getOperand(0);
3336 SDValue Y = Op.getOperand(1);
3337 SDNodeFlags Flags = Op->getFlags();
3338
3339 // log2(0) is -inf, which exp2 turns back into a finite result, so the core
3340 // goes infinite for inputs a ninf fpow still asserts about, like pow(0, 2).
3341 SDNodeFlags CoreFlags = Flags;
3342 CoreFlags.setNoInfs(false);
3343
3344 // Fast expansion: ignores denormals, NaN for a negative base.
3345 if (allowApproxFunc(DAG, Flags)) {
3346 SDValue Log = DAG.getNode(AMDGPUISD::LOG, SL, VT, X, CoreFlags);
3347 SDValue Mul =
3348 DAG.getNode(AMDGPUISD::FMUL_LEGACY, SL, VT, Y, Log, CoreFlags);
3349 return DAG.getNode(AMDGPUISD::EXP, SL, VT, Mul, CoreFlags);
3350 }
3351
3352 SDValue Abs = DAG.getNode(ISD::FABS, SL, VT, X, Flags);
3353 SDValue Log = DAG.getNode(ISD::FLOG2, SL, VT, Abs, CoreFlags);
3354 SDValue Mul = DAG.getNode(AMDGPUISD::FMUL_LEGACY, SL, VT, Y, Log, CoreFlags);
3355 SDValue R = DAG.getNode(ISD::FEXP2, SL, VT, Mul, CoreFlags);
3356
3357 // A base that is never negative needs neither the sign fixup nor the NaN.
3359 return R;
3360
3361 EVT SetCCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
3362
3363 // Infinities count as integers, and every f32 >= 2^24 in magnitude is even.
3364 SDValue YTrunc = DAG.getNode(ISD::FTRUNC, SL, VT, Y);
3365 SDValue YIsInt = DAG.getSetCC(SL, SetCCVT, YTrunc, Y, ISD::SETOEQ);
3366 SDValue YHalf =
3367 DAG.getNode(ISD::FMUL, SL, VT, Y, DAG.getConstantFP(0.5, SL, VT));
3368 SDValue YHalfTrunc = DAG.getNode(ISD::FTRUNC, SL, VT, YHalf);
3369 SDValue YIsOdd =
3370 DAG.getNode(ISD::AND, SL, SetCCVT, YIsInt,
3371 DAG.getSetCC(SL, SetCCVT, YHalfTrunc, YHalf, ISD::SETONE));
3372
3373 // pow(-x, odd y) = -pow(x, y). Selecting the copysign lets even y fold it.
3374 R = DAG.getNode(ISD::SELECT, SL, VT, YIsOdd,
3375 DAG.getNode(ISD::FCOPYSIGN, SL, VT, R, X), R);
3376
3377 if (Flags.hasNoNaNs())
3378 return R;
3379
3380 // A negative finite base to a non-integral power is NaN. -inf is excluded:
3381 // the core already gives pow(+inf, y). So are subnormals when flushed.
3382 FPClassTest NegFiniteMask = fcNegNormal;
3383 if (!DAG.getMachineFunction()
3385 .inputsAreZero())
3386 NegFiniteMask |= fcNegSubnormal;
3387 SDValue XNegFinite =
3388 DAG.getNode(ISD::IS_FPCLASS, SL, SetCCVT, X,
3389 DAG.getTargetConstant(NegFiniteMask, SL, MVT::i32));
3390 // Not a SETONE compare: pow(-1, NaN) needs the NaN-true behavior of !SETOEQ.
3391 SDValue NegNonInt = DAG.getNode(ISD::AND, SL, SetCCVT, XNegFinite,
3392 DAG.getNOT(SL, YIsInt, SetCCVT));
3393 SDValue NaN =
3395 return DAG.getNode(ISD::SELECT, SL, VT, NegNonInt, NaN, R);
3396}
3397
3398static bool isCtlzOpc(unsigned Opc) {
3399 return Opc == ISD::CTLZ || Opc == ISD::CTLZ_ZERO_POISON;
3400}
3401
3402static bool isCttzOpc(unsigned Opc) {
3403 return Opc == ISD::CTTZ || Opc == ISD::CTTZ_ZERO_POISON;
3404}
3405
3407 SelectionDAG &DAG) const {
3408 auto SL = SDLoc(Op);
3409 auto Opc = Op.getOpcode();
3410 auto Arg = Op.getOperand(0u);
3411 auto ResultVT = Op.getValueType();
3412
3413 if (ResultVT != MVT::i8 && ResultVT != MVT::i16)
3414 return {};
3415
3417 assert(ResultVT == Arg.getValueType());
3418
3419 const uint64_t NumBits = ResultVT.getFixedSizeInBits();
3420 SDValue NumExtBits = DAG.getConstant(32u - NumBits, SL, MVT::i32);
3421 SDValue NewOp;
3422
3423 if (Opc == ISD::CTLZ_ZERO_POISON) {
3424 NewOp = DAG.getNode(ISD::ANY_EXTEND, SL, MVT::i32, Arg);
3425 NewOp = DAG.getNode(ISD::SHL, SL, MVT::i32, NewOp, NumExtBits);
3426 NewOp = DAG.getNode(Opc, SL, MVT::i32, NewOp);
3427 } else {
3428 NewOp = DAG.getNode(ISD::ZERO_EXTEND, SL, MVT::i32, Arg);
3429 NewOp = DAG.getNode(Opc, SL, MVT::i32, NewOp);
3430 NewOp = DAG.getNode(ISD::SUB, SL, MVT::i32, NewOp, NumExtBits);
3431 }
3432
3433 return DAG.getNode(ISD::TRUNCATE, SL, ResultVT, NewOp);
3434}
3435
3437 SDLoc SL(Op);
3438 SDValue Src = Op.getOperand(0);
3439
3440 assert(isCtlzOpc(Op.getOpcode()) || isCttzOpc(Op.getOpcode()));
3441 bool Ctlz = isCtlzOpc(Op.getOpcode());
3442 unsigned NewOpc = Ctlz ? AMDGPUISD::FFBH_U32 : AMDGPUISD::FFBL_B32;
3443
3444 bool ZeroUndef = Op.getOpcode() == ISD::CTLZ_ZERO_POISON ||
3445 Op.getOpcode() == ISD::CTTZ_ZERO_POISON;
3446 bool Is64BitScalar = !Src->isDivergent() && Src.getValueType() == MVT::i64;
3447
3448 if (Src.getValueType() == MVT::i32 || Is64BitScalar) {
3449 // (ctlz hi:lo) -> (umin (ffbh src), 32)
3450 // (cttz hi:lo) -> (umin (ffbl src), 32)
3451 // (ctlz_zero_poison src) -> (ffbh src)
3452 // (cttz_zero_poison src) -> (ffbl src)
3453
3454 // 64-bit scalar version produce 32-bit result
3455 // (ctlz hi:lo) -> (umin (S_FLBIT_I32_B64 src), 64)
3456 // (cttz hi:lo) -> (umin (S_FF1_I32_B64 src), 64)
3457 // (ctlz_zero_poison src) -> (S_FLBIT_I32_B64 src)
3458 // (cttz_zero_poison src) -> (S_FF1_I32_B64 src)
3459 SDValue NewOpr = DAG.getNode(NewOpc, SL, MVT::i32, Src);
3460 if (!ZeroUndef) {
3461 const SDValue ConstVal = DAG.getConstant(
3462 Op.getValueType().getScalarSizeInBits(), SL, MVT::i32);
3463 NewOpr = DAG.getNode(ISD::UMIN, SL, MVT::i32, NewOpr, ConstVal);
3464 }
3465 return DAG.getNode(ISD::ZERO_EXTEND, SL, Src.getValueType(), NewOpr);
3466 }
3467
3468 SDValue Lo, Hi;
3469 std::tie(Lo, Hi) = split64BitValue(Src, DAG);
3470
3471 SDValue OprLo = DAG.getNode(NewOpc, SL, MVT::i32, Lo);
3472 SDValue OprHi = DAG.getNode(NewOpc, SL, MVT::i32, Hi);
3473
3474 // (ctlz hi:lo) -> (umin3 (ffbh hi), (uaddsat (ffbh lo), 32), 64)
3475 // (cttz hi:lo) -> (umin3 (uaddsat (ffbl hi), 32), (ffbl lo), 64)
3476 // (ctlz_zero_poison hi:lo) -> (umin (ffbh hi), (add (ffbh lo), 32))
3477 // (cttz_zero_poison hi:lo) -> (umin (add (ffbl hi), 32), (ffbl lo))
3478
3479 unsigned AddOpc = ZeroUndef ? ISD::ADD : ISD::UADDSAT;
3480 const SDValue Const32 = DAG.getConstant(32, SL, MVT::i32);
3481 if (Ctlz)
3482 OprLo = DAG.getNode(AddOpc, SL, MVT::i32, OprLo, Const32);
3483 else
3484 OprHi = DAG.getNode(AddOpc, SL, MVT::i32, OprHi, Const32);
3485
3486 SDValue NewOpr;
3487 NewOpr = DAG.getNode(ISD::UMIN, SL, MVT::i32, OprLo, OprHi);
3488 if (!ZeroUndef) {
3489 const SDValue Const64 = DAG.getConstant(64, SL, MVT::i32);
3490 NewOpr = DAG.getNode(ISD::UMIN, SL, MVT::i32, NewOpr, Const64);
3491 }
3492
3493 return DAG.getNode(ISD::ZERO_EXTEND, SL, MVT::i64, NewOpr);
3494}
3495
3497 SDLoc SL(Op);
3498 SDValue Src = Op.getOperand(0);
3499 assert(Src.getValueType() == MVT::i32 && "LowerCTLS only supports i32");
3500 SDValue Ffbh = DAG.getNode(
3501 ISD::INTRINSIC_WO_CHAIN, SL, MVT::i32,
3502 DAG.getTargetConstant(Intrinsic::amdgcn_sffbh, SL, MVT::i32), Src);
3503 SDValue Clamped = DAG.getNode(ISD::UMIN, SL, MVT::i32, Ffbh,
3504 DAG.getConstant(32, SL, MVT::i32));
3505 return DAG.getNode(ISD::ADD, SL, MVT::i32, Clamped,
3506 DAG.getAllOnesConstant(SL, MVT::i32));
3507}
3508
3510 EVT FP16Ty) const {
3511 assert(FP16Ty == MVT::f16 || FP16Ty == MVT::bf16);
3512 SDLoc SL(Op);
3513 SDValue Src = Op.getOperand(0);
3514 SDValue ToF32 = DAG.getNode(Op.getOpcode(), SL, MVT::f32, Src);
3515 SDValue FPRoundFlag = DAG.getIntPtrConstant(0, SL, /*isTarget=*/true);
3516 return DAG.getNode(ISD::FP_ROUND, SL, FP16Ty, ToF32, FPRoundFlag);
3517}
3518
3520 bool Signed) const {
3521 // The regular method converting a 64-bit integer to float roughly consists of
3522 // 2 steps: normalization and rounding. In fact, after normalization, the
3523 // conversion from a 64-bit integer to a float is essentially the same as the
3524 // one from a 32-bit integer. The only difference is that it has more
3525 // trailing bits to be rounded. To leverage the native 32-bit conversion, a
3526 // 64-bit integer could be preprocessed and fit into a 32-bit integer then
3527 // converted into the correct float number. The basic steps for the unsigned
3528 // conversion are illustrated in the following pseudo code:
3529 //
3530 // f32 uitofp(i64 u) {
3531 // i32 hi, lo = split(u);
3532 // // Only count the leading zeros in hi as we have native support of the
3533 // // conversion from i32 to f32. If hi is all 0s, the conversion is
3534 // // reduced to a 32-bit one automatically.
3535 // i32 shamt = clz(hi); // Return 32 if hi is all 0s.
3536 // u <<= shamt;
3537 // hi, lo = split(u);
3538 // hi |= (lo != 0) ? 1 : 0; // Adjust rounding bit in hi based on lo.
3539 // // convert it as a 32-bit integer and scale the result back.
3540 // return uitofp(hi) * 2^(32 - shamt);
3541 // }
3542 //
3543 // The signed one follows the same principle but uses 'ffbh_i32' to count its
3544 // sign bits instead. If 'ffbh_i32' is not available, its absolute value is
3545 // converted instead followed by negation based its sign bit.
3546
3547 SDLoc SL(Op);
3548 SDValue Src = Op.getOperand(0);
3549
3550 SDValue Lo, Hi;
3551 std::tie(Lo, Hi) = split64BitValue(Src, DAG);
3552 SDValue Sign;
3553 SDValue ShAmt;
3554 if (Signed && Subtarget->isGCN()) {
3555 // We also need to consider the sign bit in Lo if Hi has just sign bits,
3556 // i.e. Hi is 0 or -1. However, that only needs to take the MSB into
3557 // account. That is, the maximal shift is
3558 // - 32 if Lo and Hi have opposite signs;
3559 // - 33 if Lo and Hi have the same sign.
3560 //
3561 // Or, MaxShAmt = 33 + OppositeSign, where
3562 //
3563 // OppositeSign is defined as ((Lo ^ Hi) >> 31), which is
3564 // - -1 if Lo and Hi have opposite signs; and
3565 // - 0 otherwise.
3566 //
3567 // All in all, ShAmt is calculated as
3568 //
3569 // umin(sffbh(Hi), 33 + (Lo^Hi)>>31) - 1.
3570 //
3571 // or
3572 //
3573 // umin(sffbh(Hi) - 1, 32 + (Lo^Hi)>>31).
3574 //
3575 // to reduce the critical path.
3576 SDValue OppositeSign = DAG.getNode(
3577 ISD::SRA, SL, MVT::i32, DAG.getNode(ISD::XOR, SL, MVT::i32, Lo, Hi),
3578 DAG.getConstant(31, SL, MVT::i32));
3579 SDValue MaxShAmt =
3580 DAG.getNode(ISD::ADD, SL, MVT::i32, DAG.getConstant(32, SL, MVT::i32),
3581 OppositeSign);
3582 // Count the leading sign bits.
3583 ShAmt = DAG.getNode(
3584 ISD::INTRINSIC_WO_CHAIN, SL, MVT::i32,
3585 DAG.getTargetConstant(Intrinsic::amdgcn_sffbh, SL, MVT::i32), Hi);
3586 // Different from unsigned conversion, the shift should be one bit less to
3587 // preserve the sign bit.
3588 ShAmt = DAG.getNode(ISD::SUB, SL, MVT::i32, ShAmt,
3589 DAG.getConstant(1, SL, MVT::i32));
3590 ShAmt = DAG.getNode(ISD::UMIN, SL, MVT::i32, ShAmt, MaxShAmt);
3591 } else {
3592 if (Signed) {
3593 // Without 'ffbh_i32', only leading zeros could be counted. Take the
3594 // absolute value first.
3595 Sign = DAG.getNode(ISD::SRA, SL, MVT::i64, Src,
3596 DAG.getConstant(63, SL, MVT::i64));
3597 SDValue Abs =
3598 DAG.getNode(ISD::XOR, SL, MVT::i64,
3599 DAG.getNode(ISD::ADD, SL, MVT::i64, Src, Sign), Sign);
3600 std::tie(Lo, Hi) = split64BitValue(Abs, DAG);
3601 }
3602 // Count the leading zeros.
3603 ShAmt = DAG.getNode(ISD::CTLZ, SL, MVT::i32, Hi);
3604 // The shift amount for signed integers is [0, 32].
3605 }
3606 // Normalize the given 64-bit integer.
3607 SDValue Norm = DAG.getNode(ISD::SHL, SL, MVT::i64, Src, ShAmt);
3608 // Split it again.
3609 std::tie(Lo, Hi) = split64BitValue(Norm, DAG);
3610 // Calculate the adjust bit for rounding.
3611 // (lo != 0) ? 1 : 0 => (lo >= 1) ? 1 : 0 => umin(1, lo)
3612 SDValue Adjust = DAG.getNode(ISD::UMIN, SL, MVT::i32,
3613 DAG.getConstant(1, SL, MVT::i32), Lo);
3614 // Get the 32-bit normalized integer.
3615 Norm = DAG.getNode(ISD::OR, SL, MVT::i32, Hi, Adjust);
3616 // Convert the normalized 32-bit integer into f32.
3617
3618 bool UseLDEXP = isOperationLegal(ISD::FLDEXP, MVT::f32);
3619 unsigned Opc = Signed && UseLDEXP ? ISD::SINT_TO_FP : ISD::UINT_TO_FP;
3620 SDValue FVal = DAG.getNode(Opc, SL, MVT::f32, Norm);
3621
3622 // Finally, need to scale back the converted floating number as the original
3623 // 64-bit integer is converted as a 32-bit one.
3624 ShAmt = DAG.getNode(ISD::SUB, SL, MVT::i32, DAG.getConstant(32, SL, MVT::i32),
3625 ShAmt);
3626 // On GCN, use LDEXP directly.
3627 if (UseLDEXP)
3628 return DAG.getNode(ISD::FLDEXP, SL, MVT::f32, FVal, ShAmt);
3629
3630 // Otherwise, align 'ShAmt' to the exponent part and add it into the exponent
3631 // part directly to emulate the multiplication of 2^ShAmt. That 8-bit
3632 // exponent is enough to avoid overflowing into the sign bit.
3633 SDValue Exp = DAG.getNode(ISD::SHL, SL, MVT::i32, ShAmt,
3634 DAG.getConstant(23, SL, MVT::i32));
3635 SDValue IVal =
3636 DAG.getNode(ISD::ADD, SL, MVT::i32,
3637 DAG.getNode(ISD::BITCAST, SL, MVT::i32, FVal), Exp);
3638 if (Signed) {
3639 // Set the sign bit.
3640 Sign = DAG.getNode(ISD::SHL, SL, MVT::i32,
3641 DAG.getNode(ISD::TRUNCATE, SL, MVT::i32, Sign),
3642 DAG.getConstant(31, SL, MVT::i32));
3643 IVal = DAG.getNode(ISD::OR, SL, MVT::i32, IVal, Sign);
3644 }
3645 return DAG.getNode(ISD::BITCAST, SL, MVT::f32, IVal);
3646}
3647
3649 bool Signed) const {
3650 SDLoc SL(Op);
3651 SDValue Src = Op.getOperand(0);
3652
3653 SDValue Lo, Hi;
3654 std::tie(Lo, Hi) = split64BitValue(Src, DAG);
3655
3657 SL, MVT::f64, Hi);
3658
3659 SDValue CvtLo = DAG.getNode(ISD::UINT_TO_FP, SL, MVT::f64, Lo);
3660
3661 SDValue LdExp = DAG.getNode(ISD::FLDEXP, SL, MVT::f64, CvtHi,
3662 DAG.getConstant(32, SL, MVT::i32));
3663 // TODO: Should this propagate fast-math-flags?
3664 return DAG.getNode(ISD::FADD, SL, MVT::f64, LdExp, CvtLo);
3665}
3666
3668 bool Signed) const {
3669 EVT DestVT = Op.getValueType();
3670 SDValue Src = Op.getOperand(0);
3671 EVT SrcVT = Src.getValueType();
3672 unsigned ExtOpc = Signed ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND;
3673 unsigned CvtOpc = Signed ? ISD::SINT_TO_FP : ISD::UINT_TO_FP;
3674
3675 if (SrcVT == MVT::i16) {
3676 if (DestVT == MVT::f16)
3677 return Op;
3678
3679 SDLoc DL(Op);
3680 SDValue Ext = DAG.getNode(ExtOpc, DL, MVT::i32, Src);
3681 return DAG.getNode(CvtOpc, DL, DestVT, Ext);
3682 }
3683
3684 if (DestVT == MVT::bf16 || DestVT == MVT::f16)
3685 return LowerINT_TO_FP16(Op, DAG, DestVT);
3686
3687 if (SrcVT != MVT::i64)
3688 return Op;
3689
3690 if (DestVT == MVT::f32)
3691 return LowerINT_TO_FP32(Op, DAG, Signed);
3692
3693 assert(DestVT == MVT::f64);
3694 return LowerINT_TO_FP64(Op, DAG, Signed);
3695}
3696
3701
3706
3708 bool Signed) const {
3709 SDLoc SL(Op);
3710
3711 SDValue Src = Op.getOperand(0);
3712 EVT SrcVT = Src.getValueType();
3713
3714 assert(SrcVT == MVT::f32 || SrcVT == MVT::f64);
3715
3716 // The basic idea of converting a floating point number into a pair of 32-bit
3717 // integers is illustrated as follows:
3718 //
3719 // tf := trunc(val);
3720 // hif := floor(tf * 2^-32);
3721 // lof := tf - hif * 2^32; // lof is always positive due to floor.
3722 // hi := fptoi(hif);
3723 // lo := fptoi(lof);
3724 //
3725 SDValue Trunc = DAG.getNode(ISD::FTRUNC, SL, SrcVT, Src);
3726 SDValue Sign;
3727 if (Signed && SrcVT == MVT::f32) {
3728 // However, a 32-bit floating point number has only 23 bits mantissa and
3729 // it's not enough to hold all the significant bits of `lof` if val is
3730 // negative. To avoid the loss of precision, We need to take the absolute
3731 // value after truncating and flip the result back based on the original
3732 // signedness.
3733 Sign = DAG.getNode(ISD::SRA, SL, MVT::i32,
3734 DAG.getNode(ISD::BITCAST, SL, MVT::i32, Trunc),
3735 DAG.getConstant(31, SL, MVT::i32));
3736 Trunc = DAG.getNode(ISD::FABS, SL, SrcVT, Trunc);
3737 }
3738
3739 SDValue K0, K1;
3740 if (SrcVT == MVT::f64) {
3741 K0 = DAG.getConstantFP(
3742 llvm::bit_cast<double>(UINT64_C(/*2^-32*/ 0x3df0000000000000)), SL,
3743 SrcVT);
3744 K1 = DAG.getConstantFP(
3745 llvm::bit_cast<double>(UINT64_C(/*-2^32*/ 0xc1f0000000000000)), SL,
3746 SrcVT);
3747 } else {
3748 K0 = DAG.getConstantFP(
3749 llvm::bit_cast<float>(UINT32_C(/*2^-32*/ 0x2f800000)), SL, SrcVT);
3750 K1 = DAG.getConstantFP(
3751 llvm::bit_cast<float>(UINT32_C(/*-2^32*/ 0xcf800000)), SL, SrcVT);
3752 }
3753 // TODO: Should this propagate fast-math-flags?
3754 SDValue Mul = DAG.getNode(ISD::FMUL, SL, SrcVT, Trunc, K0);
3755
3756 SDValue FloorMul = DAG.getNode(ISD::FFLOOR, SL, SrcVT, Mul);
3757
3758 SDValue Fma = DAG.getNode(ISD::FMA, SL, SrcVT, FloorMul, K1, Trunc);
3759
3760 SDValue Hi = DAG.getNode((Signed && SrcVT == MVT::f64) ? ISD::FP_TO_SINT
3762 SL, MVT::i32, FloorMul);
3763 SDValue Lo = DAG.getNode(ISD::FP_TO_UINT, SL, MVT::i32, Fma);
3764
3765 SDValue Result = DAG.getNode(ISD::BITCAST, SL, MVT::i64,
3766 DAG.getBuildVector(MVT::v2i32, SL, {Lo, Hi}));
3767
3768 if (Signed && SrcVT == MVT::f32) {
3769 assert(Sign);
3770 // Flip the result based on the signedness, which is either all 0s or 1s.
3771 Sign = DAG.getNode(ISD::BITCAST, SL, MVT::i64,
3772 DAG.getBuildVector(MVT::v2i32, SL, {Sign, Sign}));
3773 // r := xor(r, sign) - sign;
3774 Result =
3775 DAG.getNode(ISD::SUB, SL, MVT::i64,
3776 DAG.getNode(ISD::XOR, SL, MVT::i64, Result, Sign), Sign);
3777 }
3778
3779 return Result;
3780}
3781
3783 SDLoc DL(Op);
3784 SDValue N0 = Op.getOperand(0);
3785
3786 // Convert to target node to get known bits
3787 if (N0.getValueType() == MVT::f32)
3788 return DAG.getNode(AMDGPUISD::FP_TO_FP16, DL, Op.getValueType(), N0);
3789
3790 if (Op->getFlags().hasApproximateFuncs()) {
3791 // There is a generic expand for FP_TO_FP16 with unsafe fast math.
3792 return SDValue();
3793 }
3794
3795 return LowerF64ToF16Safe(N0, DL, DAG);
3796}
3797
3798// return node in i32
3800 SelectionDAG &DAG) const {
3801 assert(Src.getSimpleValueType() == MVT::f64);
3802
3803 // f64 -> f16 conversion using round-to-nearest-even rounding mode.
3804 // TODO: We can generate better code for True16.
3805 const unsigned ExpMask = 0x7ff;
3806 const unsigned ExpBiasf64 = 1023;
3807 const unsigned ExpBiasf16 = 15;
3808 SDValue Zero = DAG.getConstant(0, DL, MVT::i32);
3809 SDValue One = DAG.getConstant(1, DL, MVT::i32);
3810 SDValue U = DAG.getNode(ISD::BITCAST, DL, MVT::i64, Src);
3811 SDValue UH = DAG.getNode(ISD::SRL, DL, MVT::i64, U,
3812 DAG.getConstant(32, DL, MVT::i64));
3813 UH = DAG.getZExtOrTrunc(UH, DL, MVT::i32);
3814 U = DAG.getZExtOrTrunc(U, DL, MVT::i32);
3815 SDValue E = DAG.getNode(ISD::SRL, DL, MVT::i32, UH,
3816 DAG.getConstant(20, DL, MVT::i64));
3817 E = DAG.getNode(ISD::AND, DL, MVT::i32, E,
3818 DAG.getConstant(ExpMask, DL, MVT::i32));
3819 // Subtract the fp64 exponent bias (1023) to get the real exponent and
3820 // add the f16 bias (15) to get the biased exponent for the f16 format.
3821 E = DAG.getNode(ISD::ADD, DL, MVT::i32, E,
3822 DAG.getConstant(-ExpBiasf64 + ExpBiasf16, DL, MVT::i32));
3823
3824 SDValue M = DAG.getNode(ISD::SRL, DL, MVT::i32, UH,
3825 DAG.getConstant(8, DL, MVT::i32));
3826 M = DAG.getNode(ISD::AND, DL, MVT::i32, M,
3827 DAG.getConstant(0xffe, DL, MVT::i32));
3828
3829 SDValue MaskedSig = DAG.getNode(ISD::AND, DL, MVT::i32, UH,
3830 DAG.getConstant(0x1ff, DL, MVT::i32));
3831 MaskedSig = DAG.getNode(ISD::OR, DL, MVT::i32, MaskedSig, U);
3832
3833 SDValue Lo40Set = DAG.getSelectCC(DL, MaskedSig, Zero, Zero, One, ISD::SETEQ);
3834 M = DAG.getNode(ISD::OR, DL, MVT::i32, M, Lo40Set);
3835
3836 // (M != 0 ? 0x0200 : 0) | 0x7c00;
3837 SDValue I = DAG.getNode(ISD::OR, DL, MVT::i32,
3838 DAG.getSelectCC(DL, M, Zero, DAG.getConstant(0x0200, DL, MVT::i32),
3839 Zero, ISD::SETNE), DAG.getConstant(0x7c00, DL, MVT::i32));
3840
3841 // N = M | (E << 12);
3842 SDValue N = DAG.getNode(ISD::OR, DL, MVT::i32, M,
3843 DAG.getNode(ISD::SHL, DL, MVT::i32, E,
3844 DAG.getConstant(12, DL, MVT::i32)));
3845
3846 // B = clamp(1-E, 0, 13);
3847 SDValue OneSubExp = DAG.getNode(ISD::SUB, DL, MVT::i32,
3848 One, E);
3849 SDValue B = DAG.getNode(ISD::SMAX, DL, MVT::i32, OneSubExp, Zero);
3850 B = DAG.getNode(ISD::SMIN, DL, MVT::i32, B,
3851 DAG.getConstant(13, DL, MVT::i32));
3852
3853 SDValue SigSetHigh = DAG.getNode(ISD::OR, DL, MVT::i32, M,
3854 DAG.getConstant(0x1000, DL, MVT::i32));
3855
3856 SDValue D = DAG.getNode(ISD::SRL, DL, MVT::i32, SigSetHigh, B);
3857 SDValue D0 = DAG.getNode(ISD::SHL, DL, MVT::i32, D, B);
3858 SDValue D1 = DAG.getSelectCC(DL, D0, SigSetHigh, One, Zero, ISD::SETNE);
3859 D = DAG.getNode(ISD::OR, DL, MVT::i32, D, D1);
3860
3861 SDValue V = DAG.getSelectCC(DL, E, One, D, N, ISD::SETLT);
3862 SDValue VLow3 = DAG.getNode(ISD::AND, DL, MVT::i32, V,
3863 DAG.getConstant(0x7, DL, MVT::i32));
3864 V = DAG.getNode(ISD::SRL, DL, MVT::i32, V,
3865 DAG.getConstant(2, DL, MVT::i32));
3866 SDValue V0 = DAG.getSelectCC(DL, VLow3, DAG.getConstant(3, DL, MVT::i32),
3867 One, Zero, ISD::SETEQ);
3868 SDValue V1 = DAG.getSelectCC(DL, VLow3, DAG.getConstant(5, DL, MVT::i32),
3869 One, Zero, ISD::SETGT);
3870 V1 = DAG.getNode(ISD::OR, DL, MVT::i32, V0, V1);
3871 V = DAG.getNode(ISD::ADD, DL, MVT::i32, V, V1);
3872
3873 V = DAG.getSelectCC(DL, E, DAG.getConstant(30, DL, MVT::i32),
3874 DAG.getConstant(0x7c00, DL, MVT::i32), V, ISD::SETGT);
3875 V = DAG.getSelectCC(DL, E, DAG.getConstant(1039, DL, MVT::i32),
3876 I, V, ISD::SETEQ);
3877
3878 // Extract the sign bit.
3879 SDValue Sign = DAG.getNode(ISD::SRL, DL, MVT::i32, UH,
3880 DAG.getConstant(16, DL, MVT::i32));
3881 Sign = DAG.getNode(ISD::AND, DL, MVT::i32, Sign,
3882 DAG.getConstant(0x8000, DL, MVT::i32));
3883
3884 return DAG.getNode(ISD::OR, DL, MVT::i32, Sign, V);
3885}
3886
3888 SelectionDAG &DAG) const {
3889 SDValue Src = Op.getOperand(0);
3890 unsigned OpOpcode = Op.getOpcode();
3891 EVT SrcVT = Src.getValueType();
3892 EVT DestVT = Op.getValueType();
3893
3894 // Will be selected natively
3895 if (SrcVT == MVT::f16 && DestVT == MVT::i16)
3896 return Op;
3897
3898 if (SrcVT == MVT::bf16 || (SrcVT == MVT::f16 && DestVT == MVT::i32)) {
3899 SDLoc DL(Op);
3900 SDValue PromotedSrc = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, Src);
3901 return DAG.getNode(Op.getOpcode(), DL, DestVT, PromotedSrc);
3902 }
3903
3904 // Promote i16 to i32
3905 if (DestVT == MVT::i16 && (SrcVT == MVT::f32 || SrcVT == MVT::f64)) {
3906 SDLoc DL(Op);
3907
3908 SDValue FpToInt32 = DAG.getNode(OpOpcode, DL, MVT::i32, Src);
3909 return DAG.getNode(ISD::TRUNCATE, DL, MVT::i16, FpToInt32);
3910 }
3911
3912 if (DestVT != MVT::i64)
3913 return Op;
3914
3915 if (SrcVT == MVT::f16 ||
3916 (SrcVT == MVT::f32 && Src.getOpcode() == ISD::FP16_TO_FP)) {
3917 SDLoc DL(Op);
3918
3919 SDValue FpToInt32 = DAG.getNode(OpOpcode, DL, MVT::i32, Src);
3920 unsigned Ext =
3922 return DAG.getNode(Ext, DL, MVT::i64, FpToInt32);
3923 }
3924
3925 if (SrcVT == MVT::f32 || SrcVT == MVT::f64)
3926 return LowerFP_TO_INT64(Op, DAG, OpOpcode == ISD::FP_TO_SINT);
3927
3928 return SDValue();
3929}
3930
3932 SelectionDAG &DAG) const {
3933 SDValue Src = Op.getOperand(0);
3934 unsigned OpOpcode = Op.getOpcode();
3935 EVT SrcVT = Src.getValueType();
3936 EVT DstVT = Op.getValueType();
3937 SDValue SatVTOp = Op.getNode()->getOperand(1);
3938 EVT SatVT = cast<VTSDNode>(SatVTOp)->getVT();
3939 SDLoc DL(Op);
3940
3941 uint64_t DstWidth = DstVT.getScalarSizeInBits();
3942 uint64_t SatWidth = SatVT.getScalarSizeInBits();
3943 assert(SatWidth <= DstWidth && "Saturation width cannot exceed result width");
3944
3945 // Scalar cases will be selected natively to v_cvt_/s_cvt_ instructions.
3946 // v2f32 -> v2i16 will be selected natively to v_cvt_pk_[iu]16_f32.
3947 if (SatWidth == DstWidth) {
3948 if ((DstVT == MVT::i32 && (SrcVT == MVT::f32 || SrcVT == MVT::f64)) ||
3949 (DstVT == MVT::i16 && (SrcVT == MVT::f16 || SrcVT == MVT::f32)) ||
3950 (DstVT == MVT::v2i16 && SrcVT == MVT::v2f32))
3951 return Op;
3952 }
3953
3954 // Vectors can only be selected natively.
3955 if (DstVT.isVector())
3956 return SDValue();
3957
3958 // Perform all saturation at selected width (i16 or i32) and truncate
3959 if (SatWidth < DstWidth && SatWidth <= 32) {
3960 // For f16 conversion with sub-i16 saturation perform saturation
3961 // at i16, if available in the target. This removes the need for extra f16
3962 // to f32 conversion. For all the others use i32.
3963 MVT ResultVT =
3964 Subtarget->has16BitInsts() && SrcVT == MVT::f16 && SatWidth < 16
3965 ? MVT::i16
3966 : MVT::i32;
3967
3968 const SDValue ResultVTOp = DAG.getValueType(ResultVT);
3969 const uint64_t ResultWidth = ResultVT.getScalarSizeInBits();
3970
3971 // First, convert input float into selected integer (i16 or i32)
3972 SDValue FpToInt = DAG.getNode(OpOpcode, DL, ResultVT, Src, ResultVTOp);
3973 SDValue IntSatVal;
3974
3975 // Then, clamp at the saturation width using either i16 or i32 instructions
3976 if (OpOpcode == ISD::FP_TO_SINT_SAT) {
3977 SDValue MinConst = DAG.getConstant(
3978 APInt::getSignedMaxValue(SatWidth).sext(ResultWidth), DL, ResultVT);
3979 SDValue MaxConst = DAG.getConstant(
3980 APInt::getSignedMinValue(SatWidth).sext(ResultWidth), DL, ResultVT);
3981 SDValue MinVal = DAG.getNode(ISD::SMIN, DL, ResultVT, FpToInt, MinConst);
3982 IntSatVal = DAG.getNode(ISD::SMAX, DL, ResultVT, MinVal, MaxConst);
3983 } else {
3984 SDValue MinConst = DAG.getConstant(
3985 APInt::getMaxValue(SatWidth).zext(ResultWidth), DL, ResultVT);
3986 IntSatVal = DAG.getNode(ISD::UMIN, DL, ResultVT, FpToInt, MinConst);
3987 }
3988
3989 // Finally, after saturating at i16 or i32 fit into the destination type
3990 return DAG.getExtOrTrunc(OpOpcode == ISD::FP_TO_SINT_SAT, IntSatVal, DL,
3991 DstVT);
3992 }
3993
3994 // SatWidth == DstWidth or SatWidth > 32
3995
3996 // Saturate at i32 for i64 dst and f16/bf16 src (will invoke f16 promotion
3997 // below)
3998 if (DstVT == MVT::i64 &&
3999 (SrcVT == MVT::f16 || SrcVT == MVT::bf16 ||
4000 (SrcVT == MVT::f32 && Src.getOpcode() == ISD::FP16_TO_FP))) {
4001 const SDValue Int32VTOp = DAG.getValueType(MVT::i32);
4002 return DAG.getNode(OpOpcode, DL, DstVT, Src, Int32VTOp);
4003 }
4004
4005 // Promote f16/bf16 src to f32 for i32 conversion
4006 if (DstVT == MVT::i32 && (SrcVT == MVT::f16 || SrcVT == MVT::bf16)) {
4007 SDValue PromotedSrc = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, Src);
4008 return DAG.getNode(Op.getOpcode(), DL, DstVT, PromotedSrc, SatVTOp);
4009 }
4010
4011 // For DstWidth < 16, promote i1 and i8 dst to i16 (if legal) with sub-i16
4012 // saturation. For DstWidth == 16, promote i16 dst to i32 with sub-i32
4013 // saturation; this covers i16.f32 and i16.f64
4014 if (DstWidth < 32) {
4015 // Note: this triggers SatWidth < DstWidth above to generate saturated
4016 // truncate by requesting MVT::i16/i32 destination with SatWidth < 16/32.
4017 MVT PromoteVT =
4018 (DstWidth < 16 && Subtarget->has16BitInsts()) ? MVT::i16 : MVT::i32;
4019 SDValue FpToInt = DAG.getNode(OpOpcode, DL, PromoteVT, Src, SatVTOp);
4020 return DAG.getNode(ISD::TRUNCATE, DL, DstVT, FpToInt);
4021 }
4022
4023 // TODO: can we implement i64 dst for f32/f64?
4024
4025 return SDValue();
4026}
4027
4029 SelectionDAG &DAG) const {
4030 EVT ExtraVT = cast<VTSDNode>(Op.getOperand(1))->getVT();
4031 MVT VT = Op.getSimpleValueType();
4032 MVT ScalarVT = VT.getScalarType();
4033
4034 assert(VT.isVector());
4035
4036 SDValue Src = Op.getOperand(0);
4037 SDLoc DL(Op);
4038
4039 // TODO: Don't scalarize on Evergreen?
4040 unsigned NElts = VT.getVectorNumElements();
4042 DAG.ExtractVectorElements(Src, Args, 0, NElts);
4043
4044 SDValue VTOp = DAG.getValueType(ExtraVT.getScalarType());
4045 for (unsigned I = 0; I < NElts; ++I)
4046 Args[I] = DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, ScalarVT, Args[I], VTOp);
4047
4048 return DAG.getBuildVector(VT, DL, Args);
4049}
4050
4051//===----------------------------------------------------------------------===//
4052// Custom DAG optimizations
4053//===----------------------------------------------------------------------===//
4054
4055static bool isU24(SDValue Op, SelectionDAG &DAG) {
4056 return AMDGPUTargetLowering::numBitsUnsigned(Op, DAG) <= 24;
4057}
4058
4059static bool isI24(SDValue Op, SelectionDAG &DAG) {
4060 EVT VT = Op.getValueType();
4061 return VT.getSizeInBits() >= 24 && // Types less than 24-bit should be treated
4062 // as unsigned 24-bit values.
4064}
4065
4068 SelectionDAG &DAG = DCI.DAG;
4069 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
4070 bool IsIntrin = Node24->getOpcode() == ISD::INTRINSIC_WO_CHAIN;
4071
4072 SDValue LHS = IsIntrin ? Node24->getOperand(1) : Node24->getOperand(0);
4073 SDValue RHS = IsIntrin ? Node24->getOperand(2) : Node24->getOperand(1);
4074 unsigned NewOpcode = Node24->getOpcode();
4075 if (IsIntrin) {
4076 unsigned IID = Node24->getConstantOperandVal(0);
4077 switch (IID) {
4078 case Intrinsic::amdgcn_mul_i24:
4079 NewOpcode = AMDGPUISD::MUL_I24;
4080 break;
4081 case Intrinsic::amdgcn_mul_u24:
4082 NewOpcode = AMDGPUISD::MUL_U24;
4083 break;
4084 case Intrinsic::amdgcn_mulhi_i24:
4085 NewOpcode = AMDGPUISD::MULHI_I24;
4086 break;
4087 case Intrinsic::amdgcn_mulhi_u24:
4088 NewOpcode = AMDGPUISD::MULHI_U24;
4089 break;
4090 default:
4091 llvm_unreachable("Expected 24-bit mul intrinsic");
4092 }
4093 }
4094
4095 APInt Demanded = APInt::getLowBitsSet(LHS.getValueSizeInBits(), 24);
4096
4097 // First try to simplify using SimplifyMultipleUseDemandedBits which allows
4098 // the operands to have other uses, but will only perform simplifications that
4099 // involve bypassing some nodes for this user.
4100 SDValue DemandedLHS = TLI.SimplifyMultipleUseDemandedBits(LHS, Demanded, DAG);
4101 SDValue DemandedRHS = TLI.SimplifyMultipleUseDemandedBits(RHS, Demanded, DAG);
4102 if (DemandedLHS || DemandedRHS)
4103 return DAG.getNode(NewOpcode, SDLoc(Node24), Node24->getVTList(),
4104 DemandedLHS ? DemandedLHS : LHS,
4105 DemandedRHS ? DemandedRHS : RHS);
4106
4107 // Now try SimplifyDemandedBits which can simplify the nodes used by our
4108 // operands if this node is the only user.
4109 if (TLI.SimplifyDemandedBits(LHS, Demanded, DCI))
4110 return SDValue(Node24, 0);
4111 if (TLI.SimplifyDemandedBits(RHS, Demanded, DCI))
4112 return SDValue(Node24, 0);
4113
4114 return SDValue();
4115}
4116
4117template <typename IntTy>
4119 uint32_t Width, const SDLoc &DL) {
4120 if (Width + Offset < 32) {
4121 uint32_t Shl = static_cast<uint32_t>(Src0) << (32 - Offset - Width);
4122 IntTy Result = static_cast<IntTy>(Shl) >> (32 - Width);
4123 if constexpr (std::is_signed_v<IntTy>) {
4124 return DAG.getSignedConstant(Result, DL, MVT::i32);
4125 } else {
4126 return DAG.getConstant(Result, DL, MVT::i32);
4127 }
4128 }
4129
4130 return DAG.getConstant(Src0 >> Offset, DL, MVT::i32);
4131}
4132
4133static bool hasVolatileUser(SDNode *Val) {
4134 for (SDNode *U : Val->users()) {
4135 if (MemSDNode *M = dyn_cast<MemSDNode>(U)) {
4136 if (M->isVolatile())
4137 return true;
4138 }
4139 }
4140
4141 return false;
4142}
4143
4145 // i32 vectors are the canonical memory type.
4146 if (VT.getScalarType() == MVT::i32 || isTypeLegal(VT))
4147 return false;
4148
4149 if (!VT.isByteSized())
4150 return false;
4151
4152 unsigned Size = VT.getStoreSize();
4153
4154 if ((Size == 1 || Size == 2 || Size == 4) && !VT.isVector())
4155 return false;
4156
4157 if (Size == 3 || (Size > 4 && (Size % 4 != 0)))
4158 return false;
4159
4160 return true;
4161}
4162
4163// Replace load of an illegal type with a bitcast from a load of a friendlier
4164// type.
4166 DAGCombinerInfo &DCI) const {
4167 if (!DCI.isBeforeLegalize())
4168 return SDValue();
4169
4171 if (!LN->isSimple() || !ISD::isNormalLoad(LN) || hasVolatileUser(LN))
4172 return SDValue();
4173
4174 SDLoc SL(N);
4175 SelectionDAG &DAG = DCI.DAG;
4176 EVT VT = LN->getMemoryVT();
4177
4178 unsigned Size = VT.getStoreSize();
4179 Align Alignment = LN->getAlign();
4180 if (Alignment < Size && isTypeLegal(VT)) {
4181 unsigned IsFast;
4182 unsigned AS = LN->getAddressSpace();
4183
4184 // Expand unaligned loads earlier than legalization. Due to visitation order
4185 // problems during legalization, the emitted instructions to pack and unpack
4186 // the bytes again are not eliminated in the case of an unaligned copy.
4188 VT, AS, Alignment, LN->getMemOperand()->getFlags(), &IsFast)) {
4189 if (VT.isVector())
4190 return SplitVectorLoad(SDValue(LN, 0), DAG);
4191
4192 SDValue Ops[2];
4193 std::tie(Ops[0], Ops[1]) = expandUnalignedLoad(LN, DAG);
4194
4195 return DAG.getMergeValues(Ops, SDLoc(N));
4196 }
4197
4198 if (!IsFast)
4199 return SDValue();
4200 }
4201
4202 if (!shouldCombineMemoryType(VT))
4203 return SDValue();
4204
4205 EVT NewVT = getEquivalentMemType(*DAG.getContext(), VT);
4206
4207 SDValue NewLoad
4208 = DAG.getLoad(NewVT, SL, LN->getChain(),
4209 LN->getBasePtr(), LN->getMemOperand());
4210
4211 SDValue BC = DAG.getNode(ISD::BITCAST, SL, VT, NewLoad);
4212 DCI.CombineTo(N, BC, NewLoad.getValue(1));
4213 return SDValue(N, 0);
4214}
4215
4216// Replace store of an illegal type with a store of a bitcast to a friendlier
4217// type.
4219 DAGCombinerInfo &DCI) const {
4220 if (!DCI.isBeforeLegalize())
4221 return SDValue();
4222
4224 if (!SN->isSimple() || !ISD::isNormalStore(SN))
4225 return SDValue();
4226
4227 EVT VT = SN->getMemoryVT();
4228 unsigned Size = VT.getStoreSize();
4229
4230 SDLoc SL(N);
4231 SelectionDAG &DAG = DCI.DAG;
4232 Align Alignment = SN->getAlign();
4233 if (Alignment < Size && isTypeLegal(VT)) {
4234 unsigned IsFast;
4235 unsigned AS = SN->getAddressSpace();
4236
4237 // Expand unaligned stores earlier than legalization. Due to visitation
4238 // order problems during legalization, the emitted instructions to pack and
4239 // unpack the bytes again are not eliminated in the case of an unaligned
4240 // copy.
4242 VT, AS, Alignment, SN->getMemOperand()->getFlags(), &IsFast)) {
4243 if (VT.isVector())
4244 return SplitVectorStore(SDValue(SN, 0), DAG);
4245
4246 return expandUnalignedStore(SN, DAG);
4247 }
4248
4249 if (!IsFast)
4250 return SDValue();
4251 }
4252
4253 if (!shouldCombineMemoryType(VT))
4254 return SDValue();
4255
4256 EVT NewVT = getEquivalentMemType(*DAG.getContext(), VT);
4257 SDValue Val = SN->getValue();
4258
4259 // DCI.AddToWorklist(Val.getNode());
4260
4261 bool OtherUses = !Val.hasOneUse();
4262 SDValue CastVal = DAG.getBitcast(NewVT, Val);
4263 if (OtherUses) {
4264 SDValue CastBack = DAG.getBitcast(VT, CastVal);
4265 DAG.ReplaceAllUsesOfValueWith(Val, CastBack);
4266 }
4267
4268 return DAG.getStore(SN->getChain(), SL, CastVal,
4269 SN->getBasePtr(), SN->getMemOperand());
4270}
4271
4272// FIXME: This should go in generic DAG combiner with an isTruncateFree check,
4273// but isTruncateFree is inaccurate for i16 now because of SALU vs. VALU
4274// issues.
4276 DAGCombinerInfo &DCI) const {
4277 SelectionDAG &DAG = DCI.DAG;
4278 SDValue N0 = N->getOperand(0);
4279
4280 // (vt2 (assertzext (truncate vt0:x), vt1)) ->
4281 // (vt2 (truncate (assertzext vt0:x, vt1)))
4282 if (N0.getOpcode() == ISD::TRUNCATE) {
4283 SDValue N1 = N->getOperand(1);
4284 EVT ExtVT = cast<VTSDNode>(N1)->getVT();
4285 SDLoc SL(N);
4286
4287 SDValue Src = N0.getOperand(0);
4288 EVT SrcVT = Src.getValueType();
4289 if (SrcVT.bitsGE(ExtVT)) {
4290 SDValue NewInReg = DAG.getNode(N->getOpcode(), SL, SrcVT, Src, N1);
4291 return DAG.getNode(ISD::TRUNCATE, SL, N->getValueType(0), NewInReg);
4292 }
4293 }
4294
4295 return SDValue();
4296}
4297
4299 SDNode *N, DAGCombinerInfo &DCI) const {
4300 unsigned IID = N->getConstantOperandVal(0);
4301 switch (IID) {
4302 case Intrinsic::amdgcn_mul_i24:
4303 case Intrinsic::amdgcn_mul_u24:
4304 case Intrinsic::amdgcn_mulhi_i24:
4305 case Intrinsic::amdgcn_mulhi_u24:
4306 return simplifyMul24(N, DCI);
4307 case Intrinsic::amdgcn_fract:
4308 case Intrinsic::amdgcn_rsq:
4309 case Intrinsic::amdgcn_rcp_legacy:
4310 case Intrinsic::amdgcn_rsq_legacy:
4311 case Intrinsic::amdgcn_rsq_clamp:
4312 case Intrinsic::amdgcn_tanh:
4313 case Intrinsic::amdgcn_prng_b32: {
4314 // FIXME: This is probably wrong. If src is an sNaN, it won't be quieted
4315 SDValue Src = N->getOperand(1);
4316 return Src.isUndef() ? Src : SDValue();
4317 }
4318 case Intrinsic::amdgcn_frexp_exp: {
4319 // frexp_exp (fneg x) -> frexp_exp x
4320 // frexp_exp (fabs x) -> frexp_exp x
4321 // frexp_exp (fneg (fabs x)) -> frexp_exp x
4322 SDValue Src = N->getOperand(1);
4323 SDValue PeekSign = peekFPSignOps(Src);
4324 if (PeekSign == Src)
4325 return SDValue();
4326 return SDValue(DCI.DAG.UpdateNodeOperands(N, N->getOperand(0), PeekSign),
4327 0);
4328 }
4329 default:
4330 return SDValue();
4331 }
4332}
4333
4334/// Split the 64-bit value \p LHS into two 32-bit components, and perform the
4335/// binary operation \p Opc to it with the corresponding constant operands.
4337 DAGCombinerInfo &DCI, const SDLoc &SL,
4338 unsigned Opc, SDValue LHS,
4339 uint32_t ValLo, uint32_t ValHi) const {
4340 SelectionDAG &DAG = DCI.DAG;
4341 SDValue Lo, Hi;
4342 std::tie(Lo, Hi) = split64BitValue(LHS, DAG);
4343
4344 SDValue LoRHS = DAG.getConstant(ValLo, SL, MVT::i32);
4345 SDValue HiRHS = DAG.getConstant(ValHi, SL, MVT::i32);
4346
4347 SDValue LoAnd = DAG.getNode(Opc, SL, MVT::i32, Lo, LoRHS);
4348 SDValue HiAnd = DAG.getNode(Opc, SL, MVT::i32, Hi, HiRHS);
4349
4350 // Re-visit the ands. It's possible we eliminated one of them and it could
4351 // simplify the vector.
4352 DCI.AddToWorklist(Lo.getNode());
4353 DCI.AddToWorklist(Hi.getNode());
4354
4355 SDValue Vec = DAG.getBuildVector(MVT::v2i32, SL, {LoAnd, HiAnd});
4356 return DAG.getNode(ISD::BITCAST, SL, MVT::i64, Vec);
4357}
4358
4360 DAGCombinerInfo &DCI) const {
4361 EVT VT = N->getValueType(0);
4362 SDValue LHS = N->getOperand(0);
4363 SDValue RHS = N->getOperand(1);
4365 SDLoc SL(N);
4366 SelectionDAG &DAG = DCI.DAG;
4367
4368 unsigned RHSVal;
4369 if (CRHS) {
4370 RHSVal = CRHS->getZExtValue();
4371 if (!RHSVal)
4372 return LHS;
4373
4374 switch (LHS->getOpcode()) {
4375 default:
4376 break;
4377 case ISD::ZERO_EXTEND:
4378 case ISD::SIGN_EXTEND:
4379 case ISD::ANY_EXTEND: {
4380 SDValue X = LHS->getOperand(0);
4381
4382 if (VT == MVT::i32 && RHSVal == 16 && X.getValueType() == MVT::i16 &&
4383 isOperationLegal(ISD::BUILD_VECTOR, MVT::v2i16)) {
4384 // Prefer build_vector as the canonical form if packed types are legal.
4385 // (shl ([asz]ext i16:x), 16 -> build_vector 0, x
4386 SDValue Vec = DAG.getBuildVector(
4387 MVT::v2i16, SL,
4388 {DAG.getConstant(0, SL, MVT::i16), LHS->getOperand(0)});
4389 return DAG.getNode(ISD::BITCAST, SL, MVT::i32, Vec);
4390 }
4391
4392 // shl (ext x) => zext (shl x), if shift does not overflow int
4393 if (VT != MVT::i64)
4394 break;
4396 unsigned LZ = Known.countMinLeadingZeros();
4397 if (LZ < RHSVal)
4398 break;
4399 EVT XVT = X.getValueType();
4400 SDValue Shl = DAG.getNode(ISD::SHL, SL, XVT, X, SDValue(CRHS, 0));
4401 return DAG.getZExtOrTrunc(Shl, SL, VT);
4402 }
4403 }
4404 }
4405
4406 if (VT.getScalarType() != MVT::i64)
4407 return SDValue();
4408
4409 // On some subtargets, 64-bit shift is a quarter rate instruction. In the
4410 // common case, splitting this into a move and a 32-bit shift is faster and
4411 // the same code size.
4412 KnownBits Known = DAG.computeKnownBits(RHS);
4413
4414 EVT ElementType = VT.getScalarType();
4415 EVT TargetScalarType = ElementType.getHalfSizedIntegerVT(*DAG.getContext());
4416 EVT TargetType = VT.changeElementType(*DAG.getContext(), TargetScalarType);
4417
4418 if (Known.getMinValue().getZExtValue() < TargetScalarType.getSizeInBits())
4419 return SDValue();
4420 SDValue ShiftAmt;
4421
4422 if (CRHS) {
4423 ShiftAmt = DAG.getConstant(RHSVal - TargetScalarType.getSizeInBits(), SL,
4424 TargetType);
4425 } else {
4426 SDValue TruncShiftAmt = DAG.getNode(ISD::TRUNCATE, SL, TargetType, RHS);
4427 const SDValue ShiftMask =
4428 DAG.getConstant(TargetScalarType.getSizeInBits() - 1, SL, TargetType);
4429 // This AND instruction will clamp out of bounds shift values.
4430 // It will also be removed during later instruction selection.
4431 ShiftAmt = DAG.getNode(ISD::AND, SL, TargetType, TruncShiftAmt, ShiftMask);
4432 }
4433
4434 SDValue Lo = DAG.getNode(ISD::TRUNCATE, SL, TargetType, LHS);
4435 SDValue NewShift =
4436 DAG.getNode(ISD::SHL, SL, TargetType, Lo, ShiftAmt, N->getFlags());
4437
4438 const SDValue Zero = DAG.getConstant(0, SL, TargetScalarType);
4439 SDValue Vec;
4440
4441 if (VT.isVector()) {
4442 EVT ConcatType = TargetType.getDoubleNumVectorElementsVT(*DAG.getContext());
4443 unsigned NElts = TargetType.getVectorNumElements();
4445 SmallVector<SDValue, 16> HiAndLoOps(NElts * 2, Zero);
4446
4447 DAG.ExtractVectorElements(NewShift, HiOps, 0, NElts);
4448 for (unsigned I = 0; I != NElts; ++I)
4449 HiAndLoOps[2 * I + 1] = HiOps[I];
4450 Vec = DAG.getNode(ISD::BUILD_VECTOR, SL, ConcatType, HiAndLoOps);
4451 } else {
4452 EVT ConcatType = EVT::getVectorVT(*DAG.getContext(), TargetType, 2);
4453 Vec = DAG.getBuildVector(ConcatType, SL, {Zero, NewShift});
4454 }
4455 return DAG.getNode(ISD::BITCAST, SL, VT, Vec);
4456}
4457
4459 DAGCombinerInfo &DCI) const {
4460 SDValue RHS = N->getOperand(1);
4462 EVT VT = N->getValueType(0);
4463 SDValue LHS = N->getOperand(0);
4464 SelectionDAG &DAG = DCI.DAG;
4465 SDLoc SL(N);
4466
4467 if (VT.getScalarType() != MVT::i64)
4468 return SDValue();
4469
4470 // For C >= 32
4471 // i64 (sra x, C) -> (build_pair (sra hi_32(x), C - 32), sra hi_32(x), 31))
4472
4473 // On some subtargets, 64-bit shift is a quarter rate instruction. In the
4474 // common case, splitting this into a move and a 32-bit shift is faster and
4475 // the same code size.
4476 KnownBits Known = DAG.computeKnownBits(RHS);
4477
4478 EVT ElementType = VT.getScalarType();
4479 EVT TargetScalarType = ElementType.getHalfSizedIntegerVT(*DAG.getContext());
4480 EVT TargetType = VT.changeElementType(*DAG.getContext(), TargetScalarType);
4481
4482 if (Known.getMinValue().getZExtValue() < TargetScalarType.getSizeInBits())
4483 return SDValue();
4484
4485 SDValue ShiftFullAmt =
4486 DAG.getConstant(TargetScalarType.getSizeInBits() - 1, SL, TargetType);
4487 SDValue ShiftAmt;
4488 if (CRHS) {
4489 unsigned RHSVal = CRHS->getZExtValue();
4490 ShiftAmt = DAG.getConstant(RHSVal - TargetScalarType.getSizeInBits(), SL,
4491 TargetType);
4492 } else if (Known.getMinValue().getZExtValue() ==
4493 (ElementType.getSizeInBits() - 1)) {
4494 ShiftAmt = ShiftFullAmt;
4495 } else {
4496 SDValue TruncShiftAmt = DAG.getNode(ISD::TRUNCATE, SL, TargetType, RHS);
4497 const SDValue ShiftMask =
4498 DAG.getConstant(TargetScalarType.getSizeInBits() - 1, SL, TargetType);
4499 // This AND instruction will clamp out of bounds shift values.
4500 // It will also be removed during later instruction selection.
4501 ShiftAmt = DAG.getNode(ISD::AND, SL, TargetType, TruncShiftAmt, ShiftMask);
4502 }
4503
4504 EVT ConcatType;
4505 SDValue Hi;
4506 SDLoc LHSSL(LHS);
4507 // Bitcast LHS into ConcatType so hi-half of source can be extracted into Hi
4508 if (VT.isVector()) {
4509 unsigned NElts = TargetType.getVectorNumElements();
4510 ConcatType = TargetType.getDoubleNumVectorElementsVT(*DAG.getContext());
4511 SDValue SplitLHS = DAG.getNode(ISD::BITCAST, LHSSL, ConcatType, LHS);
4512 SmallVector<SDValue, 8> HiOps(NElts);
4513 SmallVector<SDValue, 16> HiAndLoOps;
4514
4515 DAG.ExtractVectorElements(SplitLHS, HiAndLoOps, 0, NElts * 2);
4516 for (unsigned I = 0; I != NElts; ++I) {
4517 HiOps[I] = HiAndLoOps[2 * I + 1];
4518 }
4519 Hi = DAG.getNode(ISD::BUILD_VECTOR, LHSSL, TargetType, HiOps);
4520 } else {
4521 const SDValue One = DAG.getConstant(1, LHSSL, TargetScalarType);
4522 ConcatType = EVT::getVectorVT(*DAG.getContext(), TargetType, 2);
4523 SDValue SplitLHS = DAG.getNode(ISD::BITCAST, LHSSL, ConcatType, LHS);
4524 Hi = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, LHSSL, TargetType, SplitLHS, One);
4525 }
4526
4527 KnownBits KnownLHS = DAG.computeKnownBits(LHS);
4528 SDValue NewShift, HiShift;
4529 if (KnownLHS.isNegative()) {
4530 HiShift = DAG.getAllOnesConstant(SL, TargetType);
4531 NewShift =
4532 DAG.getNode(ISD::SRA, SL, TargetType, Hi, ShiftAmt, N->getFlags());
4533 } else if (CRHS &&
4534 CRHS->getZExtValue() == (ElementType.getSizeInBits() - 1)) {
4535 NewShift = HiShift =
4536 DAG.getNode(ISD::SRA, SL, TargetType, Hi, ShiftAmt, N->getFlags());
4537 } else {
4538 Hi = DAG.getFreeze(Hi);
4539 HiShift = DAG.getNode(ISD::SRA, SL, TargetType, Hi, ShiftFullAmt);
4540 NewShift =
4541 DAG.getNode(ISD::SRA, SL, TargetType, Hi, ShiftAmt, N->getFlags());
4542 }
4543
4544 SDValue Vec;
4545 if (VT.isVector()) {
4546 unsigned NElts = TargetType.getVectorNumElements();
4549 SmallVector<SDValue, 16> HiAndLoOps(NElts * 2);
4550
4551 DAG.ExtractVectorElements(HiShift, HiOps, 0, NElts);
4552 DAG.ExtractVectorElements(NewShift, LoOps, 0, NElts);
4553 for (unsigned I = 0; I != NElts; ++I) {
4554 HiAndLoOps[2 * I + 1] = HiOps[I];
4555 HiAndLoOps[2 * I] = LoOps[I];
4556 }
4557 Vec = DAG.getNode(ISD::BUILD_VECTOR, SL, ConcatType, HiAndLoOps);
4558 } else {
4559 Vec = DAG.getBuildVector(ConcatType, SL, {NewShift, HiShift});
4560 }
4561 return DAG.getNode(ISD::BITCAST, SL, VT, Vec);
4562}
4563
4565 DAGCombinerInfo &DCI) const {
4566 SDValue RHS = N->getOperand(1);
4568 EVT VT = N->getValueType(0);
4569 SDValue LHS = N->getOperand(0);
4570 SelectionDAG &DAG = DCI.DAG;
4571 SDLoc SL(N);
4572 unsigned RHSVal;
4573
4574 if (CRHS) {
4575 RHSVal = CRHS->getZExtValue();
4576
4577 // fold (srl (and x, c1 << c2), c2) -> (and (srl(x, c2), c1)
4578 // this improves the ability to match BFE patterns in isel.
4579 if (LHS.getOpcode() == ISD::AND) {
4580 if (auto *Mask = dyn_cast<ConstantSDNode>(LHS.getOperand(1))) {
4581 unsigned MaskIdx, MaskLen;
4582 if (Mask->getAPIntValue().isShiftedMask(MaskIdx, MaskLen) &&
4583 MaskIdx == RHSVal) {
4584 return DAG.getNode(ISD::AND, SL, VT,
4585 DAG.getNode(ISD::SRL, SL, VT, LHS.getOperand(0),
4586 N->getOperand(1)),
4587 DAG.getNode(ISD::SRL, SL, VT, LHS.getOperand(1),
4588 N->getOperand(1)));
4589 }
4590 }
4591 }
4592 }
4593
4594 if (VT.getScalarType() != MVT::i64)
4595 return SDValue();
4596
4597 // for C >= 32
4598 // i64 (srl x, C) -> (build_pair (srl hi_32(x), C - 32), 0)
4599
4600 // On some subtargets, 64-bit shift is a quarter rate instruction. In the
4601 // common case, splitting this into a move and a 32-bit shift is faster and
4602 // the same code size.
4603 KnownBits Known = DAG.computeKnownBits(RHS);
4604
4605 EVT ElementType = VT.getScalarType();
4606 EVT TargetScalarType = ElementType.getHalfSizedIntegerVT(*DAG.getContext());
4607 EVT TargetType = VT.changeElementType(*DAG.getContext(), TargetScalarType);
4608
4609 if (Known.getMinValue().getZExtValue() < TargetScalarType.getSizeInBits())
4610 return SDValue();
4611
4612 SDValue ShiftAmt;
4613 if (CRHS) {
4614 ShiftAmt = DAG.getConstant(RHSVal - TargetScalarType.getSizeInBits(), SL,
4615 TargetType);
4616 } else {
4617 SDValue TruncShiftAmt = DAG.getNode(ISD::TRUNCATE, SL, TargetType, RHS);
4618 const SDValue ShiftMask =
4619 DAG.getConstant(TargetScalarType.getSizeInBits() - 1, SL, TargetType);
4620 // This AND instruction will clamp out of bounds shift values.
4621 // It will also be removed during later instruction selection.
4622 ShiftAmt = DAG.getNode(ISD::AND, SL, TargetType, TruncShiftAmt, ShiftMask);
4623 }
4624
4625 const SDValue Zero = DAG.getConstant(0, SL, TargetScalarType);
4626 EVT ConcatType;
4627 SDValue Hi;
4628 SDLoc LHSSL(LHS);
4629 // Bitcast LHS into ConcatType so hi-half of source can be extracted into Hi
4630 if (VT.isVector()) {
4631 unsigned NElts = TargetType.getVectorNumElements();
4632 ConcatType = TargetType.getDoubleNumVectorElementsVT(*DAG.getContext());
4633 SDValue SplitLHS = DAG.getNode(ISD::BITCAST, LHSSL, ConcatType, LHS);
4634 SmallVector<SDValue, 8> HiOps(NElts);
4635 SmallVector<SDValue, 16> HiAndLoOps;
4636
4637 DAG.ExtractVectorElements(SplitLHS, HiAndLoOps, /*Start=*/0, NElts * 2);
4638 for (unsigned I = 0; I != NElts; ++I)
4639 HiOps[I] = HiAndLoOps[2 * I + 1];
4640 Hi = DAG.getNode(ISD::BUILD_VECTOR, LHSSL, TargetType, HiOps);
4641 } else {
4642 const SDValue One = DAG.getConstant(1, LHSSL, TargetScalarType);
4643 ConcatType = EVT::getVectorVT(*DAG.getContext(), TargetType, 2);
4644 SDValue SplitLHS = DAG.getNode(ISD::BITCAST, LHSSL, ConcatType, LHS);
4645 Hi = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, LHSSL, TargetType, SplitLHS, One);
4646 }
4647
4648 SDValue NewShift =
4649 DAG.getNode(ISD::SRL, SL, TargetType, Hi, ShiftAmt, N->getFlags());
4650
4651 SDValue Vec;
4652 if (VT.isVector()) {
4653 unsigned NElts = TargetType.getVectorNumElements();
4655 SmallVector<SDValue, 16> HiAndLoOps(NElts * 2, Zero);
4656
4657 DAG.ExtractVectorElements(NewShift, LoOps, 0, NElts);
4658 for (unsigned I = 0; I != NElts; ++I)
4659 HiAndLoOps[2 * I] = LoOps[I];
4660 Vec = DAG.getNode(ISD::BUILD_VECTOR, SL, ConcatType, HiAndLoOps);
4661 } else {
4662 Vec = DAG.getBuildVector(ConcatType, SL, {NewShift, Zero});
4663 }
4664 return DAG.getNode(ISD::BITCAST, SL, VT, Vec);
4665}
4666
4668 SDNode *N, DAGCombinerInfo &DCI) const {
4669 SDLoc SL(N);
4670 SelectionDAG &DAG = DCI.DAG;
4671 EVT VT = N->getValueType(0);
4672 SDValue Src = N->getOperand(0);
4673
4674 // vt1 (truncate (bitcast (build_vector vt0:x, ...))) -> vt1 (bitcast vt0:x)
4675 if (Src.getOpcode() == ISD::BITCAST && !VT.isVector()) {
4676 SDValue Vec = Src.getOperand(0);
4677 if (Vec.getOpcode() == ISD::BUILD_VECTOR) {
4678 SDValue Elt0 = Vec.getOperand(0);
4679 EVT EltVT = Elt0.getValueType();
4680 if (VT.getFixedSizeInBits() <= EltVT.getFixedSizeInBits()) {
4681 if (EltVT.isFloatingPoint()) {
4682 Elt0 = DAG.getNode(ISD::BITCAST, SL,
4683 EltVT.changeTypeToInteger(), Elt0);
4684 }
4685
4686 return DAG.getNode(ISD::TRUNCATE, SL, VT, Elt0);
4687 }
4688 }
4689 }
4690
4691 // Equivalent of above for accessing the high element of a vector as an
4692 // integer operation.
4693 // trunc (srl (bitcast (build_vector x, y))), 16 -> trunc (bitcast y)
4694 if (Src.getOpcode() == ISD::SRL && !VT.isVector()) {
4695 if (auto *K = isConstOrConstSplat(Src.getOperand(1))) {
4696 SDValue BV = stripBitcast(Src.getOperand(0));
4697 if (BV.getOpcode() == ISD::BUILD_VECTOR) {
4698 EVT SrcEltVT = BV.getOperand(0).getValueType();
4699 unsigned SrcEltSize = SrcEltVT.getSizeInBits();
4700 unsigned BitIndex = K->getZExtValue();
4701 unsigned PartIndex = BitIndex / SrcEltSize;
4702
4703 if (PartIndex * SrcEltSize == BitIndex &&
4704 PartIndex < BV.getNumOperands()) {
4705 if (SrcEltVT.getSizeInBits() == VT.getSizeInBits()) {
4706 SDValue SrcElt =
4707 DAG.getNode(ISD::BITCAST, SL, SrcEltVT.changeTypeToInteger(),
4708 BV.getOperand(PartIndex));
4709 return DAG.getNode(ISD::TRUNCATE, SL, VT, SrcElt);
4710 }
4711 }
4712 }
4713 }
4714 }
4715
4716 // Partially shrink 64-bit shifts to 32-bit if reduced to 16-bit.
4717 //
4718 // i16 (trunc (srl i64:x, K)), K <= 16 ->
4719 // i16 (trunc (srl (i32 (trunc x), K)))
4720 if (VT.getScalarSizeInBits() < 32) {
4721 EVT SrcVT = Src.getValueType();
4722 if (SrcVT.getScalarSizeInBits() > 32 &&
4723 (Src.getOpcode() == ISD::SRL ||
4724 Src.getOpcode() == ISD::SRA ||
4725 Src.getOpcode() == ISD::SHL)) {
4726 SDValue Amt = Src.getOperand(1);
4727 KnownBits Known = DAG.computeKnownBits(Amt);
4728
4729 // - For left shifts, do the transform as long as the shift
4730 // amount is still legal for i32, so when ShiftAmt < 32 (<= 31)
4731 // - For right shift, do it if ShiftAmt <= (32 - Size) to avoid
4732 // losing information stored in the high bits when truncating.
4733 const unsigned MaxCstSize =
4734 (Src.getOpcode() == ISD::SHL) ? 31 : (32 - VT.getScalarSizeInBits());
4735 if (Known.getMaxValue().ule(MaxCstSize)) {
4736 EVT MidVT = VT.isVector() ?
4737 EVT::getVectorVT(*DAG.getContext(), MVT::i32,
4738 VT.getVectorNumElements()) : MVT::i32;
4739
4740 EVT NewShiftVT = getShiftAmountTy(MidVT, DAG.getDataLayout());
4741 SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SL, MidVT,
4742 Src.getOperand(0));
4743 DCI.AddToWorklist(Trunc.getNode());
4744
4745 if (Amt.getValueType() != NewShiftVT) {
4746 Amt = DAG.getZExtOrTrunc(Amt, SL, NewShiftVT);
4747 DCI.AddToWorklist(Amt.getNode());
4748 }
4749
4750 SDValue ShrunkShift = DAG.getNode(Src.getOpcode(), SL, MidVT,
4751 Trunc, Amt);
4752 return DAG.getNode(ISD::TRUNCATE, SL, VT, ShrunkShift);
4753 }
4754 }
4755 }
4756
4757 return SDValue();
4758}
4759
4760// We need to specifically handle i64 mul here to avoid unnecessary conversion
4761// instructions. If we only match on the legalized i64 mul expansion,
4762// SimplifyDemandedBits will be unable to remove them because there will be
4763// multiple uses due to the separate mul + mulh[su].
4764static SDValue getMul24(SelectionDAG &DAG, const SDLoc &SL,
4765 SDValue N0, SDValue N1, unsigned Size, bool Signed) {
4766 if (Size <= 32) {
4767 unsigned MulOpc = Signed ? AMDGPUISD::MUL_I24 : AMDGPUISD::MUL_U24;
4768 return DAG.getNode(MulOpc, SL, MVT::i32, N0, N1);
4769 }
4770
4771 unsigned MulLoOpc = Signed ? AMDGPUISD::MUL_I24 : AMDGPUISD::MUL_U24;
4772 unsigned MulHiOpc = Signed ? AMDGPUISD::MULHI_I24 : AMDGPUISD::MULHI_U24;
4773
4774 SDValue MulLo = DAG.getNode(MulLoOpc, SL, MVT::i32, N0, N1);
4775 SDValue MulHi = DAG.getNode(MulHiOpc, SL, MVT::i32, N0, N1);
4776
4777 return DAG.getNode(ISD::BUILD_PAIR, SL, MVT::i64, MulLo, MulHi);
4778}
4779
4780/// If \p V is an add of a constant 1, returns the other operand. Otherwise
4781/// return SDValue().
4782static SDValue getAddOneOp(const SDNode *V) {
4783 if (V->getOpcode() != ISD::ADD)
4784 return SDValue();
4785
4786 return isOneConstant(V->getOperand(1)) ? V->getOperand(0) : SDValue();
4787}
4788
4790 DAGCombinerInfo &DCI) const {
4791 assert(N->getOpcode() == ISD::MUL);
4792 EVT VT = N->getValueType(0);
4793
4794 // Don't generate 24-bit multiplies on values that are in SGPRs, since
4795 // we only have a 32-bit scalar multiply (avoid values being moved to VGPRs
4796 // unnecessarily). isDivergent() is used as an approximation of whether the
4797 // value is in an SGPR.
4798 if (!N->isDivergent())
4799 return SDValue();
4800
4801 unsigned Size = VT.getSizeInBits();
4802 if (VT.isVector() || Size > 64)
4803 return SDValue();
4804
4805 SelectionDAG &DAG = DCI.DAG;
4806 SDLoc DL(N);
4807
4808 SDValue N0 = N->getOperand(0);
4809 SDValue N1 = N->getOperand(1);
4810
4811 // Undo InstCombine canonicalize X * (Y + 1) -> X * Y + X to enable mad
4812 // matching.
4813
4814 // mul x, (add y, 1) -> add (mul x, y), x
4815 auto IsFoldableAdd = [](SDValue V) -> SDValue {
4816 SDValue AddOp = getAddOneOp(V.getNode());
4817 if (!AddOp)
4818 return SDValue();
4819
4820 if (V.hasOneUse() || all_of(V->users(), [](const SDNode *U) -> bool {
4821 return U->getOpcode() == ISD::MUL;
4822 }))
4823 return AddOp;
4824
4825 return SDValue();
4826 };
4827
4828 // FIXME: The selection pattern is not properly checking for commuted
4829 // operands, so we have to place the mul in the LHS
4830 if (SDValue MulOper = IsFoldableAdd(N0)) {
4831 SDValue MulVal = DAG.getNode(N->getOpcode(), DL, VT, N1, MulOper);
4832 return DAG.getNode(ISD::ADD, DL, VT, MulVal, N1);
4833 }
4834
4835 if (SDValue MulOper = IsFoldableAdd(N1)) {
4836 SDValue MulVal = DAG.getNode(N->getOpcode(), DL, VT, N0, MulOper);
4837 return DAG.getNode(ISD::ADD, DL, VT, MulVal, N0);
4838 }
4839
4840 // There are i16 integer mul/mad.
4841 if (isTypeLegal(MVT::i16) && VT.getScalarType().bitsLE(MVT::i16))
4842 return SDValue();
4843
4844 // SimplifyDemandedBits has the annoying habit of turning useful zero_extends
4845 // in the source into any_extends if the result of the mul is truncated. Since
4846 // we can assume the high bits are whatever we want, use the underlying value
4847 // to avoid the unknown high bits from interfering.
4848 if (N0.getOpcode() == ISD::ANY_EXTEND)
4849 N0 = N0.getOperand(0);
4850
4851 if (N1.getOpcode() == ISD::ANY_EXTEND)
4852 N1 = N1.getOperand(0);
4853
4854 SDValue Mul;
4855
4856 if (Subtarget->hasMulU24() && isU24(N0, DAG) && isU24(N1, DAG)) {
4857 N0 = DAG.getZExtOrTrunc(N0, DL, MVT::i32);
4858 N1 = DAG.getZExtOrTrunc(N1, DL, MVT::i32);
4859 Mul = getMul24(DAG, DL, N0, N1, Size, false);
4860 } else if (Subtarget->hasMulI24() && isI24(N0, DAG) && isI24(N1, DAG)) {
4861 N0 = DAG.getSExtOrTrunc(N0, DL, MVT::i32);
4862 N1 = DAG.getSExtOrTrunc(N1, DL, MVT::i32);
4863 Mul = getMul24(DAG, DL, N0, N1, Size, true);
4864 } else {
4865 return SDValue();
4866 }
4867
4868 // We need to use sext even for MUL_U24, because MUL_U24 is used
4869 // for signed multiply of 8 and 16-bit types.
4870 return DAG.getSExtOrTrunc(Mul, DL, VT);
4871}
4872
4873SDValue
4875 DAGCombinerInfo &DCI) const {
4876 if (N->getValueType(0) != MVT::i32)
4877 return SDValue();
4878
4879 SelectionDAG &DAG = DCI.DAG;
4880 SDLoc DL(N);
4881
4882 bool Signed = N->getOpcode() == ISD::SMUL_LOHI;
4883 SDValue N0 = N->getOperand(0);
4884 SDValue N1 = N->getOperand(1);
4885
4886 // SimplifyDemandedBits has the annoying habit of turning useful zero_extends
4887 // in the source into any_extends if the result of the mul is truncated. Since
4888 // we can assume the high bits are whatever we want, use the underlying value
4889 // to avoid the unknown high bits from interfering.
4890 if (N0.getOpcode() == ISD::ANY_EXTEND)
4891 N0 = N0.getOperand(0);
4892 if (N1.getOpcode() == ISD::ANY_EXTEND)
4893 N1 = N1.getOperand(0);
4894
4895 // Try to use two fast 24-bit multiplies (one for each half of the result)
4896 // instead of one slow extending multiply.
4897 unsigned LoOpcode = 0;
4898 unsigned HiOpcode = 0;
4899 if (Signed) {
4900 if (Subtarget->hasMulI24() && isI24(N0, DAG) && isI24(N1, DAG)) {
4901 N0 = DAG.getSExtOrTrunc(N0, DL, MVT::i32);
4902 N1 = DAG.getSExtOrTrunc(N1, DL, MVT::i32);
4903 LoOpcode = AMDGPUISD::MUL_I24;
4904 HiOpcode = AMDGPUISD::MULHI_I24;
4905 }
4906 } else {
4907 if (Subtarget->hasMulU24() && isU24(N0, DAG) && isU24(N1, DAG)) {
4908 N0 = DAG.getZExtOrTrunc(N0, DL, MVT::i32);
4909 N1 = DAG.getZExtOrTrunc(N1, DL, MVT::i32);
4910 LoOpcode = AMDGPUISD::MUL_U24;
4911 HiOpcode = AMDGPUISD::MULHI_U24;
4912 }
4913 }
4914 if (!LoOpcode)
4915 return SDValue();
4916
4917 SDValue Lo = DAG.getNode(LoOpcode, DL, MVT::i32, N0, N1);
4918 SDValue Hi = DAG.getNode(HiOpcode, DL, MVT::i32, N0, N1);
4919 DCI.CombineTo(N, Lo, Hi);
4920 return SDValue(N, 0);
4921}
4922
4924 DAGCombinerInfo &DCI) const {
4925 EVT VT = N->getValueType(0);
4926
4927 if (!Subtarget->hasMulI24() || VT.isVector())
4928 return SDValue();
4929
4930 // Don't generate 24-bit multiplies on values that are in SGPRs, since
4931 // we only have a 32-bit scalar multiply (avoid values being moved to VGPRs
4932 // unnecessarily). isDivergent() is used as an approximation of whether the
4933 // value is in an SGPR.
4934 // This doesn't apply if no s_mul_hi is available (since we'll end up with a
4935 // valu op anyway)
4936 if (Subtarget->hasSMulHi() && !N->isDivergent())
4937 return SDValue();
4938
4939 SelectionDAG &DAG = DCI.DAG;
4940 SDLoc DL(N);
4941
4942 SDValue N0 = N->getOperand(0);
4943 SDValue N1 = N->getOperand(1);
4944
4945 if (!isI24(N0, DAG) || !isI24(N1, DAG))
4946 return SDValue();
4947
4948 N0 = DAG.getSExtOrTrunc(N0, DL, MVT::i32);
4949 N1 = DAG.getSExtOrTrunc(N1, DL, MVT::i32);
4950
4951 SDValue Mulhi = DAG.getNode(AMDGPUISD::MULHI_I24, DL, MVT::i32, N0, N1);
4952 DCI.AddToWorklist(Mulhi.getNode());
4953 return DAG.getSExtOrTrunc(Mulhi, DL, VT);
4954}
4955
4957 DAGCombinerInfo &DCI) const {
4958 EVT VT = N->getValueType(0);
4959
4960 if (VT.isVector() || VT.getSizeInBits() > 32 || !Subtarget->hasMulU24())
4961 return SDValue();
4962
4963 // Don't generate 24-bit multiplies on values that are in SGPRs, since
4964 // we only have a 32-bit scalar multiply (avoid values being moved to VGPRs
4965 // unnecessarily). isDivergent() is used as an approximation of whether the
4966 // value is in an SGPR.
4967 // This doesn't apply if no s_mul_hi is available (since we'll end up with a
4968 // valu op anyway)
4969 if (!N->isDivergent() && Subtarget->hasSMulHi())
4970 return SDValue();
4971
4972 SelectionDAG &DAG = DCI.DAG;
4973 SDLoc DL(N);
4974
4975 SDValue N0 = N->getOperand(0);
4976 SDValue N1 = N->getOperand(1);
4977
4978 if (!isU24(N0, DAG) || !isU24(N1, DAG))
4979 return SDValue();
4980
4981 N0 = DAG.getZExtOrTrunc(N0, DL, MVT::i32);
4982 N1 = DAG.getZExtOrTrunc(N1, DL, MVT::i32);
4983
4984 SDValue Mulhi = DAG.getNode(AMDGPUISD::MULHI_U24, DL, MVT::i32, N0, N1);
4985 DCI.AddToWorklist(Mulhi.getNode());
4986 return DAG.getZExtOrTrunc(Mulhi, DL, VT);
4987}
4988
4989SDValue AMDGPUTargetLowering::getFFBX_U32(SelectionDAG &DAG,
4990 SDValue Op,
4991 const SDLoc &DL,
4992 unsigned Opc) const {
4993 EVT VT = Op.getValueType();
4994 if (VT.bitsGT(MVT::i32))
4995 return SDValue();
4996
4997 if (VT != MVT::i32)
4998 Op = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i32, Op);
4999
5000 SDValue FFBX = DAG.getNode(Opc, DL, MVT::i32, Op);
5001 if (VT != MVT::i32)
5002 FFBX = DAG.getNode(ISD::TRUNCATE, DL, VT, FFBX);
5003
5004 return FFBX;
5005}
5006
5007// The native instructions return -1 on 0 input. Optimize out a select that
5008// produces -1 on 0.
5009//
5010// TODO: If zero is not undef, we could also do this if the output is compared
5011// against the bitwidth.
5012//
5013// TODO: Should probably combine against FFBH_U32 instead of ctlz directly.
5015 SDValue LHS, SDValue RHS,
5016 DAGCombinerInfo &DCI) const {
5017 if (!isNullConstant(Cond.getOperand(1)))
5018 return SDValue();
5019
5020 SelectionDAG &DAG = DCI.DAG;
5021 ISD::CondCode CCOpcode = cast<CondCodeSDNode>(Cond.getOperand(2))->get();
5022 SDValue CmpLHS = Cond.getOperand(0);
5023
5024 // select (setcc x, 0, eq), -1, (ctlz_zero_poison x) -> ffbh_u32 x
5025 // select (setcc x, 0, eq), -1, (cttz_zero_poison x) -> ffbl_u32 x
5026 if (CCOpcode == ISD::SETEQ &&
5027 (isCtlzOpc(RHS.getOpcode()) || isCttzOpc(RHS.getOpcode())) &&
5028 RHS.getOperand(0) == CmpLHS && isAllOnesConstant(LHS)) {
5029 unsigned Opc =
5030 isCttzOpc(RHS.getOpcode()) ? AMDGPUISD::FFBL_B32 : AMDGPUISD::FFBH_U32;
5031 return getFFBX_U32(DAG, CmpLHS, SL, Opc);
5032 }
5033
5034 // select (setcc x, 0, ne), (ctlz_zero_poison x), -1 -> ffbh_u32 x
5035 // select (setcc x, 0, ne), (cttz_zero_poison x), -1 -> ffbl_u32 x
5036 if (CCOpcode == ISD::SETNE &&
5037 (isCtlzOpc(LHS.getOpcode()) || isCttzOpc(LHS.getOpcode())) &&
5038 LHS.getOperand(0) == CmpLHS && isAllOnesConstant(RHS)) {
5039 unsigned Opc =
5040 isCttzOpc(LHS.getOpcode()) ? AMDGPUISD::FFBL_B32 : AMDGPUISD::FFBH_U32;
5041
5042 return getFFBX_U32(DAG, CmpLHS, SL, Opc);
5043 }
5044
5045 return SDValue();
5046}
5047
5049 unsigned Op,
5050 const SDLoc &SL,
5051 SDValue Cond,
5052 SDValue N1,
5053 SDValue N2) {
5054 SelectionDAG &DAG = DCI.DAG;
5055 EVT VT = N1.getValueType();
5056
5057 SDValue NewSelect = DAG.getNode(ISD::SELECT, SL, VT, Cond,
5058 N1.getOperand(0), N2.getOperand(0));
5059 DCI.AddToWorklist(NewSelect.getNode());
5060 return DAG.getNode(Op, SL, VT, NewSelect);
5061}
5062
5063// Pull a free FP operation out of a select so it may fold into uses.
5064//
5065// select c, (fneg x), (fneg y) -> fneg (select c, x, y)
5066// select c, (fneg x), k -> fneg (select c, x, (fneg k))
5067//
5068// select c, (fabs x), (fabs y) -> fabs (select c, x, y)
5069// select c, (fabs x), +k -> fabs (select c, x, k)
5070SDValue
5072 SDValue N) const {
5073 SelectionDAG &DAG = DCI.DAG;
5074 SDValue Cond = N.getOperand(0);
5075 SDValue LHS = N.getOperand(1);
5076 SDValue RHS = N.getOperand(2);
5077
5078 EVT VT = N.getValueType();
5079 if ((LHS.getOpcode() == ISD::FABS && RHS.getOpcode() == ISD::FABS) ||
5080 (LHS.getOpcode() == ISD::FNEG && RHS.getOpcode() == ISD::FNEG)) {
5082 return SDValue();
5083
5084 return distributeOpThroughSelect(DCI, LHS.getOpcode(),
5085 SDLoc(N), Cond, LHS, RHS);
5086 }
5087
5088 bool Inv = false;
5089 if (RHS.getOpcode() == ISD::FABS || RHS.getOpcode() == ISD::FNEG) {
5090 std::swap(LHS, RHS);
5091 Inv = true;
5092 }
5093
5094 // TODO: Support vector constants.
5096 if ((LHS.getOpcode() == ISD::FNEG || LHS.getOpcode() == ISD::FABS) && CRHS &&
5097 !selectSupportsSourceMods(N.getNode())) {
5098 SDLoc SL(N);
5099 // If one side is an fneg/fabs and the other is a constant, we can push the
5100 // fneg/fabs down. If it's an fabs, the constant needs to be non-negative.
5101 SDValue NewLHS = LHS.getOperand(0);
5102 SDValue NewRHS = RHS;
5103
5104 // Careful: if the neg can be folded up, don't try to pull it back down.
5105 bool ShouldFoldNeg = true;
5106
5107 if (NewLHS.hasOneUse()) {
5108 unsigned Opc = NewLHS.getOpcode();
5109 if (LHS.getOpcode() == ISD::FNEG && fnegFoldsIntoOp(NewLHS.getNode()))
5110 ShouldFoldNeg = false;
5111 if (LHS.getOpcode() == ISD::FABS && Opc == ISD::FMUL)
5112 ShouldFoldNeg = false;
5113 }
5114
5115 if (ShouldFoldNeg) {
5116 if (LHS.getOpcode() == ISD::FABS && CRHS->isNegative())
5117 return SDValue();
5118
5119 // We're going to be forced to use a source modifier anyway, there's no
5120 // point to pulling the negate out unless we can get a size reduction by
5121 // negating the constant.
5122 //
5123 // TODO: Generalize to use getCheaperNegatedExpression which doesn't know
5124 // about cheaper constants.
5125 if (NewLHS.getOpcode() == ISD::FABS &&
5127 return SDValue();
5128
5130 return SDValue();
5131
5132 if (LHS.getOpcode() == ISD::FNEG)
5133 NewRHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5134
5135 if (Inv)
5136 std::swap(NewLHS, NewRHS);
5137
5138 SDValue NewSelect = DAG.getNode(ISD::SELECT, SL, VT,
5139 Cond, NewLHS, NewRHS);
5140 DCI.AddToWorklist(NewSelect.getNode());
5141 return DAG.getNode(LHS.getOpcode(), SL, VT, NewSelect);
5142 }
5143 }
5144
5145 return SDValue();
5146}
5147
5149 DAGCombinerInfo &DCI) const {
5150 if (SDValue Folded = foldFreeOpFromSelect(DCI, SDValue(N, 0)))
5151 return Folded;
5152
5153 SDValue Cond = N->getOperand(0);
5154 if (Cond.getOpcode() != ISD::SETCC)
5155 return SDValue();
5156
5157 EVT VT = N->getValueType(0);
5158 SDValue LHS = Cond.getOperand(0);
5159 SDValue RHS = Cond.getOperand(1);
5160 SDValue CC = Cond.getOperand(2);
5161
5162 SDValue True = N->getOperand(1);
5163 SDValue False = N->getOperand(2);
5164
5165 if (Cond.hasOneUse()) { // TODO: Look for multiple select uses.
5166 SelectionDAG &DAG = DCI.DAG;
5167 if (DAG.isConstantValueOfAnyType(True) &&
5168 !DAG.isConstantValueOfAnyType(False)) {
5169 // Swap cmp + select pair to move constant to false input.
5170 // This will allow using VOPC cndmasks more often.
5171 // select (setcc x, y), k, x -> select (setccinv x, y), x, k
5172
5173 SDLoc SL(N);
5174 ISD::CondCode NewCC =
5175 getSetCCInverse(cast<CondCodeSDNode>(CC)->get(), LHS.getValueType());
5176
5177 SDValue NewCond = DAG.getSetCC(SL, Cond.getValueType(), LHS, RHS, NewCC);
5178 return DAG.getNode(ISD::SELECT, SL, VT, NewCond, False, True);
5179 }
5180
5181 if (VT == MVT::f32 && Subtarget->hasFminFmaxLegacy()) {
5182 SDValue MinMax = combineFMinMaxLegacy(SDLoc(N), VT, LHS, RHS, True, False,
5183 CC, N->getFlags(), DCI);
5184 // Revisit this node so we can catch min3/max3/med3 patterns.
5185 //DCI.AddToWorklist(MinMax.getNode());
5186 return MinMax;
5187 }
5188 }
5189
5190 // There's no reason to not do this if the condition has other uses.
5191 return performCtlz_CttzCombine(SDLoc(N), Cond, True, False, DCI);
5192}
5193
5194static bool isInv2Pi(const APFloat &APF) {
5195 static const APFloat KF16(APFloat::IEEEhalf(), APInt(16, 0x3118));
5196 static const APFloat KF32(APFloat::IEEEsingle(), APInt(32, 0x3e22f983));
5197 static const APFloat KF64(APFloat::IEEEdouble(), APInt(64, 0x3fc45f306dc9c882));
5198
5199 return APF.bitwiseIsEqual(KF16) ||
5200 APF.bitwiseIsEqual(KF32) ||
5201 APF.bitwiseIsEqual(KF64);
5202}
5203
5204// 0 and 1.0 / (0.5 * pi) do not have inline immmediates, so there is an
5205// additional cost to negate them.
5208 if (C->isZero())
5209 return C->isNegative() ? NegatibleCost::Cheaper : NegatibleCost::Expensive;
5210
5211 if (Subtarget->hasInv2PiInlineImm() && isInv2Pi(C->getValueAPF()))
5212 return C->isNegative() ? NegatibleCost::Cheaper : NegatibleCost::Expensive;
5213
5215}
5216
5222
5228
5229static unsigned inverseMinMax(unsigned Opc) {
5230 switch (Opc) {
5231 case ISD::FMAXNUM:
5232 return ISD::FMINNUM;
5233 case ISD::FMINNUM:
5234 return ISD::FMAXNUM;
5235 case ISD::FMAXNUM_IEEE:
5236 return ISD::FMINNUM_IEEE;
5237 case ISD::FMINNUM_IEEE:
5238 return ISD::FMAXNUM_IEEE;
5239 case ISD::FMAXIMUM:
5240 return ISD::FMINIMUM;
5241 case ISD::FMINIMUM:
5242 return ISD::FMAXIMUM;
5243 case ISD::FMAXIMUMNUM:
5244 return ISD::FMINIMUMNUM;
5245 case ISD::FMINIMUMNUM:
5246 return ISD::FMAXIMUMNUM;
5247 case AMDGPUISD::FMAX_LEGACY:
5248 return AMDGPUISD::FMIN_LEGACY;
5249 case AMDGPUISD::FMIN_LEGACY:
5250 return AMDGPUISD::FMAX_LEGACY;
5251 default:
5252 llvm_unreachable("invalid min/max opcode");
5253 }
5254}
5255
5256/// \return true if it's profitable to try to push an fneg into its source
5257/// instruction.
5259 // If the input has multiple uses and we can either fold the negate down, or
5260 // the other uses cannot, give up. This both prevents unprofitable
5261 // transformations and infinite loops: we won't repeatedly try to fold around
5262 // a negate that has no 'good' form.
5263 if (N0.hasOneUse()) {
5264 // This may be able to fold into the source, but at a code size cost. Don't
5265 // fold if the fold into the user is free.
5266 if (allUsesHaveSourceMods(N, 0))
5267 return false;
5268 } else {
5269 if (fnegFoldsIntoOp(N0.getNode()) &&
5271 return false;
5272 }
5273
5274 return true;
5275}
5276
5278 DAGCombinerInfo &DCI) const {
5279 SelectionDAG &DAG = DCI.DAG;
5280 SDValue N0 = N->getOperand(0);
5281 EVT VT = N->getValueType(0);
5282
5283 unsigned Opc = N0.getOpcode();
5284
5285 if (!shouldFoldFNegIntoSrc(N, N0))
5286 return SDValue();
5287
5288 bool MayIgnoreSignedZeroForAllUses =
5289 N0->getFlags().hasNoSignedZeros() ||
5290 (N0.hasOneUse() && N->getFlags().hasNoSignedZeros());
5291
5292 SDLoc SL(N);
5293 switch (Opc) {
5294 case ISD::FADD: {
5295 if (!MayIgnoreSignedZeroForAllUses)
5296 return SDValue();
5297
5298 // (fneg (fadd x, y)) -> (fadd (fneg x), (fneg y))
5299 SDValue LHS = N0.getOperand(0);
5300 SDValue RHS = N0.getOperand(1);
5301
5302 if (LHS.getOpcode() != ISD::FNEG)
5303 LHS = DAG.getNode(ISD::FNEG, SL, VT, LHS);
5304 else
5305 LHS = LHS.getOperand(0);
5306
5307 if (RHS.getOpcode() != ISD::FNEG)
5308 RHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5309 else
5310 RHS = RHS.getOperand(0);
5311
5312 SDValue Res = DAG.getNode(ISD::FADD, SL, VT, LHS, RHS, N0->getFlags());
5313 if (Res.getOpcode() != ISD::FADD)
5314 return SDValue(); // Op got folded away.
5315 if (!N0.hasOneUse())
5316 DAG.ReplaceAllUsesWith(N0, DAG.getNode(ISD::FNEG, SL, VT, Res));
5317 return Res;
5318 }
5319 case ISD::FMUL:
5320 case AMDGPUISD::FMUL_LEGACY: {
5321 // (fneg (fmul x, y)) -> (fmul x, (fneg y))
5322 // (fneg (fmul_legacy x, y)) -> (fmul_legacy x, (fneg y))
5323 SDValue LHS = N0.getOperand(0);
5324 SDValue RHS = N0.getOperand(1);
5325
5326 if (LHS.getOpcode() == ISD::FNEG)
5327 LHS = LHS.getOperand(0);
5328 else if (RHS.getOpcode() == ISD::FNEG)
5329 RHS = RHS.getOperand(0);
5330 else
5331 RHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5332
5333 SDValue Res = DAG.getNode(Opc, SL, VT, LHS, RHS, N0->getFlags());
5334 if (Res.getOpcode() != Opc)
5335 return SDValue(); // Op got folded away.
5336 if (!N0.hasOneUse())
5337 DAG.ReplaceAllUsesWith(N0, DAG.getNode(ISD::FNEG, SL, VT, Res));
5338 return Res;
5339 }
5340 case ISD::FMA:
5341 case ISD::FMAD: {
5342 // TODO: handle llvm.amdgcn.fma.legacy
5343 if (!MayIgnoreSignedZeroForAllUses)
5344 return SDValue();
5345
5346 // (fneg (fma x, y, z)) -> (fma x, (fneg y), (fneg z))
5347 SDValue LHS = N0.getOperand(0);
5348 SDValue MHS = N0.getOperand(1);
5349 SDValue RHS = N0.getOperand(2);
5350
5351 if (LHS.getOpcode() == ISD::FNEG)
5352 LHS = LHS.getOperand(0);
5353 else if (MHS.getOpcode() == ISD::FNEG)
5354 MHS = MHS.getOperand(0);
5355 else
5356 MHS = DAG.getNode(ISD::FNEG, SL, VT, MHS);
5357
5358 if (RHS.getOpcode() != ISD::FNEG)
5359 RHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5360 else
5361 RHS = RHS.getOperand(0);
5362
5363 SDValue Res = DAG.getNode(Opc, SL, VT, LHS, MHS, RHS);
5364 if (Res.getOpcode() != Opc)
5365 return SDValue(); // Op got folded away.
5366 if (!N0.hasOneUse())
5367 DAG.ReplaceAllUsesWith(N0, DAG.getNode(ISD::FNEG, SL, VT, Res));
5368 return Res;
5369 }
5370 case ISD::FMAXNUM:
5371 case ISD::FMINNUM:
5372 case ISD::FMAXNUM_IEEE:
5373 case ISD::FMINNUM_IEEE:
5374 case ISD::FMINIMUM:
5375 case ISD::FMAXIMUM:
5376 case ISD::FMINIMUMNUM:
5377 case ISD::FMAXIMUMNUM:
5378 case AMDGPUISD::FMAX_LEGACY:
5379 case AMDGPUISD::FMIN_LEGACY: {
5380 // fneg (fmaxnum x, y) -> fminnum (fneg x), (fneg y)
5381 // fneg (fminnum x, y) -> fmaxnum (fneg x), (fneg y)
5382 // fneg (fmax_legacy x, y) -> fmin_legacy (fneg x), (fneg y)
5383 // fneg (fmin_legacy x, y) -> fmax_legacy (fneg x), (fneg y)
5384
5385 SDValue LHS = N0.getOperand(0);
5386 SDValue RHS = N0.getOperand(1);
5387
5388 // 0 doesn't have a negated inline immediate.
5389 // TODO: This constant check should be generalized to other operations.
5391 return SDValue();
5392
5393 // Swapping min<->max flips which operand a signed zero tie selects.
5394 if ((Opc == AMDGPUISD::FMIN_LEGACY || Opc == AMDGPUISD::FMAX_LEGACY) &&
5395 !canIgnoreLegacyMinMaxTies(DAG, N0->getFlags(), LHS, RHS))
5396 return SDValue();
5397
5398 SDValue NegLHS = DAG.getNode(ISD::FNEG, SL, VT, LHS);
5399 SDValue NegRHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5400 unsigned Opposite = inverseMinMax(Opc);
5401
5402 SDValue Res = DAG.getNode(Opposite, SL, VT, NegLHS, NegRHS, N0->getFlags());
5403 if (Res.getOpcode() != Opposite)
5404 return SDValue(); // Op got folded away.
5405 if (!N0.hasOneUse())
5406 DAG.ReplaceAllUsesWith(N0, DAG.getNode(ISD::FNEG, SL, VT, Res));
5407 return Res;
5408 }
5409 case AMDGPUISD::FMED3: {
5410 // med3 sorts a NaN input as smaller than everything regardless of its sign,
5411 // so negating all operands does not sign-flip the median when an input may
5412 // be NaN.
5413 if (!N0->getFlags().hasNoNaNs())
5414 return SDValue();
5415
5416 SDValue Ops[3];
5417 for (unsigned I = 0; I < 3; ++I)
5418 Ops[I] = DAG.getNode(ISD::FNEG, SL, VT, N0->getOperand(I), N0->getFlags());
5419
5420 SDValue Res = DAG.getNode(AMDGPUISD::FMED3, SL, VT, Ops, N0->getFlags());
5421 if (Res.getOpcode() != AMDGPUISD::FMED3)
5422 return SDValue(); // Op got folded away.
5423
5424 if (!N0.hasOneUse()) {
5425 SDValue Neg = DAG.getNode(ISD::FNEG, SL, VT, Res);
5426 DAG.ReplaceAllUsesWith(N0, Neg);
5427
5428 for (SDNode *U : Neg->users())
5429 DCI.AddToWorklist(U);
5430 }
5431
5432 return Res;
5433 }
5434 case ISD::FP_EXTEND:
5435 case ISD::FTRUNC:
5436 case ISD::FRINT:
5437 case ISD::FNEARBYINT: // XXX - Should fround be handled?
5438 case ISD::FROUNDEVEN:
5439 case ISD::FSIN:
5440 case ISD::FCANONICALIZE:
5441 case AMDGPUISD::RCP:
5442 case AMDGPUISD::RCP_LEGACY:
5443 case AMDGPUISD::RCP_IFLAG:
5444 case AMDGPUISD::SIN_HW: {
5445 SDValue CvtSrc = N0.getOperand(0);
5446 if (CvtSrc.getOpcode() == ISD::FNEG) {
5447 // (fneg (fp_extend (fneg x))) -> (fp_extend x)
5448 // (fneg (rcp (fneg x))) -> (rcp x)
5449 return DAG.getNode(Opc, SL, VT, CvtSrc.getOperand(0));
5450 }
5451
5452 if (!N0.hasOneUse())
5453 return SDValue();
5454
5455 // (fneg (fp_extend x)) -> (fp_extend (fneg x))
5456 // (fneg (rcp x)) -> (rcp (fneg x))
5457 SDValue Neg = DAG.getNode(ISD::FNEG, SL, CvtSrc.getValueType(), CvtSrc);
5458 return DAG.getNode(Opc, SL, VT, Neg, N0->getFlags());
5459 }
5460 case ISD::FP_ROUND: {
5461 SDValue CvtSrc = N0.getOperand(0);
5462
5463 if (CvtSrc.getOpcode() == ISD::FNEG) {
5464 // (fneg (fp_round (fneg x))) -> (fp_round x)
5465 return DAG.getNode(ISD::FP_ROUND, SL, VT,
5466 CvtSrc.getOperand(0), N0.getOperand(1));
5467 }
5468
5469 if (!N0.hasOneUse())
5470 return SDValue();
5471
5472 // (fneg (fp_round x)) -> (fp_round (fneg x))
5473 SDValue Neg = DAG.getNode(ISD::FNEG, SL, CvtSrc.getValueType(), CvtSrc);
5474 return DAG.getNode(ISD::FP_ROUND, SL, VT, Neg, N0.getOperand(1));
5475 }
5476 case ISD::FP16_TO_FP: {
5477 // v_cvt_f32_f16 supports source modifiers on pre-VI targets without legal
5478 // f16, but legalization of f16 fneg ends up pulling it out of the source.
5479 // Put the fneg back as a legal source operation that can be matched later.
5480 SDLoc SL(N);
5481
5482 SDValue Src = N0.getOperand(0);
5483 EVT SrcVT = Src.getValueType();
5484
5485 // fneg (fp16_to_fp x) -> fp16_to_fp (xor x, 0x8000)
5486 SDValue IntFNeg = DAG.getNode(ISD::XOR, SL, SrcVT, Src,
5487 DAG.getConstant(0x8000, SL, SrcVT));
5488 return DAG.getNode(ISD::FP16_TO_FP, SL, N->getValueType(0), IntFNeg);
5489 }
5490 case ISD::SELECT: {
5491 // fneg (select c, a, b) -> select c, (fneg a), (fneg b)
5492 // TODO: Invert conditions of foldFreeOpFromSelect
5493 return SDValue();
5494 }
5495 case ISD::BITCAST: {
5496 SDLoc SL(N);
5497 SDValue BCSrc = N0.getOperand(0);
5498 if (BCSrc.getOpcode() == ISD::BUILD_VECTOR) {
5499 SDValue HighBits = BCSrc.getOperand(BCSrc.getNumOperands() - 1);
5500 if (VT != MVT::f64 || HighBits.getValueType().getSizeInBits() != 32 ||
5501 !fnegFoldsIntoOp(HighBits.getNode()))
5502 return SDValue();
5503
5504 // f64 fneg only really needs to operate on the high half of of the
5505 // register, so try to force it to an f32 operation to help make use of
5506 // source modifiers.
5507 //
5508 //
5509 // fneg (f64 (bitcast (build_vector x, y))) ->
5510 // f64 (bitcast (build_vector (bitcast i32:x to f32),
5511 // (fneg (bitcast i32:y to f32)))
5512
5513 SDValue CastHi = DAG.getNode(ISD::BITCAST, SL, MVT::f32, HighBits);
5514 SDValue NegHi = DAG.getNode(ISD::FNEG, SL, MVT::f32, CastHi);
5515 SDValue CastBack =
5516 DAG.getNode(ISD::BITCAST, SL, HighBits.getValueType(), NegHi);
5517
5519 Ops.back() = CastBack;
5520 DCI.AddToWorklist(NegHi.getNode());
5521 SDValue Build =
5522 DAG.getNode(ISD::BUILD_VECTOR, SL, BCSrc.getValueType(), Ops);
5523 SDValue Result = DAG.getNode(ISD::BITCAST, SL, VT, Build);
5524
5525 if (!N0.hasOneUse())
5526 DAG.ReplaceAllUsesWith(N0, DAG.getNode(ISD::FNEG, SL, VT, Result));
5527 return Result;
5528 }
5529
5530 if (BCSrc.getOpcode() == ISD::SELECT && VT == MVT::f32 &&
5531 BCSrc.hasOneUse()) {
5532 // fneg (bitcast (f32 (select cond, i32:lhs, i32:rhs))) ->
5533 // select cond, (bitcast i32:lhs to f32), (bitcast i32:rhs to f32)
5534
5535 // TODO: Cast back result for multiple uses is beneficial in some cases.
5536
5537 SDValue LHS =
5538 DAG.getNode(ISD::BITCAST, SL, MVT::f32, BCSrc.getOperand(1));
5539 SDValue RHS =
5540 DAG.getNode(ISD::BITCAST, SL, MVT::f32, BCSrc.getOperand(2));
5541
5542 SDValue NegLHS = DAG.getNode(ISD::FNEG, SL, MVT::f32, LHS);
5543 SDValue NegRHS = DAG.getNode(ISD::FNEG, SL, MVT::f32, RHS);
5544
5545 return DAG.getNode(ISD::SELECT, SL, MVT::f32, BCSrc.getOperand(0), NegLHS,
5546 NegRHS);
5547 }
5548
5549 return SDValue();
5550 }
5551 default:
5552 return SDValue();
5553 }
5554}
5555
5557 DAGCombinerInfo &DCI) const {
5558 SelectionDAG &DAG = DCI.DAG;
5559 SDValue N0 = N->getOperand(0);
5560
5561 if (!N0.hasOneUse())
5562 return SDValue();
5563
5564 switch (N0.getOpcode()) {
5565 case ISD::FP16_TO_FP: {
5566 assert(!isTypeLegal(MVT::f16) && "should only see if f16 is illegal");
5567 SDLoc SL(N);
5568 SDValue Src = N0.getOperand(0);
5569 EVT SrcVT = Src.getValueType();
5570
5571 // fabs (fp16_to_fp x) -> fp16_to_fp (and x, 0x7fff)
5572 SDValue IntFAbs = DAG.getNode(ISD::AND, SL, SrcVT, Src,
5573 DAG.getConstant(0x7fff, SL, SrcVT));
5574 return DAG.getNode(ISD::FP16_TO_FP, SL, N->getValueType(0), IntFAbs);
5575 }
5576 case ISD::FP_ROUND: {
5577 SDLoc SL(N);
5578 SDValue CvtSrc = N0.getOperand(0);
5579
5580 // fabs (fp_round x) -> fp_round (fabs x)
5581 SDValue Abs = DAG.getNode(ISD::FABS, SL, CvtSrc.getValueType(), CvtSrc,
5582 N->getFlags());
5583 return DAG.getNode(ISD::FP_ROUND, SL, N->getValueType(0), Abs,
5584 N0.getOperand(1), N0->getFlags());
5585 }
5586 default:
5587 return SDValue();
5588 }
5589}
5590
5592 DAGCombinerInfo &DCI) const {
5593 const auto *CFP = dyn_cast<ConstantFPSDNode>(N->getOperand(0));
5594 if (!CFP)
5595 return SDValue();
5596
5597 std::optional<APFloat> Result = AMDGPU::evaluateRcp(CFP->getValueAPF());
5598 if (!Result)
5599 return SDValue();
5600
5601 return DCI.DAG.getConstantFP(*Result, SDLoc(N), N->getValueType(0));
5602}
5603
5605 if (!Subtarget->isGCN())
5606 return false;
5607
5610 auto &ST = DAG.getSubtarget<GCNSubtarget>();
5611 const auto *TII = ST.getInstrInfo();
5612
5613 if (!ST.hasVMovB64Inst() || (!SDConstant && !SDFPConstant))
5614 return false;
5615
5616 if (ST.has64BitLiterals())
5617 return true;
5618
5619 if (SDConstant) {
5620 const APInt &APVal = SDConstant->getAPIntValue();
5621 return isUInt<32>(APVal.getZExtValue()) || TII->isInlineConstant(APVal);
5622 }
5623
5624 APInt Val = SDFPConstant->getValueAPF().bitcastToAPInt();
5625 return isUInt<32>(Val.getZExtValue()) || TII->isInlineConstant(Val);
5626}
5627
5629 DAGCombinerInfo &DCI) const {
5630 SelectionDAG &DAG = DCI.DAG;
5631 SDLoc DL(N);
5632
5633 switch(N->getOpcode()) {
5634 default:
5635 break;
5636 case ISD::BITCAST: {
5637 EVT DestVT = N->getValueType(0);
5638
5639 // Push casts through vector builds. This helps avoid emitting a large
5640 // number of copies when materializing floating point vector constants.
5641 //
5642 // vNt1 bitcast (vNt0 (build_vector t0:x, t0:y)) =>
5643 // vnt1 = build_vector (t1 (bitcast t0:x)), (t1 (bitcast t0:y))
5644 if (DestVT.isVector()) {
5645 SDValue Src = N->getOperand(0);
5646 if (Src.getOpcode() == ISD::BUILD_VECTOR &&
5649 EVT SrcVT = Src.getValueType();
5650 unsigned NElts = DestVT.getVectorNumElements();
5651
5652 if (SrcVT.getVectorNumElements() == NElts) {
5653 EVT DestEltVT = DestVT.getVectorElementType();
5654
5655 SmallVector<SDValue, 8> CastedElts;
5656 SDLoc SL(N);
5657 for (unsigned I = 0, E = SrcVT.getVectorNumElements(); I != E; ++I) {
5658 SDValue Elt = Src.getOperand(I);
5659 CastedElts.push_back(DAG.getNode(ISD::BITCAST, DL, DestEltVT, Elt));
5660 }
5661
5662 return DAG.getBuildVector(DestVT, SL, CastedElts);
5663 }
5664 }
5665 }
5666
5667 if (DestVT.getSizeInBits() != 64 || !DestVT.isVector())
5668 break;
5669
5670 // Fold bitcasts of constants.
5671 //
5672 // v2i32 (bitcast i64:k) -> build_vector lo_32(k), hi_32(k)
5673 // TODO: Generalize and move to DAGCombiner
5674 SDValue Src = N->getOperand(0);
5676 SDLoc SL(N);
5677 if (isInt64ImmLegal(C, DAG))
5678 break;
5679 uint64_t CVal = C->getZExtValue();
5680 SDValue BV = DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v2i32,
5681 DAG.getConstant(Lo_32(CVal), SL, MVT::i32),
5682 DAG.getConstant(Hi_32(CVal), SL, MVT::i32));
5683 return DAG.getNode(ISD::BITCAST, SL, DestVT, BV);
5684 }
5685
5687 const APInt &Val = C->getValueAPF().bitcastToAPInt();
5688 SDLoc SL(N);
5689 if (isInt64ImmLegal(C, DAG))
5690 break;
5691 uint64_t CVal = Val.getZExtValue();
5692 SDValue Vec = DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v2i32,
5693 DAG.getConstant(Lo_32(CVal), SL, MVT::i32),
5694 DAG.getConstant(Hi_32(CVal), SL, MVT::i32));
5695
5696 return DAG.getNode(ISD::BITCAST, SL, DestVT, Vec);
5697 }
5698
5699 break;
5700 }
5701 case ISD::SHL:
5702 case ISD::SRA:
5703 case ISD::SRL: {
5704 // Range metadata can be invalidated when loads are converted to legal types
5705 // (e.g. v2i64 -> v4i32).
5706 // Try to convert vector shl/sra/srl before type legalization so that range
5707 // metadata can be utilized.
5708 if (!(N->getValueType(0).isVector() &&
5711 break;
5712 if (N->getOpcode() == ISD::SHL)
5713 return performShlCombine(N, DCI);
5714 if (N->getOpcode() == ISD::SRA)
5715 return performSraCombine(N, DCI);
5716 return performSrlCombine(N, DCI);
5717 }
5718 case ISD::TRUNCATE:
5719 return performTruncateCombine(N, DCI);
5720 case ISD::MUL:
5721 return performMulCombine(N, DCI);
5722 case AMDGPUISD::MUL_U24:
5723 case AMDGPUISD::MUL_I24: {
5724 if (SDValue Simplified = simplifyMul24(N, DCI))
5725 return Simplified;
5726 break;
5727 }
5728 case AMDGPUISD::MULHI_I24:
5729 case AMDGPUISD::MULHI_U24:
5730 return simplifyMul24(N, DCI);
5731 case ISD::SMUL_LOHI:
5732 case ISD::UMUL_LOHI:
5733 return performMulLoHiCombine(N, DCI);
5734 case ISD::MULHS:
5735 return performMulhsCombine(N, DCI);
5736 case ISD::MULHU:
5737 return performMulhuCombine(N, DCI);
5738 case ISD::SELECT:
5739 return performSelectCombine(N, DCI);
5740 case ISD::FNEG:
5741 return performFNegCombine(N, DCI);
5742 case ISD::FABS:
5743 return performFAbsCombine(N, DCI);
5744 case AMDGPUISD::BFE_I32:
5745 case AMDGPUISD::BFE_U32: {
5746 assert(N->getValueType(0) == MVT::i32 &&
5747 "BFE_I32/BFE_U32 is a 32-bit operation");
5748 ConstantSDNode *Width = dyn_cast<ConstantSDNode>(N->getOperand(2));
5749 if (!Width)
5750 break;
5751
5752 uint32_t WidthVal = Width->getZExtValue() & 0x1f;
5753 if (WidthVal == 0)
5754 return DAG.getConstant(0, DL, MVT::i32);
5755
5757 if (!Offset)
5758 break;
5759
5760 SDValue BitsFrom = N->getOperand(0);
5761 uint32_t OffsetVal = Offset->getZExtValue() & 0x1f;
5762
5763 bool Signed = N->getOpcode() == AMDGPUISD::BFE_I32;
5764
5765 if (OffsetVal == 0) {
5766 // This is already sign / zero extended, so try to fold away extra BFEs.
5767 EVT SmallVT = EVT::getIntegerVT(*DAG.getContext(), WidthVal);
5768 if (Signed) {
5769 if (DAG.ComputeNumSignBits(BitsFrom) >= 32 - WidthVal + 1)
5770 return BitsFrom;
5771
5772 // This is a sign_extend_inreg. Replace it to take advantage of existing
5773 // DAG Combines. If not eliminated, we will match back to BFE during
5774 // selection.
5775
5776 // TODO: The sext_inreg of extended types ends, although we can could
5777 // handle them in a single BFE.
5778 return DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, MVT::i32, BitsFrom,
5779 DAG.getValueType(SmallVT));
5780 }
5781
5782 if (DAG.MaskedValueIsZero(BitsFrom,
5783 APInt::getHighBitsSet(32, 32 - WidthVal)))
5784 return BitsFrom;
5785
5786 return DAG.getZeroExtendInReg(BitsFrom, DL, SmallVT);
5787 }
5788
5789 if (ConstantSDNode *CVal = dyn_cast<ConstantSDNode>(BitsFrom)) {
5790 if (Signed) {
5791 return constantFoldBFE<int32_t>(DAG,
5792 CVal->getSExtValue(),
5793 OffsetVal,
5794 WidthVal,
5795 DL);
5796 }
5797
5798 return constantFoldBFE<uint32_t>(DAG,
5799 CVal->getZExtValue(),
5800 OffsetVal,
5801 WidthVal,
5802 DL);
5803 }
5804
5805 if ((OffsetVal + WidthVal) >= 32 &&
5806 !(OffsetVal == 16 && WidthVal == 16 && Subtarget->hasSDWA())) {
5807 SDValue ShiftVal = DAG.getConstant(OffsetVal, DL, MVT::i32);
5808 return DAG.getNode(Signed ? ISD::SRA : ISD::SRL, DL, MVT::i32,
5809 BitsFrom, ShiftVal);
5810 }
5811
5812 if (BitsFrom.hasOneUse()) {
5813 APInt Demanded = APInt::getBitsSet(32,
5814 OffsetVal,
5815 OffsetVal + WidthVal);
5816
5819 !DCI.isBeforeLegalizeOps());
5820 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
5821 if (TLI.ShrinkDemandedConstant(BitsFrom, Demanded, TLO) ||
5822 TLI.SimplifyDemandedBits(BitsFrom, Demanded, Known, TLO)) {
5823 DCI.CommitTargetLoweringOpt(TLO);
5824 }
5825 }
5826
5827 break;
5828 }
5829 case ISD::LOAD:
5830 return performLoadCombine(N, DCI);
5831 case ISD::STORE:
5832 return performStoreCombine(N, DCI);
5833 case AMDGPUISD::RCP:
5834 case AMDGPUISD::RCP_IFLAG:
5835 return performRcpCombine(N, DCI);
5836 case ISD::AssertZext:
5837 case ISD::AssertSext:
5838 return performAssertSZExtCombine(N, DCI);
5840 return performIntrinsicWOChainCombine(N, DCI);
5841 case AMDGPUISD::FMAD_FTZ: {
5842 SDValue N0 = N->getOperand(0);
5843 SDValue N1 = N->getOperand(1);
5844 SDValue N2 = N->getOperand(2);
5845 EVT VT = N->getValueType(0);
5846
5847 // FMAD_FTZ is a FMAD + flush denormals to zero.
5848 // We flush the inputs, the intermediate step, and the output.
5852 if (N0CFP && N1CFP && N2CFP) {
5853 const auto FTZ = [](const APFloat &V) {
5854 if (V.isDenormal()) {
5855 APFloat Zero(V.getSemantics(), 0);
5856 return V.isNegative() ? -Zero : Zero;
5857 }
5858 return V;
5859 };
5860
5861 APFloat V0 = FTZ(N0CFP->getValueAPF());
5862 APFloat V1 = FTZ(N1CFP->getValueAPF());
5863 APFloat V2 = FTZ(N2CFP->getValueAPF());
5864 V0.multiply(V1, APFloat::rmNearestTiesToEven);
5865 V0 = FTZ(V0);
5866 V0.add(V2, APFloat::rmNearestTiesToEven);
5867 return DAG.getConstantFP(FTZ(V0), DL, VT);
5868 }
5869 break;
5870 }
5871 }
5872 return SDValue();
5873}
5874
5876 SDValue Op, const APInt &OriginalDemandedBits,
5877 const APInt &OriginalDemandedElts, KnownBits &Known, TargetLoweringOpt &TLO,
5878 unsigned Depth) const {
5879 switch (Op.getOpcode()) {
5881 switch (Op.getConstantOperandVal(0)) {
5882 case Intrinsic::amdgcn_readfirstlane:
5883 case Intrinsic::amdgcn_readlane:
5884 case Intrinsic::amdgcn_wwm: {
5885 if (SimplifyDemandedBits(Op.getOperand(1), OriginalDemandedBits,
5886 OriginalDemandedElts, Known, TLO, Depth + 1))
5887 return true;
5888 break;
5889 }
5890 case Intrinsic::amdgcn_set_inactive:
5891 case Intrinsic::amdgcn_set_inactive_chain_arg: {
5892 // The result is operand 1 in active lanes and operand 2 in inactive
5893 // lanes, so the known bits are the intersection of both operands.
5894 KnownBits KnownValue, KnownInactive;
5895 if (SimplifyDemandedBits(Op.getOperand(1), OriginalDemandedBits,
5896 OriginalDemandedElts, KnownValue, TLO,
5897 Depth + 1))
5898 return true;
5899 if (SimplifyDemandedBits(Op.getOperand(2), OriginalDemandedBits,
5900 OriginalDemandedElts, KnownInactive, TLO,
5901 Depth + 1))
5902 return true;
5903 Known = KnownValue.intersectWith(KnownInactive);
5904 break;
5905 }
5906 default:
5907 break;
5908 }
5909 break;
5910 }
5911 default:
5912 break;
5913 }
5914
5915 return false;
5916}
5917
5918//===----------------------------------------------------------------------===//
5919// Helper functions
5920//===----------------------------------------------------------------------===//
5921
5923 const TargetRegisterClass *RC,
5924 Register Reg, EVT VT,
5925 const SDLoc &SL,
5926 bool RawReg) const {
5928 MachineRegisterInfo &MRI = MF.getRegInfo();
5929 Register VReg;
5930
5931 if (!MRI.isLiveIn(Reg)) {
5932 VReg = MRI.createVirtualRegister(RC);
5933 MRI.addLiveIn(Reg, VReg);
5934 } else {
5935 VReg = MRI.getLiveInVirtReg(Reg);
5936 }
5937
5938 if (RawReg)
5939 return DAG.getRegister(VReg, VT);
5940
5941 return DAG.getCopyFromReg(DAG.getEntryNode(), SL, VReg, VT);
5942}
5943
5944// This may be called multiple times, and nothing prevents creating multiple
5945// objects at the same offset. See if we already defined this object.
5947 int64_t Offset) {
5948 for (int I = MFI.getObjectIndexBegin(); I < 0; ++I) {
5949 if (MFI.getObjectOffset(I) == Offset) {
5950 assert(MFI.getObjectSize(I) == Size);
5951 return I;
5952 }
5953 }
5954
5955 return MFI.CreateFixedObject(Size, Offset, true);
5956}
5957
5959 EVT VT,
5960 const SDLoc &SL,
5961 int64_t Offset) const {
5963 MachineFrameInfo &MFI = MF.getFrameInfo();
5964 int FI = getOrCreateFixedStackObject(MFI, VT.getStoreSize(), Offset);
5965
5966 auto SrcPtrInfo = MachinePointerInfo::getStack(MF, Offset);
5967 SDValue Ptr = DAG.getFrameIndex(FI, MVT::i32);
5968
5969 return DAG.getLoad(VT, SL, DAG.getEntryNode(), Ptr, SrcPtrInfo, Align(4),
5972}
5973
5975 const SDLoc &SL,
5976 SDValue Chain,
5977 SDValue ArgVal,
5978 int64_t Offset) const {
5982
5983 SDValue Ptr = DAG.getConstant(Offset, SL, MVT::i32);
5984 // Stores to the argument stack area are relative to the stack pointer.
5985 SDValue SP =
5986 DAG.getCopyFromReg(Chain, SL, Info->getStackPtrOffsetReg(), MVT::i32);
5987 Ptr = DAG.getNode(ISD::ADD, SL, MVT::i32, SP, Ptr);
5988 SDValue Store = DAG.getStore(Chain, SL, ArgVal, Ptr, DstInfo, Align(4),
5990 return Store;
5991}
5992
5994 const TargetRegisterClass *RC,
5995 EVT VT, const SDLoc &SL,
5996 const ArgDescriptor &Arg) const {
5997 assert(Arg && "Attempting to load missing argument");
5998
5999 SDValue V = Arg.isRegister() ?
6000 CreateLiveInRegister(DAG, RC, Arg.getRegister(), VT, SL) :
6001 loadStackInputValue(DAG, VT, SL, Arg.getStackOffset());
6002
6003 if (!Arg.isMasked())
6004 return V;
6005
6006 unsigned Mask = Arg.getMask();
6007 unsigned Shift = llvm::countr_zero<unsigned>(Mask);
6008 V = DAG.getNode(ISD::SRL, SL, VT, V,
6009 DAG.getShiftAmountConstant(Shift, VT, SL));
6010 return DAG.getNode(ISD::AND, SL, VT, V,
6011 DAG.getConstant(Mask >> Shift, SL, VT));
6012}
6013
6015 uint64_t ExplicitKernArgSize, const ImplicitParameter Param) const {
6016 unsigned ExplicitArgOffset = Subtarget->getExplicitKernelArgOffset();
6017 const Align Alignment = Subtarget->getAlignmentForImplicitArgPtr();
6018 uint64_t ArgOffset =
6019 alignTo(ExplicitKernArgSize, Alignment) + ExplicitArgOffset;
6020 switch (Param) {
6021 case FIRST_IMPLICIT:
6022 return ArgOffset;
6023 case PRIVATE_BASE:
6025 case SHARED_BASE:
6026 return ArgOffset + AMDGPU::ImplicitArg::SHARED_BASE_OFFSET;
6027 case QUEUE_PTR:
6028 return ArgOffset + AMDGPU::ImplicitArg::QUEUE_PTR_OFFSET;
6029 }
6030 llvm_unreachable("unexpected implicit parameter type");
6031}
6032
6039
6041 SelectionDAG &DAG, int Enabled,
6042 int &RefinementSteps,
6043 bool &UseOneConstNR,
6044 bool Reciprocal) const {
6045 EVT VT = Operand.getValueType();
6046
6047 if (VT == MVT::f32) {
6048 RefinementSteps = 0;
6049 return DAG.getNode(AMDGPUISD::RSQ, SDLoc(Operand), VT, Operand);
6050 }
6051
6052 // TODO: There is also f64 rsq instruction, but the documentation is less
6053 // clear on its precision.
6054
6055 return SDValue();
6056}
6057
6059 SelectionDAG &DAG, int Enabled,
6060 int &RefinementSteps) const {
6061 EVT VT = Operand.getValueType();
6062
6063 if (VT == MVT::f32) {
6064 // Reciprocal, < 1 ulp error.
6065 //
6066 // This reciprocal approximation converges to < 0.5 ulp error with one
6067 // newton rhapson performed with two fused multiple adds (FMAs).
6068
6069 RefinementSteps = 0;
6070 return DAG.getNode(AMDGPUISD::RCP, SDLoc(Operand), VT, Operand);
6071 }
6072
6073 // TODO: There is also f64 rcp instruction, but the documentation is less
6074 // clear on its precision.
6075
6076 return SDValue();
6077}
6078
6079static unsigned workitemIntrinsicDim(unsigned ID) {
6080 switch (ID) {
6081 case Intrinsic::amdgcn_workitem_id_x:
6082 return 0;
6083 case Intrinsic::amdgcn_workitem_id_y:
6084 return 1;
6085 case Intrinsic::amdgcn_workitem_id_z:
6086 return 2;
6087 default:
6088 llvm_unreachable("not a workitem intrinsic");
6089 }
6090}
6091
6093 const SDValue Op, KnownBits &Known,
6094 const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth) const {
6095
6096 Known.resetAll(); // Don't know anything.
6097
6098 unsigned Opc = Op.getOpcode();
6099
6100 switch (Opc) {
6101 default:
6102 break;
6103 case AMDGPUISD::CARRY:
6104 case AMDGPUISD::BORROW: {
6105 Known.Zero = APInt::getHighBitsSet(32, 31);
6106 break;
6107 }
6108
6109 case AMDGPUISD::BFE_I32:
6110 case AMDGPUISD::BFE_U32: {
6111 ConstantSDNode *CWidth = dyn_cast<ConstantSDNode>(Op.getOperand(2));
6112 if (!CWidth)
6113 return;
6114
6115 uint32_t Width = CWidth->getZExtValue() & 0x1f;
6116
6117 if (Opc == AMDGPUISD::BFE_U32)
6118 Known.Zero = APInt::getHighBitsSet(32, 32 - Width);
6119
6120 break;
6121 }
6122 case AMDGPUISD::FP_TO_FP16: {
6123 unsigned BitWidth = Known.getBitWidth();
6124
6125 // High bits are zero.
6127 break;
6128 }
6129 case AMDGPUISD::MUL_U24:
6130 case AMDGPUISD::MUL_I24: {
6131 KnownBits LHSKnown = DAG.computeKnownBits(Op.getOperand(0), Depth + 1);
6132 KnownBits RHSKnown = DAG.computeKnownBits(Op.getOperand(1), Depth + 1);
6133 unsigned BitWidth = Op.getScalarValueSizeInBits();
6134
6135 // Sign/Zero extend from 24 bits.
6136 if (Opc == AMDGPUISD::MUL_I24) {
6137 LHSKnown = LHSKnown.trunc(24).sext(BitWidth);
6138 RHSKnown = RHSKnown.trunc(24).sext(BitWidth);
6139 } else {
6140 LHSKnown = LHSKnown.trunc(24).zext(BitWidth);
6141 RHSKnown = RHSKnown.trunc(24).zext(BitWidth);
6142 }
6143
6144 // TODO: SelfMultiply can be poison, but not undef.
6145 bool SelfMultiply = Op.getOperand(0) == Op.getOperand(1);
6146 if (SelfMultiply)
6147 SelfMultiply &= DAG.isGuaranteedNotToBeUndefOrPoison(
6148 Op.getOperand(0), DemandedElts, UndefPoisonKind::UndefOrPoison,
6149 Depth + 1);
6150
6151 Known = KnownBits::mul(LHSKnown, RHSKnown, SelfMultiply);
6152 break;
6153 }
6154 case AMDGPUISD::PERM: {
6155 ConstantSDNode *CMask = dyn_cast<ConstantSDNode>(Op.getOperand(2));
6156 if (!CMask)
6157 return;
6158
6159 KnownBits LHSKnown = DAG.computeKnownBits(Op.getOperand(0), Depth + 1);
6160 KnownBits RHSKnown = DAG.computeKnownBits(Op.getOperand(1), Depth + 1);
6161 unsigned Sel = CMask->getZExtValue();
6162
6163 for (unsigned I = 0; I < 32; I += 8) {
6164 unsigned SelBits = Sel & 0xff;
6165 if (SelBits < 4) {
6166 SelBits *= 8;
6167 Known.One |= ((RHSKnown.One.getZExtValue() >> SelBits) & 0xff) << I;
6168 Known.Zero |= ((RHSKnown.Zero.getZExtValue() >> SelBits) & 0xff) << I;
6169 } else if (SelBits < 7) {
6170 SelBits = (SelBits & 3) * 8;
6171 Known.One |= ((LHSKnown.One.getZExtValue() >> SelBits) & 0xff) << I;
6172 Known.Zero |= ((LHSKnown.Zero.getZExtValue() >> SelBits) & 0xff) << I;
6173 } else if (SelBits == 0x0c) {
6174 Known.Zero |= 0xFFull << I;
6175 } else if (SelBits > 0x0c) {
6176 Known.One |= 0xFFull << I;
6177 }
6178 Sel >>= 8;
6179 }
6180 break;
6181 }
6182 case AMDGPUISD::BUFFER_LOAD_UBYTE: {
6183 Known.Zero.setHighBits(24);
6184 break;
6185 }
6186 case AMDGPUISD::BUFFER_LOAD_USHORT: {
6187 Known.Zero.setHighBits(16);
6188 break;
6189 }
6190 case AMDGPUISD::LDS: {
6191 auto *GA = cast<GlobalAddressSDNode>(Op.getOperand(0).getNode());
6192 Align Alignment = GA->getGlobal()->getPointerAlignment(DAG.getDataLayout());
6193
6194 Known.Zero.setHighBits(16);
6195 Known.Zero.setLowBits(Log2(Alignment));
6196 break;
6197 }
6198 case AMDGPUISD::SMIN3:
6199 case AMDGPUISD::SMAX3:
6200 case AMDGPUISD::SMED3:
6201 case AMDGPUISD::UMIN3:
6202 case AMDGPUISD::UMAX3:
6203 case AMDGPUISD::UMED3: {
6204 KnownBits Known2 = DAG.computeKnownBits(Op.getOperand(2), Depth + 1);
6205 if (Known2.isUnknown())
6206 break;
6207
6208 KnownBits Known1 = DAG.computeKnownBits(Op.getOperand(1), Depth + 1);
6209 if (Known1.isUnknown())
6210 break;
6211
6212 KnownBits Known0 = DAG.computeKnownBits(Op.getOperand(0), Depth + 1);
6213 if (Known0.isUnknown())
6214 break;
6215
6216 // TODO: Handle LeadZero/LeadOne from UMIN/UMAX handling.
6217 Known.Zero = Known0.Zero & Known1.Zero & Known2.Zero;
6218 Known.One = Known0.One & Known1.One & Known2.One;
6219 break;
6220 }
6222 unsigned IID = Op.getConstantOperandVal(0);
6223 switch (IID) {
6224 case Intrinsic::amdgcn_workitem_id_x:
6225 case Intrinsic::amdgcn_workitem_id_y:
6226 case Intrinsic::amdgcn_workitem_id_z: {
6227 unsigned MaxValue = Subtarget->getMaxWorkitemID(
6229 Known.Zero.setHighBits(llvm::countl_zero(MaxValue));
6230 break;
6231 }
6232 case Intrinsic::amdgcn_readfirstlane:
6233 case Intrinsic::amdgcn_readlane:
6234 // Result is the data operand's value from some lane.
6235 Known = DAG.computeKnownBits(Op.getOperand(1), DemandedElts, Depth + 1);
6236 break;
6237 default:
6238 break;
6239 }
6240 }
6241 }
6242}
6243
6245 SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG,
6246 unsigned Depth) const {
6247 switch (Op.getOpcode()) {
6248 case AMDGPUISD::BFE_I32: {
6249 ConstantSDNode *Width = dyn_cast<ConstantSDNode>(Op.getOperand(2));
6250 if (!Width)
6251 return 1;
6252
6253 unsigned SignBits = 32 - (Width->getZExtValue() & 0x1f) + 1;
6254 if (!isNullConstant(Op.getOperand(1)))
6255 return SignBits;
6256
6257 // TODO: Could probably figure something out with non-0 offsets.
6258 unsigned Op0SignBits = DAG.ComputeNumSignBits(Op.getOperand(0), Depth + 1);
6259 return std::max(SignBits, Op0SignBits);
6260 }
6261
6262 case AMDGPUISD::BFE_U32: {
6263 ConstantSDNode *Width = dyn_cast<ConstantSDNode>(Op.getOperand(2));
6264 return Width ? 32 - (Width->getZExtValue() & 0x1f) : 1;
6265 }
6266
6267 case AMDGPUISD::CARRY:
6268 case AMDGPUISD::BORROW:
6269 return 31;
6270 case AMDGPUISD::BUFFER_LOAD_BYTE:
6271 return 25;
6272 case AMDGPUISD::BUFFER_LOAD_SHORT:
6273 return 17;
6274 case AMDGPUISD::BUFFER_LOAD_UBYTE:
6275 return 24;
6276 case AMDGPUISD::BUFFER_LOAD_USHORT:
6277 return 16;
6278 case AMDGPUISD::FP_TO_FP16:
6279 return 16;
6280 case AMDGPUISD::SMIN3:
6281 case AMDGPUISD::SMAX3:
6282 case AMDGPUISD::SMED3:
6283 case AMDGPUISD::UMIN3:
6284 case AMDGPUISD::UMAX3:
6285 case AMDGPUISD::UMED3: {
6286 unsigned Tmp2 = DAG.ComputeNumSignBits(Op.getOperand(2), Depth + 1);
6287 if (Tmp2 == 1)
6288 return 1; // Early out.
6289
6290 unsigned Tmp1 = DAG.ComputeNumSignBits(Op.getOperand(1), Depth + 1);
6291 if (Tmp1 == 1)
6292 return 1; // Early out.
6293
6294 unsigned Tmp0 = DAG.ComputeNumSignBits(Op.getOperand(0), Depth + 1);
6295 if (Tmp0 == 1)
6296 return 1; // Early out.
6297
6298 return std::min({Tmp0, Tmp1, Tmp2});
6299 }
6300 default:
6301 return 1;
6302 }
6303}
6304
6306 GISelValueTracking &Analysis, Register R, const APInt &DemandedElts,
6307 const MachineRegisterInfo &MRI, unsigned Depth) const {
6308 const MachineInstr *MI = MRI.getVRegDef(R);
6309 if (!MI)
6310 return 1;
6311
6312 // TODO: Check range metadata on MMO.
6313 switch (MI->getOpcode()) {
6314 case AMDGPU::G_AMDGPU_BUFFER_LOAD_SBYTE:
6315 return 25;
6316 case AMDGPU::G_AMDGPU_BUFFER_LOAD_SSHORT:
6317 return 17;
6318 case AMDGPU::G_AMDGPU_BUFFER_LOAD_UBYTE:
6319 return 24;
6320 case AMDGPU::G_AMDGPU_BUFFER_LOAD_USHORT:
6321 return 16;
6322 case AMDGPU::G_AMDGPU_SMED3:
6323 case AMDGPU::G_AMDGPU_UMED3: {
6324 auto [Dst, Src0, Src1, Src2] = MI->getFirst4Regs();
6325 unsigned Tmp2 = Analysis.computeNumSignBits(Src2, DemandedElts, Depth + 1);
6326 if (Tmp2 == 1)
6327 return 1;
6328 unsigned Tmp1 = Analysis.computeNumSignBits(Src1, DemandedElts, Depth + 1);
6329 if (Tmp1 == 1)
6330 return 1;
6331 unsigned Tmp0 = Analysis.computeNumSignBits(Src0, DemandedElts, Depth + 1);
6332 if (Tmp0 == 1)
6333 return 1;
6334 return std::min({Tmp0, Tmp1, Tmp2});
6335 }
6336 default:
6337 return 1;
6338 }
6339}
6340
6342 SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG,
6343 UndefPoisonKind Kind, bool ConsiderFlags, unsigned Depth) const {
6344 unsigned Opcode = Op.getOpcode();
6345 switch (Opcode) {
6346 case AMDGPUISD::BFE_I32:
6347 case AMDGPUISD::BFE_U32:
6348 return false;
6349 }
6351 Op, DemandedElts, DAG, Kind, ConsiderFlags, Depth);
6352}
6353
6355 SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, bool SNaN,
6356 unsigned Depth) const {
6357 unsigned Opcode = Op.getOpcode();
6358 switch (Opcode) {
6359 case AMDGPUISD::FMIN_LEGACY:
6360 case AMDGPUISD::FMAX_LEGACY:
6361 return DAG.isKnownNeverNaN(Op.getOperand(0), SNaN, Depth + 1) &&
6362 DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1);
6363 case AMDGPUISD::FMUL_LEGACY:
6364 case AMDGPUISD::CVT_PKRTZ_F16_F32: {
6365 if (SNaN)
6366 return true;
6367 return DAG.isKnownNeverNaN(Op.getOperand(0), SNaN, Depth + 1) &&
6368 DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1);
6369 }
6370 case AMDGPUISD::FMED3:
6371 case AMDGPUISD::FMIN3:
6372 case AMDGPUISD::FMAX3:
6373 case AMDGPUISD::FMINIMUM3:
6374 case AMDGPUISD::FMAXIMUM3:
6375 case AMDGPUISD::FMAD_FTZ: {
6376 if (SNaN)
6377 return true;
6378 return DAG.isKnownNeverNaN(Op.getOperand(0), SNaN, Depth + 1) &&
6379 DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1) &&
6380 DAG.isKnownNeverNaN(Op.getOperand(2), SNaN, Depth + 1);
6381 }
6382 case AMDGPUISD::CVT_F32_UBYTE0:
6383 case AMDGPUISD::CVT_F32_UBYTE1:
6384 case AMDGPUISD::CVT_F32_UBYTE2:
6385 case AMDGPUISD::CVT_F32_UBYTE3:
6386 return true;
6387
6388 case AMDGPUISD::RCP:
6389 case AMDGPUISD::RSQ:
6390 case AMDGPUISD::RCP_LEGACY:
6391 case AMDGPUISD::RSQ_CLAMP: {
6392 if (SNaN)
6393 return true;
6394
6395 // TODO: Need is known positive check.
6396 return false;
6397 }
6398 case ISD::FLDEXP:
6399 case AMDGPUISD::FRACT: {
6400 if (SNaN)
6401 return true;
6402 return DAG.isKnownNeverNaN(Op.getOperand(0), SNaN, Depth + 1);
6403 }
6404 case AMDGPUISD::DIV_SCALE:
6405 case AMDGPUISD::DIV_FMAS:
6406 case AMDGPUISD::DIV_FIXUP:
6407 // TODO: Refine on operands.
6408 return SNaN;
6409 case AMDGPUISD::SIN_HW:
6410 case AMDGPUISD::COS_HW: {
6411 // TODO: Need check for infinity
6412 return SNaN;
6413 }
6415 unsigned IntrinsicID = Op.getConstantOperandVal(0);
6416 // TODO: Handle more intrinsics
6417 switch (IntrinsicID) {
6418 case Intrinsic::amdgcn_cubeid:
6419 case Intrinsic::amdgcn_cvt_off_f32_i4:
6420 return true;
6421
6422 case Intrinsic::amdgcn_frexp_mant: {
6423 if (SNaN)
6424 return true;
6425 return DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1);
6426 }
6427 case Intrinsic::amdgcn_cvt_pkrtz: {
6428 if (SNaN)
6429 return true;
6430 return DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1) &&
6431 DAG.isKnownNeverNaN(Op.getOperand(2), SNaN, Depth + 1);
6432 }
6433 case Intrinsic::amdgcn_rcp:
6434 case Intrinsic::amdgcn_rsq:
6435 case Intrinsic::amdgcn_rcp_legacy:
6436 case Intrinsic::amdgcn_rsq_legacy:
6437 case Intrinsic::amdgcn_rsq_clamp:
6438 case Intrinsic::amdgcn_tanh: {
6439 if (SNaN)
6440 return true;
6441
6442 // TODO: Need is known positive check.
6443 return false;
6444 }
6445 case Intrinsic::amdgcn_trig_preop:
6446 case Intrinsic::amdgcn_fdot2:
6447 // TODO: Refine on operand
6448 return SNaN;
6449 case Intrinsic::amdgcn_fma_legacy:
6450 if (SNaN)
6451 return true;
6452 return DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1) &&
6453 DAG.isKnownNeverNaN(Op.getOperand(2), SNaN, Depth + 1) &&
6454 DAG.isKnownNeverNaN(Op.getOperand(3), SNaN, Depth + 1);
6455 default:
6456 return false;
6457 }
6458 }
6459 default:
6460 return false;
6461 }
6462}
6463
6465 Register N0, Register N1) const {
6466 return MRI.hasOneNonDBGUse(N0); // FIXME: handle regbanks
6467}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU address space definition.
static LLVM_READONLY bool hasSourceMods(const MachineInstr &MI)
static bool isInv2Pi(const APFloat &APF)
static LLVM_READONLY bool opMustUseVOP3Encoding(const MachineInstr &MI, const MachineRegisterInfo &MRI)
returns true if the operation will definitely need to use a 64-bit encoding, and thus will use a VOP3...
static unsigned inverseMinMax(unsigned Opc)
unsigned Imm
static SDValue extractF64Exponent(SDValue Hi, const SDLoc &SL, SelectionDAG &DAG)
static unsigned workitemIntrinsicDim(unsigned ID)
static int getOrCreateFixedStackObject(MachineFrameInfo &MFI, unsigned Size, int64_t Offset)
static SDValue constantFoldBFE(SelectionDAG &DAG, IntTy Src0, uint32_t Offset, uint32_t Width, const SDLoc &DL)
static SDValue getMad(SelectionDAG &DAG, const SDLoc &SL, EVT VT, SDValue X, SDValue Y, SDValue C, SDNodeFlags Flags=SDNodeFlags())
static SDValue getAddOneOp(const SDNode *V)
If V is an add of a constant 1, returns the other operand.
static bool canIgnoreLegacyMinMaxTies(const SelectionDAG &DAG, SDNodeFlags Flags, SDValue LHS, SDValue RHS)
static LLVM_READONLY bool selectSupportsSourceMods(const SDNode *N)
Return true if v_cndmask_b32 will support fabs/fneg source modifiers for the type for ISD::SELECT.
static cl::opt< bool > AMDGPUBypassSlowDiv("amdgpu-bypass-slow-div", cl::desc("Skip 64-bit divide for dynamic 32-bit values"), cl::init(true))
static SDValue getMul24(SelectionDAG &DAG, const SDLoc &SL, SDValue N0, SDValue N1, unsigned Size, bool Signed)
static bool fnegFoldsIntoOp(const SDNode *N)
static bool isI24(SDValue Op, SelectionDAG &DAG)
static bool isCttzOpc(unsigned Opc)
static bool isU24(SDValue Op, SelectionDAG &DAG)
static bool valueIsKnownNeverF32Denorm(SDValue Src)
Return true if it's known that Src can never be an f32 denormal value.
static SDValue distributeOpThroughSelect(TargetLowering::DAGCombinerInfo &DCI, unsigned Op, const SDLoc &SL, SDValue Cond, SDValue N1, SDValue N2)
static SDValue peekFNeg(SDValue Val)
static SDValue simplifyMul24(SDNode *Node24, TargetLowering::DAGCombinerInfo &DCI)
static bool isCtlzOpc(unsigned Opc)
static LLVM_READNONE bool fnegFoldsIntoOpcode(unsigned Opc)
static bool hasVolatileUser(SDNode *Val)
Interface definition of the TargetLowering class that is common to all AMD GPUs.
Contains the definition of a TargetInstrInfo class that is common to all AMD GPUs.
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Function Alias Analysis Results
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
block Block Frequency Analysis
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
#define LLVM_READNONE
Definition Compiler.h:331
#define LLVM_READONLY
Definition Compiler.h:338
Provides analysis for querying information about KnownBits during GISel passes.
const HexagonInstrInfo * TII
static MaybeAlign getAlign(Value *Ptr)
IRTranslator LLVM IR MI
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
#define G(x, y, z)
Definition MD5.cpp:55
#define T
#define P(N)
const SmallVectorImpl< MachineOperand > & Cond
#define CH(x, y, z)
Definition SHA256.cpp:34
Func MI getDebugLoc()))
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
Value * RHS
Value * LHS
BinaryOperator * Mul
static CCAssignFn * CCAssignFnForCall(CallingConv::ID CC, bool IsVarArg)
static CCAssignFn * CCAssignFnForReturn(CallingConv::ID CC, bool IsVarArg)
unsigned allocateBarrierGlobal(const DataLayout &DL, const GlobalVariable &GV)
static std::optional< uint32_t > get32BitAbsoluteAddress(const GlobalValue &GV, unsigned AS)
unsigned allocateLDSGlobal(const DataLayout &DL, const GlobalVariable &GV)
static unsigned numBitsSigned(SDValue Op, SelectionDAG &DAG)
unsigned ComputeNumSignBitsForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth=0) const override
This method can be implemented by targets that want to expose additional information about sign bits ...
SDValue performMulhuCombine(SDNode *N, DAGCombinerInfo &DCI) const
EVT getTypeForExtReturn(LLVMContext &Context, EVT VT, ISD::NodeType ExtendKind) const override
Return the type that should be used to zero or sign extend a zeroext/signext integer return value.
SDValue SplitVectorLoad(SDValue Op, SelectionDAG &DAG) const
Split a vector load into 2 loads of half the vector.
SDValue LowerCONCAT_VECTORS(SDValue Op, SelectionDAG &DAG) const
SDValue performLoadCombine(SDNode *N, DAGCombinerInfo &DCI) const
void analyzeFormalArgumentsCompute(CCState &State, const SmallVectorImpl< ISD::InputArg > &Ins) const
The SelectionDAGBuilder will automatically promote function arguments with illegal types.
SDValue LowerF64ToF16Safe(SDValue Src, const SDLoc &DL, SelectionDAG &DAG) const
SDValue LowerFROUND(SDValue Op, SelectionDAG &DAG) const
SDValue storeStackInputValue(SelectionDAG &DAG, const SDLoc &SL, SDValue Chain, SDValue ArgVal, int64_t Offset) const
bool storeOfVectorConstantIsCheap(bool IsZero, EVT MemVT, unsigned NumElem, unsigned AS) const override
Return true if it is expected to be cheaper to do a store of vector constant with the given size and ...
SDValue LowerEXTRACT_SUBVECTOR(SDValue Op, SelectionDAG &DAG) const
void computeKnownBitsForTargetNode(const SDValue Op, KnownBits &Known, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth=0) const override
Determine which of the bits specified in Mask are known to be either zero or one and return them in t...
bool shouldCombineMemoryType(EVT VT) const
SDValue splitBinaryBitConstantOpImpl(DAGCombinerInfo &DCI, const SDLoc &SL, unsigned Opc, SDValue LHS, uint32_t ValLo, uint32_t ValHi) const
Split the 64-bit value LHS into two 32-bit components, and perform the binary operation Opc to it wit...
SDValue lowerUnhandledCall(CallLoweringInfo &CLI, SmallVectorImpl< SDValue > &InVals, StringRef Reason) const
virtual SDValue LowerGlobalAddress(AMDGPUMachineFunctionInfo *MFI, SDValue Op, SelectionDAG &DAG) const
SDValue performAssertSZExtCombine(SDNode *N, DAGCombinerInfo &DCI) const
bool isTruncateFree(EVT Src, EVT Dest) const override
bool aggressivelyPreferBuildVectorSources(EVT VecVT) const override
SDValue LowerFCEIL(SDValue Op, SelectionDAG &DAG) const
TargetLowering::NegatibleCost getConstantNegateCost(const ConstantFPSDNode *C) const
SDValue LowerFLOGUnsafe(SDValue Op, const SDLoc &SL, SelectionDAG &DAG, bool IsLog10, SDNodeFlags Flags) const
SDValue combineFMinMaxLegacy(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, SDValue True, SDValue False, SDValue CC, SDNodeFlags Flags, DAGCombinerInfo &DCI) const
Flags must be the select flags, not the compare (SELECT_CC flags come from the fcmp and say nothing a...
SDValue performMulhsCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue lowerFEXPUnsafeImpl(SDValue Op, const SDLoc &SL, SelectionDAG &DAG, SDNodeFlags Flags, bool IsExp10) const
bool isSDNodeAlwaysUniform(const SDNode *N) const override
bool isDesirableToCommuteWithShift(const SDNode *N, CombineLevel Level) const override
Return true if it is profitable to move this shift by a constant amount through its operand,...
SDValue performShlCombine(SDNode *N, DAGCombinerInfo &DCI) const
bool isCheapToSpeculateCtlz(Type *Ty) const override
Return true if it is cheap to speculate a call to intrinsic ctlz.
SDValue LowerSDIVREM(SDValue Op, SelectionDAG &DAG) const
bool isFNegFree(EVT VT) const override
Return true if an fneg operation is free to the point where it is never worthwhile to replace it with...
SDValue LowerFLOG10(SDValue Op, SelectionDAG &DAG) const
SDValue LowerINT_TO_FP64(SDValue Op, SelectionDAG &DAG, bool Signed) const
unsigned computeNumSignBitsForTargetInstr(GISelValueTracking &Analysis, Register R, const APInt &DemandedElts, const MachineRegisterInfo &MRI, unsigned Depth=0) const override
This method can be implemented by targets that want to expose additional information about sign bits ...
SDValue LowerOperation(SDValue Op, SelectionDAG &DAG) const override
This callback is invoked for operations that are unsupported by the target, which are registered to u...
SDValue LowerFP_TO_FP16(SDValue Op, SelectionDAG &DAG) const
SDValue addTokenForArgument(SDValue Chain, SelectionDAG &DAG, MachineFrameInfo &MFI, int ClobberedFI) const
bool isConstantCheaperToNegate(SDValue N) const
bool isReassocProfitable(MachineRegisterInfo &MRI, Register N0, Register N1) const override
bool isKnownNeverNaNForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, bool SNaN=false, unsigned Depth=0) const override
If SNaN is false,.
static bool needsDenormHandlingF32(const SelectionDAG &DAG, SDValue Src, SDNodeFlags Flags)
SDValue lowerFPOW(SDValue Op, SelectionDAG &DAG) const
uint32_t getImplicitParameterOffset(const MachineFunction &MF, const ImplicitParameter Param) const
Helper function that returns the byte offset of the given type of implicit parameter.
SDValue lowerFEXPF64(SDValue Op, SelectionDAG &DAG) const
SDValue LowerFFLOOR(SDValue Op, SelectionDAG &DAG) const
SDValue performSelectCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue performFNegCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue LowerFP_TO_INT(SDValue Op, SelectionDAG &DAG) const
bool isConstantCostlierToNegate(SDValue N) const
SDValue loadInputValue(SelectionDAG &DAG, const TargetRegisterClass *RC, EVT VT, const SDLoc &SL, const ArgDescriptor &Arg) const
SDValue lowerFEXP10Unsafe(SDValue Op, const SDLoc &SL, SelectionDAG &DAG, SDNodeFlags Flags) const
Emit approx-funcs appropriate lowering for exp10.
bool shouldReduceLoadWidth(SDNode *Load, ISD::LoadExtType ExtType, EVT ExtVT, std::optional< unsigned > ByteOffset) const override
Return true if it is profitable to reduce a load to a smaller type.
SDValue LowerUINT_TO_FP(SDValue Op, SelectionDAG &DAG) const
bool canCreateUndefOrPoisonForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, UndefPoisonKind Kind, bool ConsiderFlags, unsigned Depth) const override
Return true if Op can create undef or poison from non-undef & non-poison operands.
bool isCheapToSpeculateCttz(Type *Ty) const override
Return true if it is cheap to speculate a call to intrinsic cttz.
SDValue performCtlz_CttzCombine(const SDLoc &SL, SDValue Cond, SDValue LHS, SDValue RHS, DAGCombinerInfo &DCI) const
SDValue performSraCombine(SDNode *N, DAGCombinerInfo &DCI) const
bool isSelectSupported(SelectSupportKind) const override
bool isZExtFree(Type *Src, Type *Dest) const override
Return true if any actual instruction that defines a value of type FromTy implicitly zero-extends the...
SDValue lowerFEXP2(SDValue Op, SelectionDAG &DAG) const
SDValue LowerCall(CallLoweringInfo &CLI, SmallVectorImpl< SDValue > &InVals) const override
This hook must be implemented to lower calls into the specified DAG.
SDValue performSrlCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue lowerFEXP(SDValue Op, SelectionDAG &DAG) const
SDValue getIsLtSmallestNormal(SelectionDAG &DAG, SDValue Op, SDNodeFlags Flags) const
SDValue getIsFinite(SelectionDAG &DAG, SDValue Op, SDNodeFlags Flags) const
bool isLoadBitCastBeneficial(EVT, EVT, const SelectionDAG &DAG, const MachineMemOperand &MMO) const final
Return true if the following transform is beneficial: fold (conv (load x)) -> (load (conv*)x) On arch...
std::pair< SDValue, SDValue > splitVector(const SDValue &N, const SDLoc &DL, const EVT &LoVT, const EVT &HighVT, SelectionDAG &DAG) const
Split a vector value into two parts of types LoVT and HiVT.
AMDGPUTargetLowering(const TargetMachine &TM, const TargetSubtargetInfo &STI, const AMDGPUSubtarget &AMDGPUSTI)
SDValue LowerFLOGCommon(SDValue Op, SelectionDAG &DAG) const
SDValue foldFreeOpFromSelect(TargetLowering::DAGCombinerInfo &DCI, SDValue N) const
SDValue LowerINT_TO_FP32(SDValue Op, SelectionDAG &DAG, bool Signed) const
bool isFAbsFree(EVT VT) const override
Return true if an fabs operation is free to the point where it is never worthwhile to replace it with...
bool isInt64ImmLegal(SDNode *Val, SelectionDAG &DAG) const
Check whether value Val can be supported by v_mov_b64, for the current target.
SDValue loadStackInputValue(SelectionDAG &DAG, EVT VT, const SDLoc &SL, int64_t Offset) const
Similar to CreateLiveInRegister, except value maybe loaded from a stack slot rather than passed in a ...
SDValue LowerFLOG2(SDValue Op, SelectionDAG &DAG) const
static EVT getEquivalentMemType(LLVMContext &Context, EVT VT)
SDValue LowerCTLS(SDValue Op, SelectionDAG &DAG) const
Split a vector store into multiple scalar stores.
SDValue getSqrtEstimate(SDValue Operand, SelectionDAG &DAG, int Enabled, int &RefinementSteps, bool &UseOneConstNR, bool Reciprocal) const override
Hooks for building estimates in place of slower divisions and square roots.
SDValue performTruncateCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue LowerSINT_TO_FP(SDValue Op, SelectionDAG &DAG) const
static SDValue stripBitcast(SDValue Val)
SDValue LowerBlockAddress(SDValue Op, SelectionDAG &DAG) const
SDValue CreateLiveInRegister(SelectionDAG &DAG, const TargetRegisterClass *RC, Register Reg, EVT VT, const SDLoc &SL, bool RawReg=false) const
Helper function that adds Reg to the LiveIn list of the DAG's MachineFunction.
SDValue SplitVectorStore(SDValue Op, SelectionDAG &DAG) const
Split a vector store into 2 stores of half the vector.
SDValue LowerCTLZ_CTTZ(SDValue Op, SelectionDAG &DAG) const
SDValue getNegatedExpression(SDValue Op, SelectionDAG &DAG, bool LegalOperations, bool ForCodeSize, NegatibleCost &Cost, unsigned Depth) const override
Return the newly negated expression if the cost is not expensive and set the cost in Cost to indicate...
std::pair< SDValue, SDValue > split64BitValue(SDValue Op, SelectionDAG &DAG) const
Return 64-bit value Op as two 32-bit integers.
SDValue performMulCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue combineFMinMaxLegacyImpl(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, SDValue True, SDValue False, SDValue CC, SDNodeFlags Flags, DAGCombinerInfo &DCI) const
SDValue getRecipEstimate(SDValue Operand, SelectionDAG &DAG, int Enabled, int &RefinementSteps) const override
Return a reciprocal estimate value for the input operand.
SDValue LowerFNEARBYINT(SDValue Op, SelectionDAG &DAG) const
SDValue LowerSIGN_EXTEND_INREG(SDValue Op, SelectionDAG &DAG) const
static CCAssignFn * CCAssignFnForReturn(CallingConv::ID CC, bool IsVarArg)
std::pair< SDValue, SDValue > getScaledLogInput(SelectionDAG &DAG, const SDLoc SL, SDValue Op, SDNodeFlags Flags) const
If denormal handling is required return the scaled input to FLOG2, and the check for denormal range.
static CCAssignFn * CCAssignFnForCall(CallingConv::ID CC, bool IsVarArg)
Selects the correct CCAssignFn for a given CallingConvention value.
bool SimplifyDemandedBitsForTargetNode(SDValue Op, const APInt &OriginalDemandedBits, const APInt &OriginalDemandedElts, KnownBits &Known, TargetLoweringOpt &TLO, unsigned Depth) const override
Attempt to simplify any target nodes based on the demanded bits/elts, returning true on success.
static bool allUsesHaveSourceMods(const SDNode *N, unsigned CostThreshold=4)
SDValue LowerFROUNDEVEN(SDValue Op, SelectionDAG &DAG) const
bool isFPImmLegal(const APFloat &Imm, EVT VT, bool ForCodeSize) const override
Returns true if the target can instruction select the specified FP immediate natively.
static unsigned numBitsUnsigned(SDValue Op, SelectionDAG &DAG)
SDValue lowerFEXPUnsafe(SDValue Op, const SDLoc &SL, SelectionDAG &DAG, SDNodeFlags Flags) const
SDValue LowerFTRUNC(SDValue Op, SelectionDAG &DAG) const
SDValue LowerDYNAMIC_STACKALLOC(SDValue Op, SelectionDAG &DAG) const
static bool allowApproxFunc(const SelectionDAG &DAG, SDNodeFlags Flags)
bool ShouldShrinkFPConstant(EVT VT) const override
If true, then instruction selection should seek to shrink the FP constant of the specified type to a ...
SDValue LowerReturn(SDValue Chain, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, const SmallVectorImpl< SDValue > &OutVals, const SDLoc &DL, SelectionDAG &DAG) const override
This hook must be implemented to lower outgoing return values, described by the Outs array,...
SDValue performStoreCombine(SDNode *N, DAGCombinerInfo &DCI) const
void ReplaceNodeResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
This callback is invoked when a node result type is illegal for the target, and the operation was reg...
SDValue performRcpCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue getLoHalf64(SDValue Op, SelectionDAG &DAG) const
SDValue lowerCTLZResults(SDValue Op, SelectionDAG &DAG) const
SDValue LowerFP_TO_INT_SAT(SDValue Op, SelectionDAG &DAG) const
SDValue performFAbsCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue LowerFP_TO_INT64(SDValue Op, SelectionDAG &DAG, bool Signed) const
static bool shouldFoldFNegIntoSrc(SDNode *FNeg, SDValue FNegSrc)
bool isNarrowingProfitable(SDNode *N, EVT SrcVT, EVT DestVT) const override
Return true if it's profitable to narrow operations of type SrcVT to DestVT.
SDValue LowerFRINT(SDValue Op, SelectionDAG &DAG) const
SDValue performIntrinsicWOChainCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue LowerUDIVREM(SDValue Op, SelectionDAG &DAG) const
SDValue performMulLoHiCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const override
This method will be invoked for all target nodes and for any target-independent nodes that the target...
SDValue lowerINT_TO_FPImpl(SDValue Op, SelectionDAG &DAG, bool Signed) const
void LowerUDIVREM64(SDValue Op, SelectionDAG &DAG, SmallVectorImpl< SDValue > &Results) const
SDValue WidenOrSplitVectorLoad(SDValue Op, SelectionDAG &DAG) const
Widen a suitably aligned v3 load.
SDValue LowerDIVREMToFloat(SDValue Op, SelectionDAG &DAG, bool sign) const
std::pair< EVT, EVT > getSplitDestVTs(const EVT &VT, SelectionDAG &DAG) const
Split a vector type into two parts.
SDValue getHiHalf64(SDValue Op, SelectionDAG &DAG) const
SDValue LowerINT_TO_FP16(SDValue Op, SelectionDAG &DAG, EVT FP16Ty) const
unsigned getVectorIdxWidth(const DataLayout &) const override
Returns the type to be used for the index operand vector operations.
static const fltSemantics & IEEEsingle()
Definition APFloat.h:304
static const fltSemantics & IEEEdouble()
Definition APFloat.h:305
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:361
static const fltSemantics & IEEEhalf()
Definition APFloat.h:302
static APFloat getQNaN(const fltSemantics &Sem, bool Negative=false, const APInt *payload=nullptr)
Factory for QNaN values.
Definition APFloat.h:1224
bool bitwiseIsEqual(const APFloat &RHS) const
Definition APFloat.h:1548
static APFloat getSmallestNormalized(const fltSemantics &Sem, bool Negative=false)
Returns the smallest (by magnitude) normalized finite number in the given semantics.
Definition APFloat.h:1262
APInt bitcastToAPInt() const
Definition APFloat.h:1475
static APFloat getInf(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Infinity.
Definition APFloat.h:1202
Class for arbitrary precision integers.
Definition APInt.h:78
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1560
static APInt getMaxValue(unsigned numBits)
Gets maximum unsigned value of APInt for specific bit width.
Definition APInt.h:202
static APInt getBitsSet(unsigned numBits, unsigned loBit, unsigned hiBit)
Get a value with a block of bits set.
Definition APInt.h:254
static APInt getSignedMaxValue(unsigned numBits)
Gets maximum signed value of APInt for a specific bit width.
Definition APInt.h:205
static APInt getSignedMinValue(unsigned numBits)
Gets minimum signed value of APInt for a specific bit width.
Definition APInt.h:215
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:302
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
Definition APInt.h:292
This class represents an incoming formal argument to a Function.
Definition Argument.h:32
const BlockAddress * getBlockAddress() const
CCState - This class holds information needed while lowering arguments and return values.
static CCValAssign getCustomMem(unsigned ValNo, MVT ValVT, int64_t Offset, MVT LocVT, LocInfo HTP)
const APFloat & getValueAPF() const
bool isNegative() const
Return true if the value is negative.
uint64_t getZExtValue() const
const APInt & getAPIntValue() const
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Diagnostic information for unsupported feature in backend.
const DataLayout & getDataLayout() const
Get the data layout of the module this function belongs to.
Definition Function.cpp:360
iterator_range< arg_iterator > args()
Definition Function.h:877
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LLVM_ABI void diagnose(const DiagnosticInfo &DI)
Report a message to the currently installed diagnostic handler.
This class is used to represent ISD::LOAD nodes.
const SDValue & getBasePtr() const
Machine Value Type.
static auto integer_fixedlen_vector_valuetypes()
uint64_t getScalarSizeInBits() const
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool isInteger() const
Return true if this is an integer or a vector integer type.
static auto integer_valuetypes()
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
LLVM_ABI int CreateFixedObject(uint64_t Size, int64_t SPOffset, bool IsImmutable, bool isAliased=false)
Create a new object at a fixed location on the stack.
int64_t getObjectSize(int ObjectIdx) const
Return the size of the specified object.
int64_t getObjectOffset(int ObjectIdx) const
Return the assigned stack offset of the specified object from the incoming stack pointer.
int getObjectIndexBegin() const
Return the minimum frame object index.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
DenormalMode getDenormalMode(const fltSemantics &FPType) const
Returns the denormal handling type for the default rounding mode of the function.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Representation of each machine instruction.
A description of a memory reference used in the backend.
@ MODereferenceable
The memory access is dereferenceable (i.e., doesn't trap).
@ MOInvariant
The memory access always returns the same value (or traps).
Flags getFlags() const
Return the raw flags of the source value,.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLVM_ABI bool isLiveIn(Register Reg) const
LLVM_ABI Register getLiveInVirtReg(MCRegister PReg) const
getLiveInVirtReg - If PReg is a live-in physical register, return the corresponding live-in virtual r...
void addLiveIn(MCRegister Reg, Register vreg=Register())
addLiveIn - Add the specified register as a live-in.
This is an abstract virtual class for memory operations.
unsigned getAddressSpace() const
Return the address space for the associated pointer.
Align getAlign() const
bool isSimple() const
Returns true if the memory operation is neither atomic or volatile.
MachineMemOperand * getMemOperand() const
Return the unique MachineMemOperand object describing the memory reference performed by operation.
const SDValue & getChain() const
bool isInvariant() const
EVT getMemoryVT() const
Return the type of the in-memory value.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
const DebugLoc & getDebugLoc() const
Represents one node in the SelectionDAG.
ArrayRef< SDUse > ops() const
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
bool hasOneUse() const
Return true if there is exactly one use of this node.
SDNodeFlags getFlags() const
SDVTList getVTList() const
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
iterator_range< user_iterator > users()
Represents a use of a SDNode.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
SDNode * getNode() const
get the SDNode which holds the desired result
bool hasOneUse() const
Return true if there is exactly one node using value ResNo of Node, in exactly one operand.
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
TypeSize getValueSizeInBits() const
Returns the size of the value in bits.
const SDValue & getOperand(unsigned i) const
unsigned getOpcode() const
unsigned getNumOperands() const
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
SIModeRegisterDefaults getMode() const
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
LLVM_ABI bool isKnownNeverLogicalZero(SDValue Op, const APInt &DemandedElts, unsigned Depth=0) const
Test whether the given floating point SDValue (or all elements of it, if it is a vector) is known to ...
SDValue getExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT, unsigned Opcode)
Convert Op, which must be of integer type, to the integer type VT, by either any/sign/zero-extending ...
LLVM_ABI unsigned ComputeMaxSignificantBits(SDValue Op, unsigned Depth=0) const
Get the upper bound on bit size for this Value Op as a signed integer.
const SDValue & getRoot() const
Return the root tag of the SelectionDAG.
const TargetSubtargetInfo & getSubtarget() const
LLVM_ABI SDValue getMergeValues(ArrayRef< SDValue > Ops, const SDLoc &dl)
Create a MERGE_VALUES node from the given operands.
LLVM_ABI SDVTList getVTList(EVT VT)
Return an SDVTList that represents the list of values specified.
LLVM_ABI SDValue getShiftAmountConstant(uint64_t Val, EVT VT, const SDLoc &DL)
LLVM_ABI SDValue getAllOnesConstant(const SDLoc &DL, EVT VT, bool IsTarget=false, bool IsOpaque=false)
LLVM_ABI void ExtractVectorElements(SDValue Op, SmallVectorImpl< SDValue > &Args, unsigned Start=0, unsigned Count=0, EVT EltVT=EVT())
Append the extracted elements from Start to Count out of the vector Op in Args.
LLVM_ABI SDValue getFreeze(SDValue V)
Return a freeze using the SDLoc of the value operand.
LLVM_ABI SDValue getConstantFP(double Val, const SDLoc &DL, EVT VT, bool isTarget=false)
Create a ConstantFPSDNode wrapping a constant value.
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
SDValue getSetCC(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, ISD::CondCode Cond, SDValue Chain=SDValue(), bool IsSignaling=false, SDNodeFlags Flags={})
Helper function to make it easier to build SetCC's if you just have an ISD::CondCode instead of an SD...
LLVM_ABI SDValue getNOT(const SDLoc &DL, SDValue Val, EVT VT)
Create a bitwise NOT operation as (XOR Val, -1).
const TargetLowering & getTargetLoweringInfo() const
SDValue getCALLSEQ_END(SDValue Chain, SDValue Op1, SDValue Op2, SDValue InGlue, const SDLoc &DL)
Return a new CALLSEQ_END node, which always must have a glue result (to ensure it's not CSE'd).
SDValue getBuildVector(EVT VT, const SDLoc &DL, ArrayRef< SDValue > Ops)
Return an ISD::BUILD_VECTOR node.
LLVM_ABI SDValue getTruncStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, SDValue Offset, MachinePointerInfo PtrInfo, EVT SVT, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getBitcast(EVT VT, SDValue V)
Return a bitcast using the SDLoc of the value operand, and casting to the provided type.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
SDValue getSelect(const SDLoc &DL, EVT VT, SDValue Cond, SDValue LHS, SDValue RHS, SDNodeFlags Flags=SDNodeFlags())
Helper function to make it easier to build Select's if you just have operands and don't want to check...
LLVM_ABI SDValue getZeroExtendInReg(SDValue Op, const SDLoc &DL, EVT VT)
Return the expression required to zero extend the Op value assuming it was the smaller SrcTy value.
const DataLayout & getDataLayout() const
LLVM_ABI SDValue getStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Helper function to build ISD::STORE nodes.
LLVM_ABI SDValue getConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
Create a ConstantSDNode wrapping a constant value.
LLVM_ABI void ReplaceAllUsesWith(SDValue From, SDValue To)
Modify anything using 'From' to use 'To' instead.
LLVM_ABI SDValue getExtLoad(ISD::LoadExtType ExtType, const SDLoc &dl, EVT VT, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, EVT MemVT, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getSignedConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
SDValue getCALLSEQ_START(SDValue Chain, uint64_t InSize, uint64_t OutSize, const SDLoc &DL)
Return a new CALLSEQ_START node, that starts new call frame, in which InSize bytes are set up inside ...
bool isConstantValueOfAnyType(SDValue N) const
SDValue getSelectCC(const SDLoc &DL, SDValue LHS, SDValue RHS, SDValue True, SDValue False, ISD::CondCode Cond, SDNodeFlags Flags=SDNodeFlags())
Helper function to make it easier to build SelectCC's if you just have an ISD::CondCode instead of an...
LLVM_ABI SDValue getSExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either sign-extending or trunca...
LLVM_ABI SDValue getLoad(EVT VT, const SDLoc &dl, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Loads are not normal binary operators: their result type is not determined by their operands,...
LLVM_ABI KnownFPClass computeKnownFPClass(SDValue Op, FPClassTest InterestedClasses, unsigned Depth=0) const
Determine floating-point class information about Op.
LLVM_ABI bool isGuaranteedNotToBeUndefOrPoison(SDValue Op, UndefPoisonKind Kind=UndefPoisonKind::UndefOrPoison, unsigned Depth=0) const
Return true if this function can prove that Op is never poison and, Kind can be used to track poison ...
LLVM_ABI SDValue getIntPtrConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI SDValue getValueType(EVT)
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
LLVM_ABI bool isKnownNeverNaN(SDValue Op, const APInt &DemandedElts, bool SNaN=false, unsigned Depth=0) const
Test whether the given SDValue (or all elements of it, if it is a vector) is known to never be NaN in...
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI unsigned ComputeNumSignBits(SDValue Op, unsigned Depth=0) const
Return the number of times the sign bit of the register is replicated into the other bits.
SDValue getTargetBlockAddress(const BlockAddress *BA, EVT VT, int64_t Offset=0, unsigned TargetFlags=0)
LLVM_ABI SDValue getVectorIdxConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI void ReplaceAllUsesOfValueWith(SDValue From, SDValue To)
Replace any uses of From with To, leaving uses of other values produced by From.getNode() alone.
MachineFunction & getMachineFunction() const
SDValue getPOISON(EVT VT)
Return a POISON node. POISON does not have a useful SDLoc.
LLVM_ABI SDValue getFrameIndex(int FI, EVT VT, bool isTarget=false)
LLVM_ABI KnownBits computeKnownBits(SDValue Op, unsigned Depth=0) const
Determine which bits of Op are known to be either zero or one and return them in Known.
LLVM_ABI SDValue getZExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either zero-extending or trunca...
LLVM_ABI bool MaskedValueIsZero(SDValue Op, const APInt &Mask, unsigned Depth=0) const
Return true if 'Op & Mask' is known to be zero.
SDValue getObjectPtrOffset(const SDLoc &SL, SDValue Ptr, TypeSize Offset)
Create an add instruction with appropriate flags when used for addressing some offset of an object.
LLVMContext * getContext() const
const SDValue & setRoot(SDValue N)
Set the current root tag of the SelectionDAG.
LLVM_ABI SDNode * UpdateNodeOperands(SDNode *N, SDValue Op)
Mutate the specified node in-place to have the specified operands.
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
LLVM_ABI std::pair< SDValue, SDValue > SplitScalar(const SDValue &N, const SDLoc &DL, const EVT &LoVT, const EVT &HiVT)
Split the scalar node with EXTRACT_ELEMENT using the provided VTs and return the low/high part.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
This class is used to represent ISD::STORE nodes.
const SDValue & getBasePtr() const
const SDValue & getValue() const
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
void setOperationAction(unsigned Op, MVT VT, LegalizeAction Action)
Indicate that the specified operation does not work with the specified type and indicate what to do a...
void setMaxDivRemBitWidthSupported(unsigned SizeInBits)
Set the size in bits of the maximum div/rem the backend supports.
bool PredictableSelectIsExpensive
Tells the code generator that select is more expensive than a branch if the branch is usually predict...
virtual bool shouldReduceLoadWidth(SDNode *Load, ISD::LoadExtType ExtTy, EVT NewVT, std::optional< unsigned > ByteOffset=std::nullopt) const
Return true if it is profitable to reduce a load to a smaller type.
unsigned MaxStoresPerMemcpyOptSize
Likewise for functions with the OptSize attribute.
const TargetMachine & getTargetMachine() const
virtual unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain targets require unusual breakdowns of certain types.
unsigned MaxGluedStoresPerMemcpy
Specify max number of store instructions to glue in inlined memcpy.
virtual MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain combinations of ABIs, Targets and features require that types are legal for some operations a...
void addBypassSlowDiv(unsigned int SlowBitWidth, unsigned int FastBitWidth)
Tells the code generator which bitwidths to bypass.
void setMaxLargeFPConvertBitWidthSupported(unsigned SizeInBits)
Set the size in bits of the maximum fp to/from int conversion the backend supports.
void setMaxAtomicSizeInBitsSupported(unsigned SizeInBits)
Set the maximum atomic operation size supported by the backend.
SelectSupportKind
Enum that describes what type of support for selects the target has.
virtual bool allowsMisalignedMemoryAccesses(EVT, unsigned AddrSpace=0, Align Alignment=Align(1), MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *=nullptr) const
Determine if the target supports unaligned memory accesses.
unsigned MaxStoresPerMemsetOptSize
Likewise for functions with the OptSize attribute.
EVT getShiftAmountTy(EVT LHSTy, const DataLayout &DL) const
Returns the type for the shift amount of a shift opcode.
unsigned MaxStoresPerMemmove
Specify maximum number of store instructions per memmove call.
virtual EVT getSetCCResultType(const DataLayout &DL, LLVMContext &Context, EVT VT) const
Return the ValueType of the result of SETCC operations.
unsigned MaxStoresPerMemmoveOptSize
Likewise for functions with the OptSize attribute.
bool isTypeLegal(EVT VT) const
Return true if the target has native support for the specified value type.
void setSupportsUnalignedAtomics(bool UnalignedSupported)
Sets whether unaligned atomic operations are supported.
bool isOperationLegal(unsigned Op, EVT VT) const
Return true if the specified operation is legal on this target.
unsigned MaxStoresPerMemset
Specify maximum number of store instructions per memset call.
void setTruncStoreAction(MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified truncating store does not work with the specified type and indicate what ...
void setMinCmpXchgSizeInBits(unsigned SizeInBits)
Sets the minimum cmpxchg or ll/sc size supported by the backend.
void AddPromotedToType(unsigned Opc, MVT OrigVT, MVT DestVT)
If Opc/OrigVT is specified as being promoted, the promotion code defaults to trying a larger integer/...
void setTargetDAGCombine(ArrayRef< ISD::NodeType > NTs)
Targets should invoke this method for each target independent node that they want to provide a custom...
void setLoadExtAction(unsigned ExtType, MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified load with extension does not work with the specified type and indicate wh...
unsigned GatherAllAliasesMaxDepth
Depth that GatherAllAliases should continue looking for chain dependencies when trying to find a more...
NegatibleCost
Enum that specifies when a float negation is beneficial.
bool allowsMemoryAccessForAlignment(LLVMContext &Context, const DataLayout &DL, EVT VT, unsigned AddrSpace=0, Align Alignment=Align(1), MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *Fast=nullptr) const
This function returns true if the memory access is aligned or if the target allows this specific unal...
unsigned MaxStoresPerMemcpy
Specify maximum number of store instructions per memcpy call.
void setSchedulingPreference(Sched::Preference Pref)
Specify the target scheduling preference.
void setJumpIsExpensive(bool isExpensive=true)
Tells the code generator not to expand logic operations on comparison predicates into separate sequen...
This class defines information used to lower LLVM code to legal SelectionDAG operators that the targe...
SDValue scalarizeVectorStore(StoreSDNode *ST, SelectionDAG &DAG) const
SDValue SimplifyMultipleUseDemandedBits(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, SelectionDAG &DAG, unsigned Depth=0) const
More limited version of SimplifyDemandedBits that can be used to "lookthrough" ops that don't contrib...
SDValue expandUnalignedStore(StoreSDNode *ST, SelectionDAG &DAG) const
Expands an unaligned store to 2 half-size stores for integer values, and possibly more for vectors.
bool ShrinkDemandedConstant(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, TargetLoweringOpt &TLO) const
Check to see if the specified operand of the specified instruction is a constant integer.
std::pair< SDValue, SDValue > expandUnalignedLoad(LoadSDNode *LD, SelectionDAG &DAG) const
Expands an unaligned load to 2 half-size loads for an integer, and possibly more for vectors.
virtual SDValue getNegatedExpression(SDValue Op, SelectionDAG &DAG, bool LegalOps, bool OptForSize, NegatibleCost &Cost, unsigned Depth=0) const
Return the newly negated expression if the cost is not expensive and set the cost in Cost to indicate...
std::pair< SDValue, SDValue > scalarizeVectorLoad(LoadSDNode *LD, SelectionDAG &DAG) const
Turn load of vector type into a load of the individual elements.
bool SimplifyDemandedBits(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, KnownBits &Known, TargetLoweringOpt &TLO, unsigned Depth=0, bool AssumeSingleUse=false) const
Look at Op.
TargetLowering(const TargetLowering &)=delete
virtual bool canCreateUndefOrPoisonForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, UndefPoisonKind Kind, bool ConsiderFlags, unsigned Depth) const
Return true if Op can create undef or poison from non-undef & non-poison operands.
Primary interface to the complete machine description for the target machine.
TargetSubtargetInfo - Generic base class for all target subtargets.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
LLVM Value Representation.
Definition Value.h:75
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ CONSTANT_ADDRESS_32BIT
Address space for 32-bit constant memory.
@ BARRIER
Address space for modeling barrier IDs as addresses.
@ REGION_ADDRESS
Address space for region memory. (GDS)
@ LOCAL_ADDRESS
Address space for local memory.
@ CONSTANT_ADDRESS
Address space for constant memory (VTX2).
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
bool isIntrinsicAlwaysUniform(unsigned IntrID)
TargetExtType * isNamedBarrier(const GlobalVariable &GV)
std::optional< APFloat > evaluateRcp(const APFloat &Val)
Evaluate the constant-folded result of v_rcp for Val, accounting for the hardware's denormal flushing...
bool isUniformMMO(const MachineMemOperand *MMO)
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
@ AMDGPU_VS
Used for Mesa vertex shaders, or AMDPAL last shader stage before rasterization (vertex shader if tess...
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ AMDGPU_Gfx
Used for AMD graphics targets.
@ AMDGPU_CS_ChainPreserve
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_HS
Used for Mesa/AMDPAL hull shaders (= tessellation control shaders).
@ AMDGPU_GS
Used for Mesa/AMDPAL geometry shaders.
@ AMDGPU_CS_Chain
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_PS
Used for Mesa/AMDPAL pixel shaders.
@ Cold
Attempts to make code in the caller as efficient as possible under the assumption that the call is no...
Definition CallingConv.h:47
@ SPIR_KERNEL
Used for SPIR kernel functions.
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ AMDGPU_ES
Used for AMDPAL shader stage before geometry shader if geometry is in use.
@ AMDGPU_LS
Used for AMDPAL vertex shader if tessellation is in use.
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
NodeType
ISD::NodeType enum - This enum defines the target-independent operators for a SelectionDAG.
Definition ISDOpcodes.h:43
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
Definition ISDOpcodes.h:837
@ SMUL_LOHI
SMUL_LOHI/UMUL_LOHI - Multiply two integers of type iN, producing a signed/unsigned value of type i[2...
Definition ISDOpcodes.h:277
@ INSERT_SUBVECTOR
INSERT_SUBVECTOR(VECTOR1, VECTOR2, IDX) - Returns a vector with VECTOR2 inserted into VECTOR1.
Definition ISDOpcodes.h:605
@ BSWAP
Byte Swap and Counting operators.
Definition ISDOpcodes.h:797
@ ATOMIC_STORE
OUTCHAIN = ATOMIC_STORE(INCHAIN, val, ptr) This corresponds to "store atomic" instruction.
@ ADDC
Carry-setting nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:296
@ FMAD
FMAD - Perform a * b + c, while getting the same result as the separately rounded operations.
Definition ISDOpcodes.h:527
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:266
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:871
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
Definition ISDOpcodes.h:523
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:898
@ CONCAT_VECTORS
CONCAT_VECTORS(VECTOR0, VECTOR1, ...) - Given a number of values of vector type with the same length ...
Definition ISDOpcodes.h:589
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:420
@ ABS
ABS - Determine the unsigned absolute value of a signed integer value of the same bitwidth.
Definition ISDOpcodes.h:757
@ SDIVREM
SDIVREM/UDIVREM - Divide two integers and produce both a quotient and remainder result.
Definition ISDOpcodes.h:282
@ FP16_TO_FP
FP16_TO_FP, FP_TO_FP16 - These operators are used to perform promotions and truncation for half-preci...
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ BUILD_PAIR
BUILD_PAIR - This is the opposite of EXTRACT_ELEMENT in some ways.
Definition ISDOpcodes.h:256
@ FLDEXP
FLDEXP - ldexp, inspired by libm (op0 * 2**op1).
@ CTLZ_ZERO_POISON
Definition ISDOpcodes.h:806
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:862
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ BRIND
BRIND - Indirect branch.
@ BR_JT
BR_JT - Jumptable branch.
@ FCANONICALIZE
Returns platform specific canonical encoding of a floating point number.
Definition ISDOpcodes.h:546
@ IS_FPCLASS
Performs a check of floating point class property, defined by IEEE-754.
Definition ISDOpcodes.h:553
@ SELECT
Select(COND, TRUEVAL, FALSEVAL).
Definition ISDOpcodes.h:814
@ ATOMIC_LOAD
Val, OUTCHAIN = ATOMIC_LOAD(INCHAIN, ptr) This corresponds to "load atomic" instruction.
@ EXTRACT_ELEMENT
EXTRACT_ELEMENT - This is used to get the lower or upper (determined by a Constant,...
Definition ISDOpcodes.h:249
@ CTLS
Count leading redundant sign bits.
Definition ISDOpcodes.h:810
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:714
@ STRICT_FP16_TO_FP
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:779
@ VECTOR_SHUFFLE
VECTOR_SHUFFLE(VEC1, VEC2) - Returns a vector, of the same type as VEC1/VEC2.
Definition ISDOpcodes.h:659
@ EXTRACT_SUBVECTOR
EXTRACT_SUBVECTOR(VECTOR, IDX) - Returns a subvector from VECTOR.
Definition ISDOpcodes.h:619
@ FMINNUM_IEEE
FMINNUM_IEEE/FMAXNUM_IEEE - Perform floating-point minimumNumber or maximumNumber on two values,...
@ EntryToken
EntryToken - This is the marker used to indicate the start of a region.
Definition ISDOpcodes.h:50
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:581
@ CopyToReg
CopyToReg - This node has three operands: a chain, a register number to set to this value,...
Definition ISDOpcodes.h:226
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:868
@ SELECT_CC
Select with condition operator - This selects between a true value and a false value (ops #2 and #3) ...
Definition ISDOpcodes.h:829
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ DYNAMIC_STACKALLOC
DYNAMIC_STACKALLOC - Allocate some number of bytes on the stack aligned to a specified boundary.
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
Definition ISDOpcodes.h:906
@ SMIN
[US]{MIN/MAX} - Binary minimum or maximum of signed or unsigned integers.
Definition ISDOpcodes.h:737
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:996
@ VSELECT
Select with a vector condition (op #0) and two vector operands (ops #1 and #2), returning a vector re...
Definition ISDOpcodes.h:823
@ UADDO_CARRY
Carry-using nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:331
@ INLINEASM_BR
INLINEASM_BR - Branching version of inline asm. Used by asm-goto.
@ FMINIMUM
FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum that also treat -0.0 as less than 0....
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:944
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:749
@ TRAP
TRAP - Trapping instruction.
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
Definition ISDOpcodes.h:207
@ ADDE
Carry-using nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:306
@ INSERT_VECTOR_ELT
INSERT_VECTOR_ELT(VECTOR, VAL, IDX) - Returns VECTOR with the element at IDX replaced with VAL.
Definition ISDOpcodes.h:570
@ TokenFactor
TokenFactor - This node takes multiple tokens as input and produces a single token result.
Definition ISDOpcodes.h:55
@ CTTZ_ZERO_POISON
Bit counting operators with a poisoned result for zero inputs.
Definition ISDOpcodes.h:805
@ FFREXP
FFREXP - frexp, extract fractional and exponent component of a floating-point value.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:977
@ ADDRSPACECAST
ADDRSPACECAST - This operator converts between pointers of different address spaces.
@ INLINEASM
INLINEASM - Represents an inline asm block.
@ FP_TO_SINT_SAT
FP_TO_[US]INT_SAT - Convert floating point value in operand 0 to a signed or unsigned scalar integer ...
Definition ISDOpcodes.h:963
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:874
@ AssertSext
AssertSext, AssertZext - These nodes record if a register contains a value that has already been zero...
Definition ISDOpcodes.h:64
@ FCOPYSIGN
FCOPYSIGN(X, Y) - Return the value of X with the sign of Y.
Definition ISDOpcodes.h:539
@ FMINIMUMNUM
FMINIMUMNUM/FMAXIMUMNUM - minimumnum/maximumnum that is same with FMINNUM_IEEE and FMAXNUM_IEEE besid...
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
Definition ISDOpcodes.h:215
@ BUILD_VECTOR
BUILD_VECTOR(ELT0, ELT1, ELT2, ELT3,...) - Return a fixed-width vector with the specified,...
Definition ISDOpcodes.h:561
bool isNormalStore(const SDNode *N)
Returns true if the specified node is a non-truncating and unindexed store.
LLVM_ABI CondCode getSetCCInverse(CondCode Operation, EVT Type)
Return the operation corresponding to !(X op Y), where 'op' is a valid SetCC operation.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LoadExtType
LoadExtType enum - This enum defines the three variants of LOADEXT (load with extension).
bool isNormalLoad(const SDNode *N)
Returns true if the specified node is a non-extending and unindexed load.
initializer< Ty > init(const Ty &Val)
constexpr double ln2
constexpr double ln10
constexpr float log2ef
Definition MathExtras.h:52
constexpr double log2e
This is an optimization pass for GlobalISel generic memory operations.
@ Offset
Definition DWP.cpp:577
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
InstructionCost Cost
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
@ Known
Known to have no common set bits.
LLVM_ABI void ComputeValueVTs(const TargetLowering &TLI, const DataLayout &DL, Type *Ty, SmallVectorImpl< EVT > &ValueVTs, SmallVectorImpl< EVT > *MemVTs=nullptr, SmallVectorImpl< TypeSize > *Offsets=nullptr, TypeSize StartingOffset=TypeSize::getZero())
ComputeValueVTs - Given an LLVM IR type, compute a sequence of EVTs that represent all the individual...
Definition Analysis.cpp:121
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
bool CCAssignFn(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
CCAssignFn - This function assigns a location for Val, updating State to reflect the change.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
SDValue peekFPSignOps(SDValue Val)
Strip fabs/fneg/fcopysign from a value to get the underlying source.
LLVM_ABI ConstantFPSDNode * isConstOrConstSplatFP(SDValue N, bool AllowUndefs=false)
Returns the SDNode if it is a constant splat BuildVector or constant float.
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
Definition MathExtras.h:380
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
int countl_zero(T Val)
Count number of 0's from the most significant bit to the least stopping at the first 1.
Definition bit.h:263
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
FPClassTest
Floating-point class tests, supported by 'is_fpclass' intrinsic.
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
CombineLevel
Definition DAGCombine.h:15
@ AfterLegalizeDAG
Definition DAGCombine.h:19
@ BeforeLegalizeTypes
Definition DAGCombine.h:16
@ AfterLegalizeTypes
Definition DAGCombine.h:17
To bit_cast(const From &from) noexcept
Definition bit.h:90
@ Mul
Product of integers.
@ Add
Sum of integers.
@ Fast
Assign the register banks as fast as possible (default).
DWARFExpression::Operation Op
LLVM_ABI ConstantSDNode * isConstOrConstSplat(SDValue N, bool AllowUndefs=false, bool AllowTruncation=false)
Returns the SDNode if it is a constant splat BuildVector or constant int.
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI bool isOneConstant(SDValue V)
Returns true if V is a constant integer one.
UndefPoisonKind
Enumeration to track whether we are interested in Undef, Poison, or both.
Definition UndefPoison.h:20
Align commonAlignment(Align A, uint64_t Offset)
Returns the alignment that satisfies both alignments.
Definition Alignment.h:201
static cl::opt< unsigned > CostThreshold("dfa-cost-threshold", cl::desc("Maximum cost accepted for the transformation"), cl::Hidden, cl::init(50))
APFloat neg(APFloat X)
Returns the negated value of the argument.
Definition APFloat.h:1727
unsigned Log2(Align A)
Returns the log2 of the alignment.
Definition Alignment.h:197
LLVM_ABI bool isAllOnesConstant(SDValue V)
Returns true if V is an integer constant with all bits set.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
MCRegister getRegister() const
unsigned getStackOffset() const
DenormalModeKind Input
Denormal treatment kind for floating point instruction inputs in the default floating-point environme...
@ PreserveSign
The sign of a flushed-to-zero number is preserved in the sign of 0.
constexpr bool inputsAreZero() const
Return true if input denormals must be implicitly treated as 0.
static constexpr DenormalMode getPreserveSign()
Extended Value Type.
Definition ValueTypes.h:35
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
Definition ValueTypes.h:418
EVT getPow2VectorType(LLVMContext &Context) const
Widens the length of the given vector EVT up to the nearest power of 2 and returns that type.
Definition ValueTypes.h:508
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition ValueTypes.h:145
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
Definition ValueTypes.h:70
EVT changeTypeToInteger() const
Return the type converted to an equivalently sized integer or vector with integer element type.
Definition ValueTypes.h:129
bool bitsGT(EVT VT) const
Return true if this has more bits than VT.
Definition ValueTypes.h:307
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
Definition ValueTypes.h:155
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
bool isByteSized() const
Return true if the bit size is a multiple of 8.
Definition ValueTypes.h:266
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
EVT getHalfSizedIntegerVT(LLVMContext &Context) const
Finds the smallest simple value type that is greater than or equal to half the width of this EVT.
Definition ValueTypes.h:453
bool isPow2VectorType() const
Returns true if the given vector is a power of 2.
Definition ValueTypes.h:501
TypeSize getStoreSizeInBits() const
Return the number of bits overwritten by a store of the specified value type.
Definition ValueTypes.h:435
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
static EVT getIntegerVT(LLVMContext &Context, unsigned BitWidth)
Returns the EVT that represents an integer with the given number of bits.
Definition ValueTypes.h:61
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
Definition ValueTypes.h:404
EVT getRoundIntegerType(LLVMContext &Context) const
Rounds the bit-width of the given integer EVT up to the nearest power of two (and at least to eight),...
Definition ValueTypes.h:442
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
Definition ValueTypes.h:346
bool bitsGE(EVT VT) const
Return true if this has no less bits than VT.
Definition ValueTypes.h:315
EVT getVectorElementType() const
Given a vector type, return the type of each element.
Definition ValueTypes.h:351
bool isExtended() const
Test if the given EVT is extended (as opposed to being simple).
Definition ValueTypes.h:150
EVT changeElementType(LLVMContext &Context, EVT EltVT) const
Return a VT for a type whose attributes match ourselves with the exception of the element type that i...
Definition ValueTypes.h:121
LLVM_ABI const fltSemantics & getFltSemantics() const
Returns an APFloat semantics tag appropriate for the value type.
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Definition ValueTypes.h:359
bool bitsLE(EVT VT) const
Return true if this has no more bits than VT.
Definition ValueTypes.h:331
bool isInteger() const
Return true if this is an integer or a vector integer type.
Definition ValueTypes.h:160
InputArg - This struct carries flags and type information about a single incoming (formal) argument o...
MVT VT
Legalized type of this argument part.
bool isUnknown() const
Returns true if we don't know any bits.
Definition KnownBits.h:64
KnownBits trunc(unsigned BitWidth) const
Return known bits for a truncation of the value we're tracking.
Definition KnownBits.h:165
KnownBits zext(unsigned BitWidth) const
Return known bits for a zero extension of the value we're tracking.
Definition KnownBits.h:176
unsigned countMaxActiveBits() const
Returns the maximum number of bits needed to represent all possible unsigned values with these known ...
Definition KnownBits.h:310
KnownBits intersectWith(const KnownBits &RHS) const
Returns KnownBits information that is known to be true for both this and RHS.
Definition KnownBits.h:325
KnownBits sext(unsigned BitWidth) const
Return known bits for a sign extension of the value we're tracking.
Definition KnownBits.h:184
unsigned countMinLeadingZeros() const
Returns the minimum number of leading zero bits.
Definition KnownBits.h:262
bool isNegative() const
Returns true if this value is known to be negative.
Definition KnownBits.h:103
static LLVM_ABI KnownBits mul(const KnownBits &LHS, const KnownBits &RHS, bool NoUndefSelfMultiply=false)
Compute known bits resulting from multiplying LHS and RHS.
bool signBitIsZeroOrNaN() const
Return true if the sign bit must be 0, ignoring the sign of nans.
Matching combinators.
This class contains a discriminated union of information about pointers in memory operands,...
LLVM_ABI bool isDereferenceable(unsigned Size, LLVMContext &C, const DataLayout &DL) const
Return true if memory region [V, V+Offset+Size) is known to be dereferenceable.
static LLVM_ABI MachinePointerInfo getStack(MachineFunction &MF, int64_t Offset, uint8_t ID=0)
Stack pointer relative access.
MachinePointerInfo getWithOffset(int64_t O) const
These are IR-level optimization flags that may be propagated to SDNodes.
void setAllowContract(bool b)
bool hasNoSignedZeros() const
This represents a list of ValueType's that has been intern'd by a SelectionDAG.
DenormalMode FP32Denormals
If this is set, neither input or output denormals are flushed for most f32 instructions.
This structure contains all information that is necessary for lowering calls.
SmallVector< ISD::InputArg, 32 > Ins
LLVM_ABI void AddToWorklist(SDNode *N)
LLVM_ABI SDValue CombineTo(SDNode *N, ArrayRef< SDValue > To, bool AddTo=true)
LLVM_ABI void CommitTargetLoweringOpt(const TargetLoweringOpt &TLO)
A convenience struct that encapsulates a DAG, and two SDValues for returning information from TargetL...