LLVM 24.0.0git
ConstantFolding.cpp
Go to the documentation of this file.
1//===-- ConstantFolding.cpp - Fold instructions into constants ------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file defines routines for folding instructions into constants.
10//
11// Also, to supplement the basic IR ConstantExpr simplifications,
12// this file defines some additional folding routines that can make use of
13// DataLayout information. These functions cannot go in IR due to library
14// dependency issues.
15//
16//===----------------------------------------------------------------------===//
17
19#include "llvm/ADT/APFloat.h"
20#include "llvm/ADT/APInt.h"
21#include "llvm/ADT/APSInt.h"
22#include "llvm/ADT/ArrayRef.h"
23#include "llvm/ADT/DenseMap.h"
24#include "llvm/ADT/STLExtras.h"
27#include "llvm/ADT/StringRef.h"
32#include "llvm/Config/config.h"
33#include "llvm/IR/Constant.h"
35#include "llvm/IR/Constants.h"
36#include "llvm/IR/DataLayout.h"
38#include "llvm/IR/Function.h"
39#include "llvm/IR/GlobalValue.h"
41#include "llvm/IR/InstrTypes.h"
42#include "llvm/IR/Instruction.h"
45#include "llvm/IR/Intrinsics.h"
46#include "llvm/IR/IntrinsicsAArch64.h"
47#include "llvm/IR/IntrinsicsAMDGPU.h"
48#include "llvm/IR/IntrinsicsARM.h"
49#include "llvm/IR/IntrinsicsNVPTX.h"
50#include "llvm/IR/IntrinsicsWebAssembly.h"
51#include "llvm/IR/IntrinsicsX86.h"
53#include "llvm/IR/Operator.h"
54#include "llvm/IR/Type.h"
55#include "llvm/IR/Value.h"
60#include <cassert>
61#include <cerrno>
62#include <cfenv>
63#include <cmath>
64#include <cstdint>
65
66using namespace llvm;
67
69 "disable-fp-call-folding",
70 cl::desc("Disable constant-folding of FP intrinsics and libcalls."),
71 cl::init(false), cl::Hidden);
72
73namespace {
74
75//===----------------------------------------------------------------------===//
76// Constant Folding internal helper functions
77//===----------------------------------------------------------------------===//
78
79static Constant *foldConstVectorToAPInt(APInt &Result, Type *DestTy,
80 Constant *C, Type *SrcEltTy,
81 unsigned NumSrcElts,
82 const DataLayout &DL) {
83 // Now that we know that the input value is a vector of integers, just shift
84 // and insert them into our result.
85 unsigned BitShift = DL.getTypeSizeInBits(SrcEltTy);
86 for (unsigned i = 0; i != NumSrcElts; ++i) {
87 Constant *Element;
88 if (DL.isLittleEndian())
89 Element = C->getAggregateElement(NumSrcElts - i - 1);
90 else
91 Element = C->getAggregateElement(i);
92
93 if (isa_and_nonnull<UndefValue>(Element)) {
94 Result <<= BitShift;
95 continue;
96 }
97
98 auto *ElementCI = dyn_cast_or_null<ConstantInt>(Element);
99 if (!ElementCI)
100 return ConstantExpr::getBitCast(C, DestTy);
101
102 Result <<= BitShift;
103 Result |= ElementCI->getValue().zext(Result.getBitWidth());
104 }
105
106 return nullptr;
107}
108
109/// Check whether folding this bitcast into a byte vector would mix poison and
110/// non-poison bits in the same output lane. While integer types track poison on
111/// a per-value basis, byte types track it on a per-bit basis. However,
112/// `ConstantByte` cannot represent values with both poison and non-poison bits.
113///
114/// Source elements are grouped by the output lane they map to. Returns true if
115/// any group contains both poison and non-poison elements.
116static bool foldMixesPoisonBits(Constant *C, unsigned NumSrcElt,
117 unsigned NumDstElt) {
118 // If element counts don't divide evenly, bail out if a poison source element
119 // might span multiple destination lanes.
120 if (NumSrcElt % NumDstElt != 0)
121 return C->containsPoisonElement();
122 unsigned Ratio = NumSrcElt / NumDstElt;
123 for (unsigned i = 0; i != NumSrcElt; i += Ratio) {
124 bool HasPoison = false;
125 bool HasNonPoison = false;
126 for (unsigned j = 0; j != Ratio; ++j) {
127 Constant *Src = C->getAggregateElement(i + j);
128 // Conservatively bail out.
129 if (!Src)
130 return true;
131 if (isa<PoisonValue>(Src))
132 HasPoison = true;
133 else
134 HasNonPoison = true;
135 }
136 if (HasPoison && HasNonPoison)
137 return true;
138 }
139 return false;
140}
141
142/// Track which destination lanes of a bitcast are produced from poison bytes.
143/// A destination lane is marked if any source element mapped to it is poison.
144/// Returns false if an aggregate element cannot be inspected. The caller should
145/// bail out of folding.
146static bool computePoisonDstLanes(Constant *C, unsigned NumSrcElt,
147 unsigned NumDstElt,
148 SmallBitVector &PoisonDstElts) {
149 // If element counts don't divide evenly, bail out if a poison source element
150 // might span multiple destination lanes.
151 if ((NumDstElt < NumSrcElt ? NumSrcElt % NumDstElt : NumDstElt % NumSrcElt))
152 return !C->containsPoisonElement();
153 if (NumDstElt < NumSrcElt) {
154 unsigned Ratio = NumSrcElt / NumDstElt;
155 for (unsigned i = 0; i != NumDstElt; ++i) {
156 for (unsigned j = 0; j != Ratio; ++j) {
157 Constant *Src = C->getAggregateElement(i * Ratio + j);
158 if (!Src)
159 return false;
160 if (isa<PoisonValue>(Src)) {
161 PoisonDstElts[i] = true;
162 break;
163 }
164 }
165 }
166 } else {
167 unsigned Ratio = NumDstElt / NumSrcElt;
168 for (unsigned i = 0; i != NumSrcElt; ++i) {
169 Constant *Src = C->getAggregateElement(i);
170 if (!Src)
171 return false;
172 if (isa<PoisonValue>(Src))
173 PoisonDstElts.set(i * Ratio, (i + 1) * Ratio);
174 }
175 }
176 return true;
177}
178
179/// Constant fold bitcast, symbolically evaluating it with DataLayout.
180/// This always returns a non-null constant, but it may be a
181/// ConstantExpr if unfoldable.
182Constant *FoldBitCast(Constant *C, Type *DestTy, const DataLayout &DL) {
183 assert(CastInst::castIsValid(Instruction::BitCast, C, DestTy) &&
184 "Invalid constantexpr bitcast!");
185
186 // Catch the obvious splat cases.
187 if (Constant *Res = ConstantFoldLoadFromUniformValue(C, DestTy, DL))
188 return Res;
189
190 if (auto *VTy = dyn_cast<VectorType>(C->getType())) {
191 // Handle a vector->scalar integer/fp cast.
192 if (isa<IntegerType>(DestTy) || DestTy->isFloatingPointTy()) {
193 unsigned NumSrcElts = cast<FixedVectorType>(VTy)->getNumElements();
194 Type *SrcEltTy = VTy->getElementType();
195
196 // Bitcasting a byte containing any poison bit to an integer or fp type
197 // yields poison.
198 if (SrcEltTy->isByteTy() && C->containsPoisonElement())
199 return PoisonValue::get(DestTy);
200
201 // If the vector is a vector of floating point or bytes, convert it to a
202 // vector of int to simplify things.
203 if (SrcEltTy->isFloatingPointTy() || SrcEltTy->isByteTy()) {
204 unsigned Width = SrcEltTy->getPrimitiveSizeInBits();
205 auto *SrcIVTy = FixedVectorType::get(
206 IntegerType::get(C->getContext(), Width), NumSrcElts);
207 // Ask IR to do the conversion now that #elts line up.
208 C = ConstantExpr::getBitCast(C, SrcIVTy);
209 }
210
211 APInt Result(DL.getTypeSizeInBits(DestTy), 0);
212 if (Constant *CE = foldConstVectorToAPInt(Result, DestTy, C,
213 SrcEltTy, NumSrcElts, DL))
214 return CE;
215
216 if (isa<IntegerType>(DestTy))
217 return ConstantInt::get(DestTy, Result);
218
219 APFloat FP(DestTy->getFltSemantics(), Result);
220 return ConstantFP::get(DestTy->getContext(), FP);
221 }
222 }
223
224 // The code below only handles casts to vectors currently.
225 auto *DestVTy = dyn_cast<VectorType>(DestTy);
226 if (!DestVTy)
227 return ConstantExpr::getBitCast(C, DestTy);
228
229 // If this is a scalar -> vector cast, convert the input into a <1 x scalar>
230 // vector so the code below can handle it uniformly.
231 if (!isa<VectorType>(C->getType()) &&
233 Constant *Ops = C; // don't take the address of C!
234 return FoldBitCast(ConstantVector::get(Ops), DestTy, DL);
235 }
236
237 // Some of what follows may extend to cover scalable vectors but the current
238 // implementation is fixed length specific.
239 if (!isa<FixedVectorType>(C->getType()))
240 return ConstantExpr::getBitCast(C, DestTy);
241
242 // If this is a bitcast from constant vector -> vector, fold it.
245 return ConstantExpr::getBitCast(C, DestTy);
246
247 // If the element types match, IR can fold it.
248 unsigned NumDstElt = cast<FixedVectorType>(DestVTy)->getNumElements();
249 unsigned NumSrcElt = cast<FixedVectorType>(C->getType())->getNumElements();
250 if (NumDstElt == NumSrcElt)
251 return ConstantExpr::getBitCast(C, DestTy);
252
253 Type *SrcEltTy = cast<VectorType>(C->getType())->getElementType();
254 Type *DstEltTy = DestVTy->getElementType();
255
256 // Otherwise, we're changing the number of elements in a vector, which
257 // requires endianness information to do the right thing. For example,
258 // bitcast (<2 x i64> <i64 0, i64 1> to <4 x i32>)
259 // folds to (little endian):
260 // <4 x i32> <i32 0, i32 0, i32 1, i32 0>
261 // and to (big endian):
262 // <4 x i32> <i32 0, i32 0, i32 0, i32 1>
263
264 // First thing is first. We only want to think about integer here, so if
265 // we have something in FP form, recast it as integer.
266 if (DstEltTy->isFloatingPointTy()) {
267 // Fold to an vector of integers with same size as our FP type.
268 unsigned FPWidth = DstEltTy->getPrimitiveSizeInBits();
269 auto *DestIVTy = FixedVectorType::get(
270 IntegerType::get(C->getContext(), FPWidth), NumDstElt);
271 // Recursively handle this integer conversion, if possible.
272 C = FoldBitCast(C, DestIVTy, DL);
273
274 // Finally, IR can handle this now that #elts line up.
275 return ConstantExpr::getBitCast(C, DestTy);
276 }
277
278 // Handle byte destination type by folding through integers.
279 if (DstEltTy->isByteTy()) {
280 // When combining elements into larger byte values, bail out if the fold
281 // mixes poison and non-poison bits in the same destination element. Byte
282 // types track poison per bit, and no constant value can represent that.
283 if (NumDstElt < NumSrcElt && foldMixesPoisonBits(C, NumSrcElt, NumDstElt))
284 return ConstantExpr::getBitCast(C, DestTy);
285
286 // Fold to a vector of integers with same size as the byte type.
287 unsigned ByteWidth = DstEltTy->getPrimitiveSizeInBits();
288 auto *DestIVTy = FixedVectorType::get(
289 IntegerType::get(C->getContext(), ByteWidth), NumDstElt);
290 C = FoldBitCast(C, DestIVTy, DL);
291 return ConstantExpr::getBitCast(C, DestTy);
292 }
293
294 // Okay, we know the destination is integer, if the input is FP, convert
295 // it to integer first.
296 if (SrcEltTy->isFloatingPointTy()) {
297 unsigned FPWidth = SrcEltTy->getPrimitiveSizeInBits();
298 auto *SrcIVTy = FixedVectorType::get(
299 IntegerType::get(C->getContext(), FPWidth), NumSrcElt);
300 // Ask IR to do the conversion now that #elts line up.
301 C = ConstantExpr::getBitCast(C, SrcIVTy);
302 assert((isa<ConstantVector>(C) || // FIXME: Remove ConstantVector.
304 "Constant folding cannot fail for plain fp->int bitcast!");
305 }
306
307 // Handle byte source type by folding through integers. Byte types track
308 // poison per bit, so any poison bit makes the destination lane poison.
309 // Record which destination lanes contain poison bits, before the generic
310 // fold below refines them to undef/zero, so they can be restored.
311 SmallBitVector PoisonDstElts(NumDstElt);
312 if (SrcEltTy->isByteTy()) {
313 if (!computePoisonDstLanes(C, NumSrcElt, NumDstElt, PoisonDstElts))
314 return ConstantExpr::getBitCast(C, DestTy);
315
316 unsigned ByteWidth = SrcEltTy->getPrimitiveSizeInBits();
317 auto *SrcIVTy = FixedVectorType::get(
318 IntegerType::get(C->getContext(), ByteWidth), NumSrcElt);
319 // Ask IR to do the conversion now that #elts line up.
320 C = ConstantExpr::getBitCast(C, SrcIVTy);
321 assert((isa<ConstantVector>(C) || // FIXME: Remove ConstantVector.
323 "Constant folding cannot fail for plain byte->int bitcast!");
324 }
325
326 // Now we know that the input and output vectors are both integer vectors
327 // of the same size, and that their #elements is not the same.
328 // Use data buffer for easy non-integer element ratio vectors handling,
329 // For example: <4 x i24> to <3 x i32>.
330 bool isLittleEndian = DL.isLittleEndian();
331 unsigned SrcBitSize = SrcEltTy->getPrimitiveSizeInBits();
332 unsigned DstBitSize = DstEltTy->getPrimitiveSizeInBits();
334 unsigned SrcElt = 0;
335
336 APInt Buffer(2 * std::max(SrcBitSize, DstBitSize), 0);
337 APInt UndefMask(Buffer.getBitWidth(), 0);
338 APInt PoisonMask(Buffer.getBitWidth(), 0);
339 unsigned BufferBitSize = 0;
340
341 while (Result.size() != NumDstElt) {
342 // Load SrcElts into Buffer.
343 while (BufferBitSize < DstBitSize) {
344 Constant *Element = C->getAggregateElement(SrcElt++);
345 if (!Element) // Reject constantexpr elements
346 return ConstantExpr::getBitCast(C, DestTy);
347
348 // Shift Buffer & Masks to fit next SrcElt.
349 if (!isLittleEndian) {
350 Buffer <<= SrcBitSize;
351 UndefMask <<= SrcBitSize;
352 PoisonMask <<= SrcBitSize;
353 }
354
355 APInt SrcValue;
356 unsigned BitPosition = isLittleEndian ? BufferBitSize : 0;
357 if (isa<UndefValue>(Element)) {
358 // Set masks fragments bits.
359 UndefMask.setBits(BitPosition, BitPosition + SrcBitSize);
360 if (isa<PoisonValue>(Element))
361 PoisonMask.setBits(BitPosition, BitPosition + SrcBitSize);
362 SrcValue = APInt::getZero(SrcBitSize);
363 } else {
364 auto *Src = dyn_cast<ConstantInt>(Element);
365 if (!Src)
366 return ConstantExpr::getBitCast(C, DestTy);
367 SrcValue = Src->getValue();
368 }
369
370 // Insert src element bits into Buffer on correct position.
371 Buffer.insertBits(SrcValue, BitPosition);
372 BufferBitSize += SrcBitSize;
373 }
374
375 // Create DstElts from Buffer.
376 while (BufferBitSize >= DstBitSize) {
377 unsigned ShiftAmt = isLittleEndian ? 0 : BufferBitSize - DstBitSize;
378 // Emit undef/poison, if all undef mask fragment bits are set.
379 if (UndefMask.extractBits(DstBitSize, ShiftAmt).isAllOnes()) {
380 // Push poison, if any bit in poison mask fragment is set.
381 if (!PoisonMask.extractBits(DstBitSize, ShiftAmt).isZero()) {
382 Result.push_back(PoisonValue::get(DstEltTy));
383 } else {
384 Result.push_back(UndefValue::get(DstEltTy));
385 }
386 } else {
387 // Create and push DstElt.
388 APInt Elt = Buffer.extractBits(DstBitSize, ShiftAmt);
389 Result.push_back(ConstantInt::get(DstEltTy, Elt));
390 }
391
392 // Shift unused Buffer fragment to lower bits.
393 if (isLittleEndian) {
394 Buffer.lshrInPlace(DstBitSize);
395 UndefMask.lshrInPlace(DstBitSize);
396 PoisonMask.lshrInPlace(DstBitSize);
397 }
398 BufferBitSize -= DstBitSize;
399 }
400 }
401
402 // Restore destination lanes whose source bytes contained poison bits.
403 for (unsigned I : PoisonDstElts.set_bits())
404 Result[I] = PoisonValue::get(DstEltTy);
405
406 return ConstantVector::get(Result);
407}
408
409} // end anonymous namespace
410
411/// If this constant is a constant offset from a global, return the global and
412/// the constant. Because of constantexprs, this function is recursive.
414 APInt &Offset, const DataLayout &DL,
415 DSOLocalEquivalent **DSOEquiv) {
416 if (DSOEquiv)
417 *DSOEquiv = nullptr;
418
419 // Trivial case, constant is the global.
420 if ((GV = dyn_cast<GlobalValue>(C))) {
421 unsigned BitWidth = DL.getIndexTypeSizeInBits(GV->getType());
422 Offset = APInt(BitWidth, 0);
423 return true;
424 }
425
426 if (auto *FoundDSOEquiv = dyn_cast<DSOLocalEquivalent>(C)) {
427 if (DSOEquiv)
428 *DSOEquiv = FoundDSOEquiv;
429 GV = FoundDSOEquiv->getGlobalValue();
430 unsigned BitWidth = DL.getIndexTypeSizeInBits(GV->getType());
431 Offset = APInt(BitWidth, 0);
432 return true;
433 }
434
435 // Otherwise, if this isn't a constant expr, bail out.
436 auto *CE = dyn_cast<ConstantExpr>(C);
437 if (!CE) return false;
438
439 // Look through ptr->int and ptr->ptr casts.
440 if (CE->getOpcode() == Instruction::PtrToInt ||
441 CE->getOpcode() == Instruction::PtrToAddr)
442 return IsConstantOffsetFromGlobal(CE->getOperand(0), GV, Offset, DL,
443 DSOEquiv);
444
445 // i32* getelementptr ([5 x i32]* @a, i32 0, i32 5)
446 auto *GEP = dyn_cast<GEPOperator>(CE);
447 if (!GEP)
448 return false;
449
450 unsigned BitWidth = DL.getIndexTypeSizeInBits(GEP->getType());
451 APInt TmpOffset(BitWidth, 0);
452
453 // If the base isn't a global+constant, we aren't either.
454 if (!IsConstantOffsetFromGlobal(CE->getOperand(0), GV, TmpOffset, DL,
455 DSOEquiv))
456 return false;
457
458 // Otherwise, add any offset that our operands provide.
459 if (!GEP->accumulateConstantOffset(DL, TmpOffset))
460 return false;
461
462 Offset = TmpOffset;
463 return true;
464}
465
467 const DataLayout &DL) {
468 do {
469 Type *SrcTy = C->getType();
470 if (SrcTy == DestTy)
471 return C;
472
473 TypeSize DestSize = DL.getTypeSizeInBits(DestTy);
474 TypeSize SrcSize = DL.getTypeSizeInBits(SrcTy);
475 if (!TypeSize::isKnownGE(SrcSize, DestSize))
476 return nullptr;
477
478 // Catch the obvious splat cases (since all-zeros can coerce non-integral
479 // pointers legally).
480 if (Constant *Res = ConstantFoldLoadFromUniformValue(C, DestTy, DL))
481 return Res;
482
483 // If the type sizes are the same and a cast is legal, just directly
484 // cast the constant.
485 // But be careful not to coerce non-integral pointers illegally.
486 if (SrcSize == DestSize &&
487 DL.isNonIntegralPointerType(SrcTy->getScalarType()) ==
488 DL.isNonIntegralPointerType(DestTy->getScalarType())) {
489 Instruction::CastOps Cast = Instruction::BitCast;
490 // If we are going from a pointer to int or vice versa, we spell the cast
491 // differently.
492 if (SrcTy->isIntegerTy() && DestTy->isPointerTy())
493 Cast = Instruction::IntToPtr;
494 else if (SrcTy->isPointerTy() && DestTy->isIntegerTy())
495 Cast = Instruction::PtrToInt;
496
497 if (CastInst::castIsValid(Cast, C, DestTy))
498 return ConstantFoldCastOperand(Cast, C, DestTy, DL);
499 }
500
501 // If this isn't an aggregate type, there is nothing we can do to drill down
502 // and find a bitcastable constant.
503 if (!SrcTy->isAggregateType() && !SrcTy->isVectorTy())
504 return nullptr;
505
506 // We're simulating a load through a pointer that was bitcast to point to
507 // a different type, so we can try to walk down through the initial
508 // elements of an aggregate to see if some part of the aggregate is
509 // castable to implement the "load" semantic model.
510 if (SrcTy->isStructTy()) {
511 // Struct types might have leading zero-length elements like [0 x i32],
512 // which are certainly not what we are looking for, so skip them.
513 unsigned Elem = 0;
514 Constant *ElemC;
515 do {
516 ElemC = C->getAggregateElement(Elem++);
517 } while (ElemC && DL.getTypeSizeInBits(ElemC->getType()).isZero());
518 C = ElemC;
519 } else {
520 // For non-byte-sized vector elements, the first element is not
521 // necessarily located at the vector base address.
522 if (auto *VT = dyn_cast<VectorType>(SrcTy))
523 if (!DL.typeSizeEqualsStoreSize(VT->getElementType()))
524 return nullptr;
525
526 C = C->getAggregateElement(0u);
527 }
528 } while (C);
529
530 return nullptr;
531}
532
533namespace {
534
535/// Recursive helper to read bits out of global. C is the constant being copied
536/// out of. ByteOffset is an offset into C. CurPtr is the pointer to copy
537/// results into and BytesLeft is the number of bytes left in
538/// the CurPtr buffer. DL is the DataLayout. When IsByteLoad is true, do not
539/// unwrap inttoptr constant expressions. The caller would reconstruct those
540/// bits as a ConstantByte, dropping the pointer's provenance.
541bool ReadDataFromGlobal(Constant *C, uint64_t ByteOffset, unsigned char *CurPtr,
542 unsigned BytesLeft, const DataLayout &DL,
543 bool IsByteLoad = false) {
544 assert(ByteOffset <= DL.getTypeAllocSize(C->getType()) &&
545 "Out of range access");
546
547 // Reading type padding, return zero.
548 if (ByteOffset >= DL.getTypeStoreSize(C->getType()))
549 return true;
550
551 // If this element is zero or undefined, we can just return since *CurPtr is
552 // zero initialized.
554 return true;
555
556 auto *CI = dyn_cast<ConstantInt>(C);
557 if (CI && CI->getType()->isIntegerTy()) {
558 if ((CI->getBitWidth() & 7) != 0)
559 return false;
560 const APInt &Val = CI->getValue();
561 unsigned IntBytes = unsigned(CI->getBitWidth()/8);
562
563 for (unsigned i = 0; i != BytesLeft && ByteOffset != IntBytes; ++i) {
564 unsigned n = ByteOffset;
565 if (!DL.isLittleEndian())
566 n = IntBytes - n - 1;
567 CurPtr[i] = Val.extractBits(8, n * 8).getZExtValue();
568 ++ByteOffset;
569 }
570 return true;
571 }
572
573 auto *CFP = dyn_cast<ConstantFP>(C);
574 if (CFP && CFP->getType()->isFloatingPointTy()) {
575 if (CFP->getType()->isDoubleTy()) {
576 C = FoldBitCast(C, Type::getInt64Ty(C->getContext()), DL);
577 return ReadDataFromGlobal(C, ByteOffset, CurPtr, BytesLeft, DL,
578 IsByteLoad);
579 }
580 if (CFP->getType()->isFloatTy()){
581 C = FoldBitCast(C, Type::getInt32Ty(C->getContext()), DL);
582 return ReadDataFromGlobal(C, ByteOffset, CurPtr, BytesLeft, DL,
583 IsByteLoad);
584 }
585 if (CFP->getType()->isHalfTy()){
586 C = FoldBitCast(C, Type::getInt16Ty(C->getContext()), DL);
587 return ReadDataFromGlobal(C, ByteOffset, CurPtr, BytesLeft, DL,
588 IsByteLoad);
589 }
590 return false;
591 }
592
593 if (auto *CS = dyn_cast<ConstantStruct>(C)) {
594 const StructLayout *SL = DL.getStructLayout(CS->getType());
595 unsigned Index = SL->getElementContainingOffset(ByteOffset);
596 uint64_t CurEltOffset = SL->getElementOffset(Index);
597 ByteOffset -= CurEltOffset;
598
599 while (true) {
600 // If the element access is to the element itself and not to tail padding,
601 // read the bytes from the element.
602 uint64_t EltSize = DL.getTypeAllocSize(CS->getOperand(Index)->getType());
603
604 if (ByteOffset < EltSize &&
605 !ReadDataFromGlobal(CS->getOperand(Index), ByteOffset, CurPtr,
606 BytesLeft, DL, IsByteLoad))
607 return false;
608
609 ++Index;
610
611 // Check to see if we read from the last struct element, if so we're done.
612 if (Index == CS->getType()->getNumElements())
613 return true;
614
615 // If we read all of the bytes we needed from this element we're done.
616 uint64_t NextEltOffset = SL->getElementOffset(Index);
617
618 if (BytesLeft <= NextEltOffset - CurEltOffset - ByteOffset)
619 return true;
620
621 // Move to the next element of the struct.
622 CurPtr += NextEltOffset - CurEltOffset - ByteOffset;
623 BytesLeft -= NextEltOffset - CurEltOffset - ByteOffset;
624 ByteOffset = 0;
625 CurEltOffset = NextEltOffset;
626 }
627 // not reached.
628 }
629
633 uint64_t NumElts, EltSize;
634 Type *EltTy;
635 if (auto *AT = dyn_cast<ArrayType>(C->getType())) {
636 NumElts = AT->getNumElements();
637 EltTy = AT->getElementType();
638 EltSize = DL.getTypeAllocSize(EltTy);
639 } else {
640 NumElts = cast<FixedVectorType>(C->getType())->getNumElements();
641 EltTy = cast<FixedVectorType>(C->getType())->getElementType();
642 // TODO: For non-byte-sized vectors, current implementation assumes there is
643 // padding to the next byte boundary between elements.
644 if (!DL.typeSizeEqualsStoreSize(EltTy))
645 return false;
646
647 EltSize = DL.getTypeStoreSize(EltTy);
648 }
649 uint64_t Index = ByteOffset / EltSize;
650 uint64_t Offset = ByteOffset - Index * EltSize;
651
652 for (; Index != NumElts; ++Index) {
653 if (!ReadDataFromGlobal(C->getAggregateElement(Index), Offset, CurPtr,
654 BytesLeft, DL, IsByteLoad))
655 return false;
656
657 uint64_t BytesWritten = EltSize - Offset;
658 assert(BytesWritten <= EltSize && "Not indexing into this element?");
659 if (BytesWritten >= BytesLeft)
660 return true;
661
662 Offset = 0;
663 BytesLeft -= BytesWritten;
664 CurPtr += BytesWritten;
665 }
666 return true;
667 }
668
669 if (auto *CE = dyn_cast<ConstantExpr>(C)) {
670 if (CE->getOpcode() == Instruction::IntToPtr &&
671 CE->getOperand(0)->getType() == DL.getIntPtrType(CE->getType())) {
672 // Folding byte loads through the integer operand would rebuild the result
673 // as a `ConstantByte`, dropping the pointer's provenance.
674 if (IsByteLoad)
675 return false;
676 return ReadDataFromGlobal(CE->getOperand(0), ByteOffset, CurPtr,
677 BytesLeft, DL, IsByteLoad);
678 }
679 }
680
681 // Otherwise, unknown initializer type.
682 return false;
683}
684
685/// OrigLoadTy is the original type being loaded, while LoadTy is the type
686/// currently being folded (which may be integer type mapped from OrigLoadTy).
687Constant *FoldReinterpretLoadFromConst(Constant *C, Type *LoadTy,
688 Type *OrigLoadTy, int64_t Offset,
689 const DataLayout &DL) {
690 // Bail out early. Not expect to load from scalable global variable.
691 if (isa<ScalableVectorType>(LoadTy))
692 return nullptr;
693
694 auto *IntType = dyn_cast<IntegerType>(LoadTy);
695
696 // If this isn't an integer load we can't fold it directly.
697 if (!IntType) {
698 // If this is a non-integer load, we can try folding it as an int load and
699 // then bitcast the result. This can be useful for union cases. Note
700 // that address spaces don't matter here since we're not going to result in
701 // an actual new load.
702 if (!LoadTy->isFloatingPointTy() && !LoadTy->isPointerTy() &&
703 !LoadTy->isByteTy() && !LoadTy->isVectorTy())
704 return nullptr;
705
706 Type *MapTy = Type::getIntNTy(C->getContext(),
707 DL.getTypeSizeInBits(LoadTy).getFixedValue());
708 if (Constant *Res =
709 FoldReinterpretLoadFromConst(C, MapTy, OrigLoadTy, Offset, DL)) {
710 if (Res->isNullValue() && !LoadTy->isX86_AMXTy())
711 // Materializing a zero can be done trivially without a bitcast
712 return Constant::getNullValue(LoadTy);
713 Type *CastTy = LoadTy->isPtrOrPtrVectorTy() ? DL.getIntPtrType(LoadTy) : LoadTy;
714 Res = FoldBitCast(Res, CastTy, DL);
715 if (LoadTy->isPtrOrPtrVectorTy()) {
716 // For vector of pointer, we needed to first convert to a vector of integer, then do vector inttoptr
717 if (Res->isNullValue() && !LoadTy->isX86_AMXTy())
718 return Constant::getNullValue(LoadTy);
719 if (DL.isNonIntegralPointerType(LoadTy->getScalarType()))
720 // Be careful not to replace a load of an addrspace value with an inttoptr here
721 return nullptr;
722 Res = ConstantExpr::getIntToPtr(Res, LoadTy);
723 }
724 return Res;
725 }
726 return nullptr;
727 }
728
729 unsigned BytesLoaded = (IntType->getBitWidth() + 7) / 8;
730 // Allow folding of large type loads (e.g. <16 x double>).
731 if (BytesLoaded > 128 || BytesLoaded == 0)
732 return nullptr;
733
734 // For scalar integer load, use smaller limit to avoid regression during
735 // memcmp expansion. Codegen may generate inefficient string operations.
736 if (BytesLoaded > 32 && OrigLoadTy->isIntegerTy())
737 return nullptr;
738
739 // If we're not accessing anything in this constant, the result is undefined.
740 if (Offset <= -1 * static_cast<int64_t>(BytesLoaded))
741 return PoisonValue::get(IntType);
742
743 // TODO: We should be able to support scalable types.
744 TypeSize InitializerSize = DL.getTypeAllocSize(C->getType());
745 if (InitializerSize.isScalable())
746 return nullptr;
747
748 // If we're not accessing anything in this constant, the result is undefined.
749 if (Offset >= (int64_t)InitializerSize.getFixedValue())
750 return PoisonValue::get(IntType);
751
752 SmallVector<unsigned char, 64> RawBytes(BytesLoaded);
753 unsigned char *CurPtr = RawBytes.data();
754 unsigned BytesLeft = BytesLoaded;
755
756 // If we're loading off the beginning of the global, some bytes may be valid.
757 if (Offset < 0) {
758 CurPtr += -Offset;
759 BytesLeft += Offset;
760 Offset = 0;
761 }
762
763 if (!ReadDataFromGlobal(C, Offset, CurPtr, BytesLeft, DL,
764 /*IsByteLoad=*/OrigLoadTy->isByteOrByteVectorTy()))
765 return nullptr;
766
767 APInt ResultVal = APInt(IntType->getBitWidth(), 0);
768 if (DL.isLittleEndian()) {
769 ResultVal = RawBytes[BytesLoaded - 1];
770 for (unsigned i = 1; i != BytesLoaded; ++i) {
771 ResultVal <<= 8;
772 ResultVal |= RawBytes[BytesLoaded - 1 - i];
773 }
774 } else {
775 ResultVal = RawBytes[0];
776 for (unsigned i = 1; i != BytesLoaded; ++i) {
777 ResultVal <<= 8;
778 ResultVal |= RawBytes[i];
779 }
780 }
781
782 return ConstantInt::get(IntType->getContext(), ResultVal);
783}
784
785} // anonymous namespace
786
787// If GV is a constant with an initializer read its representation starting
788// at Offset and return it as a constant array of unsigned char. Otherwise
789// return null.
792 if (!GV->isConstant() || !GV->hasDefinitiveInitializer())
793 return nullptr;
794
795 const DataLayout &DL = GV->getDataLayout();
796 Constant *Init = const_cast<Constant *>(GV->getInitializer());
797 TypeSize InitSize = DL.getTypeAllocSize(Init->getType());
798 if (InitSize < Offset)
799 return nullptr;
800
801 uint64_t NBytes = InitSize - Offset;
802 if (NBytes > UINT16_MAX)
803 // Bail for large initializers in excess of 64K to avoid allocating
804 // too much memory.
805 // Offset is assumed to be less than or equal than InitSize (this
806 // is enforced in ReadDataFromGlobal).
807 return nullptr;
808
809 SmallVector<unsigned char, 256> RawBytes(static_cast<size_t>(NBytes));
810 unsigned char *CurPtr = RawBytes.data();
811
812 if (!ReadDataFromGlobal(Init, Offset, CurPtr, NBytes, DL))
813 return nullptr;
814
815 return ConstantDataArray::get(GV->getContext(), RawBytes);
816}
817
818/// If this Offset points exactly to the start of an aggregate element, return
819/// that element, otherwise return nullptr.
821 const DataLayout &DL) {
822 if (Offset.isZero())
823 return Base;
824
826 return nullptr;
827
828 Type *ElemTy = Base->getType();
829 SmallVector<APInt> Indices = DL.getGEPIndicesForOffset(ElemTy, Offset);
830 if (!Offset.isZero() || !Indices[0].isZero())
831 return nullptr;
832
833 Constant *C = Base;
834 for (const APInt &Index : drop_begin(Indices)) {
835 if (Index.isNegative() || Index.getActiveBits() >= 32)
836 return nullptr;
837
838 C = C->getAggregateElement(Index.getZExtValue());
839 if (!C)
840 return nullptr;
841 }
842
843 return C;
844}
845
847 const APInt &Offset,
848 const DataLayout &DL) {
849 if (Constant *AtOffset = getConstantAtOffset(C, Offset, DL))
850 if (Constant *Result = ConstantFoldLoadThroughBitcast(AtOffset, Ty, DL))
851 return Result;
852
853 // Explicitly check for out-of-bounds access, so we return poison even if the
854 // constant is a uniform value.
855 TypeSize Size = DL.getTypeAllocSize(C->getType());
856 if (!Size.isScalable() && Offset.sge(Size.getFixedValue()))
857 return PoisonValue::get(Ty);
858
859 // Try an offset-independent fold of a uniform value.
860 if (Constant *Result = ConstantFoldLoadFromUniformValue(C, Ty, DL))
861 return Result;
862
863 // Try hard to fold loads from bitcasted strange and non-type-safe things.
864 if (Offset.getSignificantBits() <= 64)
865 if (Constant *Result =
866 FoldReinterpretLoadFromConst(C, Ty, Ty, Offset.getSExtValue(), DL))
867 return Result;
868
869 return nullptr;
870}
871
876
879 const DataLayout &DL) {
880 // We can only fold loads from constant globals with a definitive initializer.
881 // Check this upfront, to skip expensive offset calculations.
883 if (!GV || !GV->isConstant() || !GV->hasDefinitiveInitializer())
884 return nullptr;
885
886 C = cast<Constant>(C->stripAndAccumulateConstantOffsets(
887 DL, Offset, /* AllowNonInbounds */ true));
888
889 if (C == GV)
890 if (Constant *Result = ConstantFoldLoadFromConst(GV->getInitializer(), Ty,
891 Offset, DL))
892 return Result;
893
894 // If this load comes from anywhere in a uniform constant global, the value
895 // is always the same, regardless of the loaded offset.
896 return ConstantFoldLoadFromUniformValue(GV->getInitializer(), Ty, DL);
897}
898
900 const DataLayout &DL) {
901 APInt Offset(DL.getIndexTypeSizeInBits(C->getType()), 0);
902 return ConstantFoldLoadFromConstPtr(C, Ty, std::move(Offset), DL);
903}
904
906 const DataLayout &DL) {
907 if (isa<PoisonValue>(C))
908 return PoisonValue::get(Ty);
909 if (isa<UndefValue>(C))
910 return UndefValue::get(Ty);
911 // If padding is needed when storing C to memory, then it isn't considered as
912 // uniform.
913 if (!DL.typeSizeEqualsStoreSize(C->getType()))
914 return nullptr;
915 if (C->isNullValue() && !Ty->isX86_AMXTy())
916 return Constant::getNullValue(Ty);
917 if (C->isAllOnesValue() &&
918 (Ty->isIntOrIntVectorTy() || Ty->isByteOrByteVectorTy() ||
919 Ty->isFPOrFPVectorTy()))
920 return Constant::getAllOnesValue(Ty);
921 return nullptr;
922}
923
924namespace {
925
926/// One of Op0/Op1 is a constant expression.
927/// Attempt to symbolically evaluate the result of a binary operator merging
928/// these together. If target data info is available, it is provided as DL,
929/// otherwise DL is null.
930Constant *SymbolicallyEvaluateBinop(unsigned Opc, Constant *Op0, Constant *Op1,
931 const DataLayout &DL) {
932 // SROA
933
934 // Fold (and 0xffffffff00000000, (shl x, 32)) -> shl.
935 // Fold (lshr (or X, Y), 32) -> (lshr [X/Y], 32) if one doesn't contribute
936 // bits.
937
938 if (Opc == Instruction::And) {
939 KnownBits Known0 = computeKnownBits(Op0, DL);
940 KnownBits Known1 = computeKnownBits(Op1, DL);
941 if ((Known1.One | Known0.Zero).isAllOnes()) {
942 // All the bits of Op0 that the 'and' could be masking are already zero.
943 return Op0;
944 }
945 if ((Known0.One | Known1.Zero).isAllOnes()) {
946 // All the bits of Op1 that the 'and' could be masking are already zero.
947 return Op1;
948 }
949
950 Known0 &= Known1;
951 if (Known0.isConstant())
952 return ConstantInt::get(Op0->getType(), Known0.getConstant());
953 }
954
955 // If the constant expr is something like &A[123] - &A[4].f, fold this into a
956 // constant. This happens frequently when iterating over a global array.
957 if (Opc == Instruction::Sub) {
958 GlobalValue *GV1, *GV2;
959 APInt Offs1, Offs2;
960
961 if (IsConstantOffsetFromGlobal(Op0, GV1, Offs1, DL))
962 if (IsConstantOffsetFromGlobal(Op1, GV2, Offs2, DL) && GV1 == GV2) {
963 unsigned OpSize = DL.getTypeSizeInBits(Op0->getType());
964
965 // (&GV+C1) - (&GV+C2) -> C1-C2, pointer arithmetic cannot overflow.
966 // PtrToInt may change the bitwidth so we have convert to the right size
967 // first.
968 return ConstantInt::get(Op0->getType(), Offs1.zextOrTrunc(OpSize) -
969 Offs2.zextOrTrunc(OpSize));
970 }
971 }
972
973 return nullptr;
974}
975
976/// If array indices are not pointer-sized integers, explicitly cast them so
977/// that they aren't implicitly casted by the getelementptr.
978Constant *CastGEPIndices(Type *SrcElemTy, ArrayRef<Constant *> Ops,
979 Type *ResultTy, GEPNoWrapFlags NW,
980 std::optional<ConstantRange> InRange,
981 const DataLayout &DL, const TargetLibraryInfo *TLI) {
982 Type *IntIdxTy = DL.getIndexType(ResultTy);
983 Type *IntIdxScalarTy = IntIdxTy->getScalarType();
984
985 bool Any = false;
987 for (unsigned i = 1, e = Ops.size(); i != e; ++i) {
988 if ((i == 1 ||
990 SrcElemTy, Ops.slice(1, i - 1)))) &&
991 Ops[i]->getType()->getScalarType() != IntIdxScalarTy) {
992 Any = true;
993 Type *NewType =
994 Ops[i]->getType()->isVectorTy() ? IntIdxTy : IntIdxScalarTy;
996 CastInst::getCastOpcode(Ops[i], true, NewType, true), Ops[i], NewType,
997 DL);
998 if (!NewIdx)
999 return nullptr;
1000 NewIdxs.push_back(NewIdx);
1001 } else
1002 NewIdxs.push_back(Ops[i]);
1003 }
1004
1005 if (!Any)
1006 return nullptr;
1007
1008 Constant *C =
1009 ConstantExpr::getGetElementPtr(SrcElemTy, Ops[0], NewIdxs, NW, InRange);
1010 return ConstantFoldConstant(C, DL, TLI);
1011}
1012
1013/// If we can symbolically evaluate the GEP constant expression, do so.
1014Constant *SymbolicallyEvaluateGEP(const GEPOperator *GEP,
1016 const DataLayout &DL,
1017 const TargetLibraryInfo *TLI) {
1018 Type *SrcElemTy = GEP->getSourceElementType();
1019 Type *ResTy = GEP->getType();
1020 if (!SrcElemTy->isSized() || isa<ScalableVectorType>(SrcElemTy))
1021 return nullptr;
1022
1023 if (Constant *C = CastGEPIndices(SrcElemTy, Ops, ResTy, GEP->getNoWrapFlags(),
1024 GEP->getInRange(), DL, TLI))
1025 return C;
1026
1027 Constant *Ptr = Ops[0];
1028 if (!Ptr->getType()->isPointerTy())
1029 return nullptr;
1030
1031 Type *IntIdxTy = DL.getIndexType(Ptr->getType());
1032
1033 for (unsigned i = 1, e = Ops.size(); i != e; ++i)
1034 if (!isa<ConstantInt>(Ops[i]) || !Ops[i]->getType()->isIntegerTy())
1035 return nullptr;
1036
1037 unsigned BitWidth = DL.getTypeSizeInBits(IntIdxTy);
1038 APInt Offset = APInt(
1039 BitWidth,
1040 DL.getIndexedOffsetInType(
1041 SrcElemTy, ArrayRef((Value *const *)Ops.data() + 1, Ops.size() - 1)),
1042 /*isSigned=*/true, /*implicitTrunc=*/true);
1043
1044 std::optional<ConstantRange> InRange = GEP->getInRange();
1045 if (InRange)
1046 InRange = InRange->sextOrTrunc(BitWidth);
1047
1048 // If this is a GEP of a GEP, fold it all into a single GEP.
1049 GEPNoWrapFlags NW = GEP->getNoWrapFlags();
1050 bool Overflow = false;
1051 while (auto *GEP = dyn_cast<GEPOperator>(Ptr)) {
1052 NW &= GEP->getNoWrapFlags();
1053
1054 SmallVector<Value *, 4> NestedOps(llvm::drop_begin(GEP->operands()));
1055
1056 // Do not try the incorporate the sub-GEP if some index is not a number.
1057 bool AllConstantInt = true;
1058 for (Value *NestedOp : NestedOps)
1059 if (!isa<ConstantInt>(NestedOp)) {
1060 AllConstantInt = false;
1061 break;
1062 }
1063 if (!AllConstantInt)
1064 break;
1065
1066 // Adjust inrange offset and intersect inrange attributes
1067 if (auto GEPRange = GEP->getInRange()) {
1068 auto AdjustedGEPRange = GEPRange->sextOrTrunc(BitWidth).subtract(Offset);
1069 InRange =
1070 InRange ? InRange->intersectWith(AdjustedGEPRange) : AdjustedGEPRange;
1071 }
1072
1073 Ptr = cast<Constant>(GEP->getOperand(0));
1074 SrcElemTy = GEP->getSourceElementType();
1075 Offset = Offset.sadd_ov(
1076 APInt(BitWidth, DL.getIndexedOffsetInType(SrcElemTy, NestedOps),
1077 /*isSigned=*/true, /*implicitTrunc=*/true),
1078 Overflow);
1079 }
1080
1081 // Preserving nusw (without inbounds) also requires that the offset
1082 // additions did not overflow.
1083 if (NW.hasNoUnsignedSignedWrap() && !NW.isInBounds() && Overflow)
1085
1086 // If the base value for this address is a literal integer value, fold the
1087 // getelementptr to the resulting integer value casted to the pointer type.
1088 APInt BaseIntVal(DL.getPointerTypeSizeInBits(Ptr->getType()), 0);
1089 if (auto *CE = dyn_cast<ConstantExpr>(Ptr)) {
1090 if (CE->getOpcode() == Instruction::IntToPtr) {
1091 if (auto *Base = dyn_cast<ConstantInt>(CE->getOperand(0)))
1092 BaseIntVal = Base->getValue().zextOrTrunc(BaseIntVal.getBitWidth());
1093 }
1094 }
1095
1096 if ((Ptr->isNullValue() || BaseIntVal != 0) &&
1097 !DL.mustNotIntroduceIntToPtr(Ptr->getType())) {
1098
1099 // If the index size is smaller than the pointer size, add to the low
1100 // bits only.
1101 BaseIntVal.insertBits(BaseIntVal.trunc(BitWidth) + Offset, 0);
1102 Constant *C = ConstantInt::get(Ptr->getContext(), BaseIntVal);
1103 return ConstantExpr::getIntToPtr(C, ResTy);
1104 }
1105
1106 // Try to infer inbounds for GEPs of globals.
1107 if (!NW.isInBounds() && Offset.isNonNegative()) {
1108 bool CanBeNull;
1109 uint64_t DerefBytes = Ptr->getPointerDereferenceableBytes(
1110 DL, CanBeNull, /*CanBeFreed=*/nullptr);
1111 if (DerefBytes != 0 && !CanBeNull && Offset.sle(DerefBytes))
1113 }
1114
1115 // nusw + nneg -> nuw
1116 if (NW.hasNoUnsignedSignedWrap() && Offset.isNonNegative())
1118
1119 // Otherwise canonicalize this to a single ptradd.
1120 LLVMContext &Ctx = Ptr->getContext();
1121 return ConstantExpr::getPtrAdd(Ptr, ConstantInt::get(Ctx, Offset), NW,
1122 InRange);
1123}
1124
1125/// Attempt to constant fold an instruction with the
1126/// specified opcode and operands. If successful, the constant result is
1127/// returned, if not, null is returned. Note that this function can fail when
1128/// attempting to fold instructions like loads and stores, which have no
1129/// constant expression form.
1130Constant *ConstantFoldInstOperandsImpl(const Value *InstOrCE, unsigned Opcode,
1132 const DataLayout &DL,
1133 const TargetLibraryInfo *TLI,
1134 bool AllowNonDeterministic) {
1135 Type *DestTy = InstOrCE->getType();
1136
1137 if (Instruction::isUnaryOp(Opcode))
1138 return ConstantFoldUnaryOpOperand(Opcode, Ops[0], DL);
1139
1140 if (Instruction::isBinaryOp(Opcode)) {
1141 switch (Opcode) {
1142 default:
1143 break;
1144 case Instruction::FAdd:
1145 case Instruction::FSub:
1146 case Instruction::FMul:
1147 case Instruction::FDiv:
1148 case Instruction::FRem:
1149 // Handle floating point instructions separately to account for denormals
1150 // TODO: If a constant expression is being folded rather than an
1151 // instruction, denormals will not be flushed/treated as zero
1152 if (const auto *I = dyn_cast<Instruction>(InstOrCE)) {
1153 return ConstantFoldFPInstOperands(Opcode, Ops[0], Ops[1], DL, I,
1154 AllowNonDeterministic);
1155 }
1156 }
1157 return ConstantFoldBinaryOpOperands(Opcode, Ops[0], Ops[1], DL);
1158 }
1159
1160 if (Instruction::isCast(Opcode))
1161 return ConstantFoldCastOperand(Opcode, Ops[0], DestTy, DL);
1162
1163 if (auto *GEP = dyn_cast<GEPOperator>(InstOrCE)) {
1164 Type *SrcElemTy = GEP->getSourceElementType();
1166 return nullptr;
1167
1168 if (Constant *C = SymbolicallyEvaluateGEP(GEP, Ops, DL, TLI))
1169 return C;
1170
1171 return ConstantExpr::getGetElementPtr(SrcElemTy, Ops[0], Ops.slice(1),
1172 GEP->getNoWrapFlags(),
1173 GEP->getInRange());
1174 }
1175
1176 if (auto *CE = dyn_cast<ConstantExpr>(InstOrCE))
1177 return CE->getWithOperands(Ops);
1178
1179 switch (Opcode) {
1180 default: return nullptr;
1181 case Instruction::ICmp:
1182 case Instruction::FCmp: {
1183 auto *C = cast<CmpInst>(InstOrCE);
1184 return ConstantFoldCompareInstOperands(C->getPredicate(), Ops[0], Ops[1],
1185 DL, TLI, C);
1186 }
1187 case Instruction::Freeze:
1188 return isGuaranteedNotToBeUndefOrPoison(Ops[0]) ? Ops[0] : nullptr;
1189 case Instruction::Call:
1190 if (auto *F = dyn_cast<Function>(Ops.back())) {
1191 const auto *Call = cast<CallBase>(InstOrCE);
1193 return ConstantFoldCall(Call, F, Ops.slice(0, Ops.size() - 1), TLI,
1194 AllowNonDeterministic);
1195 }
1196 return nullptr;
1197 case Instruction::Select:
1198 return ConstantFoldSelectInstruction(Ops[0], Ops[1], Ops[2]);
1199 case Instruction::ExtractElement:
1201 case Instruction::ExtractValue:
1203 Ops[0], cast<ExtractValueInst>(InstOrCE)->getIndices());
1204 case Instruction::InsertElement:
1205 return ConstantExpr::getInsertElement(Ops[0], Ops[1], Ops[2]);
1206 case Instruction::InsertValue:
1208 Ops[0], Ops[1], cast<InsertValueInst>(InstOrCE)->getIndices());
1209 case Instruction::ShuffleVector:
1211 Ops[0], Ops[1], cast<ShuffleVectorInst>(InstOrCE)->getShuffleMask());
1212 case Instruction::Load: {
1213 const auto *LI = dyn_cast<LoadInst>(InstOrCE);
1214 if (LI->isVolatile())
1215 return nullptr;
1216 return ConstantFoldLoadFromConstPtr(Ops[0], LI->getType(), DL);
1217 }
1218 }
1219}
1220
1221} // end anonymous namespace
1222
1223//===----------------------------------------------------------------------===//
1224// Constant Folding public APIs
1225//===----------------------------------------------------------------------===//
1226
1227namespace {
1228
1229Constant *
1230ConstantFoldConstantImpl(const Constant *C, const DataLayout &DL,
1231 const TargetLibraryInfo *TLI,
1234 return const_cast<Constant *>(C);
1235
1237 for (const Use &OldU : C->operands()) {
1238 Constant *OldC = cast<Constant>(&OldU);
1239 Constant *NewC = OldC;
1240 // Recursively fold the ConstantExpr's operands. If we have already folded
1241 // a ConstantExpr, we don't have to process it again.
1242 if (isa<ConstantVector>(OldC) || isa<ConstantExpr>(OldC)) {
1243 auto It = FoldedOps.find(OldC);
1244 if (It == FoldedOps.end()) {
1245 NewC = ConstantFoldConstantImpl(OldC, DL, TLI, FoldedOps);
1246 FoldedOps.insert({OldC, NewC});
1247 } else {
1248 NewC = It->second;
1249 }
1250 }
1251 Ops.push_back(NewC);
1252 }
1253
1254 if (auto *CE = dyn_cast<ConstantExpr>(C)) {
1255 if (Constant *Res = ConstantFoldInstOperandsImpl(
1256 CE, CE->getOpcode(), Ops, DL, TLI, /*AllowNonDeterministic=*/true))
1257 return Res;
1258 return const_cast<Constant *>(C);
1259 }
1260
1262 return ConstantVector::get(Ops);
1263}
1264
1265} // end anonymous namespace
1266
1268 const DataLayout &DL,
1269 const TargetLibraryInfo *TLI) {
1270 // Handle PHI nodes quickly here...
1271 if (auto *PN = dyn_cast<PHINode>(I)) {
1272 Constant *CommonValue = nullptr;
1273
1275 for (Value *Incoming : PN->incoming_values()) {
1276 // If the incoming value is undef then skip it. Note that while we could
1277 // skip the value if it is equal to the phi node itself we choose not to
1278 // because that would break the rule that constant folding only applies if
1279 // all operands are constants.
1280 if (isa<UndefValue>(Incoming))
1281 continue;
1282 // If the incoming value is not a constant, then give up.
1283 auto *C = dyn_cast<Constant>(Incoming);
1284 if (!C)
1285 return nullptr;
1286 // Fold the PHI's operands.
1287 C = ConstantFoldConstantImpl(C, DL, TLI, FoldedOps);
1288 // If the incoming value is a different constant to
1289 // the one we saw previously, then give up.
1290 if (CommonValue && C != CommonValue)
1291 return nullptr;
1292 CommonValue = C;
1293 }
1294
1295 // If we reach here, all incoming values are the same constant or undef.
1296 return CommonValue ? CommonValue : UndefValue::get(PN->getType());
1297 }
1298
1299 // Scan the operand list, checking to see if they are all constants, if so,
1300 // hand off to ConstantFoldInstOperandsImpl.
1301 if (!all_of(I->operands(), [](const Use &U) { return isa<Constant>(U); }))
1302 return nullptr;
1303
1306 for (const Use &OpU : I->operands()) {
1307 auto *Op = cast<Constant>(&OpU);
1308 // Fold the Instruction's operands.
1309 Op = ConstantFoldConstantImpl(Op, DL, TLI, FoldedOps);
1310 Ops.push_back(Op);
1311 }
1312
1313 return ConstantFoldInstOperands(I, Ops, DL, TLI);
1314}
1315
1317 const TargetLibraryInfo *TLI) {
1319 return ConstantFoldConstantImpl(C, DL, TLI, FoldedOps);
1320}
1321
1324 const DataLayout &DL,
1325 const TargetLibraryInfo *TLI,
1326 bool AllowNonDeterministic) {
1327 return ConstantFoldInstOperandsImpl(I, I->getOpcode(), Ops, DL, TLI,
1328 AllowNonDeterministic);
1329}
1330
1332 unsigned IntPredicate, Constant *Ops0, Constant *Ops1, const DataLayout &DL,
1333 const TargetLibraryInfo *TLI, const Instruction *I) {
1334 CmpInst::Predicate Predicate = (CmpInst::Predicate)IntPredicate;
1335 // fold: icmp (inttoptr x), null -> icmp x, 0
1336 // fold: icmp null, (inttoptr x) -> icmp 0, x
1337 // fold: icmp (ptrtoint x), 0 -> icmp x, null
1338 // fold: icmp 0, (ptrtoint x) -> icmp null, x
1339 // fold: icmp (inttoptr x), (inttoptr y) -> icmp trunc/zext x, trunc/zext y
1340 // fold: icmp (ptrtoint x), (ptrtoint y) -> icmp x, y
1341 //
1342 // FIXME: The following comment is out of data and the DataLayout is here now.
1343 // ConstantExpr::getCompare cannot do this, because it doesn't have DL
1344 // around to know if bit truncation is happening.
1345 if (auto *CE0 = dyn_cast<ConstantExpr>(Ops0)) {
1346 if (Ops1->isNullValue()) {
1347 if (CE0->getOpcode() == Instruction::IntToPtr) {
1348 Type *IntPtrTy = DL.getIntPtrType(CE0->getType());
1349 // Convert the integer value to the right size to ensure we get the
1350 // proper extension or truncation.
1351 if (Constant *C = ConstantFoldIntegerCast(CE0->getOperand(0), IntPtrTy,
1352 /*IsSigned*/ false, DL)) {
1353 Constant *Null = Constant::getNullValue(C->getType());
1354 return ConstantFoldCompareInstOperands(Predicate, C, Null, DL, TLI);
1355 }
1356 }
1357
1358 // icmp only compares the address part of the pointer, so only do this
1359 // transform if the integer size matches the address size.
1360 if (CE0->getOpcode() == Instruction::PtrToInt ||
1361 CE0->getOpcode() == Instruction::PtrToAddr) {
1362 Type *AddrTy = DL.getAddressType(CE0->getOperand(0)->getType());
1363 if (CE0->getType() == AddrTy) {
1364 Constant *C = CE0->getOperand(0);
1365 Constant *Null = Constant::getNullValue(C->getType());
1366 return ConstantFoldCompareInstOperands(Predicate, C, Null, DL, TLI);
1367 }
1368 }
1369 }
1370
1371 if (auto *CE1 = dyn_cast<ConstantExpr>(Ops1)) {
1372 if (CE0->getOpcode() == CE1->getOpcode()) {
1373 if (CE0->getOpcode() == Instruction::IntToPtr) {
1374 Type *IntPtrTy = DL.getIntPtrType(CE0->getType());
1375
1376 // Convert the integer value to the right size to ensure we get the
1377 // proper extension or truncation.
1378 Constant *C0 = ConstantFoldIntegerCast(CE0->getOperand(0), IntPtrTy,
1379 /*IsSigned*/ false, DL);
1380 Constant *C1 = ConstantFoldIntegerCast(CE1->getOperand(0), IntPtrTy,
1381 /*IsSigned*/ false, DL);
1382 if (C0 && C1)
1383 return ConstantFoldCompareInstOperands(Predicate, C0, C1, DL, TLI);
1384 }
1385
1386 // icmp only compares the address part of the pointer, so only do this
1387 // transform if the integer size matches the address size.
1388 if (CE0->getOpcode() == Instruction::PtrToInt ||
1389 CE0->getOpcode() == Instruction::PtrToAddr) {
1390 Type *AddrTy = DL.getAddressType(CE0->getOperand(0)->getType());
1391 if (CE0->getType() == AddrTy &&
1392 CE0->getOperand(0)->getType() == CE1->getOperand(0)->getType()) {
1394 Predicate, CE0->getOperand(0), CE1->getOperand(0), DL, TLI);
1395 }
1396 }
1397 }
1398 }
1399
1400 // Convert pointer comparison (base+offset1) pred (base+offset2) into
1401 // offset1 pred offset2, for the case where the offset is inbounds. This
1402 // only works for equality and unsigned comparison, as inbounds permits
1403 // crossing the sign boundary. However, the offset comparison itself is
1404 // signed.
1405 if (Ops0->getType()->isPointerTy() && !ICmpInst::isSigned(Predicate)) {
1406 unsigned IndexWidth = DL.getIndexTypeSizeInBits(Ops0->getType());
1407 APInt Offset0(IndexWidth, 0);
1408 bool IsEqPred = ICmpInst::isEquality(Predicate);
1409 Value *Stripped0 = Ops0->stripAndAccumulateConstantOffsets(
1410 DL, Offset0, /*AllowNonInbounds=*/IsEqPred,
1411 /*AllowInvariantGroup=*/false, /*ExternalAnalysis=*/nullptr,
1412 /*LookThroughIntToPtr=*/IsEqPred);
1413 APInt Offset1(IndexWidth, 0);
1414 Value *Stripped1 = Ops1->stripAndAccumulateConstantOffsets(
1415 DL, Offset1, /*AllowNonInbounds=*/IsEqPred,
1416 /*AllowInvariantGroup=*/false, /*ExternalAnalysis=*/nullptr,
1417 /*LookThroughIntToPtr=*/IsEqPred);
1418 if (Stripped0 == Stripped1)
1419 return ConstantInt::getBool(
1420 Ops0->getContext(),
1421 ICmpInst::compare(Offset0, Offset1,
1422 ICmpInst::getSignedPredicate(Predicate)));
1423 }
1424 } else if (isa<ConstantExpr>(Ops1)) {
1425 // If RHS is a constant expression, but the left side isn't, swap the
1426 // operands and try again.
1427 Predicate = ICmpInst::getSwappedPredicate(Predicate);
1428 return ConstantFoldCompareInstOperands(Predicate, Ops1, Ops0, DL, TLI);
1429 }
1430
1431 if (CmpInst::isFPPredicate(Predicate)) {
1432 // Flush any denormal constant float input according to denormal handling
1433 // mode.
1434 Ops0 = FlushFPConstant(Ops0, I, /*IsOutput=*/false);
1435 if (!Ops0)
1436 return nullptr;
1437 Ops1 = FlushFPConstant(Ops1, I, /*IsOutput=*/false);
1438 if (!Ops1)
1439 return nullptr;
1440 }
1441
1442 return ConstantFoldCompareInstruction(Predicate, Ops0, Ops1);
1443}
1444
1446 const DataLayout &DL) {
1448
1449 return ConstantFoldUnaryInstruction(Opcode, Op);
1450}
1451
1453 Constant *RHS,
1454 const DataLayout &DL) {
1456 if (isa<ConstantExpr>(LHS) || isa<ConstantExpr>(RHS))
1457 if (Constant *C = SymbolicallyEvaluateBinop(Opcode, LHS, RHS, DL))
1458 return C;
1459
1461 return ConstantExpr::get(Opcode, LHS, RHS);
1462 return ConstantFoldBinaryInstruction(Opcode, LHS, RHS);
1463}
1464
1467 switch (Mode) {
1469 return nullptr;
1470 case DenormalMode::IEEE:
1471 return ConstantFP::get(Ty, APF);
1473 return ConstantFP::get(
1474 Ty, APFloat::getZero(APF.getSemantics(), APF.isNegative()));
1476 return ConstantFP::get(Ty, APFloat::getZero(APF.getSemantics(), false));
1477 default:
1478 break;
1479 }
1480
1481 llvm_unreachable("unknown denormal mode");
1482}
1483
1484/// Return the denormal mode that can be assumed when executing a floating point
1485/// operation at \p CtxI.
1487 if (!CtxI || !CtxI->getParent() || !CtxI->getFunction())
1488 return DenormalMode::getDynamic();
1489 return CtxI->getFunction()->getDenormalMode(
1490 Ty->getScalarType()->getFltSemantics());
1491}
1492
1494 const Instruction *Inst,
1495 bool IsOutput) {
1496 const APFloat &APF = CFP->getValueAPF();
1497 if (!APF.isDenormal())
1498 return CFP;
1499
1501 return flushDenormalConstant(CFP->getType(), APF,
1502 IsOutput ? Mode.Output : Mode.Input);
1503}
1504
1506 bool IsOutput) {
1507 if (ConstantFP *CFP = dyn_cast<ConstantFP>(Operand))
1508 return flushDenormalConstantFP(CFP, Inst, IsOutput);
1509
1511 return Operand;
1512
1513 Type *Ty = Operand->getType();
1514 VectorType *VecTy = dyn_cast<VectorType>(Ty);
1515 if (VecTy) {
1516 if (auto *Splat = dyn_cast_or_null<ConstantFP>(Operand->getSplatValue())) {
1517 ConstantFP *Folded = flushDenormalConstantFP(Splat, Inst, IsOutput);
1518 if (!Folded)
1519 return nullptr;
1520 return ConstantVector::getSplat(VecTy->getElementCount(), Folded);
1521 }
1522
1523 Ty = VecTy->getElementType();
1524 }
1525
1526 if (isa<ConstantExpr>(Operand))
1527 return Operand;
1528
1529 if (const auto *CV = dyn_cast<ConstantVector>(Operand)) {
1531 for (unsigned i = 0, e = CV->getNumOperands(); i != e; ++i) {
1532 Constant *Element = CV->getAggregateElement(i);
1533 if (isa<UndefValue>(Element)) {
1534 NewElts.push_back(Element);
1535 continue;
1536 }
1537
1538 ConstantFP *CFP = dyn_cast<ConstantFP>(Element);
1539 if (!CFP)
1540 return nullptr;
1541
1542 ConstantFP *Folded = flushDenormalConstantFP(CFP, Inst, IsOutput);
1543 if (!Folded)
1544 return nullptr;
1545 NewElts.push_back(Folded);
1546 }
1547
1548 return ConstantVector::get(NewElts);
1549 }
1550
1551 if (const auto *CDV = dyn_cast<ConstantDataVector>(Operand)) {
1553 for (unsigned I = 0, E = CDV->getNumElements(); I < E; ++I) {
1554 const APFloat &Elt = CDV->getElementAsAPFloat(I);
1555 if (!Elt.isDenormal()) {
1556 NewElts.push_back(ConstantFP::get(Ty, Elt));
1557 } else {
1558 DenormalMode Mode = getInstrDenormalMode(Inst, Ty);
1559 ConstantFP *Folded =
1560 flushDenormalConstant(Ty, Elt, IsOutput ? Mode.Output : Mode.Input);
1561 if (!Folded)
1562 return nullptr;
1563 NewElts.push_back(Folded);
1564 }
1565 }
1566
1567 return ConstantVector::get(NewElts);
1568 }
1569
1570 return nullptr;
1571}
1572
1574 Constant *RHS, const DataLayout &DL,
1575 const Instruction *I,
1576 bool AllowNonDeterministic) {
1577 if (Instruction::isBinaryOp(Opcode)) {
1578 // Flush denormal inputs if needed.
1579 Constant *Op0 = FlushFPConstant(LHS, I, /* IsOutput */ false);
1580 if (!Op0)
1581 return nullptr;
1582 Constant *Op1 = FlushFPConstant(RHS, I, /* IsOutput */ false);
1583 if (!Op1)
1584 return nullptr;
1585
1586 // If nsz or an algebraic FMF flag is set, the result of the FP operation
1587 // may change due to future optimization. Don't constant fold them if
1588 // non-deterministic results are not allowed.
1589 if (!AllowNonDeterministic)
1591 if (FP->hasNoSignedZeros() || FP->hasAllowReassoc() ||
1592 FP->hasAllowContract() || FP->hasAllowReciprocal())
1593 return nullptr;
1594
1595 // Calculate constant result.
1596 Constant *C = ConstantFoldBinaryOpOperands(Opcode, Op0, Op1, DL);
1597 if (!C)
1598 return nullptr;
1599
1600 // Flush denormal output if needed.
1601 C = FlushFPConstant(C, I, /* IsOutput */ true);
1602 if (!C)
1603 return nullptr;
1604
1605 // The precise NaN value is non-deterministic.
1606 if (!AllowNonDeterministic && C->isNaN())
1607 return nullptr;
1608
1609 return C;
1610 }
1611 // If instruction lacks a parent/function and the denormal mode cannot be
1612 // determined, use the default (IEEE).
1613 return ConstantFoldBinaryOpOperands(Opcode, LHS, RHS, DL);
1614}
1615
1617 Type *DestTy, const DataLayout &DL) {
1618 assert(Instruction::isCast(Opcode));
1619
1620 if (auto *CE = dyn_cast<ConstantExpr>(C))
1621 if (CE->isCast())
1622 if (unsigned NewOp = CastInst::isEliminableCastPair(
1623 Instruction::CastOps(CE->getOpcode()),
1624 Instruction::CastOps(Opcode), CE->getOperand(0)->getType(),
1625 C->getType(), DestTy, &DL))
1626 return ConstantFoldCastOperand(NewOp, CE->getOperand(0), DestTy, DL);
1627
1628 switch (Opcode) {
1629 default:
1630 llvm_unreachable("Missing case");
1631 case Instruction::PtrToAddr:
1632 case Instruction::PtrToInt:
1633 if (auto *CE = dyn_cast<ConstantExpr>(C)) {
1634 Constant *FoldedValue = nullptr;
1635 // If the input is an inttoptr, eliminate the pair. This requires knowing
1636 // the width of a pointer, so it can't be done in ConstantExpr::getCast.
1637 if (CE->getOpcode() == Instruction::IntToPtr) {
1638 // zext/trunc the inttoptr to pointer/address size.
1639 Type *MidTy = Opcode == Instruction::PtrToInt
1640 ? DL.getAddressType(CE->getType())
1641 : DL.getIntPtrType(CE->getType());
1642 FoldedValue = ConstantFoldIntegerCast(CE->getOperand(0), MidTy,
1643 /*IsSigned=*/false, DL);
1644 } else if (auto *GEP = dyn_cast<GEPOperator>(CE)) {
1645 // If we have GEP, we can perform the following folds:
1646 // (ptrtoint/ptrtoaddr (gep null, x)) -> x
1647 // (ptrtoint/ptrtoaddr (gep (gep null, x), y) -> x + y, etc.
1648 unsigned BitWidth = DL.getIndexTypeSizeInBits(GEP->getType());
1649 APInt BaseOffset(BitWidth, 0);
1650 auto *Base = cast<Constant>(GEP->stripAndAccumulateConstantOffsets(
1651 DL, BaseOffset, /*AllowNonInbounds=*/true));
1652 if (Base->isNullValue()) {
1653 FoldedValue = ConstantInt::get(CE->getContext(), BaseOffset);
1654 } else {
1655 // ptrtoint/ptrtoaddr (gep i8, Ptr, (sub 0, V))
1656 // -> sub (ptrtoint/ptrtoaddr Ptr), V
1657 if (GEP->getNumIndices() == 1 &&
1658 GEP->getSourceElementType()->isIntegerTy(8)) {
1659 auto *Ptr = cast<Constant>(GEP->getPointerOperand());
1660 auto *Sub = dyn_cast<ConstantExpr>(GEP->getOperand(1));
1661 Type *IntIdxTy = DL.getIndexType(Ptr->getType());
1662 if (Sub && Sub->getType() == IntIdxTy &&
1663 Sub->getOpcode() == Instruction::Sub &&
1664 Sub->getOperand(0)->isNullValue())
1665 FoldedValue = ConstantExpr::getSub(
1666 ConstantExpr::getCast(Opcode, Ptr, IntIdxTy),
1667 Sub->getOperand(1));
1668 }
1669 }
1670 }
1671 if (FoldedValue) {
1672 // Do a zext or trunc to get to the ptrtoint/ptrtoaddr dest size.
1673 return ConstantFoldIntegerCast(FoldedValue, DestTy, /*IsSigned=*/false,
1674 DL);
1675 }
1676 }
1677 break;
1678 case Instruction::IntToPtr:
1679 // If the input is a ptrtoint, turn the pair into a ptr to ptr bitcast if
1680 // the int size is >= the ptr size and the address spaces are the same.
1681 // This requires knowing the width of a pointer, so it can't be done in
1682 // ConstantExpr::getCast.
1683 if (auto *CE = dyn_cast<ConstantExpr>(C)) {
1684 if (CE->getOpcode() == Instruction::PtrToInt) {
1685 Constant *SrcPtr = CE->getOperand(0);
1686 unsigned SrcPtrSize = DL.getPointerTypeSizeInBits(SrcPtr->getType());
1687 unsigned MidIntSize = CE->getType()->getScalarSizeInBits();
1688
1689 if (MidIntSize >= SrcPtrSize) {
1690 unsigned SrcAS = SrcPtr->getType()->getPointerAddressSpace();
1691 if (SrcAS == DestTy->getPointerAddressSpace())
1692 return FoldBitCast(CE->getOperand(0), DestTy, DL);
1693 }
1694 }
1695 }
1696 break;
1697 case Instruction::Trunc:
1698 case Instruction::ZExt:
1699 case Instruction::SExt:
1700 case Instruction::FPTrunc:
1701 case Instruction::FPExt:
1702 case Instruction::UIToFP:
1703 case Instruction::SIToFP:
1704 case Instruction::FPToUI:
1705 case Instruction::FPToSI:
1706 case Instruction::AddrSpaceCast:
1707 break;
1708 case Instruction::BitCast:
1709 return FoldBitCast(C, DestTy, DL);
1710 }
1711
1713 return ConstantExpr::getCast(Opcode, C, DestTy);
1714 return ConstantFoldCastInstruction(Opcode, C, DestTy);
1715}
1716
1718 bool IsSigned, const DataLayout &DL) {
1719 Type *SrcTy = C->getType();
1720 if (SrcTy == DestTy)
1721 return C;
1722 if (SrcTy->getScalarSizeInBits() > DestTy->getScalarSizeInBits())
1723 return ConstantFoldCastOperand(Instruction::Trunc, C, DestTy, DL);
1724 if (IsSigned)
1725 return ConstantFoldCastOperand(Instruction::SExt, C, DestTy, DL);
1726 return ConstantFoldCastOperand(Instruction::ZExt, C, DestTy, DL);
1727}
1728
1729//===----------------------------------------------------------------------===//
1730// Constant Folding for Calls
1731//
1732
1733/// Returns true if the intrinsic can be constant folded, given \p IsStrictFP.
1734static bool canConstantFoldIntrinsic(Intrinsic::ID ID, bool IsStrictFP) {
1735 switch (ID) {
1736 // Operations that do not operate floating-point numbers and do not depend on
1737 // FP environment can be folded even in strictfp functions.
1738 case Intrinsic::bswap:
1739 case Intrinsic::ctpop:
1740 case Intrinsic::ctlz:
1741 case Intrinsic::cttz:
1742 case Intrinsic::fshl:
1743 case Intrinsic::fshr:
1744 case Intrinsic::clmul:
1745 case Intrinsic::pdep:
1746 case Intrinsic::pext:
1747 case Intrinsic::launder_invariant_group:
1748 case Intrinsic::strip_invariant_group:
1749 case Intrinsic::masked_load:
1750 case Intrinsic::get_active_lane_mask:
1751 case Intrinsic::abs:
1752 case Intrinsic::smax:
1753 case Intrinsic::smin:
1754 case Intrinsic::umax:
1755 case Intrinsic::umin:
1756 case Intrinsic::scmp:
1757 case Intrinsic::ucmp:
1758 case Intrinsic::sadd_with_overflow:
1759 case Intrinsic::uadd_with_overflow:
1760 case Intrinsic::ssub_with_overflow:
1761 case Intrinsic::usub_with_overflow:
1762 case Intrinsic::smul_with_overflow:
1763 case Intrinsic::umul_with_overflow:
1764 case Intrinsic::sadd_sat:
1765 case Intrinsic::uadd_sat:
1766 case Intrinsic::ssub_sat:
1767 case Intrinsic::usub_sat:
1768 case Intrinsic::smul_fix:
1769 case Intrinsic::smul_fix_sat:
1770 case Intrinsic::bitreverse:
1771 case Intrinsic::is_constant:
1772 case Intrinsic::vector_reduce_add:
1773 case Intrinsic::vector_reduce_mul:
1774 case Intrinsic::vector_reduce_and:
1775 case Intrinsic::vector_reduce_or:
1776 case Intrinsic::vector_reduce_xor:
1777 case Intrinsic::vector_reduce_smin:
1778 case Intrinsic::vector_reduce_smax:
1779 case Intrinsic::vector_reduce_umin:
1780 case Intrinsic::vector_reduce_umax:
1781 case Intrinsic::vector_extract:
1782 case Intrinsic::vector_insert:
1783 case Intrinsic::vector_interleave2:
1784 case Intrinsic::vector_interleave3:
1785 case Intrinsic::vector_interleave4:
1786 case Intrinsic::vector_interleave5:
1787 case Intrinsic::vector_interleave6:
1788 case Intrinsic::vector_interleave7:
1789 case Intrinsic::vector_interleave8:
1790 case Intrinsic::vector_deinterleave2:
1791 case Intrinsic::vector_deinterleave3:
1792 case Intrinsic::vector_deinterleave4:
1793 case Intrinsic::vector_deinterleave5:
1794 case Intrinsic::vector_deinterleave6:
1795 case Intrinsic::vector_deinterleave7:
1796 case Intrinsic::vector_deinterleave8:
1797 // Target intrinsics
1798 case Intrinsic::amdgcn_perm:
1799 case Intrinsic::amdgcn_wave_reduce_umin:
1800 case Intrinsic::amdgcn_wave_reduce_umax:
1801 case Intrinsic::amdgcn_wave_reduce_max:
1802 case Intrinsic::amdgcn_wave_reduce_min:
1803 case Intrinsic::amdgcn_wave_reduce_and:
1804 case Intrinsic::amdgcn_wave_reduce_or:
1805 case Intrinsic::amdgcn_s_wqm:
1806 case Intrinsic::amdgcn_s_quadmask:
1807 case Intrinsic::amdgcn_s_bitreplicate:
1808 case Intrinsic::arm_mve_vctp8:
1809 case Intrinsic::arm_mve_vctp16:
1810 case Intrinsic::arm_mve_vctp32:
1811 case Intrinsic::arm_mve_vctp64:
1812 case Intrinsic::aarch64_sve_convert_from_svbool:
1813 case Intrinsic::wasm_alltrue:
1814 case Intrinsic::wasm_anytrue:
1815 case Intrinsic::wasm_dot:
1816 // WebAssembly float semantics are always known
1817 case Intrinsic::wasm_trunc_signed:
1818 case Intrinsic::wasm_trunc_unsigned:
1819 return true;
1820
1821 // Floating point operations cannot be folded in strictfp functions in
1822 // general case. They can be folded if FP environment is known to compiler.
1823 case Intrinsic::minnum:
1824 case Intrinsic::maxnum:
1825 case Intrinsic::minimum:
1826 case Intrinsic::maximum:
1827 case Intrinsic::minimumnum:
1828 case Intrinsic::maximumnum:
1829 case Intrinsic::log:
1830 case Intrinsic::log2:
1831 case Intrinsic::log10:
1832 case Intrinsic::exp:
1833 case Intrinsic::exp2:
1834 case Intrinsic::exp10:
1835 case Intrinsic::sqrt:
1836 case Intrinsic::sin:
1837 case Intrinsic::cos:
1838 case Intrinsic::sincos:
1839 case Intrinsic::sinh:
1840 case Intrinsic::cosh:
1841 case Intrinsic::atan:
1842 case Intrinsic::pow:
1843 case Intrinsic::powi:
1844 case Intrinsic::ldexp:
1845 case Intrinsic::fma:
1846 case Intrinsic::fmuladd:
1847 case Intrinsic::frexp:
1848 case Intrinsic::fptoui_sat:
1849 case Intrinsic::fptosi_sat:
1850 case Intrinsic::amdgcn_cos:
1851 case Intrinsic::amdgcn_cubeid:
1852 case Intrinsic::amdgcn_cubema:
1853 case Intrinsic::amdgcn_cubesc:
1854 case Intrinsic::amdgcn_cubetc:
1855 case Intrinsic::amdgcn_fmul_legacy:
1856 case Intrinsic::amdgcn_fma_legacy:
1857 case Intrinsic::amdgcn_fract:
1858 case Intrinsic::amdgcn_sin:
1859 // The intrinsics below depend on rounding mode in MXCSR.
1860 case Intrinsic::x86_sse_cvtss2si:
1861 case Intrinsic::x86_sse_cvtss2si64:
1862 case Intrinsic::x86_sse_cvttss2si:
1863 case Intrinsic::x86_sse_cvttss2si64:
1864 case Intrinsic::x86_sse2_cvtsd2si:
1865 case Intrinsic::x86_sse2_cvtsd2si64:
1866 case Intrinsic::x86_sse2_cvttsd2si:
1867 case Intrinsic::x86_sse2_cvttsd2si64:
1868 case Intrinsic::x86_avx512_vcvtss2si32:
1869 case Intrinsic::x86_avx512_vcvtss2si64:
1870 case Intrinsic::x86_avx512_cvttss2si:
1871 case Intrinsic::x86_avx512_cvttss2si64:
1872 case Intrinsic::x86_avx512_vcvtsd2si32:
1873 case Intrinsic::x86_avx512_vcvtsd2si64:
1874 case Intrinsic::x86_avx512_cvttsd2si:
1875 case Intrinsic::x86_avx512_cvttsd2si64:
1876 case Intrinsic::x86_avx512_vcvtss2usi32:
1877 case Intrinsic::x86_avx512_vcvtss2usi64:
1878 case Intrinsic::x86_avx512_cvttss2usi:
1879 case Intrinsic::x86_avx512_cvttss2usi64:
1880 case Intrinsic::x86_avx512_vcvtsd2usi32:
1881 case Intrinsic::x86_avx512_vcvtsd2usi64:
1882 case Intrinsic::x86_avx512_cvttsd2usi:
1883 case Intrinsic::x86_avx512_cvttsd2usi64:
1884
1885 // NVVM FMax intrinsics
1886 case Intrinsic::nvvm_fmax_d:
1887 case Intrinsic::nvvm_fmax_f:
1888 case Intrinsic::nvvm_fmax_ftz_f:
1889 case Intrinsic::nvvm_fmax_ftz_nan_f:
1890 case Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_f:
1891 case Intrinsic::nvvm_fmax_ftz_xorsign_abs_f:
1892 case Intrinsic::nvvm_fmax_nan_f:
1893 case Intrinsic::nvvm_fmax_nan_xorsign_abs_f:
1894 case Intrinsic::nvvm_fmax_xorsign_abs_f:
1895
1896 // NVVM FMin intrinsics
1897 case Intrinsic::nvvm_fmin_d:
1898 case Intrinsic::nvvm_fmin_f:
1899 case Intrinsic::nvvm_fmin_ftz_f:
1900 case Intrinsic::nvvm_fmin_ftz_nan_f:
1901 case Intrinsic::nvvm_fmin_ftz_nan_xorsign_abs_f:
1902 case Intrinsic::nvvm_fmin_ftz_xorsign_abs_f:
1903 case Intrinsic::nvvm_fmin_nan_f:
1904 case Intrinsic::nvvm_fmin_nan_xorsign_abs_f:
1905 case Intrinsic::nvvm_fmin_xorsign_abs_f:
1906
1907 // NVVM float/double to int32/uint32 conversion intrinsics
1908 case Intrinsic::nvvm_f2i_rm:
1909 case Intrinsic::nvvm_f2i_rn:
1910 case Intrinsic::nvvm_f2i_rp:
1911 case Intrinsic::nvvm_f2i_rz:
1912 case Intrinsic::nvvm_f2i_rm_ftz:
1913 case Intrinsic::nvvm_f2i_rn_ftz:
1914 case Intrinsic::nvvm_f2i_rp_ftz:
1915 case Intrinsic::nvvm_f2i_rz_ftz:
1916 case Intrinsic::nvvm_f2ui_rm:
1917 case Intrinsic::nvvm_f2ui_rn:
1918 case Intrinsic::nvvm_f2ui_rp:
1919 case Intrinsic::nvvm_f2ui_rz:
1920 case Intrinsic::nvvm_f2ui_rm_ftz:
1921 case Intrinsic::nvvm_f2ui_rn_ftz:
1922 case Intrinsic::nvvm_f2ui_rp_ftz:
1923 case Intrinsic::nvvm_f2ui_rz_ftz:
1924 case Intrinsic::nvvm_d2i_rm:
1925 case Intrinsic::nvvm_d2i_rn:
1926 case Intrinsic::nvvm_d2i_rp:
1927 case Intrinsic::nvvm_d2i_rz:
1928 case Intrinsic::nvvm_d2ui_rm:
1929 case Intrinsic::nvvm_d2ui_rn:
1930 case Intrinsic::nvvm_d2ui_rp:
1931 case Intrinsic::nvvm_d2ui_rz:
1932
1933 // NVVM float/double to int64/uint64 conversion intrinsics
1934 case Intrinsic::nvvm_f2ll_rm:
1935 case Intrinsic::nvvm_f2ll_rn:
1936 case Intrinsic::nvvm_f2ll_rp:
1937 case Intrinsic::nvvm_f2ll_rz:
1938 case Intrinsic::nvvm_f2ll_rm_ftz:
1939 case Intrinsic::nvvm_f2ll_rn_ftz:
1940 case Intrinsic::nvvm_f2ll_rp_ftz:
1941 case Intrinsic::nvvm_f2ll_rz_ftz:
1942 case Intrinsic::nvvm_f2ull_rm:
1943 case Intrinsic::nvvm_f2ull_rn:
1944 case Intrinsic::nvvm_f2ull_rp:
1945 case Intrinsic::nvvm_f2ull_rz:
1946 case Intrinsic::nvvm_f2ull_rm_ftz:
1947 case Intrinsic::nvvm_f2ull_rn_ftz:
1948 case Intrinsic::nvvm_f2ull_rp_ftz:
1949 case Intrinsic::nvvm_f2ull_rz_ftz:
1950 case Intrinsic::nvvm_d2ll_rm:
1951 case Intrinsic::nvvm_d2ll_rn:
1952 case Intrinsic::nvvm_d2ll_rp:
1953 case Intrinsic::nvvm_d2ll_rz:
1954 case Intrinsic::nvvm_d2ull_rm:
1955 case Intrinsic::nvvm_d2ull_rn:
1956 case Intrinsic::nvvm_d2ull_rp:
1957 case Intrinsic::nvvm_d2ull_rz:
1958
1959 // NVVM math intrinsics:
1960 case Intrinsic::nvvm_ceil_d:
1961 case Intrinsic::nvvm_ceil_f:
1962 case Intrinsic::nvvm_ceil_ftz_f:
1963
1964 case Intrinsic::nvvm_fabs:
1965 case Intrinsic::nvvm_fabs_ftz:
1966
1967 case Intrinsic::nvvm_floor_d:
1968 case Intrinsic::nvvm_floor_f:
1969 case Intrinsic::nvvm_floor_ftz_f:
1970
1971 case Intrinsic::nvvm_rcp_rm_d:
1972 case Intrinsic::nvvm_rcp_rm_f:
1973 case Intrinsic::nvvm_rcp_rm_ftz_f:
1974 case Intrinsic::nvvm_rcp_rn_d:
1975 case Intrinsic::nvvm_rcp_rn_f:
1976 case Intrinsic::nvvm_rcp_rn_ftz_f:
1977 case Intrinsic::nvvm_rcp_rp_d:
1978 case Intrinsic::nvvm_rcp_rp_f:
1979 case Intrinsic::nvvm_rcp_rp_ftz_f:
1980 case Intrinsic::nvvm_rcp_rz_d:
1981 case Intrinsic::nvvm_rcp_rz_f:
1982 case Intrinsic::nvvm_rcp_rz_ftz_f:
1983
1984 case Intrinsic::nvvm_round_d:
1985 case Intrinsic::nvvm_round_f:
1986 case Intrinsic::nvvm_round_ftz_f:
1987
1988 case Intrinsic::nvvm_saturate_d:
1989 case Intrinsic::nvvm_saturate_f:
1990 case Intrinsic::nvvm_saturate_ftz_f:
1991
1992 case Intrinsic::nvvm_sqrt_f:
1993 case Intrinsic::nvvm_sqrt_rn_d:
1994 case Intrinsic::nvvm_sqrt_rn_f:
1995 case Intrinsic::nvvm_sqrt_rn_ftz_f:
1996 return !IsStrictFP;
1997
1998 // NVVM add intrinsics with explicit rounding modes
1999 case Intrinsic::nvvm_add_rm_d:
2000 case Intrinsic::nvvm_add_rn_d:
2001 case Intrinsic::nvvm_add_rp_d:
2002 case Intrinsic::nvvm_add_rz_d:
2003 case Intrinsic::nvvm_add_rm_f:
2004 case Intrinsic::nvvm_add_rn_f:
2005 case Intrinsic::nvvm_add_rp_f:
2006 case Intrinsic::nvvm_add_rz_f:
2007 case Intrinsic::nvvm_add_rm_ftz_f:
2008 case Intrinsic::nvvm_add_rn_ftz_f:
2009 case Intrinsic::nvvm_add_rp_ftz_f:
2010 case Intrinsic::nvvm_add_rz_ftz_f:
2011
2012 // NVVM div intrinsics with explicit rounding modes
2013 case Intrinsic::nvvm_div_rm_d:
2014 case Intrinsic::nvvm_div_rn_d:
2015 case Intrinsic::nvvm_div_rp_d:
2016 case Intrinsic::nvvm_div_rz_d:
2017 case Intrinsic::nvvm_div_rm_f:
2018 case Intrinsic::nvvm_div_rn_f:
2019 case Intrinsic::nvvm_div_rp_f:
2020 case Intrinsic::nvvm_div_rz_f:
2021 case Intrinsic::nvvm_div_rm_ftz_f:
2022 case Intrinsic::nvvm_div_rn_ftz_f:
2023 case Intrinsic::nvvm_div_rp_ftz_f:
2024 case Intrinsic::nvvm_div_rz_ftz_f:
2025
2026 // NVVM mul intrinsics with explicit rounding modes
2027 case Intrinsic::nvvm_mul_rm_d:
2028 case Intrinsic::nvvm_mul_rn_d:
2029 case Intrinsic::nvvm_mul_rp_d:
2030 case Intrinsic::nvvm_mul_rz_d:
2031 case Intrinsic::nvvm_mul_rm_f:
2032 case Intrinsic::nvvm_mul_rn_f:
2033 case Intrinsic::nvvm_mul_rp_f:
2034 case Intrinsic::nvvm_mul_rz_f:
2035 case Intrinsic::nvvm_mul_rm_ftz_f:
2036 case Intrinsic::nvvm_mul_rn_ftz_f:
2037 case Intrinsic::nvvm_mul_rp_ftz_f:
2038 case Intrinsic::nvvm_mul_rz_ftz_f:
2039
2040 // NVVM fma intrinsics with explicit rounding modes
2041 case Intrinsic::nvvm_fma_rm_d:
2042 case Intrinsic::nvvm_fma_rn_d:
2043 case Intrinsic::nvvm_fma_rp_d:
2044 case Intrinsic::nvvm_fma_rz_d:
2045 case Intrinsic::nvvm_fma_rm_f:
2046 case Intrinsic::nvvm_fma_rn_f:
2047 case Intrinsic::nvvm_fma_rp_f:
2048 case Intrinsic::nvvm_fma_rz_f:
2049 case Intrinsic::nvvm_fma_rm_ftz_f:
2050 case Intrinsic::nvvm_fma_rn_ftz_f:
2051 case Intrinsic::nvvm_fma_rp_ftz_f:
2052 case Intrinsic::nvvm_fma_rz_ftz_f:
2053
2054 // Sign operations are actually bitwise operations, they do not raise
2055 // exceptions even for SNANs.
2056 case Intrinsic::fabs:
2057 case Intrinsic::copysign:
2058 case Intrinsic::is_fpclass:
2059 // Non-constrained variants of rounding operations means default FP
2060 // environment, they can be folded in any case.
2061 case Intrinsic::ceil:
2062 case Intrinsic::floor:
2063 case Intrinsic::round:
2064 case Intrinsic::roundeven:
2065 case Intrinsic::trunc:
2066 case Intrinsic::nearbyint:
2067 case Intrinsic::rint:
2068 case Intrinsic::canonicalize:
2069
2070 // Constrained intrinsics can be folded if FP environment is known
2071 // to compiler.
2072 case Intrinsic::experimental_constrained_fma:
2073 case Intrinsic::experimental_constrained_fmuladd:
2074 case Intrinsic::experimental_constrained_fadd:
2075 case Intrinsic::experimental_constrained_fsub:
2076 case Intrinsic::experimental_constrained_fmul:
2077 case Intrinsic::experimental_constrained_fdiv:
2078 case Intrinsic::experimental_constrained_frem:
2079 case Intrinsic::experimental_constrained_ceil:
2080 case Intrinsic::experimental_constrained_floor:
2081 case Intrinsic::experimental_constrained_round:
2082 case Intrinsic::experimental_constrained_roundeven:
2083 case Intrinsic::experimental_constrained_trunc:
2084 case Intrinsic::experimental_constrained_nearbyint:
2085 case Intrinsic::experimental_constrained_rint:
2086 case Intrinsic::experimental_constrained_fcmp:
2087 case Intrinsic::experimental_constrained_fcmps:
2088
2089 case Intrinsic::experimental_cttz_elts:
2090 return true;
2091 default:
2092 return false;
2093 }
2094}
2095
2096/// Given a function's return type and its operands, determine if any of them of
2097/// of floating-point type.
2099 return RetTy->isFloatingPointTy() || any_of(Ops, [](Value *V) {
2100 return V->getType()->isFloatingPointTy();
2101 });
2102}
2103
2105 if (Call->isNoBuiltin())
2106 return false;
2107 if (Call->getFunctionType() != F->getFunctionType())
2108 return false;
2109
2110 // Allow FP calls (both libcalls and intrinsics) to avoid being folded.
2111 // This can be useful for GPU targets or in cross-compilation scenarios
2112 // when the exact target FP behaviour is required, and the host compiler's
2113 // behaviour may be slightly different from the device's run-time behaviour.
2116 F->getReturnType(),
2117 ArrayRef<Value *>((Value *const *)(F->arg_begin()), F->arg_size())))
2118 return false;
2119
2120 if (F->getIntrinsicID() != Intrinsic::not_intrinsic)
2121 return canConstantFoldIntrinsic(F->getIntrinsicID(), Call->isStrictFP());
2122
2123 if (!F->hasName() || Call->isStrictFP())
2124 return false;
2125
2126 // In these cases, the check of the length is required. We don't want to
2127 // return true for a name like "cos\0blah" which strcmp would return equal to
2128 // "cos", but has length 8.
2129 StringRef Name = F->getName();
2130 switch (Name[0]) {
2131 default:
2132 return false;
2133 // clang-format off
2134 case 'a':
2135 return Name == "acos" || Name == "acosf" ||
2136 Name == "asin" || Name == "asinf" ||
2137 Name == "atan" || Name == "atanf" ||
2138 Name == "atan2" || Name == "atan2f";
2139 case 'c':
2140 return Name == "ceil" || Name == "ceilf" ||
2141 Name == "cos" || Name == "cosf" ||
2142 Name == "cosh" || Name == "coshf";
2143 case 'e':
2144 return Name == "exp" || Name == "expf" || Name == "exp2" ||
2145 Name == "exp2f" || Name == "erf" || Name == "erff";
2146 case 'f':
2147 return Name == "fabs" || Name == "fabsf" ||
2148 Name == "floor" || Name == "floorf" ||
2149 Name == "fmod" || Name == "fmodf";
2150 case 'i':
2151 return Name == "ilogb" || Name == "ilogbf";
2152 case 'l':
2153 return Name == "log" || Name == "logf" || Name == "logl" ||
2154 Name == "log2" || Name == "log2f" || Name == "log10" ||
2155 Name == "log10f" || Name == "logb" || Name == "logbf" ||
2156 Name == "log1p" || Name == "log1pf";
2157 case 'n':
2158 return Name == "nearbyint" || Name == "nearbyintf" || Name == "nextafter" ||
2159 Name == "nextafterf" || Name == "nexttoward" ||
2160 Name == "nexttowardf";
2161 case 'p':
2162 return Name == "pow" || Name == "powf";
2163 case 'r':
2164 return Name == "remainder" || Name == "remainderf" ||
2165 Name == "rint" || Name == "rintf" ||
2166 Name == "round" || Name == "roundf" ||
2167 Name == "roundeven" || Name == "roundevenf";
2168 case 's':
2169 return Name == "sin" || Name == "sinf" ||
2170 Name == "sinh" || Name == "sinhf" ||
2171 Name == "sqrt" || Name == "sqrtf";
2172 case 't':
2173 return Name == "tan" || Name == "tanf" ||
2174 Name == "tanh" || Name == "tanhf" ||
2175 Name == "trunc" || Name == "truncf";
2176 case '_':
2177 // Check for various function names that get used for the math functions
2178 // when the header files are preprocessed with the macro
2179 // __FINITE_MATH_ONLY__ enabled.
2180 // The '12' here is the length of the shortest name that can match.
2181 // We need to check the size before looking at Name[1] and Name[2]
2182 // so we may as well check a limit that will eliminate mismatches.
2183 if (Name.size() < 12 || Name[1] != '_')
2184 return false;
2185 switch (Name[2]) {
2186 default:
2187 return false;
2188 case 'a':
2189 return Name == "__acos_finite" || Name == "__acosf_finite" ||
2190 Name == "__asin_finite" || Name == "__asinf_finite" ||
2191 Name == "__atan2_finite" || Name == "__atan2f_finite";
2192 case 'c':
2193 return Name == "__cosh_finite" || Name == "__coshf_finite";
2194 case 'e':
2195 return Name == "__exp_finite" || Name == "__expf_finite" ||
2196 Name == "__exp2_finite" || Name == "__exp2f_finite";
2197 case 'l':
2198 return Name == "__log_finite" || Name == "__logf_finite" ||
2199 Name == "__log10_finite" || Name == "__log10f_finite";
2200 case 'p':
2201 return Name == "__pow_finite" || Name == "__powf_finite";
2202 case 's':
2203 return Name == "__sinh_finite" || Name == "__sinhf_finite";
2204 }
2205 // clang-format on
2206 }
2207}
2208
2209namespace {
2210
2211Constant *GetConstantFoldFPValue(double V, Type *Ty) {
2212 if (Ty->isHalfTy() || Ty->isFloatTy()) {
2213 APFloat APF(V);
2214 bool unused;
2215 APF.convert(Ty->getFltSemantics(), APFloat::rmNearestTiesToEven, &unused);
2216 return ConstantFP::get(Ty->getContext(), APF);
2217 }
2218 if (Ty->isDoubleTy())
2219 return ConstantFP::get(Ty->getContext(), APFloat(V));
2220 llvm_unreachable("Can only constant fold half/float/double");
2221}
2222
2223#if defined(HAS_IEE754_FLOAT128) && defined(HAS_LOGF128)
2224Constant *GetConstantFoldFPValue128(float128 V, Type *Ty) {
2225 if (Ty->isFP128Ty())
2226 return ConstantFP::get(Ty, V);
2227 llvm_unreachable("Can only constant fold fp128");
2228}
2229#endif
2230
2231/// Clear the floating-point exception state.
2232inline void llvm_fenv_clearexcept() {
2233#if HAVE_DECL_FE_ALL_EXCEPT
2234 feclearexcept(FE_ALL_EXCEPT);
2235#endif
2236 errno = 0;
2237}
2238
2239/// Test if a floating-point exception was raised.
2240inline bool llvm_fenv_testexcept() {
2241 int errno_val = errno;
2242 if (errno_val == ERANGE || errno_val == EDOM)
2243 return true;
2244#if HAVE_DECL_FE_ALL_EXCEPT && HAVE_DECL_FE_INEXACT
2245 if (fetestexcept(FE_ALL_EXCEPT & ~FE_INEXACT))
2246 return true;
2247#endif
2248 return false;
2249}
2250
2251static APFloat FTZPreserveSign(const APFloat &V) {
2252 if (V.isDenormal())
2253 return APFloat::getZero(V.getSemantics(), V.isNegative());
2254 return V;
2255}
2256
2257static APFloat FlushToPositiveZero(const APFloat &V) {
2258 if (V.isDenormal())
2259 return APFloat::getZero(V.getSemantics(), false);
2260 return V;
2261}
2262
2263static APFloat FlushWithDenormKind(const APFloat &V,
2264 DenormalMode::DenormalModeKind DenormKind) {
2267 switch (DenormKind) {
2269 return V;
2271 return FTZPreserveSign(V);
2273 return FlushToPositiveZero(V);
2274 default:
2275 llvm_unreachable("Invalid denormal mode!");
2276 }
2277}
2278
2279Constant *ConstantFoldFP(double (*NativeFP)(double), const APFloat &V, Type *Ty,
2280 DenormalMode DenormMode = DenormalMode::getIEEE()) {
2281 if (!DenormMode.isValid() ||
2282 DenormMode.Input == DenormalMode::DenormalModeKind::Dynamic ||
2283 DenormMode.Output == DenormalMode::DenormalModeKind::Dynamic)
2284 return nullptr;
2285
2286 llvm_fenv_clearexcept();
2287 auto Input = FlushWithDenormKind(V, DenormMode.Input);
2288 double Result = NativeFP(Input.convertToDouble());
2289 if (llvm_fenv_testexcept()) {
2290 llvm_fenv_clearexcept();
2291 return nullptr;
2292 }
2293
2294 Constant *Output = GetConstantFoldFPValue(Result, Ty);
2295 if (DenormMode.Output == DenormalMode::DenormalModeKind::IEEE)
2296 return Output;
2297 const auto *CFP = static_cast<ConstantFP *>(Output);
2298 const auto Res = FlushWithDenormKind(CFP->getValueAPF(), DenormMode.Output);
2299 return ConstantFP::get(Ty->getContext(), Res);
2300}
2301
2302#if defined(HAS_IEE754_FLOAT128) && defined(HAS_LOGF128)
2303Constant *ConstantFoldFP128(float128 (*NativeFP)(float128), const APFloat &V,
2304 Type *Ty) {
2305 llvm_fenv_clearexcept();
2306 float128 Result = NativeFP(V.convertToQuad());
2307 if (llvm_fenv_testexcept()) {
2308 llvm_fenv_clearexcept();
2309 return nullptr;
2310 }
2311
2312 return GetConstantFoldFPValue128(Result, Ty);
2313}
2314#endif
2315
2316Constant *ConstantFoldBinaryFP(double (*NativeFP)(double, double),
2317 const APFloat &V, const APFloat &W, Type *Ty) {
2318 llvm_fenv_clearexcept();
2319 double Result = NativeFP(V.convertToDouble(), W.convertToDouble());
2320 if (llvm_fenv_testexcept()) {
2321 llvm_fenv_clearexcept();
2322 return nullptr;
2323 }
2324
2325 return GetConstantFoldFPValue(Result, Ty);
2326}
2327
2328Constant *constantFoldVectorReduce(Intrinsic::ID IID, Constant *Op) {
2329 auto *OpVT = cast<VectorType>(Op->getType());
2330
2331 // This is the same as the underlying binops - poison propagates.
2332 if (Op->containsPoisonElement())
2333 return PoisonValue::get(OpVT->getElementType());
2334
2335 // Shortcut non-accumulating reductions.
2336 if (Constant *SplatVal = Op->getSplatValue()) {
2337 switch (IID) {
2338 case Intrinsic::vector_reduce_and:
2339 case Intrinsic::vector_reduce_or:
2340 case Intrinsic::vector_reduce_smin:
2341 case Intrinsic::vector_reduce_smax:
2342 case Intrinsic::vector_reduce_umin:
2343 case Intrinsic::vector_reduce_umax:
2344 return SplatVal;
2345 case Intrinsic::vector_reduce_add:
2346 if (SplatVal->isNullValue())
2347 return SplatVal;
2348 break;
2349 case Intrinsic::vector_reduce_mul:
2350 if (SplatVal->isNullValue() || SplatVal->isOneValue())
2351 return SplatVal;
2352 break;
2353 case Intrinsic::vector_reduce_xor:
2354 if (SplatVal->isNullValue())
2355 return SplatVal;
2356 if (OpVT->getElementCount().isKnownMultipleOf(2))
2357 return Constant::getNullValue(OpVT->getElementType());
2358 break;
2359 }
2360 }
2361
2363 if (!VT)
2364 return nullptr;
2365
2366 auto *EltC = dyn_cast_or_null<ConstantInt>(Op->getAggregateElement(0U));
2367 if (!EltC)
2368 return nullptr;
2369
2370 APInt Acc = EltC->getValue();
2371 for (unsigned I = 1, E = VT->getNumElements(); I != E; I++) {
2372 if (!(EltC = dyn_cast_or_null<ConstantInt>(Op->getAggregateElement(I))))
2373 return nullptr;
2374 const APInt &X = EltC->getValue();
2375 switch (IID) {
2376 case Intrinsic::vector_reduce_add:
2377 Acc = Acc + X;
2378 break;
2379 case Intrinsic::vector_reduce_mul:
2380 Acc = Acc * X;
2381 break;
2382 case Intrinsic::vector_reduce_and:
2383 Acc = Acc & X;
2384 break;
2385 case Intrinsic::vector_reduce_or:
2386 Acc = Acc | X;
2387 break;
2388 case Intrinsic::vector_reduce_xor:
2389 Acc = Acc ^ X;
2390 break;
2391 case Intrinsic::vector_reduce_smin:
2392 Acc = APIntOps::smin(Acc, X);
2393 break;
2394 case Intrinsic::vector_reduce_smax:
2395 Acc = APIntOps::smax(Acc, X);
2396 break;
2397 case Intrinsic::vector_reduce_umin:
2398 Acc = APIntOps::umin(Acc, X);
2399 break;
2400 case Intrinsic::vector_reduce_umax:
2401 Acc = APIntOps::umax(Acc, X);
2402 break;
2403 }
2404 }
2405
2406 return ConstantInt::get(Op->getContext(), Acc);
2407}
2408
2409/// Attempt to fold an SSE floating point to integer conversion of a constant
2410/// floating point. If roundTowardZero is false, the default IEEE rounding is
2411/// used (toward nearest, ties to even). This matches the behavior of the
2412/// non-truncating SSE instructions in the default rounding mode. The desired
2413/// integer type Ty is used to select how many bits are available for the
2414/// result. Returns null if the conversion cannot be performed, otherwise
2415/// returns the Constant value resulting from the conversion.
2416Constant *ConstantFoldSSEConvertToInt(const APFloat &Val, bool roundTowardZero,
2417 Type *Ty, bool IsSigned) {
2418 // All of these conversion intrinsics form an integer of at most 64bits.
2419 unsigned ResultWidth = Ty->getIntegerBitWidth();
2420 assert(ResultWidth <= 64 &&
2421 "Can only constant fold conversions to 64 and 32 bit ints");
2422
2423 uint64_t UIntVal;
2424 bool isExact = false;
2428 Val.convertToInteger(MutableArrayRef(UIntVal), ResultWidth,
2429 IsSigned, mode, &isExact);
2430 if (status != APFloat::opOK &&
2431 (!roundTowardZero || status != APFloat::opInexact))
2432 return nullptr;
2433 return ConstantInt::get(Ty, UIntVal, IsSigned);
2434}
2435
2436double getValueAsDouble(ConstantFP *Op) {
2437 Type *Ty = Op->getType();
2438
2439 if (Ty->isBFloatTy() || Ty->isHalfTy() || Ty->isFloatTy() || Ty->isDoubleTy())
2440 return Op->getValueAPF().convertToDouble();
2441
2442 bool unused;
2443 APFloat APF = Op->getValueAPF();
2445 return APF.convertToDouble();
2446}
2447
2448static bool getConstIntOrUndef(Value *Op, const APInt *&C) {
2449 if (auto *CI = dyn_cast<ConstantInt>(Op)) {
2450 C = &CI->getValue();
2451 return true;
2452 }
2453 if (isa<UndefValue>(Op)) {
2454 C = nullptr;
2455 return true;
2456 }
2457 return false;
2458}
2459
2460/// Checks if the given intrinsic call, which evaluates to constant, is allowed
2461/// to be folded.
2462///
2463/// \param CI Constrained intrinsic call.
2464/// \param St Exception flags raised during constant evaluation.
2465static bool mayFoldConstrained(ConstrainedFPIntrinsic *CI,
2466 APFloat::opStatus St) {
2467 std::optional<RoundingMode> ORM = CI->getRoundingMode();
2468 std::optional<fp::ExceptionBehavior> EB = CI->getExceptionBehavior();
2469
2470 // If the operation does not change exception status flags, it is safe
2471 // to fold.
2472 if (St == APFloat::opStatus::opOK)
2473 return true;
2474
2475 // If evaluation raised FP exception, the result can depend on rounding
2476 // mode. If the latter is unknown, folding is not possible.
2477 if (ORM == RoundingMode::Dynamic)
2478 return false;
2479
2480 // If FP exceptions are ignored, fold the call, even if such exception is
2481 // raised.
2482 if (EB && *EB != fp::ExceptionBehavior::ebStrict)
2483 return true;
2484
2485 // Leave the calculation for runtime so that exception flags be correctly set
2486 // in hardware.
2487 return false;
2488}
2489
2490/// Returns the rounding mode that should be used for constant evaluation.
2491static RoundingMode
2492getEvaluationRoundingMode(const ConstrainedFPIntrinsic *CI) {
2493 std::optional<RoundingMode> ORM = CI->getRoundingMode();
2494 if (!ORM || *ORM == RoundingMode::Dynamic)
2495 // Even if the rounding mode is unknown, try evaluating the operation.
2496 // If it does not raise inexact exception, rounding was not applied,
2497 // so the result is exact and does not depend on rounding mode. Whether
2498 // other FP exceptions are raised, it does not depend on rounding mode.
2500 return *ORM;
2501}
2502
2503/// Try to constant fold llvm.canonicalize for the given caller and value.
2504static Constant *constantFoldCanonicalize(const Type *Ty, const APFloat &Src,
2505 const Function *CtxF = nullptr) {
2506 // Zero, positive and negative, is always OK to fold.
2507 if (Src.isZero()) {
2508 // Get a fresh 0, since ppc_fp128 does have non-canonical zeros.
2509 return ConstantFP::get(
2510 Ty->getContext(),
2511 APFloat::getZero(Src.getSemantics(), Src.isNegative()));
2512 }
2513
2514 if (!Ty->isIEEELikeFPTy())
2515 return nullptr;
2516
2517 // Zero is always canonical and the sign must be preserved.
2518 //
2519 // Denorms and nans may have special encodings, but it should be OK to fold a
2520 // totally average number.
2521 if (Src.isNormal() || Src.isInfinity())
2522 return ConstantFP::get(Ty->getContext(), Src);
2523
2524 if (Src.isDenormal() && CtxF) {
2525 DenormalMode DenormMode = CtxF->getDenormalMode(Src.getSemantics());
2526
2527 if (DenormMode == DenormalMode::getIEEE())
2528 return ConstantFP::get(Ty->getContext(), Src);
2529
2530 if (DenormMode.Input == DenormalMode::Dynamic)
2531 return nullptr;
2532
2533 // If we know if either input or output is flushed, we can fold.
2534 if ((DenormMode.Input == DenormalMode::Dynamic &&
2535 DenormMode.Output == DenormalMode::IEEE) ||
2536 (DenormMode.Input == DenormalMode::IEEE &&
2537 DenormMode.Output == DenormalMode::Dynamic))
2538 return nullptr;
2539
2540 bool IsPositive =
2541 (!Src.isNegative() || DenormMode.Input == DenormalMode::PositiveZero ||
2542 (DenormMode.Output == DenormalMode::PositiveZero &&
2543 DenormMode.Input == DenormalMode::IEEE));
2544
2545 return ConstantFP::get(Ty->getContext(),
2546 APFloat::getZero(Src.getSemantics(), !IsPositive));
2547 }
2548
2549 return nullptr;
2550}
2551
2552static Constant *ConstantFoldScalarCall1(StringRef Name,
2553 Intrinsic::ID IntrinsicID, Type *Ty,
2554 ArrayRef<Constant *> Operands,
2555 const TargetLibraryInfo *TLI = nullptr,
2556 const CallBase *Call = nullptr) {
2557 assert(Operands.size() == 1 && "Wrong number of operands.");
2558
2559 if (IntrinsicID == Intrinsic::is_constant) {
2560 // We know we have a "Constant" argument. But we want to only
2561 // return true for manifest constants, not those that depend on
2562 // constants with unknowable values, e.g. GlobalValue or BlockAddress.
2563 if (Operands[0]->isManifestConstant())
2564 return ConstantInt::getTrue(Ty->getContext());
2565 return nullptr;
2566 }
2567
2568 if (isa<UndefValue>(Operands[0])) {
2569 // cosine(arg) is between -1 and 1. cosine(invalid arg) is NaN.
2570 // ctpop() is between 0 and bitwidth, pick 0 for undef.
2571 // fptoui.sat and fptosi.sat can always fold to zero (for a zero input).
2572 if (IntrinsicID == Intrinsic::cos ||
2573 IntrinsicID == Intrinsic::ctpop ||
2574 IntrinsicID == Intrinsic::fptoui_sat ||
2575 IntrinsicID == Intrinsic::fptosi_sat ||
2576 IntrinsicID == Intrinsic::canonicalize)
2577 return Constant::getNullValue(Ty);
2578 if (IntrinsicID == Intrinsic::bswap ||
2579 IntrinsicID == Intrinsic::bitreverse ||
2580 IntrinsicID == Intrinsic::launder_invariant_group ||
2581 IntrinsicID == Intrinsic::strip_invariant_group)
2582 return Operands[0];
2583 }
2584
2585 if (isa<ConstantPointerNull>(Operands[0])) {
2586 // launder(null) == null == strip(null) iff in addrspace 0
2587 if (IntrinsicID == Intrinsic::launder_invariant_group ||
2588 IntrinsicID == Intrinsic::strip_invariant_group) {
2589 // If instruction is not yet put in a basic block (e.g. when cloning
2590 // a function during inlining), Call's caller may not be available.
2591 // So check Call's BB first before querying Call->getCaller.
2592 const Function *Caller =
2593 Call && Call->getParent() ? Call->getCaller() : nullptr;
2594 if (Caller &&
2596 Caller, Operands[0]->getType()->getPointerAddressSpace())) {
2597 return Operands[0];
2598 }
2599 return nullptr;
2600 }
2601 }
2602
2603 if (auto *Op = dyn_cast<ConstantFP>(Operands[0])) {
2604 APFloat U = Op->getValueAPF();
2605
2606 if (IntrinsicID == Intrinsic::wasm_trunc_signed ||
2607 IntrinsicID == Intrinsic::wasm_trunc_unsigned) {
2608 bool Signed = IntrinsicID == Intrinsic::wasm_trunc_signed;
2609
2610 if (U.isNaN())
2611 return nullptr;
2612
2613 unsigned Width = Ty->getIntegerBitWidth();
2614 APSInt Int(Width, !Signed);
2615 bool IsExact = false;
2617 U.convertToInteger(Int, APFloat::rmTowardZero, &IsExact);
2618
2620 return ConstantInt::get(Ty, Int);
2621
2622 return nullptr;
2623 }
2624
2625 if (IntrinsicID == Intrinsic::fptoui_sat ||
2626 IntrinsicID == Intrinsic::fptosi_sat) {
2627 // convertToInteger() already has the desired saturation semantics.
2628 APSInt Int(Ty->getIntegerBitWidth(),
2629 IntrinsicID == Intrinsic::fptoui_sat);
2630 bool IsExact;
2631 U.convertToInteger(Int, APFloat::rmTowardZero, &IsExact);
2632 return ConstantInt::get(Ty, Int);
2633 }
2634
2635 if (IntrinsicID == Intrinsic::canonicalize) {
2636 const Function *CtxF =
2637 Call && Call->getParent() ? Call->getFunction() : nullptr;
2638 return constantFoldCanonicalize(Ty, U, CtxF);
2639 }
2640
2641#if defined(HAS_IEE754_FLOAT128) && defined(HAS_LOGF128)
2642 if (Ty->isFP128Ty()) {
2643 if (IntrinsicID == Intrinsic::log) {
2644 float128 Result = logf128(Op->getValueAPF().convertToQuad());
2645 return GetConstantFoldFPValue128(Result, Ty);
2646 }
2647
2648 LibFunc Fp128Func = NotLibFunc;
2649 if (TLI && TLI->getLibFunc(Name, Fp128Func) && TLI->has(Fp128Func) &&
2650 Fp128Func == LibFunc_logl)
2651 return ConstantFoldFP128(logf128, Op->getValueAPF(), Ty);
2652 }
2653#endif
2654
2655 if (!Ty->isHalfTy() && !Ty->isFloatTy() && !Ty->isDoubleTy() &&
2656 !Ty->isIntegerTy())
2657 return nullptr;
2658
2659 // Use internal versions of these intrinsics.
2660
2661 if (IntrinsicID == Intrinsic::nearbyint || IntrinsicID == Intrinsic::rint ||
2662 IntrinsicID == Intrinsic::roundeven) {
2663 U.roundToIntegral(APFloat::rmNearestTiesToEven);
2664 return ConstantFP::get(Ty, U);
2665 }
2666
2667 if (IntrinsicID == Intrinsic::round) {
2668 U.roundToIntegral(APFloat::rmNearestTiesToAway);
2669 return ConstantFP::get(Ty, U);
2670 }
2671
2672 if (IntrinsicID == Intrinsic::roundeven) {
2673 U.roundToIntegral(APFloat::rmNearestTiesToEven);
2674 return ConstantFP::get(Ty, U);
2675 }
2676
2677 if (IntrinsicID == Intrinsic::ceil) {
2678 U.roundToIntegral(APFloat::rmTowardPositive);
2679 return ConstantFP::get(Ty, U);
2680 }
2681
2682 if (IntrinsicID == Intrinsic::floor) {
2683 U.roundToIntegral(APFloat::rmTowardNegative);
2684 return ConstantFP::get(Ty, U);
2685 }
2686
2687 if (IntrinsicID == Intrinsic::trunc) {
2688 U.roundToIntegral(APFloat::rmTowardZero);
2689 return ConstantFP::get(Ty, U);
2690 }
2691
2692 if (IntrinsicID == Intrinsic::fabs) {
2693 U.clearSign();
2694 return ConstantFP::get(Ty, U);
2695 }
2696
2697 if (IntrinsicID == Intrinsic::amdgcn_fract) {
2698 // The v_fract instruction behaves like the OpenCL spec, which defines
2699 // fract(x) as fmin(x - floor(x), 0x1.fffffep-1f): "The min() operator is
2700 // there to prevent fract(-small) from returning 1.0. It returns the
2701 // largest positive floating-point number less than 1.0."
2702 APFloat FloorU(U);
2703 FloorU.roundToIntegral(APFloat::rmTowardNegative);
2704 APFloat FractU(U - FloorU);
2705 APFloat AlmostOne(U.getSemantics(), 1);
2706 AlmostOne.next(/*nextDown*/ true);
2707 return ConstantFP::get(Ty, minimum(FractU, AlmostOne));
2708 }
2709
2710 // Rounding operations (floor, trunc, ceil, round and nearbyint) do not
2711 // raise FP exceptions, unless the argument is signaling NaN.
2712
2714 std::optional<APFloat::roundingMode> RM;
2715 switch (IntrinsicID) {
2716 default:
2717 break;
2718 case Intrinsic::experimental_constrained_nearbyint:
2719 case Intrinsic::experimental_constrained_rint: {
2720 RM = CI->getRoundingMode();
2721 if (!RM || *RM == RoundingMode::Dynamic)
2722 return nullptr;
2723 break;
2724 }
2725 case Intrinsic::experimental_constrained_round:
2727 break;
2728 case Intrinsic::experimental_constrained_ceil:
2730 break;
2731 case Intrinsic::experimental_constrained_floor:
2733 break;
2734 case Intrinsic::experimental_constrained_trunc:
2736 break;
2737 }
2738 if (RM) {
2739 if (U.isFinite()) {
2740 APFloat::opStatus St = U.roundToIntegral(*RM);
2741 if (IntrinsicID == Intrinsic::experimental_constrained_rint &&
2742 St == APFloat::opInexact) {
2743 std::optional<fp::ExceptionBehavior> EB =
2745 if (EB == fp::ebStrict)
2746 return nullptr;
2747 }
2748 } else if (U.isSignaling()) {
2749 std::optional<fp::ExceptionBehavior> EB = CI->getExceptionBehavior();
2750 if (EB && *EB != fp::ebIgnore)
2751 return nullptr;
2752 U = APFloat::getQNaN(U.getSemantics());
2753 }
2754 return ConstantFP::get(Ty, U);
2755 }
2756 }
2757
2758 // NVVM float/double to signed/unsigned int32/int64 conversions:
2759 switch (IntrinsicID) {
2760 // f2i
2761 case Intrinsic::nvvm_f2i_rm:
2762 case Intrinsic::nvvm_f2i_rn:
2763 case Intrinsic::nvvm_f2i_rp:
2764 case Intrinsic::nvvm_f2i_rz:
2765 case Intrinsic::nvvm_f2i_rm_ftz:
2766 case Intrinsic::nvvm_f2i_rn_ftz:
2767 case Intrinsic::nvvm_f2i_rp_ftz:
2768 case Intrinsic::nvvm_f2i_rz_ftz:
2769 // f2ui
2770 case Intrinsic::nvvm_f2ui_rm:
2771 case Intrinsic::nvvm_f2ui_rn:
2772 case Intrinsic::nvvm_f2ui_rp:
2773 case Intrinsic::nvvm_f2ui_rz:
2774 case Intrinsic::nvvm_f2ui_rm_ftz:
2775 case Intrinsic::nvvm_f2ui_rn_ftz:
2776 case Intrinsic::nvvm_f2ui_rp_ftz:
2777 case Intrinsic::nvvm_f2ui_rz_ftz:
2778 // d2i
2779 case Intrinsic::nvvm_d2i_rm:
2780 case Intrinsic::nvvm_d2i_rn:
2781 case Intrinsic::nvvm_d2i_rp:
2782 case Intrinsic::nvvm_d2i_rz:
2783 // d2ui
2784 case Intrinsic::nvvm_d2ui_rm:
2785 case Intrinsic::nvvm_d2ui_rn:
2786 case Intrinsic::nvvm_d2ui_rp:
2787 case Intrinsic::nvvm_d2ui_rz:
2788 // f2ll
2789 case Intrinsic::nvvm_f2ll_rm:
2790 case Intrinsic::nvvm_f2ll_rn:
2791 case Intrinsic::nvvm_f2ll_rp:
2792 case Intrinsic::nvvm_f2ll_rz:
2793 case Intrinsic::nvvm_f2ll_rm_ftz:
2794 case Intrinsic::nvvm_f2ll_rn_ftz:
2795 case Intrinsic::nvvm_f2ll_rp_ftz:
2796 case Intrinsic::nvvm_f2ll_rz_ftz:
2797 // f2ull
2798 case Intrinsic::nvvm_f2ull_rm:
2799 case Intrinsic::nvvm_f2ull_rn:
2800 case Intrinsic::nvvm_f2ull_rp:
2801 case Intrinsic::nvvm_f2ull_rz:
2802 case Intrinsic::nvvm_f2ull_rm_ftz:
2803 case Intrinsic::nvvm_f2ull_rn_ftz:
2804 case Intrinsic::nvvm_f2ull_rp_ftz:
2805 case Intrinsic::nvvm_f2ull_rz_ftz:
2806 // d2ll
2807 case Intrinsic::nvvm_d2ll_rm:
2808 case Intrinsic::nvvm_d2ll_rn:
2809 case Intrinsic::nvvm_d2ll_rp:
2810 case Intrinsic::nvvm_d2ll_rz:
2811 // d2ull
2812 case Intrinsic::nvvm_d2ull_rm:
2813 case Intrinsic::nvvm_d2ull_rn:
2814 case Intrinsic::nvvm_d2ull_rp:
2815 case Intrinsic::nvvm_d2ull_rz: {
2816 // In float-to-integer conversion, NaN inputs are converted to 0.
2817 if (U.isNaN()) {
2818 // In float-to-integer conversion, NaN inputs are converted to 0
2819 // when the source and destination bitwidths are both less than 64.
2820 if (nvvm::FPToIntegerIntrinsicNaNZero(IntrinsicID))
2821 return ConstantInt::get(Ty, 0);
2822
2823 // Otherwise, the most significant bit is set.
2824 unsigned BitWidth = Ty->getIntegerBitWidth();
2825 uint64_t Val = 1ULL << (BitWidth - 1);
2826 return ConstantInt::get(Ty, APInt(BitWidth, Val, /*IsSigned=*/false));
2827 }
2828
2829 APFloat::roundingMode RMode =
2831 bool IsFTZ = nvvm::FPToIntegerIntrinsicShouldFTZ(IntrinsicID);
2832 bool IsSigned = nvvm::FPToIntegerIntrinsicResultIsSigned(IntrinsicID);
2833
2834 APSInt ResInt(Ty->getIntegerBitWidth(), !IsSigned);
2835 auto FloatToRound = IsFTZ ? FTZPreserveSign(U) : U;
2836
2837 // Return max/min value for integers if the result is +/-inf or
2838 // is too large to fit in the result's integer bitwidth.
2839 bool IsExact = false;
2840 FloatToRound.convertToInteger(ResInt, RMode, &IsExact);
2841 return ConstantInt::get(Ty, ResInt);
2842 }
2843 }
2844
2845 /// We only fold functions with finite arguments. Folding NaN and inf is
2846 /// likely to be aborted with an exception anyway, and some host libms
2847 /// have known errors raising exceptions.
2848 if (!U.isFinite())
2849 return nullptr;
2850
2851 /// Currently APFloat versions of these functions do not exist, so we use
2852 /// the host native double versions. Float versions are not called
2853 /// directly but for all these it is true (float)(f((double)arg)) ==
2854 /// f(arg). Long double not supported yet.
2855 const APFloat &APF = Op->getValueAPF();
2856
2857 switch (IntrinsicID) {
2858 default: break;
2859 case Intrinsic::log:
2860 if (U.isZero())
2861 return ConstantFP::getInfinity(Ty, true);
2862 if (U.isNegative())
2863 return ConstantFP::getNaN(Ty);
2864 if (U.isOne())
2865 return ConstantFP::getZero(Ty);
2866 return ConstantFoldFP(log, APF, Ty);
2867 case Intrinsic::log2:
2868 if (U.isZero())
2869 return ConstantFP::getInfinity(Ty, true);
2870 if (U.isNegative())
2871 return ConstantFP::getNaN(Ty);
2872 if (U.isOne())
2873 return ConstantFP::getZero(Ty);
2874 // TODO: What about hosts that lack a C99 library?
2875 return ConstantFoldFP(log2, APF, Ty);
2876 case Intrinsic::log10:
2877 if (U.isZero())
2878 return ConstantFP::getInfinity(Ty, true);
2879 if (U.isNegative())
2880 return ConstantFP::getNaN(Ty);
2881 if (U.isOne())
2882 return ConstantFP::getZero(Ty);
2883 // TODO: What about hosts that lack a C99 library?
2884 return ConstantFoldFP(log10, APF, Ty);
2885 case Intrinsic::exp:
2886 return ConstantFoldFP(exp, APF, Ty);
2887 case Intrinsic::exp2:
2888 // Fold exp2(x) as pow(2, x), in case the host lacks a C99 library.
2889 return ConstantFoldBinaryFP(pow, APFloat(2.0), APF, Ty);
2890 case Intrinsic::exp10:
2891 // Fold exp10(x) as pow(10, x), in case the host lacks a C99 library.
2892 return ConstantFoldBinaryFP(pow, APFloat(10.0), APF, Ty);
2893 case Intrinsic::sin:
2894 return ConstantFoldFP(sin, APF, Ty);
2895 case Intrinsic::cos:
2896 return ConstantFoldFP(cos, APF, Ty);
2897 case Intrinsic::sinh:
2898 return ConstantFoldFP(sinh, APF, Ty);
2899 case Intrinsic::cosh:
2900 return ConstantFoldFP(cosh, APF, Ty);
2901 case Intrinsic::atan:
2902 // Implement optional behavior from C's Annex F for +/-0.0.
2903 if (U.isZero())
2904 return ConstantFP::get(Ty, U);
2905 return ConstantFoldFP(atan, APF, Ty);
2906 case Intrinsic::sqrt:
2907 return ConstantFoldFP(sqrt, APF, Ty);
2908
2909 // NVVM Intrinsics:
2910 case Intrinsic::nvvm_ceil_ftz_f:
2911 case Intrinsic::nvvm_ceil_f:
2912 case Intrinsic::nvvm_ceil_d:
2913 return ConstantFoldFP(
2914 ceil, APF, Ty,
2916 nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID)));
2917
2918 case Intrinsic::nvvm_fabs_ftz:
2919 case Intrinsic::nvvm_fabs:
2920 return ConstantFoldFP(
2921 fabs, APF, Ty,
2923 nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID)));
2924
2925 case Intrinsic::nvvm_floor_ftz_f:
2926 case Intrinsic::nvvm_floor_f:
2927 case Intrinsic::nvvm_floor_d:
2928 return ConstantFoldFP(
2929 floor, APF, Ty,
2931 nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID)));
2932
2933 case Intrinsic::nvvm_rcp_rm_ftz_f:
2934 case Intrinsic::nvvm_rcp_rn_ftz_f:
2935 case Intrinsic::nvvm_rcp_rp_ftz_f:
2936 case Intrinsic::nvvm_rcp_rz_ftz_f:
2937 case Intrinsic::nvvm_rcp_rm_d:
2938 case Intrinsic::nvvm_rcp_rm_f:
2939 case Intrinsic::nvvm_rcp_rn_d:
2940 case Intrinsic::nvvm_rcp_rn_f:
2941 case Intrinsic::nvvm_rcp_rp_d:
2942 case Intrinsic::nvvm_rcp_rp_f:
2943 case Intrinsic::nvvm_rcp_rz_d:
2944 case Intrinsic::nvvm_rcp_rz_f: {
2945 APFloat::roundingMode RoundMode = nvvm::GetRCPRoundingMode(IntrinsicID);
2946 bool IsFTZ = nvvm::RCPShouldFTZ(IntrinsicID);
2947
2948 auto Denominator = IsFTZ ? FTZPreserveSign(APF) : APF;
2950 APFloat::opStatus Status = Res.divide(Denominator, RoundMode);
2951
2953 if (IsFTZ)
2954 Res = FTZPreserveSign(Res);
2955 return ConstantFP::get(Ty, Res);
2956 }
2957 return nullptr;
2958 }
2959
2960 case Intrinsic::nvvm_round_ftz_f:
2961 case Intrinsic::nvvm_round_f:
2962 case Intrinsic::nvvm_round_d: {
2963 // nvvm_round is lowered to PTX cvt.rni, which will round to nearest
2964 // integer, choosing even integer if source is equidistant between two
2965 // integers, so the semantics are closer to "rint" rather than "round".
2966 bool IsFTZ = nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID);
2967 auto V = IsFTZ ? FTZPreserveSign(APF) : APF;
2969 return ConstantFP::get(Ty, V);
2970 }
2971
2972 case Intrinsic::nvvm_saturate_ftz_f:
2973 case Intrinsic::nvvm_saturate_d:
2974 case Intrinsic::nvvm_saturate_f: {
2975 bool IsFTZ = nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID);
2976 auto V = IsFTZ ? FTZPreserveSign(APF) : APF;
2977 if (V.isNegative() || V.isZero() || V.isNaN())
2978 return ConstantFP::getZero(Ty);
2980 if (V > One)
2981 return ConstantFP::get(Ty, One);
2982 return ConstantFP::get(Ty, APF);
2983 }
2984
2985 case Intrinsic::nvvm_sqrt_rn_ftz_f:
2986 case Intrinsic::nvvm_sqrt_f:
2987 case Intrinsic::nvvm_sqrt_rn_d:
2988 case Intrinsic::nvvm_sqrt_rn_f:
2989 if (APF.isNegative())
2990 return nullptr;
2991 return ConstantFoldFP(
2992 sqrt, APF, Ty,
2994 nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID)));
2995
2996 // AMDGCN Intrinsics:
2997 case Intrinsic::amdgcn_cos:
2998 case Intrinsic::amdgcn_sin: {
2999 double V = getValueAsDouble(Op);
3000 if (V < -256.0 || V > 256.0)
3001 // The gfx8 and gfx9 architectures handle arguments outside the range
3002 // [-256, 256] differently. This should be a rare case so bail out
3003 // rather than trying to handle the difference.
3004 return nullptr;
3005 bool IsCos = IntrinsicID == Intrinsic::amdgcn_cos;
3006 double V4 = V * 4.0;
3007 if (V4 == floor(V4)) {
3008 // Force exact results for quarter-integer inputs.
3009 const double SinVals[4] = { 0.0, 1.0, 0.0, -1.0 };
3010 V = SinVals[((int)V4 + (IsCos ? 1 : 0)) & 3];
3011 } else {
3012 if (IsCos)
3013 V = cos(V * 2.0 * numbers::pi);
3014 else
3015 V = sin(V * 2.0 * numbers::pi);
3016 }
3017 return GetConstantFoldFPValue(V, Ty);
3018 }
3019 }
3020
3021 if (!TLI)
3022 return nullptr;
3023
3024 LibFunc Func = NotLibFunc;
3025 if (!TLI->getLibFunc(Name, Func))
3026 return nullptr;
3027
3028 switch (Func) {
3029 default:
3030 break;
3031 case LibFunc_acos:
3032 case LibFunc_acosf:
3033 case LibFunc_acos_finite:
3034 case LibFunc_acosf_finite:
3035 if (TLI->has(Func))
3036 return ConstantFoldFP(acos, APF, Ty);
3037 break;
3038 case LibFunc_asin:
3039 case LibFunc_asinf:
3040 case LibFunc_asin_finite:
3041 case LibFunc_asinf_finite:
3042 if (TLI->has(Func))
3043 return ConstantFoldFP(asin, APF, Ty);
3044 break;
3045 case LibFunc_atan:
3046 case LibFunc_atanf:
3047 // Implement optional behavior from C's Annex F for +/-0.0.
3048 if (U.isZero())
3049 return ConstantFP::get(Ty, U);
3050 if (TLI->has(Func))
3051 return ConstantFoldFP(atan, APF, Ty);
3052 break;
3053 case LibFunc_ceil:
3054 case LibFunc_ceilf:
3055 if (TLI->has(Func)) {
3056 U.roundToIntegral(APFloat::rmTowardPositive);
3057 return ConstantFP::get(Ty, U);
3058 }
3059 break;
3060 case LibFunc_cos:
3061 case LibFunc_cosf:
3062 if (TLI->has(Func))
3063 return ConstantFoldFP(cos, APF, Ty);
3064 break;
3065 case LibFunc_cosh:
3066 case LibFunc_coshf:
3067 case LibFunc_cosh_finite:
3068 case LibFunc_coshf_finite:
3069 if (TLI->has(Func))
3070 return ConstantFoldFP(cosh, APF, Ty);
3071 break;
3072 case LibFunc_exp:
3073 case LibFunc_expf:
3074 case LibFunc_exp_finite:
3075 case LibFunc_expf_finite:
3076 if (TLI->has(Func))
3077 return ConstantFoldFP(exp, APF, Ty);
3078 break;
3079 case LibFunc_exp2:
3080 case LibFunc_exp2f:
3081 case LibFunc_exp2_finite:
3082 case LibFunc_exp2f_finite:
3083 if (TLI->has(Func))
3084 // Fold exp2(x) as pow(2, x), in case the host lacks a C99 library.
3085 return ConstantFoldBinaryFP(pow, APFloat(2.0), APF, Ty);
3086 break;
3087 case LibFunc_fabs:
3088 case LibFunc_fabsf:
3089 if (TLI->has(Func)) {
3090 U.clearSign();
3091 return ConstantFP::get(Ty, U);
3092 }
3093 break;
3094 case LibFunc_floor:
3095 case LibFunc_floorf:
3096 if (TLI->has(Func)) {
3097 U.roundToIntegral(APFloat::rmTowardNegative);
3098 return ConstantFP::get(Ty, U);
3099 }
3100 break;
3101 case LibFunc_log:
3102 case LibFunc_logf:
3103 case LibFunc_log_finite:
3104 case LibFunc_logf_finite:
3105 if (!APF.isNegative() && !APF.isZero() && TLI->has(Func))
3106 return ConstantFoldFP(log, APF, Ty);
3107 break;
3108 case LibFunc_log2:
3109 case LibFunc_log2f:
3110 case LibFunc_log2_finite:
3111 case LibFunc_log2f_finite:
3112 if (!APF.isNegative() && !APF.isZero() && TLI->has(Func))
3113 // TODO: What about hosts that lack a C99 library?
3114 return ConstantFoldFP(log2, APF, Ty);
3115 break;
3116 case LibFunc_log10:
3117 case LibFunc_log10f:
3118 case LibFunc_log10_finite:
3119 case LibFunc_log10f_finite:
3120 if (!APF.isNegative() && !APF.isZero() && TLI->has(Func))
3121 // TODO: What about hosts that lack a C99 library?
3122 return ConstantFoldFP(log10, APF, Ty);
3123 break;
3124 case LibFunc_ilogb:
3125 case LibFunc_ilogbf:
3126 if (!APF.isZero() && TLI->has(Func))
3127 return ConstantInt::get(Ty, ilogb(APF), true);
3128 break;
3129 case LibFunc_logb:
3130 case LibFunc_logbf:
3131 if (!APF.isZero() && TLI->has(Func))
3132 return ConstantFoldFP(logb, APF, Ty);
3133 break;
3134 case LibFunc_log1p:
3135 case LibFunc_log1pf:
3136 // Implement optional behavior from C's Annex F for +/-0.0.
3137 if (U.isZero())
3138 return ConstantFP::get(Ty, U);
3139 if (APF > APFloat::getOne(APF.getSemantics(), true) && TLI->has(Func))
3140 return ConstantFoldFP(log1p, APF, Ty);
3141 break;
3142 case LibFunc_logl:
3143 return nullptr;
3144 case LibFunc_erf:
3145 case LibFunc_erff:
3146 if (TLI->has(Func))
3147 return ConstantFoldFP(erf, APF, Ty);
3148 break;
3149 case LibFunc_nearbyint:
3150 case LibFunc_nearbyintf:
3151 case LibFunc_rint:
3152 case LibFunc_rintf:
3153 case LibFunc_roundeven:
3154 case LibFunc_roundevenf:
3155 if (TLI->has(Func)) {
3156 U.roundToIntegral(APFloat::rmNearestTiesToEven);
3157 return ConstantFP::get(Ty, U);
3158 }
3159 break;
3160 case LibFunc_round:
3161 case LibFunc_roundf:
3162 if (TLI->has(Func)) {
3163 U.roundToIntegral(APFloat::rmNearestTiesToAway);
3164 return ConstantFP::get(Ty, U);
3165 }
3166 break;
3167 case LibFunc_sin:
3168 case LibFunc_sinf:
3169 if (TLI->has(Func))
3170 return ConstantFoldFP(sin, APF, Ty);
3171 break;
3172 case LibFunc_sinh:
3173 case LibFunc_sinhf:
3174 case LibFunc_sinh_finite:
3175 case LibFunc_sinhf_finite:
3176 if (TLI->has(Func))
3177 return ConstantFoldFP(sinh, APF, Ty);
3178 break;
3179 case LibFunc_sqrt:
3180 case LibFunc_sqrtf:
3181 if (!APF.isNegative() && TLI->has(Func))
3182 return ConstantFoldFP(sqrt, APF, Ty);
3183 break;
3184 case LibFunc_tan:
3185 case LibFunc_tanf:
3186 if (TLI->has(Func))
3187 return ConstantFoldFP(tan, APF, Ty);
3188 break;
3189 case LibFunc_tanh:
3190 case LibFunc_tanhf:
3191 if (TLI->has(Func))
3192 return ConstantFoldFP(tanh, APF, Ty);
3193 break;
3194 case LibFunc_trunc:
3195 case LibFunc_truncf:
3196 if (TLI->has(Func)) {
3197 U.roundToIntegral(APFloat::rmTowardZero);
3198 return ConstantFP::get(Ty, U);
3199 }
3200 break;
3201 }
3202 return nullptr;
3203 }
3204
3205 if (auto *Op = dyn_cast<ConstantInt>(Operands[0])) {
3206 switch (IntrinsicID) {
3207 case Intrinsic::bswap:
3208 return ConstantInt::get(Ty->getContext(), Op->getValue().byteSwap());
3209 case Intrinsic::ctpop:
3210 return ConstantInt::get(Ty, Op->getValue().popcount());
3211 case Intrinsic::bitreverse:
3212 return ConstantInt::get(Ty->getContext(), Op->getValue().reverseBits());
3213 case Intrinsic::amdgcn_s_wqm: {
3214 uint64_t Val = Op->getZExtValue();
3215 Val |= (Val & 0x5555555555555555ULL) << 1 |
3216 ((Val >> 1) & 0x5555555555555555ULL);
3217 Val |= (Val & 0x3333333333333333ULL) << 2 |
3218 ((Val >> 2) & 0x3333333333333333ULL);
3219 return ConstantInt::get(Ty, Val);
3220 }
3221
3222 case Intrinsic::amdgcn_s_quadmask: {
3223 uint64_t Val = Op->getZExtValue();
3224 uint64_t QuadMask = 0;
3225 for (unsigned I = 0; I < Op->getBitWidth() / 4; ++I, Val >>= 4) {
3226 if (!(Val & 0xF))
3227 continue;
3228
3229 QuadMask |= (1ULL << I);
3230 }
3231 return ConstantInt::get(Ty, QuadMask);
3232 }
3233
3234 case Intrinsic::amdgcn_s_bitreplicate: {
3235 uint64_t Val = Op->getZExtValue();
3236 Val = (Val & 0x000000000000FFFFULL) | (Val & 0x00000000FFFF0000ULL) << 16;
3237 Val = (Val & 0x000000FF000000FFULL) | (Val & 0x0000FF000000FF00ULL) << 8;
3238 Val = (Val & 0x000F000F000F000FULL) | (Val & 0x00F000F000F000F0ULL) << 4;
3239 Val = (Val & 0x0303030303030303ULL) | (Val & 0x0C0C0C0C0C0C0C0CULL) << 2;
3240 Val = (Val & 0x1111111111111111ULL) | (Val & 0x2222222222222222ULL) << 1;
3241 Val = Val | Val << 1;
3242 return ConstantInt::get(Ty, Val);
3243 }
3244 }
3245 }
3246
3247 if (Operands[0]->getType()->isVectorTy()) {
3248 auto *Op = cast<Constant>(Operands[0]);
3249 switch (IntrinsicID) {
3250 default: break;
3251 case Intrinsic::vector_reduce_add:
3252 case Intrinsic::vector_reduce_mul:
3253 case Intrinsic::vector_reduce_and:
3254 case Intrinsic::vector_reduce_or:
3255 case Intrinsic::vector_reduce_xor:
3256 case Intrinsic::vector_reduce_smin:
3257 case Intrinsic::vector_reduce_smax:
3258 case Intrinsic::vector_reduce_umin:
3259 case Intrinsic::vector_reduce_umax:
3260 if (Constant *C = constantFoldVectorReduce(IntrinsicID, Operands[0]))
3261 return C;
3262 break;
3263 case Intrinsic::x86_sse_cvtss2si:
3264 case Intrinsic::x86_sse_cvtss2si64:
3265 case Intrinsic::x86_sse2_cvtsd2si:
3266 case Intrinsic::x86_sse2_cvtsd2si64:
3267 if (ConstantFP *FPOp =
3268 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
3269 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
3270 /*roundTowardZero=*/false, Ty,
3271 /*IsSigned*/true);
3272 break;
3273 case Intrinsic::x86_sse_cvttss2si:
3274 case Intrinsic::x86_sse_cvttss2si64:
3275 case Intrinsic::x86_sse2_cvttsd2si:
3276 case Intrinsic::x86_sse2_cvttsd2si64:
3277 if (ConstantFP *FPOp =
3278 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
3279 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
3280 /*roundTowardZero=*/true, Ty,
3281 /*IsSigned*/true);
3282 break;
3283
3284 case Intrinsic::wasm_anytrue:
3285 return Op->isNullValue() ? ConstantInt::get(Ty, 0)
3286 : ConstantInt::get(Ty, 1);
3287
3288 case Intrinsic::wasm_alltrue:
3289 // Check each element individually
3290 unsigned E = cast<FixedVectorType>(Op->getType())->getNumElements();
3291 for (unsigned I = 0; I != E; ++I) {
3292 Constant *Elt = Op->getAggregateElement(I);
3293 // Return false as soon as we find a non-true element.
3294 if (Elt && Elt->isNullValue())
3295 return ConstantInt::get(Ty, 0);
3296 // Bail as soon as we find an element we cannot prove to be true.
3297 if (!Elt || !isa<ConstantInt>(Elt))
3298 return nullptr;
3299 }
3300
3301 return ConstantInt::get(Ty, 1);
3302 }
3303 }
3304
3305 return nullptr;
3306}
3307
3308static Constant *evaluateCompare(const APFloat &Op1, const APFloat &Op2,
3312 FCmpInst::Predicate Cond = FCmp->getPredicate();
3313 if (FCmp->isSignaling()) {
3314 if (Op1.isNaN() || Op2.isNaN())
3316 } else {
3317 if (Op1.isSignaling() || Op2.isSignaling())
3319 }
3320 bool Result = FCmpInst::compare(Op1, Op2, Cond);
3321 if (mayFoldConstrained(const_cast<ConstrainedFPCmpIntrinsic *>(FCmp), St))
3322 return ConstantInt::get(Call->getType()->getScalarType(), Result);
3323 return nullptr;
3324}
3325
3326static Constant *ConstantFoldNextToward(const APFloat &Op0, const APFloat &Op1,
3327 const Type *RetTy) {
3328 assert(RetTy != nullptr);
3329 bool LosesInfo;
3330
3331 if (Op1.isSignaling())
3332 return nullptr;
3333 if (Op1.isNaN()) {
3334 APFloat Ret(Op1);
3335 Ret.convert(RetTy->getFltSemantics(), detail::rmNearestTiesToEven,
3336 &LosesInfo);
3337 return ConstantFP::get(RetTy->getContext(), Ret);
3338 }
3339
3340 // Recall that the second argument of nexttoward is always a long double,
3341 // so we may need to promote the first argument for comparisons to be valid.
3342 APFloat PromotedOp0(Op0);
3343 PromotedOp0.convert(Op1.getSemantics(), detail::rmNearestTiesToEven,
3344 &LosesInfo);
3345 assert(!LosesInfo && "Unexpected lossy promotion");
3346 const APFloat::cmpResult Result = PromotedOp0.compare(Op1);
3347
3348 // When equal, the standard says we must return the second argument.
3349 // This allows nice behavior such as nexttoward(0.0, -0.0) = -0.0 and
3350 // nexttoward(-0.0, 0.0) = 0.0
3351 if (Result == detail::cmpEqual) {
3352 APFloat Ret(Op1);
3353 Ret.convert(RetTy->getFltSemantics(), detail::rmNearestTiesToEven,
3354 &LosesInfo);
3355 return ConstantFP::get(RetTy->getContext(), Ret);
3356 }
3357
3358 APFloat Next(Op0);
3359 Next.next(/*nextDown=*/Result == APFloat::cmpGreaterThan);
3360 if (Next.isZero() || Next.isDenormal() || Next.isSignaling())
3361 return nullptr;
3362 return ConstantFP::get(RetTy->getContext(), Next);
3363}
3364
3365static Constant *ConstantFoldLibCall2(StringRef Name, Type *Ty,
3366 ArrayRef<Constant *> Operands,
3367 const TargetLibraryInfo *TLI = nullptr) {
3368 if (!TLI)
3369 return nullptr;
3370
3371 LibFunc Func = NotLibFunc;
3372 if (!TLI->getLibFunc(Name, Func))
3373 return nullptr;
3374
3375 const auto *Op1 = dyn_cast<ConstantFP>(Operands[0]);
3376 if (!Op1)
3377 return nullptr;
3378
3379 const auto *Op2 = dyn_cast<ConstantFP>(Operands[1]);
3380 if (!Op2)
3381 return nullptr;
3382
3383 const APFloat &Op1V = Op1->getValueAPF();
3384 const APFloat &Op2V = Op2->getValueAPF();
3385
3386 switch (Func) {
3387 default:
3388 break;
3389 case LibFunc_pow:
3390 case LibFunc_powf:
3391 case LibFunc_pow_finite:
3392 case LibFunc_powf_finite:
3393 if (TLI->has(Func))
3394 return ConstantFoldBinaryFP(pow, Op1V, Op2V, Ty);
3395 break;
3396 case LibFunc_fmod:
3397 case LibFunc_fmodf:
3398 if (TLI->has(Func)) {
3399 APFloat V = Op1->getValueAPF();
3400 if (APFloat::opStatus::opOK == V.mod(Op2->getValueAPF()))
3401 return ConstantFP::get(Ty, V);
3402 }
3403 break;
3404 case LibFunc_remainder:
3405 case LibFunc_remainderf:
3406 if (TLI->has(Func)) {
3407 APFloat V = Op1->getValueAPF();
3408 if (APFloat::opStatus::opOK == V.remainder(Op2->getValueAPF()))
3409 return ConstantFP::get(Ty, V);
3410 }
3411 break;
3412 case LibFunc_atan2:
3413 case LibFunc_atan2f:
3414 // atan2(+/-0.0, +/-0.0) is known to raise an exception on some libm
3415 // (Solaris), so we do not assume a known result for that.
3416 if (Op1V.isZero() && Op2V.isZero())
3417 return nullptr;
3418 [[fallthrough]];
3419 case LibFunc_atan2_finite:
3420 case LibFunc_atan2f_finite:
3421 if (TLI->has(Func))
3422 return ConstantFoldBinaryFP(atan2, Op1V, Op2V, Ty);
3423 break;
3424 case LibFunc_nextafter:
3425 case LibFunc_nextafterf:
3426 case LibFunc_nexttoward:
3427 case LibFunc_nexttowardf:
3428 if (TLI->has(Func))
3429 return ConstantFoldNextToward(Op1V, Op2V, Ty);
3430 break;
3431 }
3432
3433 return nullptr;
3434}
3435
3436static Constant *ConstantFoldIntrinsicCall2(Intrinsic::ID IntrinsicID, Type *Ty,
3437 ArrayRef<Constant *> Operands,
3438 const CallBase *Call = nullptr) {
3439 assert(Operands.size() == 2 && "Wrong number of operands.");
3440
3441 if (Ty->isFloatingPointTy()) {
3442 // TODO: We should have undef handling for all of the FP intrinsics that
3443 // are attempted to be folded in this function.
3444 bool IsOp0Undef = isa<UndefValue>(Operands[0]);
3445 bool IsOp1Undef = isa<UndefValue>(Operands[1]);
3446 switch (IntrinsicID) {
3447 case Intrinsic::maxnum:
3448 case Intrinsic::minnum:
3449 case Intrinsic::maximum:
3450 case Intrinsic::minimum:
3451 case Intrinsic::maximumnum:
3452 case Intrinsic::minimumnum:
3453 case Intrinsic::nvvm_fmax_d:
3454 case Intrinsic::nvvm_fmin_d:
3455 // If one argument is undef, return the other argument.
3456 if (IsOp0Undef)
3457 return Operands[1];
3458 if (IsOp1Undef)
3459 return Operands[0];
3460 break;
3461
3462 case Intrinsic::nvvm_fmax_f:
3463 case Intrinsic::nvvm_fmax_ftz_f:
3464 case Intrinsic::nvvm_fmax_ftz_nan_f:
3465 case Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_f:
3466 case Intrinsic::nvvm_fmax_ftz_xorsign_abs_f:
3467 case Intrinsic::nvvm_fmax_nan_f:
3468 case Intrinsic::nvvm_fmax_nan_xorsign_abs_f:
3469 case Intrinsic::nvvm_fmax_xorsign_abs_f:
3470
3471 case Intrinsic::nvvm_fmin_f:
3472 case Intrinsic::nvvm_fmin_ftz_f:
3473 case Intrinsic::nvvm_fmin_ftz_nan_f:
3474 case Intrinsic::nvvm_fmin_ftz_nan_xorsign_abs_f:
3475 case Intrinsic::nvvm_fmin_ftz_xorsign_abs_f:
3476 case Intrinsic::nvvm_fmin_nan_f:
3477 case Intrinsic::nvvm_fmin_nan_xorsign_abs_f:
3478 case Intrinsic::nvvm_fmin_xorsign_abs_f:
3479 // If one arg is undef, the other arg can be returned only if it is
3480 // constant, as we may need to flush it to sign-preserving zero or
3481 // canonicalize the NaN.
3482 if (!IsOp0Undef && !IsOp1Undef)
3483 break;
3484 if (auto *Op = dyn_cast<ConstantFP>(Operands[IsOp0Undef ? 1 : 0])) {
3485 if (Op->isNaN()) {
3486 APInt NVCanonicalNaN(32, 0x7fffffff);
3487 return ConstantFP::get(
3488 Ty, APFloat(Ty->getFltSemantics(), NVCanonicalNaN));
3489 }
3490 if (nvvm::FMinFMaxShouldFTZ(IntrinsicID))
3491 return ConstantFP::get(Ty, FTZPreserveSign(Op->getValueAPF()));
3492 else
3493 return Op;
3494 }
3495 break;
3496 }
3497 }
3498
3499 if (const auto *Op1 = dyn_cast<ConstantFP>(Operands[0])) {
3500 const APFloat &Op1V = Op1->getValueAPF();
3501
3502 if (const auto *Op2 = dyn_cast<ConstantFP>(Operands[1])) {
3503 if (Op2->getType() != Op1->getType())
3504 return nullptr;
3505 const APFloat &Op2V = Op2->getValueAPF();
3506
3507 if (const auto *ConstrIntr =
3509 RoundingMode RM = getEvaluationRoundingMode(ConstrIntr);
3510 APFloat Res = Op1V;
3512 switch (IntrinsicID) {
3513 default:
3514 return nullptr;
3515 case Intrinsic::experimental_constrained_fadd:
3516 St = Res.add(Op2V, RM);
3517 break;
3518 case Intrinsic::experimental_constrained_fsub:
3519 St = Res.subtract(Op2V, RM);
3520 break;
3521 case Intrinsic::experimental_constrained_fmul:
3522 St = Res.multiply(Op2V, RM);
3523 break;
3524 case Intrinsic::experimental_constrained_fdiv:
3525 St = Res.divide(Op2V, RM);
3526 break;
3527 case Intrinsic::experimental_constrained_frem:
3528 St = Res.mod(Op2V);
3529 break;
3530 case Intrinsic::experimental_constrained_fcmp:
3531 case Intrinsic::experimental_constrained_fcmps:
3532 return evaluateCompare(Op1V, Op2V, ConstrIntr);
3533 }
3534 if (mayFoldConstrained(const_cast<ConstrainedFPIntrinsic *>(ConstrIntr),
3535 St))
3536 return ConstantFP::get(Ty, Res);
3537 return nullptr;
3538 }
3539
3540 switch (IntrinsicID) {
3541 default:
3542 break;
3543 case Intrinsic::copysign:
3544 return ConstantFP::get(Ty, APFloat::copySign(Op1V, Op2V));
3545 case Intrinsic::minnum:
3546 return ConstantFP::get(Ty, minnum(Op1V, Op2V));
3547 case Intrinsic::maxnum:
3548 return ConstantFP::get(Ty, maxnum(Op1V, Op2V));
3549 case Intrinsic::minimum:
3550 return ConstantFP::get(Ty, minimum(Op1V, Op2V));
3551 case Intrinsic::maximum:
3552 return ConstantFP::get(Ty, maximum(Op1V, Op2V));
3553 case Intrinsic::minimumnum:
3554 return ConstantFP::get(Ty, minimumnum(Op1V, Op2V));
3555 case Intrinsic::maximumnum:
3556 return ConstantFP::get(Ty, maximumnum(Op1V, Op2V));
3557
3558 case Intrinsic::nvvm_fmax_d:
3559 case Intrinsic::nvvm_fmax_f:
3560 case Intrinsic::nvvm_fmax_ftz_f:
3561 case Intrinsic::nvvm_fmax_ftz_nan_f:
3562 case Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_f:
3563 case Intrinsic::nvvm_fmax_ftz_xorsign_abs_f:
3564 case Intrinsic::nvvm_fmax_nan_f:
3565 case Intrinsic::nvvm_fmax_nan_xorsign_abs_f:
3566 case Intrinsic::nvvm_fmax_xorsign_abs_f:
3567
3568 case Intrinsic::nvvm_fmin_d:
3569 case Intrinsic::nvvm_fmin_f:
3570 case Intrinsic::nvvm_fmin_ftz_f:
3571 case Intrinsic::nvvm_fmin_ftz_nan_f:
3572 case Intrinsic::nvvm_fmin_ftz_nan_xorsign_abs_f:
3573 case Intrinsic::nvvm_fmin_ftz_xorsign_abs_f:
3574 case Intrinsic::nvvm_fmin_nan_f:
3575 case Intrinsic::nvvm_fmin_nan_xorsign_abs_f:
3576 case Intrinsic::nvvm_fmin_xorsign_abs_f: {
3577
3578 bool ShouldCanonicalizeNaNs = !(IntrinsicID == Intrinsic::nvvm_fmax_d ||
3579 IntrinsicID == Intrinsic::nvvm_fmin_d);
3580 bool IsFTZ = nvvm::FMinFMaxShouldFTZ(IntrinsicID);
3581 bool IsNaNPropagating = nvvm::FMinFMaxPropagatesNaNs(IntrinsicID);
3582 bool IsXorSignAbs = nvvm::FMinFMaxIsXorSignAbs(IntrinsicID);
3583
3584 APFloat A = IsFTZ ? FTZPreserveSign(Op1V) : Op1V;
3585 APFloat B = IsFTZ ? FTZPreserveSign(Op2V) : Op2V;
3586
3587 bool XorSign = false;
3588 if (IsXorSignAbs) {
3589 XorSign = A.isNegative() ^ B.isNegative();
3590 A = abs(A);
3591 B = abs(B);
3592 }
3593
3594 bool IsFMax = false;
3595 switch (IntrinsicID) {
3596 case Intrinsic::nvvm_fmax_d:
3597 case Intrinsic::nvvm_fmax_f:
3598 case Intrinsic::nvvm_fmax_ftz_f:
3599 case Intrinsic::nvvm_fmax_ftz_nan_f:
3600 case Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_f:
3601 case Intrinsic::nvvm_fmax_ftz_xorsign_abs_f:
3602 case Intrinsic::nvvm_fmax_nan_f:
3603 case Intrinsic::nvvm_fmax_nan_xorsign_abs_f:
3604 case Intrinsic::nvvm_fmax_xorsign_abs_f:
3605 IsFMax = true;
3606 break;
3607 }
3608 APFloat Res =
3609 IsFMax ? (IsNaNPropagating ? maximum(A, B) : maximumnum(A, B))
3610 : (IsNaNPropagating ? minimum(A, B) : minimumnum(A, B));
3611
3612 if (ShouldCanonicalizeNaNs && Res.isNaN()) {
3613 APFloat NVCanonicalNaN(Res.getSemantics(), APInt(32, 0x7fffffff));
3614 return ConstantFP::get(Ty, NVCanonicalNaN);
3615 }
3616
3617 if (IsXorSignAbs && XorSign != Res.isNegative())
3618 Res.changeSign();
3619
3620 return ConstantFP::get(Ty, Res);
3621 }
3622
3623 case Intrinsic::nvvm_add_rm_f:
3624 case Intrinsic::nvvm_add_rn_f:
3625 case Intrinsic::nvvm_add_rp_f:
3626 case Intrinsic::nvvm_add_rz_f:
3627 case Intrinsic::nvvm_add_rm_d:
3628 case Intrinsic::nvvm_add_rn_d:
3629 case Intrinsic::nvvm_add_rp_d:
3630 case Intrinsic::nvvm_add_rz_d:
3631 case Intrinsic::nvvm_add_rm_ftz_f:
3632 case Intrinsic::nvvm_add_rn_ftz_f:
3633 case Intrinsic::nvvm_add_rp_ftz_f:
3634 case Intrinsic::nvvm_add_rz_ftz_f: {
3635
3636 bool IsFTZ = nvvm::FAddShouldFTZ(IntrinsicID);
3637 APFloat A = IsFTZ ? FTZPreserveSign(Op1V) : Op1V;
3638 APFloat B = IsFTZ ? FTZPreserveSign(Op2V) : Op2V;
3639
3640 APFloat::roundingMode RoundMode =
3641 nvvm::GetFAddRoundingMode(IntrinsicID);
3642
3643 APFloat Res = A;
3644 APFloat::opStatus Status = Res.add(B, RoundMode);
3645
3646 if (!Res.isNaN() &&
3648 Res = IsFTZ ? FTZPreserveSign(Res) : Res;
3649 return ConstantFP::get(Ty, Res);
3650 }
3651 return nullptr;
3652 }
3653
3654 case Intrinsic::nvvm_mul_rm_f:
3655 case Intrinsic::nvvm_mul_rn_f:
3656 case Intrinsic::nvvm_mul_rp_f:
3657 case Intrinsic::nvvm_mul_rz_f:
3658 case Intrinsic::nvvm_mul_rm_d:
3659 case Intrinsic::nvvm_mul_rn_d:
3660 case Intrinsic::nvvm_mul_rp_d:
3661 case Intrinsic::nvvm_mul_rz_d:
3662 case Intrinsic::nvvm_mul_rm_ftz_f:
3663 case Intrinsic::nvvm_mul_rn_ftz_f:
3664 case Intrinsic::nvvm_mul_rp_ftz_f:
3665 case Intrinsic::nvvm_mul_rz_ftz_f: {
3666
3667 bool IsFTZ = nvvm::FMulShouldFTZ(IntrinsicID);
3668 APFloat A = IsFTZ ? FTZPreserveSign(Op1V) : Op1V;
3669 APFloat B = IsFTZ ? FTZPreserveSign(Op2V) : Op2V;
3670
3671 APFloat::roundingMode RoundMode =
3672 nvvm::GetFMulRoundingMode(IntrinsicID);
3673
3674 APFloat Res = A;
3675 APFloat::opStatus Status = Res.multiply(B, RoundMode);
3676
3677 if (!Res.isNaN() &&
3679 Res = IsFTZ ? FTZPreserveSign(Res) : Res;
3680 return ConstantFP::get(Ty, Res);
3681 }
3682 return nullptr;
3683 }
3684
3685 case Intrinsic::nvvm_div_rm_f:
3686 case Intrinsic::nvvm_div_rn_f:
3687 case Intrinsic::nvvm_div_rp_f:
3688 case Intrinsic::nvvm_div_rz_f:
3689 case Intrinsic::nvvm_div_rm_d:
3690 case Intrinsic::nvvm_div_rn_d:
3691 case Intrinsic::nvvm_div_rp_d:
3692 case Intrinsic::nvvm_div_rz_d:
3693 case Intrinsic::nvvm_div_rm_ftz_f:
3694 case Intrinsic::nvvm_div_rn_ftz_f:
3695 case Intrinsic::nvvm_div_rp_ftz_f:
3696 case Intrinsic::nvvm_div_rz_ftz_f: {
3697 bool IsFTZ = nvvm::FDivShouldFTZ(IntrinsicID);
3698 APFloat A = IsFTZ ? FTZPreserveSign(Op1V) : Op1V;
3699 APFloat B = IsFTZ ? FTZPreserveSign(Op2V) : Op2V;
3700 APFloat::roundingMode RoundMode =
3701 nvvm::GetFDivRoundingMode(IntrinsicID);
3702
3703 APFloat Res = A;
3704 APFloat::opStatus Status = Res.divide(B, RoundMode);
3705 if (!Res.isNaN() &&
3707 Res = IsFTZ ? FTZPreserveSign(Res) : Res;
3708 return ConstantFP::get(Ty, Res);
3709 }
3710 return nullptr;
3711 }
3712 }
3713
3714 if (!Ty->isHalfTy() && !Ty->isFloatTy() && !Ty->isDoubleTy())
3715 return nullptr;
3716
3717 switch (IntrinsicID) {
3718 default:
3719 break;
3720 case Intrinsic::pow:
3721 return ConstantFoldBinaryFP(pow, Op1V, Op2V, Ty);
3722 case Intrinsic::amdgcn_fmul_legacy:
3723 // The legacy behaviour is that multiplying +/- 0.0 by anything, even
3724 // NaN or infinity, gives +0.0.
3725 if (Op1V.isZero() || Op2V.isZero())
3726 return ConstantFP::getZero(Ty);
3727 return ConstantFP::get(Ty, Op1V * Op2V);
3728 }
3729
3730 } else if (auto *Op2C = dyn_cast<ConstantInt>(Operands[1])) {
3731 switch (IntrinsicID) {
3732 case Intrinsic::ldexp: {
3733 // APFloat::scalbn takes the exponent as `int`. Clamp wider integer
3734 // exponents into [INT_MIN, INT_MAX] so values still saturate the
3735 // result to +/-inf or +/-0.
3736 APInt Exp = Op2C->getValue();
3737 Exp = Exp.getBitWidth() < 32 ? Exp.sext(32) : Exp.truncSSat(32);
3738 return ConstantFP::get(
3739 Ty->getContext(),
3740 scalbn(Op1V, Exp.getSExtValue(), APFloat::rmNearestTiesToEven));
3741 }
3742 case Intrinsic::is_fpclass: {
3743 FPClassTest Mask = static_cast<FPClassTest>(Op2C->getZExtValue());
3744 bool Result =
3745 ((Mask & fcSNan) && Op1V.isNaN() && Op1V.isSignaling()) ||
3746 ((Mask & fcQNan) && Op1V.isNaN() && !Op1V.isSignaling()) ||
3747 ((Mask & fcNegInf) && Op1V.isNegInfinity()) ||
3748 ((Mask & fcNegNormal) && Op1V.isNormal() && Op1V.isNegative()) ||
3749 ((Mask & fcNegSubnormal) && Op1V.isDenormal() && Op1V.isNegative()) ||
3750 ((Mask & fcNegZero) && Op1V.isZero() && Op1V.isNegative()) ||
3751 ((Mask & fcPosZero) && Op1V.isZero() && !Op1V.isNegative()) ||
3752 ((Mask & fcPosSubnormal) && Op1V.isDenormal() && !Op1V.isNegative()) ||
3753 ((Mask & fcPosNormal) && Op1V.isNormal() && !Op1V.isNegative()) ||
3754 ((Mask & fcPosInf) && Op1V.isPosInfinity());
3755 return ConstantInt::get(Ty, Result);
3756 }
3757 case Intrinsic::powi: {
3758 // Square-and-multiply using the operand's own semantics, matching
3759 // the multiply sequence ExpandPowI builds in SelectionDAG.
3760 int Exp = static_cast<int>(Op2C->getSExtValue());
3761 unsigned UExp = static_cast<unsigned>(Exp);
3762 if (Exp < 0)
3763 UExp = -UExp;
3764 const fltSemantics &Semantics = Op1V.getSemantics();
3765 APFloat Res = APFloat::getOne(Semantics);
3766 APFloat CurSquare = Op1V;
3767 while (UExp) {
3768 if (UExp & 1)
3769 Res = Res * CurSquare;
3770 CurSquare = CurSquare * CurSquare;
3771 UExp >>= 1;
3772 }
3773 if (Exp < 0)
3774 Res = APFloat::getOne(Semantics) / Res;
3775 return ConstantFP::get(Ty, Res);
3776 }
3777 default:
3778 break;
3779 }
3780 }
3781 return nullptr;
3782 }
3783
3784 if (Operands[0]->getType()->isIntegerTy() &&
3785 Operands[1]->getType()->isIntegerTy()) {
3786 const APInt *C0, *C1;
3787 if (!getConstIntOrUndef(Operands[0], C0) ||
3788 !getConstIntOrUndef(Operands[1], C1))
3789 return nullptr;
3790
3791 switch (IntrinsicID) {
3792 default: break;
3793 case Intrinsic::smax:
3794 case Intrinsic::smin:
3795 case Intrinsic::umax:
3796 case Intrinsic::umin:
3797 if (!C0 || !C1)
3798 return MinMaxIntrinsic::getSaturationPoint(IntrinsicID, Ty);
3799 return ConstantInt::get(
3800 Ty, ICmpInst::compare(*C0, *C1,
3801 MinMaxIntrinsic::getPredicate(IntrinsicID))
3802 ? *C0
3803 : *C1);
3804
3805 case Intrinsic::scmp:
3806 case Intrinsic::ucmp:
3807 if (!C0 || !C1)
3808 return ConstantInt::get(Ty, 0);
3809
3810 int Res;
3811 if (IntrinsicID == Intrinsic::scmp)
3812 Res = C0->sgt(*C1) ? 1 : C0->slt(*C1) ? -1 : 0;
3813 else
3814 Res = C0->ugt(*C1) ? 1 : C0->ult(*C1) ? -1 : 0;
3815 return ConstantInt::get(Ty, Res, /*IsSigned=*/true);
3816
3817 case Intrinsic::usub_with_overflow:
3818 case Intrinsic::ssub_with_overflow:
3819 // X - undef -> { 0, false }
3820 // undef - X -> { 0, false }
3821 if (!C0 || !C1)
3822 return Constant::getNullValue(Ty);
3823 [[fallthrough]];
3824 case Intrinsic::uadd_with_overflow:
3825 case Intrinsic::sadd_with_overflow:
3826 // X + undef -> { -1, false }
3827 // undef + x -> { -1, false }
3828 if (!C0 || !C1) {
3829 return ConstantStruct::get(
3830 cast<StructType>(Ty),
3831 {Constant::getAllOnesValue(Ty->getStructElementType(0)),
3832 Constant::getNullValue(Ty->getStructElementType(1))});
3833 }
3834 [[fallthrough]];
3835 case Intrinsic::smul_with_overflow:
3836 case Intrinsic::umul_with_overflow: {
3837 // undef * X -> { 0, false }
3838 // X * undef -> { 0, false }
3839 if (!C0 || !C1)
3840 return Constant::getNullValue(Ty);
3841
3842 APInt Res;
3843 bool Overflow;
3844 switch (IntrinsicID) {
3845 default: llvm_unreachable("Invalid case");
3846 case Intrinsic::sadd_with_overflow:
3847 Res = C0->sadd_ov(*C1, Overflow);
3848 break;
3849 case Intrinsic::uadd_with_overflow:
3850 Res = C0->uadd_ov(*C1, Overflow);
3851 break;
3852 case Intrinsic::ssub_with_overflow:
3853 Res = C0->ssub_ov(*C1, Overflow);
3854 break;
3855 case Intrinsic::usub_with_overflow:
3856 Res = C0->usub_ov(*C1, Overflow);
3857 break;
3858 case Intrinsic::smul_with_overflow:
3859 Res = C0->smul_ov(*C1, Overflow);
3860 break;
3861 case Intrinsic::umul_with_overflow:
3862 Res = C0->umul_ov(*C1, Overflow);
3863 break;
3864 }
3865 Constant *Ops[] = {
3866 ConstantInt::get(Ty->getContext(), Res),
3867 ConstantInt::get(Type::getInt1Ty(Ty->getContext()), Overflow)
3868 };
3870 }
3871 case Intrinsic::uadd_sat:
3872 case Intrinsic::sadd_sat:
3873 if (!C0 || !C1)
3874 return Constant::getAllOnesValue(Ty);
3875 if (IntrinsicID == Intrinsic::uadd_sat)
3876 return ConstantInt::get(Ty, C0->uadd_sat(*C1));
3877 else
3878 return ConstantInt::get(Ty, C0->sadd_sat(*C1));
3879 case Intrinsic::usub_sat:
3880 case Intrinsic::ssub_sat:
3881 if (!C0 || !C1)
3882 return Constant::getNullValue(Ty);
3883 if (IntrinsicID == Intrinsic::usub_sat)
3884 return ConstantInt::get(Ty, C0->usub_sat(*C1));
3885 else
3886 return ConstantInt::get(Ty, C0->ssub_sat(*C1));
3887 case Intrinsic::cttz:
3888 case Intrinsic::ctlz:
3889 assert(C1 && "Must be constant int");
3890
3891 // cttz(0, 1) and ctlz(0, 1) are poison.
3892 if (C1->isOne() && (!C0 || C0->isZero()))
3893 return PoisonValue::get(Ty);
3894 if (!C0)
3895 return Constant::getNullValue(Ty);
3896 if (IntrinsicID == Intrinsic::cttz)
3897 return ConstantInt::get(Ty, C0->countr_zero());
3898 else
3899 return ConstantInt::get(Ty, C0->countl_zero());
3900
3901 case Intrinsic::abs:
3902 assert(C1 && "Must be constant int");
3903 assert((C1->isOne() || C1->isZero()) && "Must be 0 or 1");
3904
3905 // Undef or minimum val operand with poison min --> poison
3906 if (C1->isOne() && (!C0 || C0->isMinSignedValue()))
3907 return PoisonValue::get(Ty);
3908
3909 // Undef operand with no poison min --> 0 (sign bit must be clear)
3910 if (!C0)
3911 return Constant::getNullValue(Ty);
3912
3913 return ConstantInt::get(Ty, C0->abs());
3914 case Intrinsic::clmul:
3915 if (!C0 || !C1)
3916 return Constant::getNullValue(Ty);
3917 return ConstantInt::get(Ty, APIntOps::clmul(*C0, *C1));
3918 case Intrinsic::pdep:
3919 if (!C0 || !C1)
3920 return Constant::getNullValue(Ty);
3921 return ConstantInt::get(Ty, APIntOps::pdep(*C0, *C1));
3922 case Intrinsic::pext:
3923 if (!C0 || !C1)
3924 return Constant::getNullValue(Ty);
3925 return ConstantInt::get(Ty, APIntOps::pext(*C0, *C1));
3926 case Intrinsic::amdgcn_wave_reduce_umin:
3927 case Intrinsic::amdgcn_wave_reduce_umax:
3928 case Intrinsic::amdgcn_wave_reduce_max:
3929 case Intrinsic::amdgcn_wave_reduce_min:
3930 case Intrinsic::amdgcn_wave_reduce_and:
3931 case Intrinsic::amdgcn_wave_reduce_or:
3932 return Operands[0];
3933 }
3934
3935 return nullptr;
3936 }
3937
3938 // Support ConstantVector in case we have an Undef in the top.
3939 if ((isa<ConstantVector>(Operands[0]) ||
3940 isa<ConstantDataVector>(Operands[0])) &&
3941 // Check for default rounding mode.
3942 // FIXME: Support other rounding modes?
3943 isa<ConstantInt>(Operands[1]) &&
3944 cast<ConstantInt>(Operands[1])->getValue() == 4) {
3945 auto *Op = cast<Constant>(Operands[0]);
3946 switch (IntrinsicID) {
3947 default: break;
3948 case Intrinsic::x86_avx512_vcvtss2si32:
3949 case Intrinsic::x86_avx512_vcvtss2si64:
3950 case Intrinsic::x86_avx512_vcvtsd2si32:
3951 case Intrinsic::x86_avx512_vcvtsd2si64:
3952 if (ConstantFP *FPOp =
3953 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
3954 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
3955 /*roundTowardZero=*/false, Ty,
3956 /*IsSigned*/true);
3957 break;
3958 case Intrinsic::x86_avx512_vcvtss2usi32:
3959 case Intrinsic::x86_avx512_vcvtss2usi64:
3960 case Intrinsic::x86_avx512_vcvtsd2usi32:
3961 case Intrinsic::x86_avx512_vcvtsd2usi64:
3962 if (ConstantFP *FPOp =
3963 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
3964 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
3965 /*roundTowardZero=*/false, Ty,
3966 /*IsSigned*/false);
3967 break;
3968 case Intrinsic::x86_avx512_cvttss2si:
3969 case Intrinsic::x86_avx512_cvttss2si64:
3970 case Intrinsic::x86_avx512_cvttsd2si:
3971 case Intrinsic::x86_avx512_cvttsd2si64:
3972 if (ConstantFP *FPOp =
3973 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
3974 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
3975 /*roundTowardZero=*/true, Ty,
3976 /*IsSigned*/true);
3977 break;
3978 case Intrinsic::x86_avx512_cvttss2usi:
3979 case Intrinsic::x86_avx512_cvttss2usi64:
3980 case Intrinsic::x86_avx512_cvttsd2usi:
3981 case Intrinsic::x86_avx512_cvttsd2usi64:
3982 if (ConstantFP *FPOp =
3983 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
3984 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
3985 /*roundTowardZero=*/true, Ty,
3986 /*IsSigned*/false);
3987 break;
3988 }
3989 }
3990
3991 if (IntrinsicID == Intrinsic::experimental_cttz_elts) {
3992 auto *FVTy = dyn_cast<FixedVectorType>(Operands[0]->getType());
3993 bool ZeroIsPoison = cast<ConstantInt>(Operands[1])->isOne();
3994 if (!FVTy)
3995 return nullptr;
3996 unsigned Width = Ty->getIntegerBitWidth();
3997 if (APInt::getMaxValue(Width).ult(FVTy->getNumElements()))
3998 return PoisonValue::get(Ty);
3999 for (unsigned I = 0; I < FVTy->getNumElements(); ++I) {
4000 Constant *Elt = Operands[0]->getAggregateElement(I);
4001 if (!Elt)
4002 return nullptr;
4003 if (isa<UndefValue>(Elt) || Elt->isNullValue())
4004 continue;
4005 return ConstantInt::get(Ty, I);
4006 }
4007 if (ZeroIsPoison)
4008 return PoisonValue::get(Ty);
4009 return ConstantInt::get(Ty, FVTy->getNumElements());
4010 }
4011 return nullptr;
4012}
4013
4014static APFloat ConstantFoldAMDGCNCubeIntrinsic(Intrinsic::ID IntrinsicID,
4015 const APFloat &S0,
4016 const APFloat &S1,
4017 const APFloat &S2) {
4018 unsigned ID;
4019 const fltSemantics &Sem = S0.getSemantics();
4020 APFloat MA(Sem), SC(Sem), TC(Sem);
4021 if (abs(S2) >= abs(S0) && abs(S2) >= abs(S1)) {
4022 if (S2.isNegative() && S2.isNonZero() && !S2.isNaN()) {
4023 // S2 < 0
4024 ID = 5;
4025 SC = -S0;
4026 } else {
4027 ID = 4;
4028 SC = S0;
4029 }
4030 MA = S2;
4031 TC = -S1;
4032 } else if (abs(S1) >= abs(S0)) {
4033 if (S1.isNegative() && S1.isNonZero() && !S1.isNaN()) {
4034 // S1 < 0
4035 ID = 3;
4036 TC = -S2;
4037 } else {
4038 ID = 2;
4039 TC = S2;
4040 }
4041 MA = S1;
4042 SC = S0;
4043 } else {
4044 if (S0.isNegative() && S0.isNonZero() && !S0.isNaN()) {
4045 // S0 < 0
4046 ID = 1;
4047 SC = S2;
4048 } else {
4049 ID = 0;
4050 SC = -S2;
4051 }
4052 MA = S0;
4053 TC = -S1;
4054 }
4055 switch (IntrinsicID) {
4056 default:
4057 llvm_unreachable("unhandled amdgcn cube intrinsic");
4058 case Intrinsic::amdgcn_cubeid:
4059 return APFloat(Sem, ID);
4060 case Intrinsic::amdgcn_cubema:
4061 return MA + MA;
4062 case Intrinsic::amdgcn_cubesc:
4063 return SC;
4064 case Intrinsic::amdgcn_cubetc:
4065 return TC;
4066 }
4067}
4068
4069static Constant *ConstantFoldAMDGCNPermIntrinsic(ArrayRef<Constant *> Operands,
4070 Type *Ty) {
4071 const APInt *C0, *C1, *C2;
4072 if (!getConstIntOrUndef(Operands[0], C0) ||
4073 !getConstIntOrUndef(Operands[1], C1) ||
4074 !getConstIntOrUndef(Operands[2], C2))
4075 return nullptr;
4076
4077 if (!C2)
4078 return UndefValue::get(Ty);
4079
4080 APInt Val(32, 0);
4081 unsigned NumUndefBytes = 0;
4082 for (unsigned I = 0; I < 32; I += 8) {
4083 unsigned Sel = C2->extractBitsAsZExtValue(8, I);
4084 unsigned B = 0;
4085
4086 if (Sel >= 13)
4087 B = 0xff;
4088 else if (Sel == 12)
4089 B = 0x00;
4090 else {
4091 const APInt *Src = ((Sel & 10) == 10 || (Sel & 12) == 4) ? C0 : C1;
4092 if (!Src)
4093 ++NumUndefBytes;
4094 else if (Sel < 8)
4095 B = Src->extractBitsAsZExtValue(8, (Sel & 3) * 8);
4096 else
4097 B = Src->extractBitsAsZExtValue(1, (Sel & 1) ? 31 : 15) * 0xff;
4098 }
4099
4100 Val.insertBits(B, I, 8);
4101 }
4102
4103 if (NumUndefBytes == 4)
4104 return UndefValue::get(Ty);
4105
4106 return ConstantInt::get(Ty, Val);
4107}
4108
4109static Constant *ConstantFoldScalarCall3(StringRef Name,
4110 Intrinsic::ID IntrinsicID, Type *Ty,
4111 ArrayRef<Constant *> Operands,
4112 const TargetLibraryInfo *TLI = nullptr,
4113 const CallBase *Call = nullptr) {
4114 assert(Operands.size() == 3 && "Wrong number of operands.");
4115
4116 if (const auto *Op1 = dyn_cast<ConstantFP>(Operands[0])) {
4117 if (const auto *Op2 = dyn_cast<ConstantFP>(Operands[1])) {
4118 if (const auto *Op3 = dyn_cast<ConstantFP>(Operands[2])) {
4119 const APFloat &C1 = Op1->getValueAPF();
4120 const APFloat &C2 = Op2->getValueAPF();
4121 const APFloat &C3 = Op3->getValueAPF();
4122
4123 if (const auto *ConstrIntr =
4125 RoundingMode RM = getEvaluationRoundingMode(ConstrIntr);
4126 APFloat Res = C1;
4128 switch (IntrinsicID) {
4129 default:
4130 return nullptr;
4131 case Intrinsic::experimental_constrained_fma:
4132 case Intrinsic::experimental_constrained_fmuladd:
4133 St = Res.fusedMultiplyAdd(C2, C3, RM);
4134 break;
4135 }
4136 if (mayFoldConstrained(
4137 const_cast<ConstrainedFPIntrinsic *>(ConstrIntr), St))
4138 return ConstantFP::get(Ty, Res);
4139 return nullptr;
4140 }
4141
4142 switch (IntrinsicID) {
4143 default: break;
4144 case Intrinsic::amdgcn_fma_legacy: {
4145 // The legacy behaviour is that multiplying +/- 0.0 by anything, even
4146 // NaN or infinity, gives +0.0.
4147 if (C1.isZero() || C2.isZero()) {
4148 // It's tempting to just return C3 here, but that would give the
4149 // wrong result if C3 was -0.0.
4150 return ConstantFP::get(Ty, APFloat(0.0f) + C3);
4151 }
4152 [[fallthrough]];
4153 }
4154 case Intrinsic::fma:
4155 case Intrinsic::fmuladd: {
4156 APFloat V = C1;
4158 return ConstantFP::get(Ty, V);
4159 }
4160
4161 case Intrinsic::nvvm_fma_rm_f:
4162 case Intrinsic::nvvm_fma_rn_f:
4163 case Intrinsic::nvvm_fma_rp_f:
4164 case Intrinsic::nvvm_fma_rz_f:
4165 case Intrinsic::nvvm_fma_rm_d:
4166 case Intrinsic::nvvm_fma_rn_d:
4167 case Intrinsic::nvvm_fma_rp_d:
4168 case Intrinsic::nvvm_fma_rz_d:
4169 case Intrinsic::nvvm_fma_rm_ftz_f:
4170 case Intrinsic::nvvm_fma_rn_ftz_f:
4171 case Intrinsic::nvvm_fma_rp_ftz_f:
4172 case Intrinsic::nvvm_fma_rz_ftz_f: {
4173 bool IsFTZ = nvvm::FMAShouldFTZ(IntrinsicID);
4174 APFloat A = IsFTZ ? FTZPreserveSign(C1) : C1;
4175 APFloat B = IsFTZ ? FTZPreserveSign(C2) : C2;
4176 APFloat C = IsFTZ ? FTZPreserveSign(C3) : C3;
4177
4178 APFloat::roundingMode RoundMode =
4179 nvvm::GetFMARoundingMode(IntrinsicID);
4180
4181 APFloat Res = A;
4182 APFloat::opStatus Status = Res.fusedMultiplyAdd(B, C, RoundMode);
4183
4184 if (!Res.isNaN() &&
4186 Res = IsFTZ ? FTZPreserveSign(Res) : Res;
4187 return ConstantFP::get(Ty, Res);
4188 }
4189 return nullptr;
4190 }
4191
4192 case Intrinsic::amdgcn_cubeid:
4193 case Intrinsic::amdgcn_cubema:
4194 case Intrinsic::amdgcn_cubesc:
4195 case Intrinsic::amdgcn_cubetc: {
4196 APFloat V = ConstantFoldAMDGCNCubeIntrinsic(IntrinsicID, C1, C2, C3);
4197 return ConstantFP::get(Ty, V);
4198 }
4199 }
4200 }
4201 }
4202 }
4203
4204 if (IntrinsicID == Intrinsic::smul_fix ||
4205 IntrinsicID == Intrinsic::smul_fix_sat) {
4206 const APInt *C0, *C1;
4207 if (!getConstIntOrUndef(Operands[0], C0) ||
4208 !getConstIntOrUndef(Operands[1], C1))
4209 return nullptr;
4210
4211 // undef * C -> 0
4212 // C * undef -> 0
4213 if (!C0 || !C1)
4214 return Constant::getNullValue(Ty);
4215
4216 // This code performs rounding towards negative infinity in case the result
4217 // cannot be represented exactly for the given scale. Targets that do care
4218 // about rounding should use a target hook for specifying how rounding
4219 // should be done, and provide their own folding to be consistent with
4220 // rounding. This is the same approach as used by
4221 // DAGTypeLegalizer::ExpandIntRes_MULFIX.
4222 unsigned Scale = cast<ConstantInt>(Operands[2])->getZExtValue();
4223 unsigned Width = C0->getBitWidth();
4224 assert(Scale < Width && "Illegal scale.");
4225 unsigned ExtendedWidth = Width * 2;
4226 APInt Product =
4227 (C0->sext(ExtendedWidth) * C1->sext(ExtendedWidth)).ashr(Scale);
4228 if (IntrinsicID == Intrinsic::smul_fix_sat) {
4229 APInt Max = APInt::getSignedMaxValue(Width).sext(ExtendedWidth);
4230 APInt Min = APInt::getSignedMinValue(Width).sext(ExtendedWidth);
4231 Product = APIntOps::smin(Product, Max);
4232 Product = APIntOps::smax(Product, Min);
4233 }
4234 return ConstantInt::get(Ty->getContext(), Product.sextOrTrunc(Width));
4235 }
4236
4237 if (IntrinsicID == Intrinsic::fshl || IntrinsicID == Intrinsic::fshr) {
4238 const APInt *C0, *C1, *C2;
4239 if (!getConstIntOrUndef(Operands[0], C0) ||
4240 !getConstIntOrUndef(Operands[1], C1) ||
4241 !getConstIntOrUndef(Operands[2], C2))
4242 return nullptr;
4243
4244 bool IsRight = IntrinsicID == Intrinsic::fshr;
4245 if (!C2)
4246 return Operands[IsRight ? 1 : 0];
4247 if (!C0 && !C1)
4248 return UndefValue::get(Ty);
4249
4250 // The shift amount is interpreted as modulo the bitwidth. If the shift
4251 // amount is effectively 0, avoid UB due to oversized inverse shift below.
4252 unsigned BitWidth = C2->getBitWidth();
4253 unsigned ShAmt = C2->urem(BitWidth);
4254 if (!ShAmt)
4255 return Operands[IsRight ? 1 : 0];
4256
4257 // (C0 << ShlAmt) | (C1 >> LshrAmt)
4258 unsigned LshrAmt = IsRight ? ShAmt : BitWidth - ShAmt;
4259 unsigned ShlAmt = !IsRight ? ShAmt : BitWidth - ShAmt;
4260 if (!C0)
4261 return ConstantInt::get(Ty, C1->lshr(LshrAmt));
4262 if (!C1)
4263 return ConstantInt::get(Ty, C0->shl(ShlAmt));
4264 return ConstantInt::get(Ty, C0->shl(ShlAmt) | C1->lshr(LshrAmt));
4265 }
4266
4267 if (IntrinsicID == Intrinsic::amdgcn_perm)
4268 return ConstantFoldAMDGCNPermIntrinsic(Operands, Ty);
4269
4270 return nullptr;
4271}
4272
4273static Constant *ConstantFoldScalarCall(StringRef Name,
4274 Intrinsic::ID IntrinsicID, Type *Ty,
4275 ArrayRef<Constant *> Operands,
4276 const TargetLibraryInfo *TLI = nullptr,
4277 const CallBase *Call = nullptr) {
4278 if (IntrinsicID != Intrinsic::not_intrinsic &&
4279 any_of(Operands, IsaPred<PoisonValue>) &&
4280 intrinsicPropagatesPoison(IntrinsicID))
4281 return PoisonValue::get(Ty);
4282
4283 if (Operands.size() == 1)
4284 return ConstantFoldScalarCall1(Name, IntrinsicID, Ty, Operands, TLI, Call);
4285
4286 if (Operands.size() == 2) {
4287 if (Constant *FoldedLibCall =
4288 ConstantFoldLibCall2(Name, Ty, Operands, TLI)) {
4289 return FoldedLibCall;
4290 }
4291 return ConstantFoldIntrinsicCall2(IntrinsicID, Ty, Operands, Call);
4292 }
4293
4294 if (Operands.size() == 3)
4295 return ConstantFoldScalarCall3(Name, IntrinsicID, Ty, Operands, TLI, Call);
4296
4297 return nullptr;
4298}
4299
4300static Constant *ConstantFoldFixedVectorCall(
4301 StringRef Name, Intrinsic::ID IntrinsicID, FixedVectorType *FVTy,
4302 ArrayRef<Constant *> Operands, const DataLayout &DL,
4303 const TargetLibraryInfo *TLI = nullptr, const CallBase *Call = nullptr) {
4305 SmallVector<Constant *, 4> Lane(Operands.size());
4306 Type *Ty = FVTy->getElementType();
4307
4308 switch (IntrinsicID) {
4309 case Intrinsic::masked_load: {
4310 auto *SrcPtr = Operands[0];
4311 auto *Mask = Operands[1];
4312 auto *Passthru = Operands[2];
4313
4314 Constant *VecData = ConstantFoldLoadFromConstPtr(SrcPtr, FVTy, DL);
4315
4316 SmallVector<Constant *, 32> NewElements;
4317 for (unsigned I = 0, E = FVTy->getNumElements(); I != E; ++I) {
4318 auto *MaskElt = Mask->getAggregateElement(I);
4319 if (!MaskElt)
4320 break;
4321 auto *PassthruElt = Passthru->getAggregateElement(I);
4322 auto *VecElt = VecData ? VecData->getAggregateElement(I) : nullptr;
4323 if (isa<UndefValue>(MaskElt)) {
4324 if (PassthruElt)
4325 NewElements.push_back(PassthruElt);
4326 else if (VecElt)
4327 NewElements.push_back(VecElt);
4328 else
4329 return nullptr;
4330 }
4331 if (MaskElt->isNullValue()) {
4332 if (!PassthruElt)
4333 return nullptr;
4334 NewElements.push_back(PassthruElt);
4335 } else if (MaskElt->isOneValue()) {
4336 if (!VecElt)
4337 return nullptr;
4338 NewElements.push_back(VecElt);
4339 } else {
4340 return nullptr;
4341 }
4342 }
4343 if (NewElements.size() != FVTy->getNumElements())
4344 return nullptr;
4345 return ConstantVector::get(NewElements);
4346 }
4347 case Intrinsic::arm_mve_vctp8:
4348 case Intrinsic::arm_mve_vctp16:
4349 case Intrinsic::arm_mve_vctp32:
4350 case Intrinsic::arm_mve_vctp64: {
4351 if (auto *Op = dyn_cast<ConstantInt>(Operands[0])) {
4352 unsigned Lanes = FVTy->getNumElements();
4353 uint64_t Limit = Op->getZExtValue();
4354
4356 for (unsigned i = 0; i < Lanes; i++) {
4357 if (i < Limit)
4359 else
4361 }
4362 return ConstantVector::get(NCs);
4363 }
4364 return nullptr;
4365 }
4366 case Intrinsic::get_active_lane_mask: {
4367 auto *Op0 = dyn_cast<ConstantInt>(Operands[0]);
4368 auto *Op1 = dyn_cast<ConstantInt>(Operands[1]);
4369 if (Op0 && Op1) {
4370 unsigned Lanes = FVTy->getNumElements();
4371 uint64_t Base = Op0->getZExtValue();
4372 uint64_t Limit = Op1->getZExtValue();
4373
4375 for (unsigned i = 0; i < Lanes; i++) {
4376 if (Base + i < Limit)
4378 else
4380 }
4381 return ConstantVector::get(NCs);
4382 }
4383 return nullptr;
4384 }
4385 case Intrinsic::vector_extract: {
4386 auto *Idx = dyn_cast<ConstantInt>(Operands[1]);
4387 Constant *Vec = Operands[0];
4388 if (!Idx || !isa<FixedVectorType>(Vec->getType()))
4389 return nullptr;
4390
4391 unsigned NumElements = FVTy->getNumElements();
4392 unsigned VecNumElements =
4393 cast<FixedVectorType>(Vec->getType())->getNumElements();
4394 unsigned StartingIndex = Idx->getZExtValue();
4395
4396 // Extracting entire vector is nop
4397 if (NumElements == VecNumElements && StartingIndex == 0)
4398 return Vec;
4399
4400 for (unsigned I = StartingIndex, E = StartingIndex + NumElements; I < E;
4401 ++I) {
4402 Constant *Elt = Vec->getAggregateElement(I);
4403 if (!Elt)
4404 return nullptr;
4405 Result[I - StartingIndex] = Elt;
4406 }
4407
4408 return ConstantVector::get(Result);
4409 }
4410 case Intrinsic::vector_insert: {
4411 Constant *Vec = Operands[0];
4412 Constant *SubVec = Operands[1];
4413 auto *Idx = dyn_cast<ConstantInt>(Operands[2]);
4414 if (!Idx || !isa<FixedVectorType>(Vec->getType()))
4415 return nullptr;
4416
4417 unsigned SubVecNumElements =
4418 cast<FixedVectorType>(SubVec->getType())->getNumElements();
4419 unsigned VecNumElements =
4420 cast<FixedVectorType>(Vec->getType())->getNumElements();
4421 unsigned IdxN = Idx->getZExtValue();
4422 // Replacing entire vector with a subvec is nop
4423 if (SubVecNumElements == VecNumElements && IdxN == 0)
4424 return SubVec;
4425
4426 for (unsigned I = 0; I < VecNumElements; ++I) {
4427 Constant *Elt;
4428 if (I < IdxN + SubVecNumElements)
4429 Elt = SubVec->getAggregateElement(I - IdxN);
4430 else
4431 Elt = Vec->getAggregateElement(I);
4432 if (!Elt)
4433 return nullptr;
4434 Result[I] = Elt;
4435 }
4436 return ConstantVector::get(Result);
4437 }
4438 case Intrinsic::vector_interleave2:
4439 case Intrinsic::vector_interleave3:
4440 case Intrinsic::vector_interleave4:
4441 case Intrinsic::vector_interleave5:
4442 case Intrinsic::vector_interleave6:
4443 case Intrinsic::vector_interleave7:
4444 case Intrinsic::vector_interleave8: {
4445 unsigned NumElements =
4446 cast<FixedVectorType>(Operands[0]->getType())->getNumElements();
4447 unsigned NumOperands = Operands.size();
4448 for (unsigned I = 0; I < NumElements; ++I) {
4449 for (unsigned J = 0; J < NumOperands; ++J) {
4450 Constant *Elt = Operands[J]->getAggregateElement(I);
4451 if (!Elt)
4452 return nullptr;
4453 Result[NumOperands * I + J] = Elt;
4454 }
4455 }
4456 return ConstantVector::get(Result);
4457 }
4458 case Intrinsic::wasm_dot: {
4459 unsigned NumElements =
4460 cast<FixedVectorType>(Operands[0]->getType())->getNumElements();
4461
4462 assert(NumElements == 8 && Result.size() == 4 &&
4463 "wasm dot takes i16x8 and produces i32x4");
4464 assert(Ty->isIntegerTy());
4465 int32_t MulVector[8];
4466
4467 for (unsigned I = 0; I < NumElements; ++I) {
4468 ConstantInt *Elt0 =
4469 dyn_cast<ConstantInt>(Operands[0]->getAggregateElement(I));
4470 ConstantInt *Elt1 =
4471 dyn_cast<ConstantInt>(Operands[1]->getAggregateElement(I));
4472
4473 if (!Elt0 || !Elt1)
4474 return nullptr;
4475
4476 MulVector[I] = Elt0->getSExtValue() * Elt1->getSExtValue();
4477 }
4478 for (unsigned I = 0; I < Result.size(); I++) {
4479 int64_t IAdd = (int64_t)MulVector[I * 2] + (int64_t)MulVector[I * 2 + 1];
4480 Result[I] = ConstantInt::getSigned(Ty, IAdd, /*ImplicitTrunc=*/true);
4481 }
4482
4483 return ConstantVector::get(Result);
4484 }
4485 default:
4486 break;
4487 }
4488
4489 for (unsigned I = 0, E = FVTy->getNumElements(); I != E; ++I) {
4490 // Gather a column of constants.
4491 for (unsigned J = 0, JE = Operands.size(); J != JE; ++J) {
4492 // Some intrinsics use a scalar type for certain arguments.
4493 if (isVectorIntrinsicWithScalarOpAtArg(IntrinsicID, J, /*TTI=*/nullptr)) {
4494 Lane[J] = Operands[J];
4495 continue;
4496 }
4497
4498 Constant *Agg = Operands[J]->getAggregateElement(I);
4499 if (!Agg)
4500 return nullptr;
4501
4502 Lane[J] = Agg;
4503 }
4504
4505 // Use the regular scalar folding to simplify this column.
4506 Constant *Folded =
4507 ConstantFoldScalarCall(Name, IntrinsicID, Ty, Lane, TLI, Call);
4508 if (!Folded)
4509 return nullptr;
4510 Result[I] = Folded;
4511 }
4512
4513 return ConstantVector::get(Result);
4514}
4515
4516static Constant *ConstantFoldScalableVectorCall(
4517 StringRef Name, Intrinsic::ID IntrinsicID, ScalableVectorType *SVTy,
4518 ArrayRef<Constant *> Operands, const DataLayout &DL,
4519 const TargetLibraryInfo *TLI, const CallBase *Call) {
4520 switch (IntrinsicID) {
4521 case Intrinsic::aarch64_sve_convert_from_svbool: {
4522 Constant *Src = Operands[0];
4523 if (!Src->isNullValue())
4524 break;
4525
4526 return ConstantInt::getFalse(SVTy);
4527 }
4528 case Intrinsic::get_active_lane_mask: {
4529 auto *Op0 = dyn_cast<ConstantInt>(Operands[0]);
4530 auto *Op1 = dyn_cast<ConstantInt>(Operands[1]);
4531 if (Op0 && Op1 && Op0->getValue().uge(Op1->getValue()))
4532 return ConstantVector::getNullValue(SVTy);
4533 break;
4534 }
4535 case Intrinsic::vector_interleave2:
4536 case Intrinsic::vector_interleave3:
4537 case Intrinsic::vector_interleave4:
4538 case Intrinsic::vector_interleave5:
4539 case Intrinsic::vector_interleave6:
4540 case Intrinsic::vector_interleave7:
4541 case Intrinsic::vector_interleave8: {
4542 Constant *SplatVal = Operands[0]->getSplatValue();
4543 if (!SplatVal)
4544 return nullptr;
4545
4546 if (!llvm::all_equal(Operands))
4547 return nullptr;
4548
4549 return ConstantVector::getSplat(SVTy->getElementCount(), SplatVal);
4550 }
4551 default:
4552 break;
4553 }
4554
4555 // If trivially vectorizable, try folding it via the scalar call if all
4556 // operands are splats.
4557
4558 // TODO: ConstantFoldFixedVectorCall should probably check this too?
4559 if (!isTriviallyVectorizable(IntrinsicID))
4560 return nullptr;
4561
4563 for (auto [I, Op] : enumerate(Operands)) {
4564 if (isVectorIntrinsicWithScalarOpAtArg(IntrinsicID, I, /*TTI=*/nullptr)) {
4565 SplatOps.push_back(Op);
4566 continue;
4567 }
4568 Constant *Splat = Op->getSplatValue();
4569 if (!Splat)
4570 return nullptr;
4571 SplatOps.push_back(Splat);
4572 }
4573 Constant *Folded = ConstantFoldScalarCall(
4574 Name, IntrinsicID, SVTy->getElementType(), SplatOps, TLI, Call);
4575 if (!Folded)
4576 return nullptr;
4577 return ConstantVector::getSplat(SVTy->getElementCount(), Folded);
4578}
4579
4580static std::pair<Constant *, Constant *>
4581ConstantFoldScalarFrexpCall(Constant *Op, Type *IntTy) {
4582 auto *ConstFP = dyn_cast<ConstantFP>(Op);
4583 if (!ConstFP)
4584 return {};
4585
4586 const APFloat &U = ConstFP->getValueAPF();
4587 int FrexpExp;
4588 APFloat FrexpMant = frexp(U, FrexpExp, APFloat::rmNearestTiesToEven);
4589 Constant *Result0 = ConstantFP::get(ConstFP->getType(), FrexpMant);
4590
4591 // The exponent is an "unspecified value" for inf/nan. We use zero to avoid
4592 // using undef.
4593 Constant *Result1 = FrexpMant.isFinite()
4594 ? ConstantInt::getSigned(IntTy, FrexpExp)
4595 : ConstantInt::getNullValue(IntTy);
4596 return {Result0, Result1};
4597}
4598
4599/// Handle intrinsics that return tuples, which may be tuples of vectors.
4600static Constant *
4601ConstantFoldStructCall(StringRef Name, Intrinsic::ID IntrinsicID,
4602 StructType *StTy, ArrayRef<Constant *> Operands,
4603 const DataLayout &DL, const TargetLibraryInfo *TLI,
4604 const CallBase *Call) {
4605
4606 switch (IntrinsicID) {
4607 case Intrinsic::frexp: {
4608 Type *Ty0 = StTy->getContainedType(0);
4609 Type *Ty1 = StTy->getContainedType(1)->getScalarType();
4610
4611 if (auto *FVTy0 = dyn_cast<FixedVectorType>(Ty0)) {
4612 SmallVector<Constant *, 4> Results0(FVTy0->getNumElements());
4613 SmallVector<Constant *, 4> Results1(FVTy0->getNumElements());
4614
4615 for (unsigned I = 0, E = FVTy0->getNumElements(); I != E; ++I) {
4616 Constant *Lane = Operands[0]->getAggregateElement(I);
4617 std::tie(Results0[I], Results1[I]) =
4618 ConstantFoldScalarFrexpCall(Lane, Ty1);
4619 if (!Results0[I])
4620 return nullptr;
4621 }
4622
4623 return ConstantStruct::get(StTy, ConstantVector::get(Results0),
4624 ConstantVector::get(Results1));
4625 }
4626
4627 auto [Result0, Result1] = ConstantFoldScalarFrexpCall(Operands[0], Ty1);
4628 if (!Result0)
4629 return nullptr;
4630 return ConstantStruct::get(StTy, Result0, Result1);
4631 }
4632 case Intrinsic::sincos: {
4633 Type *Ty = StTy->getContainedType(0);
4634 Type *TyScalar = Ty->getScalarType();
4635
4636 auto ConstantFoldScalarSincosCall =
4637 [&](Constant *Op) -> std::pair<Constant *, Constant *> {
4638 Constant *SinResult =
4639 ConstantFoldScalarCall(Name, Intrinsic::sin, TyScalar, Op, TLI, Call);
4640 Constant *CosResult =
4641 ConstantFoldScalarCall(Name, Intrinsic::cos, TyScalar, Op, TLI, Call);
4642 return std::make_pair(SinResult, CosResult);
4643 };
4644
4645 if (auto *FVTy = dyn_cast<FixedVectorType>(Ty)) {
4646 SmallVector<Constant *> SinResults(FVTy->getNumElements());
4647 SmallVector<Constant *> CosResults(FVTy->getNumElements());
4648
4649 for (unsigned I = 0, E = FVTy->getNumElements(); I != E; ++I) {
4650 Constant *Lane = Operands[0]->getAggregateElement(I);
4651 std::tie(SinResults[I], CosResults[I]) =
4652 ConstantFoldScalarSincosCall(Lane);
4653 if (!SinResults[I] || !CosResults[I])
4654 return nullptr;
4655 }
4656
4657 return ConstantStruct::get(StTy, ConstantVector::get(SinResults),
4658 ConstantVector::get(CosResults));
4659 }
4660
4661 if (!Ty->isFloatingPointTy())
4662 return nullptr;
4663
4664 auto [SinResult, CosResult] = ConstantFoldScalarSincosCall(Operands[0]);
4665 if (!SinResult || !CosResult)
4666 return nullptr;
4667 return ConstantStruct::get(StTy, SinResult, CosResult);
4668 }
4669 case Intrinsic::vector_deinterleave2:
4670 case Intrinsic::vector_deinterleave3:
4671 case Intrinsic::vector_deinterleave4:
4672 case Intrinsic::vector_deinterleave5:
4673 case Intrinsic::vector_deinterleave6:
4674 case Intrinsic::vector_deinterleave7:
4675 case Intrinsic::vector_deinterleave8: {
4676 unsigned NumResults = StTy->getNumElements();
4677 auto *Vec = Operands[0];
4678 auto *VecTy = cast<VectorType>(Vec->getType());
4679
4680 ElementCount ResultEC =
4681 VecTy->getElementCount().divideCoefficientBy(NumResults);
4682
4683 if (auto *EltC = Vec->getSplatValue()) {
4684 auto *ResultVec = ConstantVector::getSplat(ResultEC, EltC);
4685 SmallVector<Constant *, 8> Results(NumResults, ResultVec);
4686 return ConstantStruct::get(StTy, Results);
4687 }
4688
4689 if (!ResultEC.isFixed())
4690 return nullptr;
4691
4692 unsigned NumElements = ResultEC.getFixedValue();
4694 SmallVector<Constant *> Elements(NumElements);
4695 for (unsigned I = 0; I != NumResults; ++I) {
4696 for (unsigned J = 0; J != NumElements; ++J) {
4697 Constant *Elt = Vec->getAggregateElement(J * NumResults + I);
4698 if (!Elt)
4699 return nullptr;
4700 Elements[J] = Elt;
4701 }
4702 Results[I] = ConstantVector::get(Elements);
4703 }
4704 return ConstantStruct::get(StTy, Results);
4705 }
4706 default:
4707 // TODO: Constant folding of vector intrinsics that fall through here does
4708 // not work (e.g. overflow intrinsics)
4709 return ConstantFoldScalarCall(Name, IntrinsicID, StTy, Operands, TLI, Call);
4710 }
4711
4712 return nullptr;
4713}
4714
4715} // end anonymous namespace
4716
4719 const DataLayout &DL, Function *CxtF) {
4720 // In the absence of CxtF, assume strictfp conservatively.
4721 if (!canConstantFoldIntrinsic(ID, CxtF ? CxtF->isStrictFP() : true) ||
4724 Ty, ArrayRef<Value *>((Value *const *)Ops.data(), Ops.size()))))
4725 return nullptr;
4726 if (auto *FVTy = dyn_cast<FixedVectorType>(Ty))
4727 return ConstantFoldFixedVectorCall("", ID, FVTy, Ops, DL);
4728 return ConstantFoldScalarCall("", ID, Ty, Ops);
4729}
4730
4732 ArrayRef<Constant *> Operands,
4733 const TargetLibraryInfo *TLI,
4734 bool AllowNonDeterministic) {
4735 if (Call->isNoBuiltin())
4736 return nullptr;
4737 if (!F->hasName())
4738 return nullptr;
4739
4740 // If this is not an intrinsic and not recognized as a library call, bail out.
4741 Intrinsic::ID IID = F->getIntrinsicID();
4742 if (IID == Intrinsic::not_intrinsic) {
4743 if (!TLI)
4744 return nullptr;
4745 LibFunc LibF;
4746 if (!TLI->getLibFunc(*F, LibF))
4747 return nullptr;
4748 }
4749
4750 // Conservatively assume that floating-point libcalls may be
4751 // non-deterministic.
4752 Type *Ty = F->getReturnType();
4753 if (!AllowNonDeterministic && Ty->isFPOrFPVectorTy())
4754 return nullptr;
4755
4756 StringRef Name = F->getName();
4757 if (auto *FVTy = dyn_cast<FixedVectorType>(Ty))
4758 return ConstantFoldFixedVectorCall(
4759 Name, IID, FVTy, Operands, F->getDataLayout(), TLI, Call);
4760
4761 if (auto *SVTy = dyn_cast<ScalableVectorType>(Ty))
4762 return ConstantFoldScalableVectorCall(
4763 Name, IID, SVTy, Operands, F->getDataLayout(), TLI, Call);
4764
4765 if (auto *StTy = dyn_cast<StructType>(Ty))
4766 return ConstantFoldStructCall(Name, IID, StTy, Operands,
4767 F->getDataLayout(), TLI, Call);
4768
4769 // TODO: If this is a library function, we already discovered that above,
4770 // so we should pass the LibFunc, not the name (and it might be better
4771 // still to separate intrinsic handling from libcalls).
4772 return ConstantFoldScalarCall(Name, IID, Ty, Operands, TLI, Call);
4773}
4774
4776 const TargetLibraryInfo *TLI) {
4777 // FIXME: Refactor this code; this duplicates logic in LibCallsShrinkWrap
4778 // (and to some extent ConstantFoldScalarCall).
4779 if (Call->isNoBuiltin() || Call->isStrictFP())
4780 return false;
4781 Function *F = Call->getCalledFunction();
4782 if (!F)
4783 return false;
4784
4785 LibFunc Func;
4786 if (!TLI || !TLI->getLibFunc(*F, Func))
4787 return false;
4788
4789 if (Call->arg_size() == 1) {
4790 if (ConstantFP *OpC = dyn_cast<ConstantFP>(Call->getArgOperand(0))) {
4791 const APFloat &Op = OpC->getValueAPF();
4792 switch (Func) {
4793 case LibFunc_logl:
4794 case LibFunc_log:
4795 case LibFunc_logf:
4796 case LibFunc_log2l:
4797 case LibFunc_log2:
4798 case LibFunc_log2f:
4799 case LibFunc_log10l:
4800 case LibFunc_log10:
4801 case LibFunc_log10f:
4802 return Op.isNaN() || (!Op.isZero() && !Op.isNegative());
4803
4804 case LibFunc_ilogb:
4805 return !Op.isNaN() && !Op.isZero() && !Op.isInfinity();
4806
4807 case LibFunc_expl:
4808 case LibFunc_exp:
4809 case LibFunc_expf:
4810 // FIXME: These boundaries are slightly conservative.
4811 if (OpC->getType()->isDoubleTy())
4812 return !(Op < APFloat(-745.0) || Op > APFloat(709.0));
4813 if (OpC->getType()->isFloatTy())
4814 return !(Op < APFloat(-103.0f) || Op > APFloat(88.0f));
4815 break;
4816
4817 case LibFunc_exp2l:
4818 case LibFunc_exp2:
4819 case LibFunc_exp2f:
4820 // FIXME: These boundaries are slightly conservative.
4821 if (OpC->getType()->isDoubleTy())
4822 return !(Op < APFloat(-1074.0) || Op > APFloat(1023.0));
4823 if (OpC->getType()->isFloatTy())
4824 return !(Op < APFloat(-149.0f) || Op > APFloat(127.0f));
4825 break;
4826
4827 case LibFunc_sinl:
4828 case LibFunc_sin:
4829 case LibFunc_sinf:
4830 case LibFunc_cosl:
4831 case LibFunc_cos:
4832 case LibFunc_cosf:
4833 return !Op.isInfinity();
4834
4835 case LibFunc_tanl:
4836 case LibFunc_tan:
4837 case LibFunc_tanf: {
4838 // FIXME: Stop using the host math library.
4839 // FIXME: The computation isn't done in the right precision.
4840 Type *Ty = OpC->getType();
4841 if (Ty->isDoubleTy() || Ty->isFloatTy() || Ty->isHalfTy())
4842 return ConstantFoldFP(tan, OpC->getValueAPF(), Ty) != nullptr;
4843 break;
4844 }
4845
4846 case LibFunc_atan:
4847 case LibFunc_atanf:
4848 case LibFunc_atanl:
4849 // Per POSIX, this MAY fail if Op is denormal. We choose not failing.
4850 return true;
4851
4852 case LibFunc_asinl:
4853 case LibFunc_asin:
4854 case LibFunc_asinf:
4855 case LibFunc_acosl:
4856 case LibFunc_acos:
4857 case LibFunc_acosf:
4858 return !(Op < APFloat::getOne(Op.getSemantics(), true) ||
4859 Op > APFloat::getOne(Op.getSemantics()));
4860
4861 case LibFunc_sinh:
4862 case LibFunc_cosh:
4863 case LibFunc_sinhf:
4864 case LibFunc_coshf:
4865 case LibFunc_sinhl:
4866 case LibFunc_coshl:
4867 // FIXME: These boundaries are slightly conservative.
4868 if (OpC->getType()->isDoubleTy())
4869 return !(Op < APFloat(-710.0) || Op > APFloat(710.0));
4870 if (OpC->getType()->isFloatTy())
4871 return !(Op < APFloat(-89.0f) || Op > APFloat(89.0f));
4872 break;
4873
4874 case LibFunc_sqrtl:
4875 case LibFunc_sqrt:
4876 case LibFunc_sqrtf:
4877 return Op.isNaN() || Op.isZero() || !Op.isNegative();
4878
4879 // FIXME: Add more functions: sqrt_finite, atanh, expm1, log1p,
4880 // maybe others?
4881 default:
4882 break;
4883 }
4884 }
4885 }
4886
4887 if (Call->arg_size() == 2) {
4888 ConstantFP *Op0C = dyn_cast<ConstantFP>(Call->getArgOperand(0));
4889 ConstantFP *Op1C = dyn_cast<ConstantFP>(Call->getArgOperand(1));
4890 if (Op0C && Op1C) {
4891 const APFloat &Op0 = Op0C->getValueAPF();
4892 const APFloat &Op1 = Op1C->getValueAPF();
4893
4894 switch (Func) {
4895 case LibFunc_powl:
4896 case LibFunc_pow:
4897 case LibFunc_powf: {
4898 // FIXME: Stop using the host math library.
4899 // FIXME: The computation isn't done in the right precision.
4900 Type *Ty = Op0C->getType();
4901 if (Ty->isDoubleTy() || Ty->isFloatTy() || Ty->isHalfTy()) {
4902 if (Ty == Op1C->getType())
4903 return ConstantFoldBinaryFP(pow, Op0, Op1, Ty) != nullptr;
4904 }
4905 break;
4906 }
4907
4908 case LibFunc_fmodl:
4909 case LibFunc_fmod:
4910 case LibFunc_fmodf:
4911 case LibFunc_remainderl:
4912 case LibFunc_remainder:
4913 case LibFunc_remainderf:
4914 return Op0.isNaN() || Op1.isNaN() ||
4915 (!Op0.isInfinity() && !Op1.isZero());
4916
4917 case LibFunc_atan2:
4918 case LibFunc_atan2f:
4919 case LibFunc_atan2l:
4920 // Although IEEE-754 says atan2(+/-0.0, +/-0.0) are well-defined, and
4921 // GLIBC and MSVC do not appear to raise an error on those, we
4922 // cannot rely on that behavior. POSIX and C11 say that a domain error
4923 // may occur, so allow for that possibility.
4924 return !Op0.isZero() || !Op1.isZero();
4925
4926 case LibFunc_nextafter:
4927 case LibFunc_nextafterf:
4928 case LibFunc_nextafterl:
4929 case LibFunc_nexttoward:
4930 case LibFunc_nexttowardf:
4931 case LibFunc_nexttowardl: {
4932 return ConstantFoldNextToward(Op0, Op1, F->getReturnType()) != nullptr;
4933 }
4934 default:
4935 break;
4936 }
4937 }
4938 }
4939
4940 return false;
4941}
4942
4944 unsigned CastOp, const DataLayout &DL,
4945 PreservedCastFlags *Flags) {
4946 switch (CastOp) {
4947 case Instruction::BitCast:
4948 // Bitcast is always lossless.
4949 return ConstantFoldCastOperand(Instruction::BitCast, C, InvCastTo, DL);
4950 case Instruction::Trunc: {
4951 auto *ZExtC = ConstantFoldCastOperand(Instruction::ZExt, C, InvCastTo, DL);
4952 if (Flags) {
4953 // Truncation back on ZExt value is always NUW.
4954 Flags->NUW = true;
4955 // Test positivity of C.
4956 auto *SExtC =
4957 ConstantFoldCastOperand(Instruction::SExt, C, InvCastTo, DL);
4958 Flags->NSW = ZExtC == SExtC;
4959 }
4960 return ZExtC;
4961 }
4962 case Instruction::SExt:
4963 case Instruction::ZExt: {
4964 auto *InvC = ConstantExpr::getTrunc(C, InvCastTo);
4965 auto *CastInvC = ConstantFoldCastOperand(CastOp, InvC, C->getType(), DL);
4966 // Must satisfy CastOp(InvC) == C.
4967 if (!CastInvC || CastInvC != C)
4968 return nullptr;
4969 if (Flags && CastOp == Instruction::ZExt) {
4970 auto *SExtInvC =
4971 ConstantFoldCastOperand(Instruction::SExt, InvC, C->getType(), DL);
4972 // Test positivity of InvC.
4973 Flags->NNeg = CastInvC == SExtInvC;
4974 }
4975 return InvC;
4976 }
4977 case Instruction::FPExt: {
4978 Constant *InvC =
4979 ConstantFoldCastOperand(Instruction::FPTrunc, C, InvCastTo, DL);
4980 if (InvC) {
4981 Constant *CastInvC =
4982 ConstantFoldCastOperand(CastOp, InvC, C->getType(), DL);
4983 if (CastInvC == C)
4984 return InvC;
4985 }
4986 return nullptr;
4987 }
4988 default:
4989 return nullptr;
4990 }
4991}
4992
4994 const DataLayout &DL,
4995 PreservedCastFlags *Flags) {
4996 return getLosslessInvCast(C, DestTy, Instruction::ZExt, DL, Flags);
4997}
4998
5000 const DataLayout &DL,
5001 PreservedCastFlags *Flags) {
5002 return getLosslessInvCast(C, DestTy, Instruction::SExt, DL, Flags);
5003}
5004
5005void TargetFolder::anchor() {}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
constexpr LLT S1
This file declares a class to represent arbitrary precision floating point values and provide a varie...
This file implements a class to represent arbitrary precision integral constant values and operations...
This file implements the APSInt class, which is a simple class that represents an arbitrary sized int...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Function Alias Analysis Results
#define X(NUM, ENUM, NAME)
Definition ELF.h:856
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static Constant * FoldBitCast(Constant *V, Type *DestTy)
static ConstantFP * flushDenormalConstant(Type *Ty, const APFloat &APF, DenormalMode::DenormalModeKind Mode)
Constant * getConstantAtOffset(Constant *Base, APInt Offset, const DataLayout &DL)
If this Offset points exactly to the start of an aggregate element, return that element,...
static cl::opt< bool > DisableFPCallFolding("disable-fp-call-folding", cl::desc("Disable constant-folding of FP intrinsics and libcalls."), cl::init(false), cl::Hidden)
static bool canConstantFoldIntrinsic(Intrinsic::ID ID, bool IsStrictFP)
Returns true if the intrinsic can be constant folded, given IsStrictFP.
static ConstantFP * flushDenormalConstantFP(ConstantFP *CFP, const Instruction *Inst, bool IsOutput)
static bool anyTypeContainsFP(Type *RetTy, ArrayRef< Value * > Ops)
Given a function's return type and its operands, determine if any of them of of floating-point type.
static DenormalMode getInstrDenormalMode(const Instruction *CtxI, Type *Ty)
Return the denormal mode that can be assumed when executing a floating point operation at CtxI.
This file contains the declarations for the subclasses of Constant, which represent the different fla...
This file defines the DenseMap class.
Hexagon Common GEP
amode Optimize addressing mode
static constexpr Value * getValue(Ty &ValueOrUse)
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static bool InRange(int64_t Value, unsigned short Shift, int LBound, int HBound)
This file contains the definitions of the enumerations and flags associated with NVVM Intrinsics,...
if(PassOpts->AAPipeline)
const SmallVectorImpl< MachineOperand > & Cond
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
This file contains some templates that are useful if you are working with the STL at all.
This file implements the SmallBitVector class.
This file defines the SmallVector class.
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
The Input class is used to parse a yaml document into in-memory structs and vectors.
cmpResult
IEEE-754R 5.11: Floating Point Comparison Relations.
Definition APFloat.h:343
static constexpr roundingMode rmTowardZero
Definition APFloat.h:357
llvm::RoundingMode roundingMode
IEEE-754R 4.3: Rounding-direction attributes.
Definition APFloat.h:351
static const fltSemantics & IEEEdouble()
Definition APFloat.h:305
static constexpr roundingMode rmTowardNegative
Definition APFloat.h:356
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:353
static constexpr roundingMode rmTowardPositive
Definition APFloat.h:355
static constexpr roundingMode rmNearestTiesToAway
Definition APFloat.h:358
opStatus
IEEE-754R 7: Default exception handling.
Definition APFloat.h:369
static APFloat getQNaN(const fltSemantics &Sem, bool Negative=false, const APInt *payload=nullptr)
Factory for QNaN values.
Definition APFloat.h:1216
opStatus divide(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1304
void copySign(const APFloat &RHS)
Definition APFloat.h:1398
LLVM_ABI opStatus convert(const fltSemantics &ToSemantics, roundingMode RM, bool *losesInfo)
Definition APFloat.cpp:5929
opStatus subtract(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1286
bool isNegative() const
Definition APFloat.h:1575
LLVM_ABI double convertToDouble() const
Converts this APFloat to host double value.
Definition APFloat.cpp:5988
bool isPosInfinity() const
Definition APFloat.h:1588
bool isNormal() const
Definition APFloat.h:1579
bool isDenormal() const
Definition APFloat.h:1576
opStatus add(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1277
const fltSemantics & getSemantics() const
Definition APFloat.h:1583
bool isNonZero() const
Definition APFloat.h:1584
bool isFinite() const
Definition APFloat.h:1580
bool isNaN() const
Definition APFloat.h:1573
static APFloat getOne(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative One.
Definition APFloat.h:1184
opStatus multiply(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1295
bool isSignaling() const
Definition APFloat.h:1577
opStatus fusedMultiplyAdd(const APFloat &Multiplicand, const APFloat &Addend, roundingMode RM)
Definition APFloat.h:1331
bool isZero() const
Definition APFloat.h:1571
opStatus convertToInteger(MutableArrayRef< integerPart > Input, unsigned int Width, bool IsSigned, roundingMode RM, bool *IsExact) const
Definition APFloat.h:1428
opStatus mod(const APFloat &RHS)
Definition APFloat.h:1322
bool isNegInfinity() const
Definition APFloat.h:1589
opStatus roundToIntegral(roundingMode RM)
Definition APFloat.h:1344
void changeSign()
Definition APFloat.h:1393
static APFloat getZero(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Zero.
Definition APFloat.h:1175
bool isInfinity() const
Definition APFloat.h:1572
Class for arbitrary precision integers.
Definition APInt.h:78
LLVM_ABI APInt umul_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:2006
LLVM_ABI APInt usub_sat(const APInt &RHS) const
Definition APInt.cpp:2090
bool isMinSignedValue() const
Determine if this is the smallest signed value.
Definition APInt.h:424
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1565
LLVM_ABI uint64_t extractBitsAsZExtValue(unsigned numBits, unsigned bitPosition) const
Definition APInt.cpp:521
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
Definition APInt.cpp:1076
static APInt getMaxValue(unsigned numBits)
Gets maximum unsigned value of APInt for specific bit width.
Definition APInt.h:207
APInt abs() const
Get the absolute value.
Definition APInt.h:1820
LLVM_ABI APInt sadd_sat(const APInt &RHS) const
Definition APInt.cpp:2061
bool sgt(const APInt &RHS) const
Signed greater than comparison.
Definition APInt.h:1210
LLVM_ABI APInt usub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1983
bool ugt(const APInt &RHS) const
Unsigned greater than comparison.
Definition APInt.h:1191
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
Definition APInt.h:381
LLVM_ABI APInt urem(const APInt &RHS) const
Unsigned remainder operation.
Definition APInt.cpp:1692
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1513
bool ult(const APInt &RHS) const
Unsigned less than comparison.
Definition APInt.h:1120
static APInt getSignedMaxValue(unsigned numBits)
Gets maximum signed value of APInt for a specific bit width.
Definition APInt.h:210
LLVM_ABI APInt sadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1963
LLVM_ABI APInt uadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1970
unsigned countr_zero() const
Count the number of trailing zero bits.
Definition APInt.h:1664
unsigned countl_zero() const
The APInt version of std::countl_zero.
Definition APInt.h:1623
static APInt getSignedMinValue(unsigned numBits)
Gets minimum signed value of APInt for a specific bit width.
Definition APInt.h:220
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
Definition APInt.cpp:1084
LLVM_ABI APInt uadd_sat(const APInt &RHS) const
Definition APInt.cpp:2071
APInt ashr(unsigned ShiftAmt) const
Arithmetic right-shift function.
Definition APInt.h:834
LLVM_ABI APInt smul_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1995
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
Definition APInt.cpp:1028
APInt shl(unsigned shiftAmt) const
Left-shift function.
Definition APInt.h:880
bool slt(const APInt &RHS) const
Signed less than comparison.
Definition APInt.h:1139
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:201
LLVM_ABI APInt extractBits(unsigned numBits, unsigned bitPosition) const
Return an APInt with the extracted bits [bitPosition,bitPosition+numBits).
Definition APInt.cpp:483
LLVM_ABI APInt ssub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1976
bool isOne() const
Determine if this is a value of 1.
Definition APInt.h:390
APInt lshr(unsigned shiftAmt) const
Logical right-shift function.
Definition APInt.h:858
LLVM_ABI APInt ssub_sat(const APInt &RHS) const
Definition APInt.cpp:2080
An arbitrary precision integer that knows its signedness.
Definition APSInt.h:24
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
static LLVM_ABI Instruction::CastOps getCastOpcode(const Value *Val, bool SrcIsSigned, Type *Ty, bool DstIsSigned)
Returns the opcode necessary to cast Val into Ty using usual casting rules.
static LLVM_ABI unsigned isEliminableCastPair(Instruction::CastOps firstOpcode, Instruction::CastOps secondOpcode, Type *SrcTy, Type *MidTy, Type *DstTy, const DataLayout *DL)
Determine how a pair of casts can be eliminated, if they can be at all.
static LLVM_ABI bool castIsValid(Instruction::CastOps op, Type *SrcTy, Type *DstTy)
This method can be used to determine if a cast from SrcTy to DstTy using Opcode op is valid or not.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
bool isSigned() const
Definition InstrTypes.h:993
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
Definition InstrTypes.h:890
static bool isFPPredicate(Predicate P)
Definition InstrTypes.h:833
static Constant * get(LLVMContext &Context, ArrayRef< ElementTy > Elts)
get() constructor - Return a constant with array type with an element count and element type matching...
Definition Constants.h:878
static LLVM_ABI Constant * getIntToPtr(Constant *C, Type *Ty, bool OnlyIfReduced=false)
static LLVM_ABI Constant * getExtractElement(Constant *Vec, Constant *Idx, Type *OnlyIfReducedTy=nullptr)
static LLVM_ABI bool isDesirableCastOp(unsigned Opcode)
Whether creating a constant expression for this cast is desirable.
static LLVM_ABI Constant * getCast(unsigned ops, Constant *C, Type *Ty, bool OnlyIfReduced=false)
Convenience function for getting a Cast operation.
static LLVM_ABI Constant * getSub(Constant *C1, Constant *C2, bool HasNUW=false, bool HasNSW=false)
static Constant * getPtrAdd(Constant *Ptr, Constant *Offset, GEPNoWrapFlags NW=GEPNoWrapFlags::none(), std::optional< ConstantRange > InRange=std::nullopt, Type *OnlyIfReduced=nullptr)
Create a getelementptr i8, ptr, offset constant expression.
Definition Constants.h:1497
static LLVM_ABI Constant * getInsertElement(Constant *Vec, Constant *Elt, Constant *Idx, Type *OnlyIfReducedTy=nullptr)
static LLVM_ABI Constant * getShuffleVector(Constant *V1, Constant *V2, ArrayRef< int > Mask, Type *OnlyIfReducedTy=nullptr)
static bool isSupportedGetElementPtr(const Type *SrcElemTy)
Whether creating a constant expression for this getelementptr type is supported.
Definition Constants.h:1598
static LLVM_ABI Constant * get(unsigned Opcode, Constant *C1, Constant *C2, unsigned Flags=0, Type *OnlyIfReducedTy=nullptr)
get - Return a binary or shift operator constant expression, folding if possible.
static LLVM_ABI bool isDesirableBinOp(unsigned Opcode)
Whether creating a constant expression for this binary operator is desirable.
static Constant * getGetElementPtr(Type *Ty, Constant *C, ArrayRef< Constant * > IdxList, GEPNoWrapFlags NW=GEPNoWrapFlags::none(), std::optional< ConstantRange > InRange=std::nullopt, Type *OnlyIfReducedTy=nullptr)
Getelementptr form.
Definition Constants.h:1470
static LLVM_ABI Constant * getBitCast(Constant *C, Type *Ty, bool OnlyIfReduced=false)
static LLVM_ABI Constant * getTrunc(Constant *C, Type *Ty, bool OnlyIfReduced=false)
ConstantFP - Floating Point Values [float, double].
Definition Constants.h:420
const APFloat & getValueAPF() const
Definition Constants.h:463
static LLVM_ABI ConstantFP * getZero(Type *Ty, bool Negative=false)
static LLVM_ABI ConstantFP * getNaN(Type *Ty, bool Negative=false, uint64_t Payload=0)
static LLVM_ABI ConstantFP * getInfinity(Type *Ty, bool Negative=false)
This is the shared class of boolean and integer constants.
Definition Constants.h:87
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
static ConstantInt * getSigned(IntegerType *Ty, int64_t V, bool ImplicitTrunc=false)
Return a ConstantInt with the specified value for the specified type.
Definition Constants.h:135
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
int64_t getSExtValue() const
Return the constant as a 64-bit integer value after it has been sign extended as appropriate for the ...
Definition Constants.h:174
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
static LLVM_ABI Constant * get(StructType *T, ArrayRef< Constant * > V)
static LLVM_ABI Constant * getSplat(ElementCount EC, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
static LLVM_ABI Constant * get(ArrayRef< Constant * > V)
This is an important base class in LLVM.
Definition Constant.h:43
LLVM_ABI Constant * getSplatValue(bool AllowPoison=false) const
If all elements of the vector constant have the same value, return that value.
bool isNullValue() const
Return true if this is the value that would be returned by getNullValue.
Definition Constant.h:64
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
LLVM_ABI Constant * getAggregateElement(unsigned Elt) const
For aggregates (struct/array/vector) return the constant that corresponds to the specified element if...
Constrained floating point compare intrinsics.
This is the common base class for constrained floating point intrinsics.
LLVM_ABI std::optional< fp::ExceptionBehavior > getExceptionBehavior() const
LLVM_ABI std::optional< RoundingMode > getRoundingMode() const
Wrapper for a function that represents a value that functionally represents the original function.
Definition Constants.h:1143
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
iterator find(const_arg_type_t< KeyT > Val)
Definition DenseMap.h:223
iterator end()
Definition DenseMap.h:141
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Definition DenseMap.h:284
static LLVM_ABI bool compare(const APFloat &LHS, const APFloat &RHS, FCmpInst::Predicate Pred)
Return result of LHS Pred RHS comparison.
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:867
DenormalMode getDenormalMode(const fltSemantics &FPType) const
Returns the denormal handling type for the default rounding mode of the function.
Definition Function.cpp:803
bool isStrictFP() const
Determine if the function has strict floating point sematics.
Definition Function.h:636
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags inBounds()
GEPNoWrapFlags withoutNoUnsignedSignedWrap() const
static GEPNoWrapFlags noUnsignedWrap()
bool hasNoUnsignedSignedWrap() const
bool isInBounds() const
static LLVM_ABI Type * getIndexedType(Type *Ty, ArrayRef< Value * > IdxList)
Returns the result type of a getelementptr with the given source element type and indexes.
PointerType * getType() const
Global values are always pointers.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this global belongs to.
Definition Globals.cpp:205
const Constant * getInitializer() const
getInitializer - Return the initializer for this global variable.
bool isConstant() const
If the value is a global constant, its value is immutable throughout the runtime execution of the pro...
bool hasDefinitiveInitializer() const
hasDefinitiveInitializer - Whether the global variable has an initializer, and any other instances of...
static LLVM_ABI bool compare(const APInt &LHS, const APInt &RHS, ICmpInst::Predicate Pred)
Return result of LHS Pred RHS comparison.
Predicate getSignedPredicate() const
For example, EQ->EQ, SLE->SLE, UGT->SGT, etc.
bool isEquality() const
Return true if this predicate is either EQ or NE.
bool isCast() const
bool isBinaryOp() const
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
bool isUnaryOp() const
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:348
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
static APInt getSaturationPoint(Intrinsic::ID ID, unsigned numBits)
Min/max intrinsics are monotonic, they operate on a fixed-bitwidth values, so there is a certain thre...
static ICmpInst::Predicate getPredicate(Intrinsic::ID ID)
Returns the comparison predicate underlying the intrinsic.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
Class to represent scalable SIMD vectors.
This is a 'bitvector' (really, a variable-sized bit array), optimized for the case when the array is ...
SmallBitVector & set()
iterator_range< const_set_bits_iterator > set_bits() const
void push_back(const T &Elt)
pointer data()
Return a pointer to the vector's buffer, even if empty().
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Used to lazily calculate structure layout information for a target machine, based on the DataLayout s...
Definition DataLayout.h:743
LLVM_ABI unsigned getElementContainingOffset(uint64_t FixedOffset) const
Given a valid byte offset into the structure, returns the structure index that contains it.
TypeSize getElementOffset(unsigned Idx) const
Definition DataLayout.h:774
Class to represent struct types.
unsigned getNumElements() const
Random access to the elements.
Provides information about what library functions are available for the current target.
bool has(LibFunc F) const
Tests whether a library function is available.
bool getLibFunc(StringRef funcName, LibFunc &F) const
Searches for a particular function name.
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
Definition Type.cpp:310
bool isByteTy() const
True if this is an instance of ByteType.
Definition Type.h:242
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:288
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:309
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:282
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:368
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:197
bool isByteOrByteVectorTy() const
Return true if this is a byte type or a vector of byte types.
Definition Type.h:248
static LLVM_ABI IntegerType * getInt16Ty(LLVMContext &C)
Definition Type.cpp:308
bool isSized(SmallPtrSetImpl< Type * > *Visited=nullptr) const
Return true if it makes sense to take the size of this type.
Definition Type.h:326
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:232
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:306
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
Definition Type.h:186
bool isPtrOrPtrVectorTy() const
Return true if this is a pointer type or a vector of pointer types.
Definition Type.h:285
bool isX86_AMXTy() const
Return true if this is X86 AMX.
Definition Type.h:202
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:257
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:313
Type * getContainedType(unsigned i) const
This method is used to implement the type iterator (defined at the end of the file).
Definition Type.h:397
LLVM_ABI const fltSemantics & getFltSemantics() const
Definition Type.cpp:106
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:255
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:258
LLVM_ABI const Value * stripAndAccumulateConstantOffsets(const DataLayout &DL, APInt &Offset, bool AllowNonInbounds, bool AllowInvariantGroup=false, function_ref< bool(Value &Value, APInt &Offset)> ExternalAnalysis=nullptr, bool LookThroughIntToPtr=false) const
Accumulate the constant offset this value has compared to a base pointer.
LLVM_ABI uint64_t getPointerDereferenceableBytes(const DataLayout &DL, bool &CanBeNull, bool *CanBeFreed) const
Returns the number of bytes known to be dereferenceable for the pointer value.
Definition Value.cpp:909
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
Type * getElementType() const
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
Definition TypeSize.h:171
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
static constexpr bool isKnownGE(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:237
const ParentTy * getParent() const
Definition ilist_node.h:34
CallInst * Call
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt pext(const APInt &Val, const APInt &Mask)
Perform a "compress" operation, also known as pext or bext.
Definition APInt.cpp:3242
const APInt & smin(const APInt &A, const APInt &B)
Determine the smaller of two APInts considered to be signed.
Definition APInt.h:2279
const APInt & smax(const APInt &A, const APInt &B)
Determine the larger of two APInts considered to be signed.
Definition APInt.h:2284
LLVM_ABI APInt clmul(const APInt &LHS, const APInt &RHS)
Perform a carry-less multiply, also known as XOR multiplication, and return low-bits.
Definition APInt.cpp:3222
const APInt & umin(const APInt &A, const APInt &B)
Determine the smaller of two APInts considered to be unsigned.
Definition APInt.h:2289
LLVM_ABI APInt pdep(const APInt &Val, const APInt &Mask)
Perform an "expand" operation, also known as pdep or bdep.
Definition APInt.cpp:3252
const APInt & umax(const APInt &A, const APInt &B)
Determine the larger of two APInts considered to be unsigned.
Definition APInt.h:2294
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ CE
Windows NT (Windows on ARM)
Definition MCAsmInfo.h:51
initializer< Ty > init(const Ty &Val)
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:446
static constexpr cmpResult cmpEqual
Definition APFloat.h:454
@ ebStrict
This corresponds to "fpexcept.strict".
Definition FPEnv.h:42
@ ebIgnore
This corresponds to "fpexcept.ignore".
Definition FPEnv.h:40
constexpr double pi
APFloat::roundingMode GetFMARoundingMode(Intrinsic::ID IntrinsicID)
DenormalMode GetNVVMDenormMode(bool ShouldFTZ)
bool FPToIntegerIntrinsicNaNZero(Intrinsic::ID IntrinsicID)
APFloat::roundingMode GetFDivRoundingMode(Intrinsic::ID IntrinsicID)
bool FPToIntegerIntrinsicResultIsSigned(Intrinsic::ID IntrinsicID)
APFloat::roundingMode GetFPToIntegerRoundingMode(Intrinsic::ID IntrinsicID)
bool RCPShouldFTZ(Intrinsic::ID IntrinsicID)
bool FPToIntegerIntrinsicShouldFTZ(Intrinsic::ID IntrinsicID)
bool FDivShouldFTZ(Intrinsic::ID IntrinsicID)
bool FAddShouldFTZ(Intrinsic::ID IntrinsicID)
bool FMinFMaxIsXorSignAbs(Intrinsic::ID IntrinsicID)
APFloat::roundingMode GetFMulRoundingMode(Intrinsic::ID IntrinsicID)
bool UnaryMathIntrinsicShouldFTZ(Intrinsic::ID IntrinsicID)
bool FMinFMaxShouldFTZ(Intrinsic::ID IntrinsicID)
APFloat::roundingMode GetFAddRoundingMode(Intrinsic::ID IntrinsicID)
bool FMAShouldFTZ(Intrinsic::ID IntrinsicID)
bool FMulShouldFTZ(Intrinsic::ID IntrinsicID)
APFloat::roundingMode GetRCPRoundingMode(Intrinsic::ID IntrinsicID)
bool FMinFMaxPropagatesNaNs(Intrinsic::ID IntrinsicID)
NodeAddr< FuncNode * > Func
Definition RDFGraph.h:393
LLVM_ABI std::error_code status(const Twine &path, file_status &result, bool follow=true)
Get file status as if by POSIX stat().
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:315
@ Offset
Definition DWP.cpp:578
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1739
LLVM_ABI Constant * ConstantFoldLoadThroughBitcast(Constant *C, Type *DestTy, const DataLayout &DL)
ConstantFoldLoadThroughBitcast - try to cast constant to destination type returning null if unsuccess...
static double log2(double V)
LLVM_ABI Constant * ConstantFoldSelectInstruction(Constant *Cond, Constant *V1, Constant *V2)
Attempt to constant fold a select instruction with the specified operands.
LLVM_ABI Constant * ConstantFoldFPInstOperands(unsigned Opcode, Constant *LHS, Constant *RHS, const DataLayout &DL, const Instruction *I, bool AllowNonDeterministic=true)
Attempt to constant fold a floating point binary operation with the specified operands,...
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2554
LLVM_ABI bool canConstantFoldCallTo(const CallBase *Call, const Function *F)
canConstantFoldCallTo - Return true if its even possible to fold a call to the specified function.
unsigned getPointerAddressSpace(const Type *T)
Definition SPIRVUtils.h:390
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
LLVM_ABI Constant * ConstantFoldInstruction(const Instruction *I, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr)
ConstantFoldInstruction - Try to constant fold the specified instruction.
APFloat abs(APFloat X)
Returns the absolute value of the argument.
Definition APFloat.h:1713
LLVM_ABI Constant * ConstantFoldCompareInstruction(CmpInst::Predicate Predicate, Constant *C1, Constant *C2)
LLVM_ABI Constant * ConstantFoldUnaryInstruction(unsigned Opcode, Constant *V)
LLVM_ABI bool IsConstantOffsetFromGlobal(Constant *C, GlobalValue *&GV, APInt &Offset, const DataLayout &DL, DSOLocalEquivalent **DSOEquiv=nullptr)
If this constant is a constant offset from a global, return the global and the constant.
LLVM_ABI bool isMathLibCallNoop(const CallBase *Call, const TargetLibraryInfo *TLI)
Check whether the given call has no side-effects.
LLVM_ABI Constant * ReadByteArrayFromGlobal(const GlobalVariable *GV, uint64_t Offset)
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
Definition Casting.h:732
LLVM_READONLY APFloat maximum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximum semantics.
Definition APFloat.h:1793
LLVM_ABI Constant * ConstantFoldCompareInstOperands(unsigned Predicate, Constant *LHS, Constant *RHS, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, const Instruction *I=nullptr)
Attempt to constant fold a compare instruction (icmp/fcmp) with the specified operands.
int ilogb(const APFloat &Arg)
Returns the exponent of the internal representation of the APFloat.
Definition APFloat.h:1684
bool isa_and_nonnull(const Y &Val)
Definition Casting.h:676
LLVM_ABI Constant * ConstantFoldCall(const CallBase *Call, Function *F, ArrayRef< Constant * > Operands, const TargetLibraryInfo *TLI=nullptr, bool AllowNonDeterministic=true)
ConstantFoldCall - Attempt to constant fold a call to the specified function with the specified argum...
APFloat frexp(const APFloat &X, int &Exp, APFloat::roundingMode RM)
Equivalent of C standard library function.
Definition APFloat.h:1705
LLVM_ABI Constant * ConstantFoldExtractValueInstruction(Constant *Agg, ArrayRef< unsigned > Idxs)
Attempt to constant fold an extractvalue instruction with the specified operands and indices.
LLVM_ABI Constant * ConstantFoldConstant(const Constant *C, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr)
ConstantFoldConstant - Fold the constant using the specified DataLayout.
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1746
LLVM_READONLY APFloat maxnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 maxNum semantics.
Definition APFloat.h:1748
LLVM_ABI Constant * ConstantFoldLoadFromUniformValue(Constant *C, Type *Ty, const DataLayout &DL)
If C is a uniform value where all bits are the same (either all zero, all ones, all undef or all pois...
LLVM_ABI Constant * ConstantFoldUnaryOpOperand(unsigned Opcode, Constant *Op, const DataLayout &DL)
Attempt to constant fold a unary operation with the specified operand.
LLVM_ABI Constant * FlushFPConstant(Constant *Operand, const Instruction *I, bool IsOutput)
Attempt to flush float point constant according to denormal mode set in the instruction's parent func...
LLVM_ABI Constant * getLosslessUnsignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
LLVM_READONLY LLVM_ABI std::optional< APFloat > exp(const APFloat &X, RoundingMode RM=APFloat::rmNearestTiesToEven, APFloat::opStatus *Status=nullptr)
Implement IEEE 754-2019 exp functions.
Definition APFloat.cpp:6148
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
LLVM_READONLY APFloat minimumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimumNumber semantics.
Definition APFloat.h:1779
FPClassTest
Floating-point class tests, supported by 'is_fpclass' intrinsic.
APFloat scalbn(APFloat X, int Exp, APFloat::roundingMode RM)
Returns: X * 2^Exp for integral exponents.
Definition APFloat.h:1693
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
LLVM_ABI bool NullPointerIsDefined(const Function *F, unsigned AS=0)
Check whether null pointer dereferencing is considered undefined behavior for a given function or an ...
LLVM_ABI Constant * getLosslessSignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
LLVM_ABI Constant * ConstantFoldCastOperand(unsigned Opcode, Constant *C, Type *DestTy, const DataLayout &DL)
Attempt to constant fold a cast with the specified operand.
LLVM_ABI Constant * ConstantFoldLoadFromConst(Constant *C, Type *Ty, const APInt &Offset, const DataLayout &DL)
Extract value of C at the given Offset reinterpreted as Ty.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI bool intrinsicPropagatesPoison(Intrinsic::ID IID)
Return whether this intrinsic propagates poison for all operands.
LLVM_ABI Constant * ConstantFoldBinaryOpOperands(unsigned Opcode, Constant *LHS, Constant *RHS, const DataLayout &DL)
Attempt to constant fold a binary operation with the specified operands.
MutableArrayRef(T &OneElt) -> MutableArrayRef< T >
LLVM_ABI Constant * ConstantFoldIntrinsic(Intrinsic::ID ID, ArrayRef< Constant * > Ops, Type *Ty, const DataLayout &DL, Function *CxtF=nullptr)
LLVM_READONLY APFloat minnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 minNum semantics.
Definition APFloat.h:1729
@ Sub
Subtraction of integers.
LLVM_ABI bool isVectorIntrinsicWithScalarOpAtArg(Intrinsic::ID ID, unsigned ScalarOpdIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic has a scalar operand.
IntPtrTy
Definition InstrProf.h:82
DWARFExpression::Operation Op
RoundingMode
Rounding mode.
@ NearestTiesToEven
roundTiesToEven.
@ Dynamic
Denotes mode unknown at compile time.
LLVM_ABI bool isGuaranteedNotToBeUndefOrPoison(const Value *V, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Return true if this function can prove that V does not have undef bits and is never poison.
constexpr unsigned BitWidth
LLVM_ABI Constant * getLosslessInvCast(Constant *C, Type *InvCastTo, unsigned CastOp, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
Try to cast C to InvC losslessly, satisfying CastOp(InvC) equals C, or CastOp(InvC) is a refined valu...
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
Definition InstrProf.h:147
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
Definition STLExtras.h:2166
LLVM_ABI Constant * ConstantFoldCastInstruction(unsigned opcode, Constant *V, Type *DestTy)
LLVM_ABI Constant * ConstantFoldInsertValueInstruction(Constant *Agg, Constant *Val, ArrayRef< unsigned > Idxs)
Attempt to constant fold an insertvalue instruction with the specified operands and indices.
LLVM_ABI Constant * ConstantFoldLoadFromConstPtr(Constant *C, Type *Ty, APInt Offset, const DataLayout &DL)
Return the value that a load from C with offset Offset would produce if it is constant and determinab...
LLVM_ABI Constant * ConstantFoldInstOperands(const Instruction *I, ArrayRef< Constant * > Ops, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, bool AllowNonDeterministic=true)
ConstantFoldInstOperands - Attempt to constant fold an instruction with the specified operands.
LLVM_READONLY APFloat minimum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimum semantics.
Definition APFloat.h:1766
LLVM_READONLY APFloat maximumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximumNumber semantics.
Definition APFloat.h:1806
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
LLVM_ABI Constant * ConstantFoldIntegerCast(Constant *C, Type *DestTy, bool IsSigned, const DataLayout &DL)
Constant fold a zext, sext or trunc, depending on IsSigned and whether the DestTy is wider or narrowe...
LLVM_ABI bool isTriviallyVectorizable(Intrinsic::ID ID)
Identify if the intrinsic is trivially vectorizable.
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
Definition Casting.h:866
LLVM_ABI Constant * ConstantFoldBinaryInstruction(unsigned Opcode, Constant *V1, Constant *V2)
Represent subnormal handling kind for floating point instruction inputs and outputs.
DenormalModeKind Input
Denormal treatment kind for floating point instruction inputs in the default floating-point environme...
DenormalModeKind
Represent handled modes for denormal (aka subnormal) modes in the floating point environment.
@ PreserveSign
The sign of a flushed-to-zero number is preserved in the sign of 0.
@ PositiveZero
Denormals are flushed to positive zero.
@ Dynamic
Denormals have unknown treatment.
@ IEEE
IEEE-754 denormal numbers preserved.
DenormalModeKind Output
Denormal flushing mode for floating point instruction results in the default floating point environme...
static constexpr DenormalMode getDynamic()
static constexpr DenormalMode getIEEE()
bool isConstant() const
Returns true if we know the value of all bits.
Definition KnownBits.h:54
const APInt & getConstant() const
Returns the value when all bits have a known value.
Definition KnownBits.h:58