47#define DEBUG_TYPE "vector-combine"
53STATISTIC(NumVecLoad,
"Number of vector loads formed");
54STATISTIC(NumVecCmp,
"Number of vector compares formed");
55STATISTIC(NumVecBO,
"Number of vector binops formed");
56STATISTIC(NumVecCmpBO,
"Number of vector compare + binop formed");
57STATISTIC(NumShufOfBitcast,
"Number of shuffles moved after bitcast");
58STATISTIC(NumScalarOps,
"Number of scalar unary + binary ops formed");
59STATISTIC(NumScalarCmp,
"Number of scalar compares formed");
60STATISTIC(NumScalarIntrinsic,
"Number of scalar intrinsic calls formed");
64 cl::desc(
"Disable all vector combine transforms"));
68 cl::desc(
"Disable binop extract to shuffle transforms"));
72 cl::desc(
"Max number of instructions to scan for vector combining."));
74static const unsigned InvalidIndex = std::numeric_limits<unsigned>::max();
82 bool TryEarlyFoldsOnly)
85 SQ(*
DL, nullptr, &DT, &AC),
86 TryEarlyFoldsOnly(TryEarlyFoldsOnly) {}
93 const TargetTransformInfo &TTI;
94 const DominatorTree &DT;
98 const SimplifyQuery SQ;
102 bool TryEarlyFoldsOnly;
104 InstructionWorklist Worklist;
113 bool vectorizeLoadInsert(Instruction &
I);
114 bool widenSubvectorLoad(Instruction &
I);
115 ExtractElementInst *getShuffleExtract(ExtractElementInst *Ext0,
116 ExtractElementInst *Ext1,
117 unsigned PreferredExtractIndex)
const;
118 bool isExtractExtractCheap(ExtractElementInst *Ext0, ExtractElementInst *Ext1,
119 const Instruction &
I,
120 ExtractElementInst *&ConvertToShuffle,
121 unsigned PreferredExtractIndex);
124 bool foldExtractExtract(Instruction &
I);
125 bool foldInsExtFNeg(Instruction &
I);
126 bool foldInsExtBinop(Instruction &
I);
127 bool foldInsExtVectorToShuffle(Instruction &
I);
128 bool foldBitOpOfCastops(Instruction &
I);
129 bool foldBitOpOfCastConstant(Instruction &
I);
130 bool foldBitcastShuffle(Instruction &
I);
131 bool scalarizeOpOrCmp(Instruction &
I);
132 bool foldExtractedCmps(Instruction &
I);
133 bool foldSelectsFromBitcast(Instruction &
I);
134 bool foldBinopOfReductions(Instruction &
I);
135 bool foldInsertElementsToStores(Instruction &
I);
136 bool scalarizeLoad(Instruction &
I);
137 bool scalarizeLoadExtract(LoadInst *LI, VectorType *VecTy,
Value *Ptr);
138 bool scalarizeLoadBitcast(LoadInst *LI, VectorType *VecTy,
Value *Ptr);
139 bool scalarizeExtExtract(Instruction &
I);
140 bool foldConcatOfBoolMasks(Instruction &
I);
141 bool foldPermuteOfBinops(Instruction &
I);
142 bool foldShuffleOfBinops(Instruction &
I);
143 bool foldShuffleOfSelects(Instruction &
I);
144 bool foldShuffleOfCastops(Instruction &
I);
145 bool foldShuffleOfShuffles(Instruction &
I);
146 bool foldPermuteOfIntrinsic(Instruction &
I);
147 bool foldShufflesOfLengthChangingShuffles(Instruction &
I);
148 bool foldShuffleOfIntrinsics(Instruction &
I);
149 bool foldShuffleToIdentity(Instruction &
I);
150 bool foldShuffleFromReductions(Instruction &
I);
151 bool foldShuffleChainsToReduce(Instruction &
I);
152 bool foldCastFromReductions(Instruction &
I);
153 bool foldSignBitReductionCmp(Instruction &
I);
154 bool foldReductionZeroTest(Instruction &
I);
155 bool foldICmpEqZeroVectorReduce(Instruction &
I);
156 bool foldEquivalentReductionCmp(Instruction &
I);
157 bool foldReduceAddCmpZero(Instruction &
I);
158 bool foldSelectShuffle(Instruction &
I,
bool FromReduction =
false);
159 bool foldInterleaveIntrinsics(Instruction &
I);
160 bool foldDeinterleaveIntrinsics(Instruction &
I);
161 bool foldBitcastOfVPLoad(Instruction &
I);
162 bool foldBitOrderReverseAndSwap(Instruction &
I);
163 bool shrinkType(Instruction &
I);
164 bool shrinkLoadForShuffles(Instruction &
I);
165 bool shrinkPhiOfShuffles(Instruction &
I);
166 bool foldDeinterleaveInterleavePair(Instruction &
I);
168 void replaceValue(Instruction &Old,
Value &New,
bool Erase =
true) {
174 Worklist.pushUsersToWorkList(*NewI);
175 Worklist.pushValue(NewI);
192 SmallPtrSet<Value *, 4> Visited;
197 OpI,
nullptr,
nullptr, [&](
Value *V) {
202 NextInst = NextInst->getNextNode();
207 Worklist.pushUsersToWorkList(*OpI);
208 Worklist.pushValue(OpI);
226 return X->getType() ==
Y->getType() &&
235 Load->getFunction()->hasFnAttribute(Attribute::SanitizeMemTag) ||
241 Type *ScalarTy =
Load->getType()->getScalarType();
243 unsigned MinVectorSize =
TTI.getMinVectorRegisterBitWidth();
244 if (!ScalarSize || !MinVectorSize || MinVectorSize % ScalarSize != 0 ||
251bool VectorCombine::vectorizeLoadInsert(
Instruction &
I) {
277 Value *SrcPtr =
Load->getPointerOperand()->stripPointerCasts();
280 unsigned MinVecNumElts = MinVectorSize / ScalarSize;
281 auto *MinVecTy = VectorType::get(ScalarTy, MinVecNumElts,
false);
282 unsigned OffsetEltIndex = 0;
290 unsigned OffsetBitWidth =
DL->getIndexTypeSizeInBits(SrcPtr->
getType());
291 APInt
Offset(OffsetBitWidth, 0);
301 uint64_t ScalarSizeInBytes = ScalarSize / 8;
302 if (
Offset.urem(ScalarSizeInBytes) != 0)
306 APInt OffsetEltIndexAP =
Offset.udiv(ScalarSizeInBytes);
307 if (OffsetEltIndexAP.
uge(MinVecNumElts))
325 unsigned AS =
Load->getPointerAddressSpace();
344 unsigned OutputNumElts = Ty->getNumElements();
346 assert(OffsetEltIndex < MinVecNumElts &&
"Address offset too big");
347 Mask[0] = OffsetEltIndex;
354 if (OldCost < NewCost || !NewCost.
isValid())
365 replaceValue(
I, *VecLd);
373bool VectorCombine::widenSubvectorLoad(Instruction &
I) {
376 if (!Shuf->isIdentityWithPadding())
382 unsigned OpIndex =
any_of(Shuf->getShuffleMask(), [&NumOpElts](
int M) {
383 return M >= (int)(NumOpElts);
403 unsigned AS =
Load->getPointerAddressSpace();
418 if (OldCost < NewCost || !NewCost.
isValid())
425 replaceValue(
I, *VecLd);
432ExtractElementInst *VectorCombine::getShuffleExtract(
433 ExtractElementInst *Ext0, ExtractElementInst *Ext1,
437 assert(Index0C && Index1C &&
"Expected constant extract indexes");
439 unsigned Index0 = Index0C->getZExtValue();
440 unsigned Index1 = Index1C->getZExtValue();
443 if (Index0 == Index1)
467 if (PreferredExtractIndex == Index0)
469 if (PreferredExtractIndex == Index1)
473 return Index0 > Index1 ? Ext0 : Ext1;
481bool VectorCombine::isExtractExtractCheap(ExtractElementInst *Ext0,
482 ExtractElementInst *Ext1,
483 const Instruction &
I,
484 ExtractElementInst *&ConvertToShuffle,
485 unsigned PreferredExtractIndex) {
488 assert(Ext0IndexC && Ext1IndexC &&
"Expected constant extract indexes");
490 unsigned Opcode =
I.getOpcode();
503 assert((Opcode == Instruction::ICmp || Opcode == Instruction::FCmp) &&
504 "Expected a compare");
514 unsigned Ext0Index = Ext0IndexC->getZExtValue();
515 unsigned Ext1Index = Ext1IndexC->getZExtValue();
529 unsigned BestExtIndex = Extract0Cost > Extract1Cost ? Ext0Index : Ext1Index;
530 unsigned BestInsIndex = Extract0Cost > Extract1Cost ? Ext1Index : Ext0Index;
531 InstructionCost CheapExtractCost = std::min(Extract0Cost, Extract1Cost);
536 if (Ext0Src == Ext1Src && Ext0Index == Ext1Index) {
541 bool HasUseTax = Ext0 == Ext1 ? !Ext0->
hasNUses(2)
543 OldCost = CheapExtractCost + ScalarOpCost;
544 NewCost = VectorOpCost + CheapExtractCost + HasUseTax * CheapExtractCost;
548 OldCost = Extract0Cost + Extract1Cost + ScalarOpCost;
549 NewCost = VectorOpCost + CheapExtractCost +
554 ConvertToShuffle = getShuffleExtract(Ext0, Ext1, PreferredExtractIndex);
555 if (ConvertToShuffle) {
567 SmallVector<int> ShuffleMask(FixedVecTy->getNumElements(),
569 ShuffleMask[BestInsIndex] = BestExtIndex;
571 VecTy, VecTy, ShuffleMask,
CostKind, 0,
572 nullptr, {ConvertToShuffle});
575 VecTy, VecTy, {},
CostKind, 0,
nullptr,
580 LLVM_DEBUG(
dbgs() <<
"Found a binop of extractions: " <<
I <<
"\n OldCost: "
581 << OldCost <<
" vs NewCost: " << NewCost <<
"\n");
586 return OldCost < NewCost;
598 ShufMask[NewIndex] = OldIndex;
599 return Builder.CreateShuffleVector(Vec, ShufMask,
"shift");
651 V1,
"foldExtExtBinop");
656 VecBOInst->copyIRFlags(&
I);
662bool VectorCombine::foldExtractExtract(Instruction &
I) {
683 unsigned NumElts = FixedVecTy->getNumElements();
684 if (C0 >= NumElts || C1 >= NumElts)
700 ExtractElementInst *ExtractToChange;
701 if (isExtractExtractCheap(Ext0, Ext1,
I, ExtractToChange, InsertIndex))
707 if (ExtractToChange) {
708 unsigned CheapExtractIdx = ExtractToChange == Ext0 ? C1 : C0;
713 if (ExtractToChange == Ext0)
722 ? foldExtExtCmp(ExtOp0, ExtOp1, ExtIndex,
I)
723 : foldExtExtBinop(ExtOp0, ExtOp1, ExtIndex,
I);
726 replaceValue(
I, *NewExt);
732bool VectorCombine::foldInsExtFNeg(Instruction &
I) {
750 auto *DstVecScalarTy = DstVecTy->getScalarType();
752 if (!SrcVecTy || DstVecScalarTy != SrcVecTy->getScalarType())
757 unsigned NumDstElts = DstVecTy->getNumElements();
758 unsigned NumSrcElts = SrcVecTy->getNumElements();
759 if (ExtIdx > NumSrcElts || InsIdx >= NumDstElts || NumDstElts == 1)
765 SmallVector<int>
Mask(NumDstElts);
766 std::iota(
Mask.begin(),
Mask.end(), 0);
767 Mask[InsIdx] = (ExtIdx % NumDstElts) + NumDstElts;
783 bool NeedLenChg = SrcVecTy->getNumElements() != NumDstElts;
786 SmallVector<int> SrcMask;
789 SrcMask[ExtIdx % NumDstElts] = ExtIdx;
791 DstVecTy, SrcVecTy, SrcMask,
CostKind);
795 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
797 if (NewCost > OldCost)
800 Value *NewShuf, *LenChgShuf =
nullptr;
814 replaceValue(
I, *NewShuf);
820bool VectorCombine::foldInsExtBinop(Instruction &
I) {
821 BinaryOperator *VecBinOp, *SclBinOp;
853 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
855 if (NewCost > OldCost)
866 NewInst->copyIRFlags(VecBinOp);
867 NewInst->andIRFlags(SclBinOp);
872 replaceValue(
I, *NewBO);
878bool VectorCombine::foldBitOpOfCastops(Instruction &
I) {
881 if (!BinOp || !BinOp->isBitwiseLogicOp())
887 if (!LHSCast || !RHSCast) {
888 LLVM_DEBUG(
dbgs() <<
" One or both operands are not cast instructions\n");
894 if (CastOpcode != RHSCast->getOpcode())
898 switch (CastOpcode) {
899 case Instruction::BitCast:
900 case Instruction::Trunc:
901 case Instruction::SExt:
902 case Instruction::ZExt:
908 Value *LHSSrc = LHSCast->getOperand(0);
909 Value *RHSSrc = RHSCast->getOperand(0);
915 auto *SrcTy = LHSSrc->
getType();
916 auto *DstTy =
I.getType();
919 if (CastOpcode != Instruction::BitCast &&
924 if (!SrcTy->getScalarType()->isIntegerTy() ||
925 !DstTy->getScalarType()->isIntegerTy())
940 LHSCastCost + RHSCastCost;
951 if (!LHSCast->hasOneUse())
952 NewCost += LHSCastCost;
953 if (!RHSCast->hasOneUse())
954 NewCost += RHSCastCost;
957 <<
" NewCost=" << NewCost <<
"\n");
959 if (NewCost > OldCost)
964 BinOp->getName() +
".inner");
966 NewBinOp->copyIRFlags(BinOp);
980 replaceValue(
I, *Result);
989bool VectorCombine::foldBitOpOfCastConstant(Instruction &
I) {
1005 switch (CastOpcode) {
1006 case Instruction::BitCast:
1007 case Instruction::ZExt:
1008 case Instruction::SExt:
1009 case Instruction::Trunc:
1015 Value *LHSSrc = LHSCast->getOperand(0);
1017 auto *SrcTy = LHSSrc->
getType();
1018 auto *DstTy =
I.getType();
1021 if (CastOpcode != Instruction::BitCast &&
1026 if (!SrcTy->getScalarType()->isIntegerTy() ||
1027 !DstTy->getScalarType()->isIntegerTy())
1031 PreservedCastFlags RHSFlags;
1056 if (!LHSCast->hasOneUse())
1057 NewCost += LHSCastCost;
1059 LLVM_DEBUG(
dbgs() <<
"foldBitOpOfCastConstant: OldCost=" << OldCost
1060 <<
" NewCost=" << NewCost <<
"\n");
1062 if (NewCost > OldCost)
1067 LHSSrc, InvC,
I.getName() +
".inner");
1069 NewBinOp->copyIRFlags(&
I);
1089 replaceValue(
I, *Result);
1096bool VectorCombine::foldBitcastShuffle(Instruction &
I) {
1110 if (!DestTy || !SrcTy)
1113 unsigned DestEltSize = DestTy->getScalarSizeInBits();
1114 unsigned SrcEltSize = SrcTy->getScalarSizeInBits();
1115 if (SrcTy->getPrimitiveSizeInBits() % DestEltSize != 0)
1125 if (!(BCTy0 && BCTy0->getElementType() == DestTy->getElementType()) &&
1126 !(BCTy1 && BCTy1->getElementType() == DestTy->getElementType()))
1130 SmallVector<int, 16> NewMask;
1131 if (DestEltSize <= SrcEltSize) {
1134 if (SrcEltSize % DestEltSize != 0)
1136 unsigned ScaleFactor = SrcEltSize / DestEltSize;
1141 if (DestEltSize % SrcEltSize != 0)
1143 unsigned ScaleFactor = DestEltSize / SrcEltSize;
1150 unsigned NumSrcElts = SrcTy->getPrimitiveSizeInBits() / DestEltSize;
1151 auto *NewShuffleTy =
1153 auto *OldShuffleTy =
1155 unsigned NumOps = IsUnary ? 1 : 2;
1165 TargetTransformInfo::CastContextHint::None,
1170 TargetTransformInfo::CastContextHint::None,
1173 LLVM_DEBUG(
dbgs() <<
"Found a bitcasted shuffle: " <<
I <<
"\n OldCost: "
1174 << OldCost <<
" vs NewCost: " << NewCost <<
"\n");
1176 if (NewCost > OldCost || !NewCost.
isValid())
1184 replaceValue(
I, *Shuf);
1191bool VectorCombine::scalarizeOpOrCmp(Instruction &
I) {
1196 if (!UO && !BO && !CI && !
II)
1204 if (Arg->getType() !=
II->getType() &&
1214 for (User *U :
I.users())
1221 std::optional<uint64_t>
Index;
1223 auto Ops =
II ?
II->args() :
I.operands();
1232 if (OpTy->getElementCount().getKnownMinValue() <= InsIdx)
1238 else if (InsIdx != *Index)
1255 if (!
Index.has_value())
1259 Type *ScalarTy = VecTy->getScalarType();
1260 assert(VecTy->isVectorTy() &&
1263 "Unexpected types for insert element into binop or cmp");
1265 unsigned Opcode =
I.getOpcode();
1273 }
else if (UO || BO) {
1277 IntrinsicCostAttributes ScalarICA(
1278 II->getIntrinsicID(), ScalarTy,
1281 IntrinsicCostAttributes VectorICA(
1282 II->getIntrinsicID(), VecTy,
1289 Value *NewVecC =
nullptr;
1291 NewVecC =
simplifyCmpInst(CI->getPredicate(), VecCs[0], VecCs[1], SQ);
1294 simplifyUnOp(UO->getOpcode(), VecCs[0], UO->getFastMathFlags(), SQ);
1296 NewVecC =
simplifyBinOp(BO->getOpcode(), VecCs[0], VecCs[1], SQ);
1310 for (
auto [Idx,
Op, VecC, Scalar] :
enumerate(
Ops, VecCs, ScalarOps)) {
1312 II->getIntrinsicID(), Idx, &
TTI)))
1315 Instruction::InsertElement, VecTy,
CostKind, *Index, VecC, Scalar);
1316 OldCost += InsertCost;
1317 NewCost += !
Op->hasOneUse() * InsertCost;
1321 if (OldCost < NewCost || !NewCost.
isValid())
1331 ++NumScalarIntrinsic;
1334 for (
auto [OpIdx, Scalar, VecC] :
enumerate(ScalarOps, VecCs))
1341 Scalar = Builder.
CreateCmp(CI->getPredicate(), ScalarOps[0], ScalarOps[1]);
1347 Scalar->setName(
I.getName() +
".scalar");
1352 ScalarInst->copyIRFlags(&
I);
1355 replaceValue(
I, *Insert);
1362bool VectorCombine::foldExtractedCmps(Instruction &
I) {
1367 if (!BI || !
I.getType()->isIntegerTy(1))
1372 Value *
B0 =
I.getOperand(0), *
B1 =
I.getOperand(1);
1375 CmpPredicate
P0,
P1;
1394 ExtractElementInst *ConvertToShuf = getShuffleExtract(Ext0, Ext1,
CostKind);
1397 assert((ConvertToShuf == Ext0 || ConvertToShuf == Ext1) &&
1398 "Unknown ExtractElementInst");
1403 unsigned CmpOpcode =
1409 if (Index0 >= VecTy->getNumElements() || Index1 >= VecTy->getNumElements())
1421 Ext0Cost + Ext1Cost + CmpCost * 2 +
1427 int CheapIndex = ConvertToShuf == Ext0 ? Index1 : Index0;
1428 int ExpensiveIndex = ConvertToShuf == Ext0 ? Index0 : Index1;
1433 ShufMask[CheapIndex] = ExpensiveIndex;
1438 NewCost += Ext0->
hasOneUse() ? 0 : Ext0Cost;
1439 NewCost += Ext1->
hasOneUse() ? 0 : Ext1Cost;
1444 if (OldCost < NewCost || !NewCost.
isValid())
1454 Value *
LHS = ConvertToShuf == Ext0 ? Shuf : VCmp;
1455 Value *
RHS = ConvertToShuf == Ext0 ? VCmp : Shuf;
1458 replaceValue(
I, *NewExt);
1485bool VectorCombine::foldSelectsFromBitcast(Instruction &
I) {
1492 if (!SrcVecTy || !DstVecTy)
1502 if (SrcEltBits != 32 && SrcEltBits != 64)
1505 if (!DstEltTy->
isIntegerTy() || DstEltBits >= SrcEltBits)
1522 if (!ScalarSelCost.
isValid() || ScalarSelCost == 0)
1525 unsigned MinSelects = (VecSelCost.
getValue() / ScalarSelCost.
getValue()) + 1;
1528 if (!BC->hasNUsesOrMore(MinSelects))
1533 DenseMap<Value *, SmallVector<SelectInst *, 8>> CondToSelects;
1535 for (User *U : BC->users()) {
1540 for (User *ExtUser : Ext->users()) {
1544 Cond->getType()->isIntegerTy(1))
1549 if (CondToSelects.
empty())
1552 bool MadeChange =
false;
1553 Value *SrcVec = BC->getOperand(0);
1556 for (
auto [
Cond, Selects] : CondToSelects) {
1558 if (Selects.size() < MinSelects) {
1559 LLVM_DEBUG(
dbgs() <<
"VectorCombine: foldSelectsFromBitcast not "
1560 <<
"profitable (VecCost=" << VecSelCost
1561 <<
", ScalarCost=" << ScalarSelCost
1562 <<
", NumSelects=" << Selects.size() <<
")\n");
1567 auto InsertPt = std::next(BC->getIterator());
1571 InsertPt = std::next(CondInst->getIterator());
1579 for (SelectInst *Sel : Selects) {
1581 Value *Idx = Ext->getIndexOperand();
1585 replaceValue(*Sel, *NewExt);
1590 <<
" selects into vector select\n");
1604 unsigned ReductionOpc =
1610 CostBeforeReduction =
1611 TTI.getCastInstrCost(RedOp->getOpcode(), VecRedTy, ExtType,
1613 CostAfterReduction =
1614 TTI.getExtendedReductionCost(ReductionOpc, IsUnsigned,
II.getType(),
1618 if (RedOp &&
II.getIntrinsicID() == Intrinsic::vector_reduce_add &&
1624 (Op0->
getOpcode() == RedOp->getOpcode() || Op0 == Op1)) {
1631 TTI.getCastInstrCost(Op0->
getOpcode(), MulType, ExtType,
1634 TTI.getArithmeticInstrCost(Instruction::Mul, MulType,
CostKind);
1636 TTI.getCastInstrCost(RedOp->getOpcode(), VecRedTy, MulType,
1639 CostBeforeReduction = ExtCost * 2 + MulCost + Ext2Cost;
1640 CostAfterReduction =
TTI.getMulAccReductionCost(
1641 IsUnsigned, ReductionOpc,
II.getType(), ExtType,
CostKind);
1644 CostAfterReduction =
TTI.getArithmeticReductionCost(ReductionOpc, VecRedTy,
1648bool VectorCombine::foldBinopOfReductions(Instruction &
I) {
1651 if (BinOpOpc == Instruction::Sub)
1652 ReductionIID = Intrinsic::vector_reduce_add;
1656 if (ReductionIID == Intrinsic::vector_reduce_fadd ||
1657 ReductionIID == Intrinsic::vector_reduce_fmul)
1660 auto checkIntrinsicAndGetItsArgument = [](
Value *
V,
1665 if (
II->getIntrinsicID() == IID &&
II->hasOneUse())
1666 return II->getArgOperand(0);
1670 Value *V0 = checkIntrinsicAndGetItsArgument(
I.getOperand(0), ReductionIID);
1673 Value *
V1 = checkIntrinsicAndGetItsArgument(
I.getOperand(1), ReductionIID);
1678 if (
V1->getType() != VTy)
1682 unsigned ReductionOpc =
1695 CostOfRedOperand0 + CostOfRedOperand1 +
1698 if (NewCost >= OldCost || !NewCost.
isValid())
1702 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
1705 if (BinOpOpc == Instruction::Or)
1712 replaceValue(
I, *Rdx);
1721 unsigned NumScanned = 0;
1722 if (std::any_of(Begin, End, [&](
const Instruction &Instr) {
1736class ScalarizationResult {
1737 enum class StatusTy { Unsafe, Safe, SafeWithFreeze };
1742 ScalarizationResult(StatusTy Status,
Value *ToFreeze =
nullptr)
1743 : Status(Status), ToFreeze(ToFreeze) {}
1746 ScalarizationResult(
const ScalarizationResult &
Other) =
default;
1747 ~ScalarizationResult() {
1748 assert(!ToFreeze &&
"freeze() not called with ToFreeze being set");
1751 static ScalarizationResult unsafe() {
return {StatusTy::Unsafe}; }
1752 static ScalarizationResult safe() {
return {StatusTy::Safe}; }
1753 static ScalarizationResult safeWithFreeze(
Value *ToFreeze) {
1754 return {StatusTy::SafeWithFreeze, ToFreeze};
1758 bool isSafe()
const {
return Status == StatusTy::Safe; }
1760 bool isUnsafe()
const {
return Status == StatusTy::Unsafe; }
1763 bool isSafeWithFreeze()
const {
return Status == StatusTy::SafeWithFreeze; }
1768 Status = StatusTy::Unsafe;
1772 void freeze(IRBuilderBase &Builder, Instruction &UserI) {
1773 assert(isSafeWithFreeze() &&
1774 "should only be used when freezing is required");
1776 "UserI must be a user of ToFreeze");
1777 IRBuilder<>::InsertPointGuard Guard(Builder);
1782 if (
U.get() == ToFreeze)
1797 uint64_t NumElements = VecTy->getElementCount().getKnownMinValue();
1801 if (
C->getValue().ult(NumElements))
1802 return ScalarizationResult::safe();
1803 return ScalarizationResult::unsafe();
1808 return ScalarizationResult::unsafe();
1810 APInt Zero(IntWidth, 0);
1811 APInt MaxElts(IntWidth, NumElements);
1818 return ScalarizationResult::safe();
1819 return ScalarizationResult::unsafe();
1832 if (ValidIndices.
contains(IdxRange))
1833 return ScalarizationResult::safeWithFreeze(IdxBase);
1834 return ScalarizationResult::unsafe();
1854 unsigned GEPBits = GEPIndexTy->getBitWidth();
1855 uint64_t NumElements = VecTy->getElementCount().getKnownMinValue();
1857 uint64_t MaxLane = NumElements - 1;
1859 if (
C->getValue().uge(NumElements))
1861 MaxLane =
C->getZExtValue();
1864 Type *ElemTy = VecTy->getElementType();
1865 if (!
DL.typeSizeEqualsStoreSize(ElemTy))
1868 TypeSize ElemStride =
DL.getTypeStoreSize(ElemTy);
1885 unsigned WideBits = std::max(GEPBits, 128u);
1886 APInt MaxLaneValue(WideBits, MaxLane);
1887 APInt ByteOffset = MaxLaneValue;
1892 if (ByteOffset.
ugt(MaxGEPOffset))
1905 if (SrcBits >= DstBits)
1908 return Builder.CreateZExt(Idx, GEPIndexTy, Idx->
getName() +
".gepidx");
1920 C->getZExtValue() *
DL.getTypeStoreSize(ScalarType));
1957bool VectorCombine::foldInsertElementsToStores(Instruction &
I) {
1972 if (!
Insert->hasOneUse())
1976 InsertElements.
push_back({InsertVal, Idx});
1980 if (InsertElements.
empty())
1985 std::reverse(InsertElements.
begin(), InsertElements.
end());
1994 if (InsertElements.
size() == FVT->getNumElements()) {
1995 Value *FirstVal = InsertElements.
front().first;
1996 if (
all_of(InsertElements,
1997 [FirstVal](
const auto &Elt) {
return Elt.first == FirstVal; }))
2001 Value *SrcAddr =
Load->getPointerOperand()->stripPointerCasts();
2006 if (!
Load->isSimple() ||
Load->getParent() !=
SI->getParent() ||
2007 !
DL->typeSizeEqualsStoreSize(
Load->getType()->getScalarType()) ||
2008 SrcAddr !=
SI->getPointerOperand()->stripPointerCasts())
2018 for (
auto [InsertVal, Idx] : InsertElements) {
2019 auto ScalarizableIdx =
2021 if (ScalarizableIdx.isUnsafe())
2027 ScalarizableIdx.discard();
2033 ScalarizableIdx.discard();
2037 Instruction::Store,
SI->getValueOperand()->getType(),
SI->getAlign(),
2040 if (
Load->hasOneUse())
2045 for (
auto [InsertVal, Idx] : InsertElements) {
2048 Index = CIdx->getZExtValue();
2059 for (
auto [InsertVal, Idx] : InsertElements) {
2062 const Value *GEPIndices[] = {ConstantInt::get(Idx->
getType(), 0), Idx};
2067 for (
auto [InsertVal, Idx] : InsertElements) {
2069 std::max(
SI->getAlign(),
Load->getAlign()), InsertVal->
getType(), Idx,
2077 LLVM_DEBUG(
dbgs() <<
"Found an insert-elements vector store scalarization "
2080 <<
" NumInserts: " << InsertElements.size() <<
"\n"
2081 <<
" OldCost: " << OldCost <<
" vs NewCost: " << NewCost
2084 if (OldCost <= NewCost)
2087 for (
auto [InsertVal, Idx] : InsertElements) {
2088 auto ScalarizableIdx =
2090 assert(!ScalarizableIdx.isUnsafe() &&
"already checked above");
2092 if (ScalarizableIdx.isSafeWithFreeze())
2097 StoreInst *LastStore =
nullptr;
2098 for (
auto [InsertVal, Idx] : InsertElements) {
2099 auto ScalarizableIdx =
2101 if (ScalarizableIdx.isUnsafe())
2104 IntegerType *GEPIndexTy =
2109 SI->getValueOperand()->getType(),
SI->getPointerOperand(),
2110 {ConstantInt::get(GEPIdx->getType(), 0), GEPIdx});
2117 LastStore->
setMetadata(LLVMContext::MD_invariant_group,
nullptr);
2119 std::max(
SI->getAlign(),
Load->getAlign()), InsertVal->
getType(), Idx,
2124 replaceValue(
I, *LastStore);
2131bool VectorCombine::scalarizeLoad(Instruction &
I) {
2141 if (!LI->isSimple() || !
DL->typeSizeEqualsStoreSize(VecTy->getScalarType()))
2144 bool AllExtracts =
true;
2145 bool AllBitcasts =
true;
2147 unsigned NumInstChecked = 0;
2152 for (User *U : LI->users()) {
2154 if (!UI || UI->getParent() != LI->getParent())
2159 if (UI->use_empty())
2163 AllExtracts =
false;
2165 AllBitcasts =
false;
2169 for (Instruction &
I :
2170 make_range(std::next(LI->getIterator()), UI->getIterator())) {
2177 LastCheckedInst = UI;
2182 return scalarizeLoadExtract(LI, VecTy, Ptr);
2184 return scalarizeLoadBitcast(LI, VecTy, Ptr);
2189bool VectorCombine::scalarizeLoadExtract(LoadInst *LI, VectorType *VecTy,
2194 DenseMap<ExtractElementInst *, ScalarizationResult> NeedFreeze;
2195 DenseMap<ExtractElementInst *, IntegerType *> GEPIndexInfos;
2198 for (
auto &Pair : NeedFreeze)
2199 Pair.second.discard();
2207 for (User *U : LI->
users()) {
2212 if (ScalarIdx.isUnsafe())
2218 ScalarIdx.discard();
2224 if (ScalarIdx.isSafeWithFreeze()) {
2225 NeedFreeze.try_emplace(UI, ScalarIdx);
2226 ScalarIdx.discard();
2232 Index ?
Index->getZExtValue() : -1);
2238 if (!Index && UI->getIndexOperand()->getType()->getIntegerBitWidth() <
2241 Instruction::ZExt, GEPIndex, UI->getIndexOperand()->getType(),
2245 LLVM_DEBUG(
dbgs() <<
"Found all extractions of a vector load: " << *LI
2246 <<
"\n LoadExtractCost: " << OriginalCost
2247 <<
" vs ScalarizedCost: " << ScalarizedCost <<
"\n");
2249 if (ScalarizedCost >= OriginalCost)
2256 Type *ElemType = VecTy->getElementType();
2259 for (User *U : LI->
users()) {
2261 Value *Idx = EI->getIndexOperand();
2264 if (
auto It = NeedFreeze.find(EI); It != NeedFreeze.end())
2268 auto It = GEPIndexInfos.
find(EI);
2270 "Missing scalarized GEP index information");
2273 VecTy, Ptr, {ConstantInt::get(GEPIdx->
getType(), 0), GEPIdx});
2275 Builder.
CreateLoad(ElemType,
GEP, EI->getName() +
".scalar"));
2277 Align ScalarOpAlignment =
2279 NewLoad->setAlignment(ScalarOpAlignment);
2282 size_t Offset = ConstIdx->getZExtValue() *
DL->getTypeStoreSize(ElemType);
2287 replaceValue(*EI, *NewLoad,
false);
2290 FailureGuard.release();
2295bool VectorCombine::scalarizeLoadBitcast(LoadInst *LI, VectorType *VecTy,
2301 Type *TargetScalarType =
nullptr;
2302 unsigned VecBitWidth =
DL->getTypeSizeInBits(VecTy);
2304 for (User *U : LI->
users()) {
2307 Type *DestTy = BC->getDestTy();
2311 unsigned DestBitWidth =
DL->getTypeSizeInBits(DestTy);
2312 if (DestBitWidth != VecBitWidth)
2316 if (!TargetScalarType)
2317 TargetScalarType = DestTy;
2318 else if (TargetScalarType != DestTy)
2326 if (!TargetScalarType)
2334 LLVM_DEBUG(
dbgs() <<
"Found vector load feeding only bitcasts: " << *LI
2335 <<
"\n OriginalCost: " << OriginalCost
2336 <<
" vs ScalarizedCost: " << ScalarizedCost <<
"\n");
2338 if (ScalarizedCost >= OriginalCost)
2349 ScalarLoad->copyMetadata(*LI);
2352 for (User *U : LI->
users()) {
2354 replaceValue(*BC, *ScalarLoad,
false);
2360bool VectorCombine::scalarizeExtExtract(Instruction &
I) {
2375 Type *ScalarDstTy = DstTy->getElementType();
2376 if (
DL->getTypeSizeInBits(SrcTy) !=
DL->getTypeSizeInBits(ScalarDstTy))
2382 unsigned ExtCnt = 0;
2383 bool ExtLane0 =
false;
2384 for (User *U : Ext->users()) {
2398 Instruction::And, ScalarDstTy,
CostKind,
2401 (ExtCnt - ExtLane0) *
2403 Instruction::LShr, ScalarDstTy,
CostKind,
2406 if (ScalarCost > VectorCost)
2409 Value *ScalarV = Ext->getOperand(0);
2416 SmallDenseSet<ConstantInt *, 8> ExtractedLanes;
2417 bool AllExtractsTriggerUB =
true;
2418 ExtractElementInst *LastExtract =
nullptr;
2420 for (User *U : Ext->users()) {
2423 AllExtractsTriggerUB =
false;
2427 if (!LastExtract || LastExtract->
comesBefore(Extract))
2428 LastExtract = Extract;
2430 if (ExtractedLanes.
size() != DstTy->getNumElements() ||
2431 !AllExtractsTriggerUB ||
2439 uint64_t SrcEltSizeInBits =
DL->getTypeSizeInBits(SrcTy->getElementType());
2440 uint64_t TotalBits =
DL->getTypeSizeInBits(SrcTy);
2443 Value *
Mask = ConstantInt::get(PackedTy, EltBitMask);
2444 for (User *U : Ext->users()) {
2450 ? (TotalBits - SrcEltSizeInBits - Idx * SrcEltSizeInBits)
2451 : (Idx * SrcEltSizeInBits);
2454 U->replaceAllUsesWith(
And);
2462bool VectorCombine::foldConcatOfBoolMasks(Instruction &
I) {
2463 Type *Ty =
I.getType();
2468 if (
DL->isBigEndian())
2495 if (ShAmtX > ShAmtY) {
2503 uint64_t ShAmtDiff = ShAmtY - ShAmtX;
2504 unsigned NumSHL = (ShAmtX > 0) + (ShAmtY > 0);
2509 MaskTy->getNumElements() != ShAmtDiff ||
2510 MaskTy->getNumElements() > (
BitWidth / 2))
2515 Type::getIntNTy(Ty->
getContext(), ConcatTy->getNumElements());
2516 auto *MaskIntTy = Type::getIntNTy(Ty->
getContext(), ShAmtDiff);
2519 std::iota(ConcatMask.begin(), ConcatMask.end(), 0);
2536 if (Ty != ConcatIntTy)
2542 LLVM_DEBUG(
dbgs() <<
"Found a concatenation of bitcasted bool masks: " <<
I
2543 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
2546 if (NewCost > OldCost)
2556 if (Ty != ConcatIntTy) {
2566 replaceValue(
I, *Result);
2572bool VectorCombine::foldPermuteOfBinops(Instruction &
I) {
2573 BinaryOperator *BinOp;
2574 ArrayRef<int> OuterMask;
2582 Value *Op00, *Op01, *Op10, *Op11;
2583 ArrayRef<int> Mask0, Mask1;
2588 if (!Match0 && !Match1)
2601 if (!ShuffleDstTy || !BinOpTy || !Op0Ty || !Op1Ty)
2604 unsigned NumSrcElts = BinOpTy->getNumElements();
2609 any_of(OuterMask, [NumSrcElts](
int M) {
return M >= (int)NumSrcElts; }))
2613 SmallVector<int> NewMask0, NewMask1;
2614 for (
int M : OuterMask) {
2615 if (M < 0 || M >= (
int)NumSrcElts) {
2619 NewMask0.
push_back(Match0 ? Mask0[M] : M);
2620 NewMask1.
push_back(Match1 ? Mask1[M] : M);
2624 unsigned NumOpElts = Op0Ty->getNumElements();
2625 bool IsIdentity0 = ShuffleDstTy == Op0Ty &&
2626 all_of(NewMask0, [NumOpElts](
int M) {
return M < (int)NumOpElts; }) &&
2628 bool IsIdentity1 = ShuffleDstTy == Op1Ty &&
2629 all_of(NewMask1, [NumOpElts](
int M) {
return M < (int)NumOpElts; }) &&
2638 ShuffleDstTy, BinOpTy, OuterMask,
CostKind,
2639 0,
nullptr, {BinOp}, &
I);
2641 NewCost += BinOpCost;
2647 OldCost += Shuf0Cost;
2649 NewCost += Shuf0Cost;
2655 OldCost += Shuf1Cost;
2657 NewCost += Shuf1Cost;
2665 Op0Ty, NewMask0,
CostKind, 0,
nullptr, {Op00, Op01});
2669 Op1Ty, NewMask1,
CostKind, 0,
nullptr, {Op10, Op11});
2671 LLVM_DEBUG(
dbgs() <<
"Found a shuffle feeding a shuffled binop: " <<
I
2672 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
2676 if (NewCost > OldCost)
2687 NewInst->copyIRFlags(BinOp);
2691 replaceValue(
I, *NewBO);
2697bool VectorCombine::foldShuffleOfBinops(Instruction &
I) {
2698 ArrayRef<int> OldMask;
2705 if (
LHS->getOpcode() !=
RHS->getOpcode())
2709 bool IsCommutative =
false;
2718 IsCommutative = BinaryOperator::isCommutative(BO->getOpcode());
2729 if (!ShuffleDstTy || !BinResTy || !BinOpTy ||
X->getType() !=
Z->getType())
2732 bool SameBinOp =
LHS ==
RHS;
2733 unsigned NumSrcElts = BinOpTy->getNumElements();
2736 if (IsCommutative &&
X != Z &&
Y != W && (
X == W ||
Y == Z))
2739 auto ConvertToUnary = [NumSrcElts](
int &
M) {
2740 if (M >= (
int)NumSrcElts)
2744 SmallVector<int> NewMask0(OldMask);
2753 SmallVector<int> NewMask1(OldMask);
2772 ShuffleDstTy, BinResTy, OldMask,
CostKind, 0,
2782 ArrayRef<int> InnerMask;
2784 m_Mask(InnerMask)))) &&
2787 [NumSrcElts](
int M) {
return M < (int)NumSrcElts; })) {
2799 bool ReducedInstCount =
false;
2800 ReducedInstCount |= MergeInner(
X, 0, NewMask0,
CostKind);
2801 ReducedInstCount |= MergeInner(
Y, 0, NewMask1,
CostKind);
2802 ReducedInstCount |= MergeInner(Z, NumSrcElts, NewMask0,
CostKind);
2803 ReducedInstCount |= MergeInner(W, NumSrcElts, NewMask1,
CostKind);
2804 bool SingleSrcBinOp = (
X ==
Y) && (Z == W) && (NewMask0 == NewMask1);
2816 I.getType()->getScalarType()->isIntegerTy(1) &&
2820 auto *ShuffleCmpTy =
2823 SK0, ShuffleCmpTy, BinOpTy, NewMask0,
CostKind, 0,
nullptr, {
X,
Z});
2824 if (!SingleSrcBinOp)
2834 PredLHS,
CostKind, Op0Info, Op1Info);
2844 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
2851 if (ReducedInstCount ? (NewCost > OldCost) : (NewCost >= OldCost))
2860 : Builder.
CreateCmp(PredLHS, Shuf0, Shuf1);
2864 NewInst->copyIRFlags(
LHS);
2865 NewInst->andIRFlags(
RHS);
2870 replaceValue(
I, *NewBO);
2877bool VectorCombine::foldShuffleOfSelects(Instruction &
I) {
2879 Value *C1, *
T1, *F1, *C2, *T2, *F2;
2890 if (!C1VecTy || !C2VecTy || C1VecTy != C2VecTy)
2896 if (((SI0FOp ==
nullptr) != (SI1FOp ==
nullptr)) ||
2897 ((SI0FOp !=
nullptr) &&
2898 (SI0FOp->getFastMathFlags() != SI1FOp->getFastMathFlags())))
2904 auto SelOp = Instruction::Select;
2912 CostSel1 + CostSel2 +
2914 {
I.getOperand(0),
I.getOperand(1)}, &
I);
2918 Mask,
CostKind, 0,
nullptr, {C1, C2});
2928 if (!Sel1->hasOneUse())
2929 NewCost += CostSel1;
2930 if (!Sel2->hasOneUse())
2931 NewCost += CostSel2;
2934 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
2936 if (NewCost > OldCost)
2945 NewSel = Builder.
CreateSelectFMF(ShuffleCmp, ShuffleTrue, ShuffleFalse,
2946 SI0FOp->getFastMathFlags());
2948 NewSel = Builder.
CreateSelect(ShuffleCmp, ShuffleTrue, ShuffleFalse);
2953 replaceValue(
I, *NewSel);
2959bool VectorCombine::foldShuffleOfCastops(Instruction &
I) {
2961 ArrayRef<int> OldMask;
2970 if (!C0 || (IsBinaryShuffle && !C1))
2977 if (!IsBinaryShuffle && Opcode == Instruction::BitCast)
2980 if (IsBinaryShuffle) {
2981 if (C0->getSrcTy() != C1->getSrcTy())
2984 if (Opcode != C1->getOpcode()) {
2986 Opcode = Instruction::SExt;
2995 if (!ShuffleDstTy || !CastDstTy || !CastSrcTy)
2998 unsigned NumSrcElts = CastSrcTy->getNumElements();
2999 unsigned NumDstElts = CastDstTy->getNumElements();
3000 assert((NumDstElts == NumSrcElts || Opcode == Instruction::BitCast) &&
3001 "Only bitcasts expected to alter src/dst element counts");
3005 if (NumDstElts != NumSrcElts && (NumSrcElts % NumDstElts) != 0 &&
3006 (NumDstElts % NumSrcElts) != 0)
3009 SmallVector<int, 16> NewMask;
3010 if (NumSrcElts >= NumDstElts) {
3013 assert(NumSrcElts % NumDstElts == 0 &&
"Unexpected shuffle mask");
3014 unsigned ScaleFactor = NumSrcElts / NumDstElts;
3019 assert(NumDstElts % NumSrcElts == 0 &&
"Unexpected shuffle mask");
3020 unsigned ScaleFactor = NumDstElts / NumSrcElts;
3025 auto *NewShuffleDstTy =
3034 if (IsBinaryShuffle)
3049 if (IsBinaryShuffle) {
3059 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
3061 if (NewCost > OldCost)
3065 if (IsBinaryShuffle)
3075 NewInst->copyIRFlags(C0);
3076 if (IsBinaryShuffle)
3077 NewInst->andIRFlags(C1);
3081 replaceValue(
I, *Cast);
3091bool VectorCombine::foldShuffleOfShuffles(Instruction &
I) {
3092 ArrayRef<int> OuterMask;
3093 Value *OuterV0, *OuterV1;
3098 ArrayRef<int> InnerMask0, InnerMask1;
3099 Value *X0, *X1, *Y0, *Y1;
3104 if (!Match0 && !Match1)
3109 SmallVector<int, 16> PoisonMask1;
3114 InnerMask1 = PoisonMask1;
3118 X0 = Match0 ? X0 : OuterV0;
3119 Y0 = Match0 ? Y0 : OuterV0;
3120 X1 = Match1 ? X1 : OuterV1;
3121 Y1 = Match1 ? Y1 : OuterV1;
3125 if (!ShuffleDstTy || !ShuffleSrcTy || !ShuffleImmTy ||
3129 unsigned NumSrcElts = ShuffleSrcTy->getNumElements();
3130 unsigned NumImmElts = ShuffleImmTy->getNumElements();
3135 SmallVector<int, 16> NewMask(OuterMask);
3136 Value *NewX =
nullptr, *NewY =
nullptr;
3137 for (
int &M : NewMask) {
3138 Value *Src =
nullptr;
3139 if (0 <= M && M < (
int)NumImmElts) {
3143 Src =
M >= (int)NumSrcElts ? Y0 : X0;
3144 M =
M >= (int)NumSrcElts ? (M - NumSrcElts) :
M;
3146 }
else if (M >= (
int)NumImmElts) {
3151 Src =
M >= (int)NumSrcElts ? Y1 : X1;
3152 M =
M >= (int)NumSrcElts ? (M - NumSrcElts) :
M;
3156 assert(0 <= M && M < (
int)NumSrcElts &&
"Unexpected shuffle mask index");
3165 if (!NewX || NewX == Src) {
3169 if (!NewY || NewY == Src) {
3188 replaceValue(
I, *NewX);
3205 bool IsUnary =
all_of(NewMask, [&](
int M) {
return M < (int)NumSrcElts; });
3211 nullptr, {NewX, NewY});
3213 NewCost += InnerCost0;
3215 NewCost += InnerCost1;
3218 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
3220 if (NewCost > OldCost)
3224 replaceValue(
I, *Shuf);
3240bool VectorCombine::foldShufflesOfLengthChangingShuffles(Instruction &
I) {
3245 unsigned ChainLength = 0;
3246 SmallVector<int>
Mask;
3247 SmallVector<int> YMask;
3257 ArrayRef<int> OuterMask;
3258 Value *OuterV0, *OuterV1;
3259 if (ChainLength != 0 && !Trunk->
hasOneUse())
3262 m_Mask(OuterMask))))
3264 if (OuterV0->
getType() != TrunkType) {
3270 ArrayRef<int> InnerMask0, InnerMask1;
3276 bool Match0Leaf = Match0 && A0->
getType() !=
I.getType();
3277 bool Match1Leaf = Match1 && A1->
getType() !=
I.getType();
3278 if (Match0Leaf == Match1Leaf) {
3284 SmallVector<int> CommutedOuterMask;
3291 for (
int &M : CommutedOuterMask) {
3294 if (M < (
int)NumTrunkElts)
3299 OuterMask = CommutedOuterMask;
3318 int NumLeafElts = YType->getNumElements();
3319 SmallVector<int> LocalYMask(InnerMask1);
3320 for (
int &M : LocalYMask) {
3321 if (M >= NumLeafElts)
3331 Mask.assign(OuterMask);
3332 YMask.
assign(LocalYMask);
3333 OldCost = NewCost = LocalOldCost;
3340 SmallVector<int> NewYMask(YMask);
3342 for (
auto [CombinedM, LeafM] :
llvm::zip(NewYMask, LocalYMask)) {
3343 if (LeafM == -1 || CombinedM == LeafM)
3345 if (CombinedM == -1) {
3355 SmallVector<int> NewMask;
3356 NewMask.
reserve(NumTrunkElts);
3357 for (
int M : Mask) {
3358 if (M < 0 || M >=
static_cast<int>(NumTrunkElts))
3373 if (LocalNewCost >= NewCost && LocalOldCost < LocalNewCost - NewCost)
3377 if (ChainLength == 1) {
3378 dbgs() <<
"Found chain of shuffles fed by length-changing shuffles: "
3381 dbgs() <<
" next chain link: " << *Trunk <<
'\n'
3382 <<
" old cost: " << (OldCost + LocalOldCost)
3383 <<
" new cost: " << LocalNewCost <<
'\n';
3388 OldCost += LocalOldCost;
3389 NewCost = LocalNewCost;
3393 if (ChainLength <= 1)
3401 return M < 0 || M >=
static_cast<int>(NumTrunkElts);
3404 for (
int &M : Mask) {
3405 if (M >=
static_cast<int>(NumTrunkElts))
3406 M = YMask[
M - NumTrunkElts];
3410 replaceValue(
I, *Root);
3417 replaceValue(
I, *Root);
3423bool VectorCombine::foldShuffleOfIntrinsics(Instruction &
I) {
3425 ArrayRef<int> OldMask;
3435 if (IID != II1->getIntrinsicID())
3444 if (!ShuffleDstTy || !II0Ty)
3450 for (
unsigned I = 0,
E = II0->arg_size();
I !=
E; ++
I) {
3451 Value *Arg0 = II0->getArgOperand(
I);
3452 Value *Arg1 = II1->getArgOperand(
I);
3469 II0Ty, OldMask,
CostKind, 0,
nullptr, {II0, II1}, &
I);
3473 SmallDenseSet<std::pair<Value *, Value *>> SeenOperandPairs;
3474 for (
unsigned I = 0,
E = II0->arg_size();
I !=
E; ++
I) {
3476 NewArgsTy.
push_back(II0->getArgOperand(
I)->getType());
3480 ShuffleDstTy->getNumElements());
3482 std::pair<Value *, Value *> OperandPair =
3483 std::make_pair(II0->getArgOperand(
I), II1->getArgOperand(
I));
3484 if (!SeenOperandPairs.
insert(OperandPair).second) {
3490 CostKind, 0,
nullptr, {II0->getArgOperand(
I), II1->getArgOperand(
I)});
3493 IntrinsicCostAttributes NewAttr(IID, ShuffleDstTy, NewArgsTy);
3496 if (!II0->hasOneUse())
3498 if (II1 != II0 && !II1->hasOneUse())
3502 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
3505 if (NewCost > OldCost)
3509 SmallDenseMap<std::pair<Value *, Value *>,
Value *> ShuffleCache;
3510 for (
unsigned I = 0,
E = II0->arg_size();
I !=
E; ++
I)
3514 std::pair<Value *, Value *> OperandPair =
3515 std::make_pair(II0->getArgOperand(
I), II1->getArgOperand(
I));
3516 auto It = ShuffleCache.
find(OperandPair);
3517 if (It != ShuffleCache.
end()) {
3523 II1->getArgOperand(
I), OldMask);
3524 ShuffleCache[OperandPair] = Shuf;
3532 NewInst->copyIRFlags(II0);
3533 NewInst->andIRFlags(II1);
3536 replaceValue(
I, *NewIntrinsic);
3542bool VectorCombine::foldPermuteOfIntrinsic(Instruction &
I) {
3554 if (!ShuffleDstTy || !IntrinsicSrcTy)
3558 unsigned NumSrcElts = IntrinsicSrcTy->getNumElements();
3559 if (
any_of(Mask, [NumSrcElts](
int M) {
return M >= (int)NumSrcElts; }))
3572 IntrinsicSrcTy, Mask,
CostKind, 0,
nullptr, {V0}, &
I);
3576 for (
unsigned I = 0,
E = II0->arg_size();
I !=
E; ++
I) {
3578 NewArgsTy.
push_back(II0->getArgOperand(
I)->getType());
3582 ShuffleDstTy->getNumElements());
3585 ArgTy, VecTy, Mask,
CostKind, 0,
nullptr,
3586 {II0->getArgOperand(
I)});
3589 IntrinsicCostAttributes NewAttr(IID, ShuffleDstTy, NewArgsTy);
3594 if (!II0->hasOneUse())
3597 LLVM_DEBUG(
dbgs() <<
"Found a permute of intrinsic: " <<
I <<
"\n OldCost: "
3598 << OldCost <<
" vs NewCost: " << NewCost <<
"\n");
3600 if (NewCost > OldCost)
3605 for (
unsigned I = 0,
E = II0->arg_size();
I !=
E; ++
I) {
3618 NewInst->copyIRFlags(II0);
3620 replaceValue(
I, *NewIntrinsic);
3630 int M = SV->getMaskValue(Lane);
3633 if (
static_cast<unsigned>(M) < NumElts) {
3634 V = SV->getOperand(0);
3637 V = SV->getOperand(1);
3648 auto [U, Lane] = IL;
3661 unsigned NumElts = Ty->getNumElements();
3662 if (Item.
size() == NumElts || NumElts == 1 || Item.
size() % NumElts != 0)
3668 std::iota(ConcatMask.
begin(), ConcatMask.
end(), 0);
3674 unsigned NumSlices = Item.
size() / NumElts;
3679 for (
unsigned Slice = 0; Slice < NumSlices; ++Slice) {
3680 Value *SliceV = Item[Slice * NumElts].first;
3681 if (!SliceV || SliceV->
getType() != Ty)
3683 for (
unsigned Elt = 0; Elt < NumElts; ++Elt) {
3684 auto [V, Lane] = Item[Slice * NumElts + Elt];
3685 if (Lane !=
static_cast<int>(Elt) || SliceV != V)
3694 const DenseSet<std::pair<Value *, Use *>> &IdentityLeafs,
3695 const DenseSet<std::pair<Value *, Use *>> &SplatLeafs,
3696 const DenseSet<std::pair<Value *, Use *>> &ConcatLeafs,
3699 auto [FrontV, FrontLane] = Item.
front();
3701 if (IdentityLeafs.contains(std::make_pair(FrontV, From))) {
3704 if (SplatLeafs.contains(std::make_pair(FrontV, From))) {
3706 return Builder.CreateShuffleVector(FrontV, Mask);
3708 if (ConcatLeafs.contains(std::make_pair(FrontV, From))) {
3712 for (
unsigned S = 0; S <
Values.size(); ++S)
3713 Values[S] = Item[S * NumElts].first;
3715 while (
Values.size() > 1) {
3718 std::iota(Mask.begin(), Mask.end(), 0);
3720 for (
unsigned S = 0; S < NewValues.
size(); ++S)
3722 Builder.CreateShuffleVector(
Values[S * 2],
Values[S * 2 + 1], Mask);
3736 if (BCDstTy && BCSrcTy &&
3737 BCDstTy->getElementCount() != BCSrcTy->getElementCount()) {
3738 unsigned DstElts = BCDstTy->getNumElements();
3739 unsigned SrcElts = BCSrcTy->getNumElements();
3741 if (DstElts > SrcElts) {
3743 unsigned R = DstElts / SrcElts;
3744 if (Item.
size() % R != 0)
3746 for (
unsigned Idx = 0,
E = Item.
size(); Idx <
E; Idx += R) {
3747 auto [V, Lane] = Item[Idx];
3757 unsigned R = SrcElts / DstElts;
3758 for (
auto [V, Lane] : Item) {
3764 for (
unsigned J = 0; J < R; ++J)
3769 IdentityLeafs, SplatLeafs, ConcatLeafs,
3770 Builder, WorkList,
TTI);
3772 return Builder.CreateBitCast(
3777 unsigned NumOps =
I->getNumOperands() - (
II ? 1 : 0);
3779 for (
unsigned Idx = 0; Idx <
NumOps; Idx++) {
3782 Ops[Idx] =
II->getOperand(Idx);
3787 IdentityLeafs, SplatLeafs, ConcatLeafs, Builder, WorkList,
TTI);
3797 for (
const auto &Lane : Item)
3810 auto *
Value = Builder.CreateCmp(CI->getPredicate(),
Ops[0],
Ops[1]);
3820 auto *
Value = Builder.CreateCast(CI->getOpcode(),
Ops[0], DstTy);
3825 auto *
Value = Builder.CreateIntrinsic(DstTy,
II->getIntrinsicID(),
Ops);
3839bool VectorCombine::foldShuffleToIdentity(Instruction &
I) {
3841 if (!Ty ||
I.use_empty())
3845 for (
unsigned M = 0,
E = Ty->getNumElements(); M <
E; ++M)
3849 Candidates.
push_back(std::make_pair(Start, &*
I.use_begin()));
3850 DenseSet<std::pair<Value *, Use *>> IdentityLeafs, SplatLeafs, ConcatLeafs;
3851 unsigned NumVisited = 0;
3852 bool TraversedElCountChangingBitcast =
false;
3854 while (!Candidates.
empty()) {
3859 auto Item = ItemFrom.first;
3860 auto From = ItemFrom.second;
3861 auto [FrontV, FrontLane] = Item.front();
3868 if (FrontLane == 0 &&
3872 Value *FrontV = Item.front().first;
3874 E.value().second == (int)
E.index());
3876 IdentityLeafs.
insert(std::make_pair(FrontV, From));
3881 C &&
C->getSplatValue() &&
3883 Value *FrontV = Item.front().first;
3889 SplatLeafs.
insert(std::make_pair(FrontV, From));
3894 auto [FrontV, FrontLane] = Item.front();
3895 auto [
V, Lane] = IL;
3896 return !
V || (
V == FrontV && Lane == FrontLane);
3898 SplatLeafs.
insert(std::make_pair(FrontV, From));
3904 auto CheckLaneIsEquivalentToFirst = [Item](
InstLane IL) {
3905 Value *FrontV = Item.front().first;
3914 if (CI->getPredicate() !=
cast<CmpInst>(FrontV)->getPredicate())
3917 if (CI->getSrcTy()->getScalarType() !=
3922 SI->getOperand(0)->getType() !=
3929 II->getIntrinsicID() ==
3931 !
II->hasOperandBundles());
3938 BO && BO->isIntDivRem())
3945 }
else if (
isa<UnaryOperator, TruncInst, ZExtInst, SExtInst, FPToSIInst,
3946 FPToUIInst, SIToFPInst, UIToFPInst>(FrontV)) {
3953 if (BCDstTy && BCSrcTy) {
3954 ElementCount DstEC = BCDstTy->getElementCount();
3955 ElementCount SrcEC = BCSrcTy->getElementCount();
3956 if (DstEC == SrcEC) {
3959 &BitCast->getOperandUse(0));
3964 if (DstElts > SrcElts && DstElts % SrcElts == 0) {
3968 unsigned R = DstElts / SrcElts;
3970 bool Valid = Item.size() %
R == 0;
3971 for (
unsigned Idx = 0,
E = Item.size(); Valid && Idx <
E;
3973 auto [V0, L0] = Item[Idx];
3976 [](
InstLane IL) {
return IL.first !=
nullptr; })) {
3987 for (
unsigned J = 1; J <
R; ++J) {
3988 auto [VJ, LJ] = Item[Idx + J];
3989 if (!VJ || VJ != V0 || LJ != L0 + (
int)J) {
4000 TraversedElCountChangingBitcast =
true;
4001 Candidates.
emplace_back(NItem, &BitCast->getOperandUse(0));
4004 }
else if (SrcElts > DstElts && SrcElts % DstElts == 0) {
4007 unsigned R = SrcElts / DstElts;
4009 for (
auto [V, Lane] : Item) {
4015 for (
unsigned J = 0; J <
R; ++J)
4018 TraversedElCountChangingBitcast =
true;
4019 Candidates.
emplace_back(NItem, &BitCast->getOperandUse(0));
4025 &Sel->getOperandUse(0));
4027 &Sel->getOperandUse(1));
4029 &Sel->getOperandUse(2));
4033 !
II->hasOperandBundles()) {
4034 for (
unsigned Op = 0,
E =
II->getNumOperands() - 1;
Op <
E;
Op++) {
4038 Value *FrontV = Item.front().first;
4055 ConcatLeafs.
insert(std::make_pair(FrontV, From));
4062 if (NumVisited <= 1)
4068 if (NumVisited == 2 && TraversedElCountChangingBitcast)
4071 LLVM_DEBUG(
dbgs() <<
"Found a superfluous identity shuffle: " <<
I <<
"\n");
4078 ConcatLeafs, Builder, Worklist, &
TTI);
4079 replaceValue(
I, *V);
4086bool VectorCombine::foldShuffleFromReductions(Instruction &
I) {
4090 switch (
II->getIntrinsicID()) {
4091 case Intrinsic::vector_reduce_add:
4092 case Intrinsic::vector_reduce_mul:
4093 case Intrinsic::vector_reduce_and:
4094 case Intrinsic::vector_reduce_or:
4095 case Intrinsic::vector_reduce_xor:
4096 case Intrinsic::vector_reduce_smin:
4097 case Intrinsic::vector_reduce_smax:
4098 case Intrinsic::vector_reduce_umin:
4099 case Intrinsic::vector_reduce_umax:
4108 std::queue<Value *> Worklist;
4109 SmallPtrSet<Value *, 4> Visited;
4110 ShuffleVectorInst *Shuffle =
nullptr;
4114 while (!Worklist.empty()) {
4115 Value *CV = Worklist.front();
4127 if (CI->isBinaryOp()) {
4128 for (
auto *
Op : CI->operand_values())
4132 if (Shuffle && Shuffle != SV)
4149 for (
auto *V : Visited)
4150 for (
auto *U :
V->users())
4151 if (!Visited.contains(U) && U != &
I)
4154 FixedVectorType *VecType =
4158 FixedVectorType *ShuffleInputType =
4160 if (!ShuffleInputType)
4166 SmallVector<int> ConcatMask;
4168 sort(ConcatMask, [](
int X,
int Y) {
return (
unsigned)
X < (unsigned)
Y; });
4169 bool UsesSecondVec =
4170 any_of(ConcatMask, [&](
int M) {
return M >= (int)NumInputElts; });
4177 ShuffleInputType, ConcatMask,
CostKind);
4179 LLVM_DEBUG(
dbgs() <<
"Found a reduction feeding from a shuffle: " << *Shuffle
4181 LLVM_DEBUG(
dbgs() <<
" OldCost: " << OldCost <<
" vs NewCost: " << NewCost
4183 bool MadeChanges =
false;
4184 if (NewCost < OldCost) {
4188 LLVM_DEBUG(
dbgs() <<
"Created new shuffle: " << *NewShuffle <<
"\n");
4189 replaceValue(*Shuffle, *NewShuffle);
4195 MadeChanges |= foldSelectShuffle(*Shuffle,
true);
4216bool VectorCombine::foldShuffleChainsToReduce(Instruction &
I) {
4225 if (FVT->getNumElements() < 2)
4228 std::optional<Instruction::BinaryOps> CommonBinOp;
4229 std::optional<Intrinsic::ID> CommonCallOp;
4234 CommonBinOp = BO->getOpcode();
4236 CommonCallOp = MMI->getIntrinsicID();
4242 FastMathFlags CommonFMF;
4243 bool IsFloatReduction =
false;
4247 auto IsChainNode = [&](
Value *
V) {
4249 return CommonBinOp && BO->getOpcode() == *CommonBinOp;
4251 return CommonCallOp && MMI->getIntrinsicID() == *CommonCallOp;
4259 constexpr unsigned MaxChainNodes = 32;
4260 SmallSetVector<Value *, 16> Nodes;
4261 SmallSetVector<Value *, 4> Sources;
4262 unsigned NumVisited = 0;
4263 auto AddSource = [&](
Value *
V) {
4269 auto Walk = [&](
Value *
V,
auto &&Walk) ->
bool {
4272 if (++NumVisited > MaxChainNodes)
4274 if (!IsChainNode(V))
4275 return AddSource(V);
4280 if (!Walk(
U->getOperand(
I), Walk))
4289 return AddSource(V);
4291 if (!Walk(VecOpEE, Walk) || Nodes.
empty())
4298 for (
Value *V : Nodes) {
4304 if (!IsFloatReduction) {
4306 IsFloatReduction =
true;
4320 DenseMap<Value *, Demand> Demands;
4321 auto DemandOf = [&](
Value *
V) -> Demand & {
4323 Demand &
D = Demands[
V];
4324 if (
D.Lanes.getBitWidth() !=
N)
4328 DemandOf(VecOpEE).Lanes.setBit(0);
4330 Demand DV = Demands.
lookup(V);
4331 if (DV.Lanes.isZero())
4334 ArrayRef<int>
Mask = SVI->getShuffleMask();
4335 Demand &
DS = DemandOf(SVI->getOperand(0));
4336 for (
unsigned I = 0,
E =
Mask.size();
I !=
E; ++
I) {
4338 if (!DV.Lanes[
I] || Mask[
I] < 0 ||
4339 (
unsigned)Mask[
I] >=
DS.Lanes.getBitWidth())
4341 if (
DS.Lanes[Mask[
I]] || DV.Duplicates[
I])
4342 DS.Duplicates.setBit(Mask[
I]);
4343 DS.Lanes.setBit(Mask[
I]);
4347 for (
Value *
Op : {
U->getOperand(0),
U->getOperand(1)}) {
4348 Demand &DOp = DemandOf(
Op);
4350 DOp.Duplicates |= DV.Duplicates | (DOp.Lanes & DV.Lanes);
4351 DOp.Lanes |= DV.Lanes;
4358 auto CoversChain = [&](
Value *
V) {
4359 SmallVector<Value *, 8> Worklist(1, VecOpEE);
4360 SmallPtrSet<Value *, 8> Seen;
4362 while (!Worklist.empty()) {
4365 for (
unsigned I = 0;
I !=
NumOps; ++
I) {
4369 if (!Nodes.contains(
Op))
4371 Worklist.push_back(
Op);
4379 struct ReductionCut {
4383 std::optional<ReductionCut> Cut;
4384 for (
Value *S : Sources) {
4385 auto It = Demands.
find(S);
4386 if (It == Demands.
end() || It->second.Lanes.isZero())
4388 if (!IsIdempotent && !It->second.Duplicates.isZero()) {
4393 Cut = ReductionCut{S, It->second.Lanes};
4400 if (!IsIdempotent && !(Cut->Elts & It->second.Lanes).isZero()) {
4404 Cut->Elts |= It->second.Lanes;
4407 for (
Value *V : Nodes) {
4410 auto It = Demands.
find(V);
4411 if (It == Demands.
end() || !It->second.Lanes.isAllOnes())
4413 if (!IsIdempotent && !It->second.Duplicates.isZero())
4415 if (!CoversChain(V))
4417 Cut = ReductionCut{
V, It->second.Lanes};
4422 if (!Cut || Cut->Elts.popcount() < 2)
4432 for (
Value *V : Nodes)
4436 bool IsPartialReduction = !Cut->Elts.isAllOnes();
4437 FixedVectorType *ReduceVecTy =
4442 SmallVector<int> ExtractMask;
4444 if (IsPartialReduction) {
4445 for (
unsigned I = 0,
E = Cut->Elts.getBitWidth();
I !=
E; ++
I)
4447 ExtractMask.push_back(
I);
4448 unsigned SubIdx = 0, SubLen;
4449 auto SK = Cut->Elts.isShiftedMask(SubIdx, SubLen)
4453 SubIdx, ReduceVecTy);
4456 IntrinsicCostAttributes ICA(
4457 ReducedOp, ReduceVecTy->getElementType(),
4461 IsFloatReduction ? CommonFMF : FastMathFlags());
4464 LLVM_DEBUG(
dbgs() <<
"Found reduction shuffle chain: " <<
I <<
"\n OldCost : "
4465 << OrigCost <<
" vs NewCost: " << NewCost <<
"\n");
4470 if (VecOpEE->
hasOneUse() ? (NewCost > OrigCost) : (NewCost >= OrigCost))
4473 Value *ReduceInput = Cut->Src;
4474 if (IsPartialReduction)
4477 Value *ReducedResult;
4478 if (IsFloatReduction) {
4480 *CommonBinOp, ReduceVecTy->getElementType(),
false,
4483 {Identity, ReduceInput}, CommonFMF);
4488 replaceValue(
I, *ReducedResult);
4497bool VectorCombine::foldCastFromReductions(Instruction &
I) {
4502 bool TruncOnly =
false;
4505 case Intrinsic::vector_reduce_add:
4506 case Intrinsic::vector_reduce_mul:
4509 case Intrinsic::vector_reduce_and:
4510 case Intrinsic::vector_reduce_or:
4511 case Intrinsic::vector_reduce_xor:
4518 Value *ReductionSrc =
I.getOperand(0);
4530 Type *ResultTy =
I.getType();
4533 ReductionOpc, ReductionSrcTy, std::nullopt,
CostKind);
4543 if (OldCost <= NewCost || !NewCost.
isValid())
4547 II->getIntrinsicID(), {Src});
4549 replaceValue(
I, *NewCast);
4577bool VectorCombine::foldSignBitReductionCmp(Instruction &
I) {
4579 IntrinsicInst *ReduceOp;
4580 const APInt *CmpVal;
4587 case Intrinsic::vector_reduce_or:
4588 case Intrinsic::vector_reduce_umax:
4589 case Intrinsic::vector_reduce_and:
4590 case Intrinsic::vector_reduce_umin:
4591 case Intrinsic::vector_reduce_add:
4602 unsigned BitWidth = VecTy->getScalarSizeInBits();
4606 unsigned NumElts = VecTy->getNumElements();
4615 case Intrinsic::vector_reduce_or:
4616 case Intrinsic::vector_reduce_umax:
4617 TreeOpcode = Instruction::Or;
4619 case Intrinsic::vector_reduce_and:
4620 case Intrinsic::vector_reduce_umin:
4621 TreeOpcode = Instruction::And;
4623 case Intrinsic::vector_reduce_add:
4624 TreeOpcode = Instruction::Add;
4632 SmallVector<Value *, 8> Worklist;
4633 SmallVector<Value *, 8> Sources;
4635 std::optional<bool> IsAShr;
4636 constexpr unsigned MaxSources = 8;
4641 while (!Worklist.
empty() && Worklist.
size() <= MaxSources &&
4642 Sources.
size() <= MaxSources) {
4651 bool ThisIsAShr = Shr->getOpcode() == Instruction::AShr;
4653 IsAShr = ThisIsAShr;
4654 else if (*IsAShr != ThisIsAShr)
4680 if (Sources.
empty() || Sources.
size() > MaxSources ||
4681 Worklist.
size() > MaxSources || !IsAShr)
4684 unsigned NumSources = Sources.
size();
4688 if (OrigIID == Intrinsic::vector_reduce_add &&
4696 (OrigIID == Intrinsic::vector_reduce_add) ? NumSources * NumElts : 1;
4699 NegativeVal.negate();
4731 TestsNegative =
false;
4732 }
else if (*CmpVal == NegativeVal) {
4733 TestsNegative =
true;
4737 IsEq = Pred == ICmpInst::ICMP_EQ;
4738 }
else if (Pred == ICmpInst::ICMP_SLT && *CmpVal == RangeHigh) {
4740 TestsNegative = (RangeHigh == NegativeVal);
4741 }
else if (Pred == ICmpInst::ICMP_SGT && *CmpVal == RangeHigh - 1) {
4743 TestsNegative = (RangeHigh == NegativeVal);
4744 }
else if (Pred == ICmpInst::ICMP_SGT && *CmpVal == RangeLow) {
4746 TestsNegative = (RangeLow == NegativeVal);
4747 }
else if (Pred == ICmpInst::ICMP_SLT && *CmpVal == RangeLow + 1) {
4749 TestsNegative = (RangeLow == NegativeVal);
4792 enum CheckKind :
unsigned {
4799 auto RequiresOr = [](CheckKind
C) ->
bool {
return C & 0b100; };
4801 auto IsNegativeCheck = [](CheckKind
C) ->
bool {
return C & 0b010; };
4803 auto Invert = [](CheckKind
C) {
return CheckKind(
C ^ 0b011); };
4807 case Intrinsic::vector_reduce_or:
4808 case Intrinsic::vector_reduce_umax:
4809 Base = TestsNegative ? AnyNeg : AllNonNeg;
4811 case Intrinsic::vector_reduce_and:
4812 case Intrinsic::vector_reduce_umin:
4813 Base = TestsNegative ? AllNeg : AnyNonNeg;
4815 case Intrinsic::vector_reduce_add:
4816 Base = TestsNegative ? AllNeg : AllNonNeg;
4831 return ArithCost <= MinMaxCost ? std::make_pair(Arith, ArithCost)
4832 : std::make_pair(MinMax, MinMaxCost);
4836 auto [NewIID, NewCost] = RequiresOr(
Check)
4837 ? PickCheaper(Intrinsic::vector_reduce_or,
4838 Intrinsic::vector_reduce_umax)
4839 : PickCheaper(
Intrinsic::vector_reduce_and,
4843 if (NumSources > 1) {
4844 unsigned CombineOpc =
4845 RequiresOr(
Check) ? Instruction::Or : Instruction::And;
4850 LLVM_DEBUG(
dbgs() <<
"Found sign-bit reduction cmp: " <<
I <<
"\n OldCost: "
4851 << OldCost <<
" vs NewCost: " << NewCost <<
"\n");
4853 if (NewCost > OldCost)
4858 Type *ScalarTy = VecTy->getScalarType();
4861 if (NumSources == 1) {
4872 replaceValue(
I, *NewCmp);
4903bool VectorCombine::foldReductionZeroTest(Instruction &
I) {
4912 if (!
II || !
II->hasOneUse())
4915 auto ReduceID =
II->getIntrinsicID();
4916 if (ReduceID != Intrinsic::vector_reduce_or &&
4917 ReduceID != Intrinsic::vector_reduce_umax)
4920 Value *Vec =
II->getArgOperand(0);
4922 if (!VecTy || !VecTy->getElementType()->isIntegerTy())
4927 ? Intrinsic::vector_reduce_or
4942 LLVM_DEBUG(
dbgs() <<
"Found a reduction zero test: " <<
I <<
"\n OldCost: "
4943 << OldCost <<
" vs NewCost: " << NewCost <<
"\n");
4945 if (!OldCost.
isValid() || !NewCost.
isValid() || NewCost > OldCost)
4951 replaceValue(
I, *NewReduce);
4976bool VectorCombine::foldICmpEqZeroVectorReduce(Instruction &
I) {
4987 switch (
II->getIntrinsicID()) {
4988 case Intrinsic::vector_reduce_add:
4989 case Intrinsic::vector_reduce_or:
4990 case Intrinsic::vector_reduce_umin:
4991 case Intrinsic::vector_reduce_umax:
4992 case Intrinsic::vector_reduce_smin:
4993 case Intrinsic::vector_reduce_smax:
4999 Value *InnerOp =
II->getArgOperand(0);
5042 switch (
II->getIntrinsicID()) {
5043 case Intrinsic::vector_reduce_add: {
5048 unsigned NumElems = XTy->getNumElements();
5054 if (LeadingZerosX <= LostBits || LeadingZerosFX <= LostBits)
5062 case Intrinsic::vector_reduce_smin:
5063 case Intrinsic::vector_reduce_smax:
5073 LLVM_DEBUG(
dbgs() <<
"Found a reduction to 0 comparison with removable op: "
5089 case Intrinsic::vector_reduce_add:
5090 case Intrinsic::vector_reduce_or:
5096 case Intrinsic::vector_reduce_umin:
5097 case Intrinsic::vector_reduce_umax:
5098 case Intrinsic::vector_reduce_smin:
5099 case Intrinsic::vector_reduce_smax:
5111 NewReduceCost + (InnerOp->
hasOneUse() ? 0 : ExtCost);
5113 LLVM_DEBUG(
dbgs() <<
"Found a removable extension before reduction: "
5114 << *InnerOp <<
"\n OldCost: " << OldCost
5115 <<
" vs NewCost: " << NewCost <<
"\n");
5121 if (NewCost > OldCost)
5130 Builder.
CreateICmp(Pred, NewReduce, ConstantInt::getNullValue(Ty));
5131 replaceValue(
I, *NewCmp);
5162bool VectorCombine::foldEquivalentReductionCmp(Instruction &
I) {
5165 const APInt *CmpVal;
5170 if (!
II || !
II->hasOneUse())
5173 const auto IsValidOrUmaxCmp = [&]() {
5182 bool IsPositive = CmpVal->
isAllOnes() && Pred == ICmpInst::ICMP_SGT;
5184 bool IsNegative = (CmpVal->
isZero() || CmpVal->
isOne() || *CmpVal == 2) &&
5185 Pred == ICmpInst::ICMP_SLT;
5186 return IsEquality || IsPositive || IsNegative;
5189 const auto IsValidAndUminCmp = [&]() {
5194 const auto LeadingOnes = CmpVal->
countl_one();
5201 bool IsNegative = CmpVal->
isZero() && Pred == ICmpInst::ICMP_SLT;
5210 ((*CmpVal)[0] || (*CmpVal)[1]) && Pred == ICmpInst::ICMP_SGT;
5211 return IsEquality || IsNegative || IsPositive;
5219 switch (OriginalIID) {
5220 case Intrinsic::vector_reduce_or:
5221 if (!IsValidOrUmaxCmp())
5223 AlternativeIID = Intrinsic::vector_reduce_umax;
5225 case Intrinsic::vector_reduce_umax:
5226 if (!IsValidOrUmaxCmp())
5228 AlternativeIID = Intrinsic::vector_reduce_or;
5230 case Intrinsic::vector_reduce_and:
5231 if (!IsValidAndUminCmp())
5233 AlternativeIID = Intrinsic::vector_reduce_umin;
5235 case Intrinsic::vector_reduce_umin:
5236 if (!IsValidAndUminCmp())
5238 AlternativeIID = Intrinsic::vector_reduce_and;
5251 if (ReductionOpc != Instruction::ICmp)
5262 <<
"\n OrigCost: " << OrigCost
5263 <<
" vs AltCost: " << AltCost <<
"\n");
5265 if (AltCost >= OrigCost)
5269 Type *ScalarTy = VecTy->getScalarType();
5272 Builder.
CreateICmp(Pred, NewReduce, ConstantInt::get(ScalarTy, *CmpVal));
5274 replaceValue(
I, *NewCmp);
5288 unsigned Depth = 0) {
5289 constexpr unsigned MaxLocalDepth = 2;
5290 if (
Depth > MaxLocalDepth)
5293 auto NumSignBits = [&](
const Value *
X) {
5296 if (NumSignBits(V) == V->getType()->getScalarSizeInBits())
5301 return NumSignBits(
A) >= 2 && NumSignBits(
B) >= 2 &&
5312bool VectorCombine::foldReduceAddCmpZero(Instruction &
I) {
5322 if (!VecTy || VecTy->getNumElements() < 2)
5328 if (!IsNonNegative && !IsNonPositive)
5333 unsigned NumElts = VecTy->getNumElements();
5335 if (
Log2_32(NumElts) >= NumSignBits)
5338 ICmpInst::Predicate NewPred;
5340 case ICmpInst::ICMP_EQ:
5341 case ICmpInst::ICMP_ULE:
5342 case ICmpInst::ICMP_SLE:
5343 case ICmpInst::ICMP_SGE:
5344 NewPred = ICmpInst::ICMP_EQ;
5346 case ICmpInst::ICMP_NE:
5347 case ICmpInst::ICMP_UGT:
5348 case ICmpInst::ICMP_SGT:
5349 case ICmpInst::ICMP_SLT:
5350 NewPred = ICmpInst::ICMP_NE;
5360 if (!IsNonNegative &&
5361 (Pred == ICmpInst::ICMP_SGT || Pred == ICmpInst::ICMP_SLE))
5363 if (!IsNonPositive &&
5364 (Pred == ICmpInst::ICMP_SLT || Pred == ICmpInst::ICMP_SGE))
5366 if ((Pred == ICmpInst::ICMP_SGT || Pred == ICmpInst::ICMP_SLE ||
5367 Pred == ICmpInst::ICMP_SLT || Pred == ICmpInst::ICMP_SGE) &&
5368 Log2_32(NumElts) >= NumSignBits - 1)
5372 Instruction::Add, VecTy, std::nullopt,
CostKind);
5374 Instruction::Or, VecTy, std::nullopt,
CostKind);
5376 Intrinsic::umax, VecTy, FastMathFlags(),
CostKind);
5379 bool UseOr = OrCost.
isValid() && (!UmaxCost.
isValid() || OrCost <= UmaxCost);
5381 if (AltCost > OrigCost)
5387 Intrinsic::vector_reduce_umax, {VecTy}, {Vec});
5388 Worklist.pushValue(NewReduce);
5390 NewPred, NewReduce, ConstantInt::getNullValue(VecTy->getScalarType()));
5391 replaceValue(
I, *NewCmp);
5400 constexpr unsigned MaxVisited = 32;
5403 bool FoundReduction =
false;
5406 while (!WorkList.
empty()) {
5408 for (
User *U :
I->users()) {
5410 if (!UI || !Visited.
insert(UI).second)
5412 if (Visited.
size() > MaxVisited)
5418 switch (
II->getIntrinsicID()) {
5419 case Intrinsic::vector_reduce_add:
5420 case Intrinsic::vector_reduce_mul:
5421 case Intrinsic::vector_reduce_and:
5422 case Intrinsic::vector_reduce_or:
5423 case Intrinsic::vector_reduce_xor:
5424 case Intrinsic::vector_reduce_smin:
5425 case Intrinsic::vector_reduce_smax:
5426 case Intrinsic::vector_reduce_umin:
5427 case Intrinsic::vector_reduce_umax:
5428 FoundReduction =
true;
5441 return FoundReduction;
5454bool VectorCombine::foldSelectShuffle(Instruction &
I,
bool FromReduction) {
5459 if (!Op0 || !Op1 || Op0 == Op1 || !Op0->isBinaryOp() || !Op1->isBinaryOp() ||
5460 VT != Op0->getType())
5467 SmallPtrSet<Instruction *, 4> InputShuffles({SVI0A, SVI0B, SVI1A, SVI1B});
5469 if (!
I ||
I->getOperand(0)->getType() != VT)
5471 return any_of(
I->users(), [&](User *U) {
5472 return U != Op0 && U != Op1 &&
5473 !(isa<ShuffleVectorInst>(U) &&
5474 (InputShuffles.contains(cast<Instruction>(U)) ||
5475 isInstructionTriviallyDead(cast<Instruction>(U))));
5478 if (checkSVNonOpUses(SVI0A) || checkSVNonOpUses(SVI0B) ||
5479 checkSVNonOpUses(SVI1A) || checkSVNonOpUses(SVI1B))
5487 for (
auto *U :
I->users()) {
5489 if (!SV || SV->getType() != VT)
5491 if ((SV->getOperand(0) != Op0 && SV->getOperand(0) != Op1) ||
5492 (SV->getOperand(1) != Op0 && SV->getOperand(1) != Op1))
5499 if (!collectShuffles(Op0) || !collectShuffles(Op1))
5503 if (FromReduction && Shuffles.
size() > 1)
5508 if (!FromReduction) {
5509 for (
size_t Idx = 0,
E = Shuffles.
size(); Idx !=
E; ++Idx) {
5510 for (
auto *U : Shuffles[Idx]->
users()) {
5525 int MaxV1Elt = 0, MaxV2Elt = 0;
5526 unsigned NumElts = VT->getNumElements();
5527 for (ShuffleVectorInst *SVN : Shuffles) {
5528 SmallVector<int>
Mask;
5529 SVN->getShuffleMask(Mask);
5533 Value *SVOp0 = SVN->getOperand(0);
5534 Value *SVOp1 = SVN->getOperand(1);
5539 for (
int &Elem : Mask) {
5545 if (SVOp0 == Op1 && SVOp1 == Op0) {
5549 if (SVOp0 != Op0 || SVOp1 != Op1)
5555 SmallVector<int> ReconstructMask;
5556 for (
unsigned I = 0;
I <
Mask.size();
I++) {
5559 }
else if (Mask[
I] <
static_cast<int>(NumElts)) {
5560 MaxV1Elt = std::max(MaxV1Elt, Mask[
I]);
5561 auto It =
find_if(
V1, [&](
const std::pair<int, int> &
A) {
5562 return Mask[
I] ==
A.first;
5568 V1.emplace_back(Mask[
I],
V1.size());
5571 MaxV2Elt = std::max<int>(MaxV2Elt, Mask[
I] - NumElts);
5572 auto It =
find_if(V2, [&](
const std::pair<int, int> &
A) {
5573 return Mask[
I] -
static_cast<int>(NumElts) ==
A.first;
5587 sort(ReconstructMask);
5588 OrigReconstructMasks.
push_back(std::move(ReconstructMask));
5595 if (
V1.empty() || V2.
empty() ||
5596 (MaxV1Elt ==
static_cast<int>(
V1.size()) - 1 &&
5597 MaxV2Elt ==
static_cast<int>(V2.
size()) - 1))
5609 if (InputShuffles.contains(SSV))
5611 return SV->getMaskValue(M);
5619 std::pair<int, int>
Y) {
5620 int MXA = GetBaseMaskValue(
A,
X.first);
5621 int MYA = GetBaseMaskValue(
A,
Y.first);
5625 return SortBase(SVI0A,
A,
B);
5627 stable_sort(V2, [&](std::pair<int, int>
A, std::pair<int, int>
B) {
5628 return SortBase(SVI1A,
A,
B);
5633 for (
const auto &Mask : OrigReconstructMasks) {
5634 SmallVector<int> ReconstructMask;
5635 for (
int M : Mask) {
5637 auto It =
find_if(V, [M](
auto A) {
return A.second ==
M; });
5638 assert(It !=
V.end() &&
"Expected all entries in Mask");
5639 return std::distance(
V.begin(), It);
5643 else if (M <
static_cast<int>(NumElts)) {
5646 ReconstructMask.
push_back(NumElts + FindIndex(V2, M));
5649 ReconstructMasks.
push_back(std::move(ReconstructMask));
5654 SmallVector<int> V1A, V1B, V2A, V2B;
5655 for (
unsigned I = 0;
I <
V1.size();
I++) {
5659 for (
unsigned I = 0;
I < V2.
size();
I++) {
5660 V2A.
push_back(GetBaseMaskValue(SVI1A, V2[
I].first));
5661 V2B.
push_back(GetBaseMaskValue(SVI1B, V2[
I].first));
5663 while (V1A.
size() < NumElts) {
5667 while (V2A.
size() < NumElts) {
5679 VT, VT, SV->getShuffleMask(),
CostKind);
5686 unsigned ElementSize = VT->getElementType()->getPrimitiveSizeInBits();
5687 unsigned MaxVectorSize =
5689 unsigned MaxElementsInVector = MaxVectorSize / ElementSize;
5690 if (MaxElementsInVector == 0)
5699 std::set<SmallVector<int, 4>> UniqueShuffles;
5704 unsigned NumFullVectors =
Mask.size() / MaxElementsInVector;
5705 if (NumFullVectors < 2)
5706 return C + ShuffleCost;
5707 SmallVector<int, 4> SubShuffle(MaxElementsInVector);
5708 unsigned NumUniqueGroups = 0;
5709 unsigned NumGroups =
Mask.size() / MaxElementsInVector;
5712 for (
unsigned I = 0;
I < NumFullVectors; ++
I) {
5713 for (
unsigned J = 0; J < MaxElementsInVector; ++J)
5714 SubShuffle[J] = Mask[MaxElementsInVector *
I + J];
5715 if (UniqueShuffles.insert(SubShuffle).second)
5716 NumUniqueGroups += 1;
5718 return C + ShuffleCost * NumUniqueGroups / NumGroups;
5724 SmallVector<int, 16>
Mask;
5725 SV->getShuffleMask(Mask);
5726 return AddShuffleMaskAdjustedCost(
C, Mask);
5729 auto AllShufflesHaveSameOperands =
5730 [](SmallPtrSetImpl<Instruction *> &InputShuffles) {
5731 if (InputShuffles.size() < 2)
5733 ShuffleVectorInst *FirstSV =
5740 std::next(InputShuffles.begin()), InputShuffles.end(),
5741 [&](Instruction *
I) {
5742 ShuffleVectorInst *SV = dyn_cast<ShuffleVectorInst>(I);
5743 return SV && SV->getOperand(0) == In0 && SV->getOperand(1) == In1;
5752 CostBefore += std::accumulate(Shuffles.begin(), Shuffles.end(),
5754 if (AllShufflesHaveSameOperands(InputShuffles)) {
5755 UniqueShuffles.clear();
5756 CostBefore += std::accumulate(InputShuffles.begin(), InputShuffles.end(),
5759 CostBefore += std::accumulate(InputShuffles.begin(), InputShuffles.end(),
5765 FixedVectorType *Op0SmallVT =
5767 FixedVectorType *Op1SmallVT =
5772 UniqueShuffles.clear();
5773 CostAfter += std::accumulate(ReconstructMasks.begin(), ReconstructMasks.end(),
5775 std::set<SmallVector<int>> OutputShuffleMasks({V1A, V1B, V2A, V2B});
5777 std::accumulate(OutputShuffleMasks.begin(), OutputShuffleMasks.end(),
5780 LLVM_DEBUG(
dbgs() <<
"Found a binop select shuffle pattern: " <<
I <<
"\n");
5782 <<
" vs CostAfter: " << CostAfter <<
"\n");
5783 if (CostBefore < CostAfter ||
5794 if (InputShuffles.contains(SSV))
5796 return SV->getOperand(
Op);
5800 GetShuffleOperand(SVI0A, 1), V1A);
5803 GetShuffleOperand(SVI0B, 1), V1B);
5806 GetShuffleOperand(SVI1A, 1), V2A);
5809 GetShuffleOperand(SVI1B, 1), V2B);
5814 I->copyIRFlags(Op0,
true);
5819 I->copyIRFlags(Op1,
true);
5821 for (
int S = 0,
E = ReconstructMasks.size(); S !=
E; S++) {
5824 replaceValue(*Shuffles[S], *NSV,
false);
5827 Worklist.pushValue(NSV0A);
5828 Worklist.pushValue(NSV0B);
5829 Worklist.pushValue(NSV1A);
5830 Worklist.pushValue(NSV1B);
5840bool VectorCombine::shrinkType(Instruction &
I) {
5841 Value *ZExted, *OtherOperand;
5847 Value *ZExtOperand =
I.getOperand(
I.getOperand(0) == OtherOperand ? 1 : 0);
5851 unsigned BW = SmallTy->getElementType()->getPrimitiveSizeInBits();
5853 if (
I.getOpcode() == Instruction::LShr) {
5870 Instruction::ZExt, BigTy, SmallTy,
5871 TargetTransformInfo::CastContextHint::None,
CostKind);
5876 for (User *U : ZExtOperand->
users()) {
5883 ShrinkCost += ZExtCost;
5898 ShrinkCost += ZExtCost;
5905 Instruction::Trunc, SmallTy, BigTy,
5906 TargetTransformInfo::CastContextHint::None,
CostKind);
5911 if (ShrinkCost > CurrentCost)
5915 Value *Op0 = ZExted;
5918 if (
I.getOperand(0) == OtherOperand)
5925 replaceValue(
I, *NewZExtr);
5931bool VectorCombine::foldInsExtVectorToShuffle(Instruction &
I) {
5932 Value *DstVec, *SrcVec;
5943 if (!DstVecTy || !SrcVecTy ||
5949 if (InsIdx >= NumDstElts || ExtIdx >= NumSrcElts || NumDstElts == 1)
5956 bool NeedExpOrNarrow = NumSrcElts != NumDstElts;
5958 if (NeedDstSrcSwap) {
5960 Mask[InsIdx] = ExtIdx % NumDstElts;
5964 std::iota(
Mask.begin(),
Mask.end(), 0);
5965 Mask[InsIdx] = (ExtIdx % NumDstElts) + NumDstElts;
5978 SmallVector<int> ExtToVecMask;
5979 if (!NeedExpOrNarrow) {
5984 nullptr, {DstVec, SrcVec});
5990 ExtToVecMask[ExtIdx % NumDstElts] = ExtIdx;
5993 DstVecTy, SrcVecTy, ExtToVecMask,
CostKind);
5997 if (!Ext->hasOneUse())
6000 LLVM_DEBUG(
dbgs() <<
"Found a insert/extract shuffle-like pair: " <<
I
6001 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
6004 if (OldCost < NewCost)
6007 if (NeedExpOrNarrow) {
6008 if (!NeedDstSrcSwap)
6021 replaceValue(
I, *Shuf);
6045bool VectorCombine::foldDeinterleaveInterleavePair(Instruction &
I) {
6062 if (
U.getUser()->isDroppable())
6066 if (!Extract || Extract->getNumIndices() != 1)
6069 unsigned Index = *Extract->idx_begin();
6070 if (Index >= Factor || CurrentUses[Index])
6078 IntrinsicInst *Interleave =
nullptr;
6079 unsigned NumVisited = 0;
6083 return CB->arg_size();
6084 return Inst->getNumOperands();
6087 auto IsSupportedElementwise = [&](
Instruction *Inst) {
6093 if (
II->hasOperandBundles() ||
6096 }
else if (!
isa<BinaryOperator, UnaryOperator, CastInst, CmpInst,
6097 SelectInst, FreezeInst>(Inst)) {
6103 for (
unsigned Op = 0,
E = GetNumDataOperands(Inst);
Op !=
E; ++
Op) {
6106 OperandTy->getElementCount() != ResultTy->getElementCount())
6118 NumVisited += Factor;
6120 for (Use *&CurrentUse : CurrentUses) {
6121 Use *NextUse = CurrentUse->getUser()->getSingleUndroppableUse();
6127 CurrentUse = NextUse;
6132 II &&
II->getIntrinsicID() == ExpectedInterleaveIID) {
6133 if (
II->hasOperandBundles())
6136 for (
unsigned Index = 0;
Index != Factor; ++
Index)
6137 if (CurrentUses[Index]->getUser() !=
II ||
6138 CurrentUses[Index]->getOperandNo() != Index)
6146 if (!IsSupportedElementwise(FirstInst))
6149 unsigned ChainOperand = CurrentUses.front()->getOperandNo();
6150 if (
any_of(CurrentUses, [&](Use *U) {
6152 return Inst != FirstInst && (
U->getOperandNo() != ChainOperand ||
6153 !FirstInst->isSameOperationAs(Inst));
6157 auto GetSplatOrScalar = [](
Value *
V) {
6164 for (
unsigned Op = 0,
E = GetNumDataOperands(FirstInst);
Op !=
E; ++
Op) {
6165 if (
Op == ChainOperand)
6168 Value *CommonValue = GetSplatOrScalar(FirstInst->getOperand(
Op));
6169 if (!CommonValue ||
any_of(CurrentUses, [&](Use *U) {
6171 return Inst != FirstInst &&
6185 ElementCount WideEC =
6188 auto CreateWideInstruction = [&](
Instruction *NarrowInst,
6191 assert(IsSupportedElementwise(NarrowInst) &&
6192 "Expected supported elementwise");
6196 return Builder.
CreateCast(Cast->getOpcode(), NewOperands[0],
6199 return Builder.
CreateCmp(
Cmp->getPredicate(), NewOperands[0],
6203 NewOperands[0], NewOperands[1], NewOperands[2],
"",
6215 for (
const ElementwiseStep &Step : Steps) {
6217 unsigned ChainOperand = Step.front()->getOperandNo();
6222 unsigned NumOperands = GetNumDataOperands(NarrowInst);
6223 SmallVector<Value *, 4> NewOperands;
6224 NewOperands.
reserve(NumOperands);
6226 for (
unsigned Op = 0;
Op != NumOperands; ++
Op) {
6229 if (
Op == ChainOperand)
6230 Operand = WideValue;
6236 auto *WideResultTy =
6239 CreateWideInstruction(NarrowInst, NewOperands, WideResultTy);
6248 WideValue = NewValue;
6252 replaceValue(*Interleave, *WideValue);
6260bool VectorCombine::foldInterleaveIntrinsics(Instruction &
I) {
6261 const APInt *SplatVal0, *SplatVal1;
6271 auto *ExtVTy = VectorType::getExtendedElementVectorType(VTy);
6272 unsigned Width = VTy->getElementType()->getIntegerBitWidth();
6281 LLVM_DEBUG(
dbgs() <<
"VC: The cost to cast from " << *ExtVTy <<
" to "
6282 << *
I.getType() <<
" is too high.\n");
6286 APInt NewSplatVal = SplatVal1->
zext(Width * 2);
6287 NewSplatVal <<= Width;
6288 NewSplatVal |= SplatVal0->
zext(Width * 2);
6290 ExtVTy->getElementCount(), ConstantInt::get(
F.getContext(), NewSplatVal));
6325bool VectorCombine::foldDeinterleaveIntrinsics(Instruction &
I) {
6326 if (foldDeinterleaveInterleavePair(
I))
6330 if (
DL->isBigEndian())
6333 using namespace PatternMatch;
6334 Value *DeinterleavedVal;
6345 unsigned HalfElementWidth = ElementWidth / 2;
6349 std::array<ExtractValueInst *, 2> OrigFields{};
6350 for (User *Usr :
I.users()) {
6353 if (!
E ||
E->getNumIndices() != 1)
6355 unsigned Idx = *
E->idx_begin();
6357 if (Idx >= 2 || OrigFields[Idx] || !
E->hasNUses(2))
6359 OrigFields[Idx] =
E;
6363 SmallVector<Instruction *, 2> MergeInsts;
6364 for (
auto *FieldUsr : OrigFields[0]->
users()) {
6372 auto MatchMerge = [&](void) ->
bool {
6375 return match(MergeInsts[0],
6379 match(MergeInsts[1],
6384 if (!MatchMerge()) {
6385 std::swap(MergeInsts[0], MergeInsts[1]);
6400 auto *NewFieldTy = VecTy->getWithNewBitWidth(HalfElementWidth);
6410 if (OldCost <= NewCost || !NewCost.
isValid()) {
6412 dbgs() <<
"VC: New deinterleave2 sequence cost (" << NewCost <<
")"
6413 <<
" is higher than that of the old one (" << OldCost <<
")\n");
6421 Intrinsic::vector_deinterleave2, {NewVecTy}, {NewVecCast});
6422 for (
auto [Idx, MergeInst] :
enumerate(MergeInsts)) {
6424 NewField = Builder.
CreateBitCast(NewField, MergeInst->getType());
6425 replaceValue(*MergeInst, *NewField);
6431bool VectorCombine::foldBitcastOfVPLoad(Instruction &
I) {
6432 const DataLayout &
DL =
I.getDataLayout();
6447 DL.getValueOrABITypeAlignment(
II->getPointerAlignment(), OrigVecTy);
6448 ElementCount OrigVecCnt = OrigVecTy->getElementCount();
6450 ElementCount NewVecCnt = NewVecTy->getElementCount();
6462 II->getMemoryPointerParam(),
false,
6468 {Intrinsic::vp_load, NewVecTy,
II->getMemoryPointerParam(),
false,
6472 <<
" NewCost=" << NewCost <<
"\n");
6473 if (NewCost > OldCost || !NewCost.
isValid())
6480 NewVecTy, Intrinsic::vp_load,
6481 {
II->getMemoryPointerParam(), NewMask, NewEVL});
6484 0, AttrBuilder(
II->getContext()).addAlignmentAttr(OrigAlign));
6485 replaceValue(*Cast, *NewVP);
6495bool VectorCombine::foldBitOrderReverseAndSwap(Instruction &
I) {
6499 Type *Ty =
X->getType();
6500 Type *VecTy =
I.getOperand(0)->getType();
6514 if (CanUseBswap || CanUseFshl) {
6525 IntrinsicCostAttributes ICABSwap(Intrinsic::bswap, Ty, {Ty});
6526 IntrinsicCostAttributes ICABFshl(Intrinsic::fshl, Ty, {
X,
X, HalfBW},
6528 IntrinsicCostAttributes ICABRev(Intrinsic::bitreverse, Ty, {Ty});
6533 if (!InnerCall->hasOneUse())
6536 else if (!InnerBitCast->hasOneUse())
6539 <<
"\n OldCost: " << OldCost
6540 <<
" vs NewCost: " << NewCost <<
"\n");
6541 if (NewCost.isValid() && NewCost < OldCost) {
6547 Worklist.pushValue(Swap);
6549 replaceValue(
I, *BRev);
6558 Type *Ty =
I.getType();
6560 TypeSize ElementSize =
DL->getTypeStoreSize(Ty);
6563 Type *NewVecTy = VectorType::get(I8Ty, NewVecCnt);
6576 IntrinsicCostAttributes ICANew(Intrinsic::bitreverse, NewVecTy, {NewVecTy});
6579 InstructionCost NewCost = CastToVecCost + NewIntrinsicCost + CastToOrigCost;
6580 if (!InnerII->hasOneUse())
6583 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
6585 if (!NewCost.
isValid() || NewCost >= OldCost)
6593 replaceValue(
I, *CastToOrig);
6603 unsigned RawNumElements = MaxIdx + 1u;
6606 if (!
TTI.isTypeLegal(ElemTy))
6607 return RawNumElements;
6609 TypeSize ElemSize =
DL.getTypeSizeInBits(ElemTy);
6611 return RawNumElements;
6616 return RawNumElements;
6621 if (ElemsPerReg == 0 || RawNumElements <= ElemsPerReg)
6622 return RawNumElements;
6624 return alignTo(RawNumElements, ElemsPerReg);
6628bool VectorCombine::shrinkLoadForShuffles(Instruction &
I) {
6630 if (!OldLoad || !OldLoad->isSimple())
6637 unsigned const OldNumElements = OldLoadTy->getNumElements();
6643 using IndexRange = std::pair<int, int>;
6644 auto GetIndexRangeInShuffles = [&]() -> std::optional<IndexRange> {
6645 IndexRange OutputRange = IndexRange(OldNumElements, -1);
6646 for (llvm::Use &Use :
I.uses()) {
6648 User *Shuffle =
Use.getUser();
6653 return std::nullopt;
6660 for (
int Index : Mask) {
6661 if (Index >= 0 && Index <
static_cast<int>(OldNumElements)) {
6662 OutputRange.first = std::min(Index, OutputRange.first);
6663 OutputRange.second = std::max(Index, OutputRange.second);
6668 if (OutputRange.second < OutputRange.first)
6669 return std::nullopt;
6675 if (std::optional<IndexRange> Indices = GetIndexRangeInShuffles()) {
6676 unsigned const NewNumElements =
6681 if (NewNumElements < OldNumElements) {
6686 Type *ElemTy = OldLoadTy->getElementType();
6688 Value *PtrOp = OldLoad->getPointerOperand();
6691 Instruction::Load, OldLoad->getType(), OldLoad->getAlign(),
6692 OldLoad->getPointerAddressSpace(),
CostKind);
6695 OldLoad->getPointerAddressSpace(),
CostKind);
6697 using UseEntry = std::pair<ShuffleVectorInst *, std::vector<int>>;
6699 unsigned const MaxIndex = NewNumElements * 2u;
6701 for (llvm::Use &Use :
I.uses()) {
6708 ArrayRef<int> OldMask = Shuffle->getShuffleMask();
6714 for (
int Index : OldMask) {
6715 if (Index >=
static_cast<int>(MaxIndex))
6729 dbgs() <<
"Found a load used only by shufflevector instructions: "
6730 <<
I <<
"\n OldCost: " << OldCost
6731 <<
" vs NewCost: " << NewCost <<
"\n");
6733 if (OldCost < NewCost || !NewCost.
isValid())
6739 NewLoad->copyMetadata(
I);
6742 for (UseEntry &Use : NewUses) {
6743 ShuffleVectorInst *Shuffle =
Use.first;
6744 std::vector<int> &NewMask =
Use.second;
6751 replaceValue(*Shuffle, *NewShuffle,
false);
6764bool VectorCombine::shrinkPhiOfShuffles(Instruction &
I) {
6766 if (!Phi ||
Phi->getNumIncomingValues() != 2u)
6770 ArrayRef<int> Mask0;
6771 ArrayRef<int> Mask1;
6784 auto const InputNumElements = InputVT->getNumElements();
6786 if (InputNumElements >= ResultVT->getNumElements())
6791 SmallVector<int, 16> NewMask;
6794 for (
auto [
M0,
M1] :
zip(Mask0, Mask1)) {
6795 if (
M0 >= 0 &&
M1 >= 0)
6797 else if (
M0 == -1 &&
M1 == -1)
6810 int MaskOffset = NewMask[0
u];
6811 unsigned Index = (InputNumElements + MaskOffset) % InputNumElements;
6814 for (
unsigned I = 0u;
I < InputNumElements; ++
I) {
6828 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
6831 if (NewCost > OldCost)
6843 auto *NewPhi = Builder.
CreatePHI(NewShuf0->getType(), 2u);
6845 NewPhi->addIncoming(
Op,
Phi->getIncomingBlock(1u));
6851 replaceValue(*Phi, *NewShuf1);
6857bool VectorCombine::run() {
6871 auto Opcode =
I.getOpcode();
6879 if (IsFixedVectorType) {
6881 case Instruction::InsertElement:
6882 if (vectorizeLoadInsert(
I))
6885 case Instruction::ShuffleVector:
6886 if (widenSubvectorLoad(
I))
6897 if (scalarizeOpOrCmp(
I))
6899 if (scalarizeLoad(
I))
6901 if (scalarizeExtExtract(
I))
6903 if (foldInterleaveIntrinsics(
I))
6905 if (foldBitcastOfVPLoad(
I))
6909 if (foldDeinterleaveIntrinsics(
I))
6912 if (Opcode == Instruction::Store)
6913 if (foldInsertElementsToStores(
I))
6917 if (TryEarlyFoldsOnly)
6920 if (Opcode == Instruction::Call)
6921 if (foldBitOrderReverseAndSwap(
I))
6923 if (Opcode == Instruction::BitCast)
6924 if (foldBitOrderReverseAndSwap(
I))
6931 if (IsFixedVectorType) {
6933 case Instruction::InsertElement:
6934 if (foldInsExtFNeg(
I))
6936 if (foldInsExtBinop(
I))
6938 if (foldInsExtVectorToShuffle(
I))
6941 case Instruction::ShuffleVector:
6942 if (foldPermuteOfBinops(
I))
6944 if (foldShuffleOfBinops(
I))
6946 if (foldShuffleOfSelects(
I))
6948 if (foldShuffleOfCastops(
I))
6950 if (foldShuffleOfShuffles(
I))
6952 if (foldPermuteOfIntrinsic(
I))
6954 if (foldShufflesOfLengthChangingShuffles(
I))
6956 if (foldShuffleOfIntrinsics(
I))
6958 if (foldSelectShuffle(
I))
6960 if (foldShuffleToIdentity(
I))
6963 case Instruction::Load:
6964 if (shrinkLoadForShuffles(
I))
6967 case Instruction::BitCast:
6968 if (foldBitcastShuffle(
I))
6970 if (foldSelectsFromBitcast(
I))
6973 case Instruction::And:
6974 case Instruction::Or:
6975 case Instruction::Xor:
6976 if (foldBitOpOfCastops(
I))
6978 if (foldBitOpOfCastConstant(
I))
6981 case Instruction::PHI:
6982 if (shrinkPhiOfShuffles(
I))
6992 case Instruction::Call:
6993 if (foldShuffleFromReductions(
I))
6995 if (foldCastFromReductions(
I))
6998 case Instruction::ExtractElement:
6999 if (foldShuffleChainsToReduce(
I))
7002 case Instruction::ICmp:
7003 if (foldSignBitReductionCmp(
I))
7005 if (foldICmpEqZeroVectorReduce(
I))
7007 if (foldReductionZeroTest(
I))
7009 if (foldEquivalentReductionCmp(
I))
7011 if (foldReduceAddCmpZero(
I))
7014 case Instruction::FCmp:
7015 if (foldExtractExtract(
I))
7018 case Instruction::Or:
7019 if (foldConcatOfBoolMasks(
I))
7024 if (foldExtractExtract(
I))
7026 if (foldExtractedCmps(
I))
7028 if (foldBinopOfReductions(
I))
7037 bool MadeChange =
false;
7038 for (BasicBlock &BB :
F) {
7050 if (!
I->isDebugOrPseudoInst())
7051 MadeChange |= FoldInst(*
I);
7058 while (!Worklist.isEmpty()) {
7068 MadeChange |= FoldInst(*
I);
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static cl::opt< unsigned > MaxInstrsToScan("aggressive-instcombine-max-scan-instrs", cl::init(64), cl::Hidden, cl::desc("Max number of instructions to scan for aggressive instcombine."))
This is the interface for LLVM's primary stateless and local alias analysis.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static cl::opt< IntrinsicCostStrategy > IntrinsicCost("intrinsic-cost-strategy", cl::desc("Costing strategy for intrinsic instructions"), cl::init(IntrinsicCostStrategy::InstructionCost), cl::values(clEnumValN(IntrinsicCostStrategy::InstructionCost, "instruction-cost", "Use TargetTransformInfo::getInstructionCost"), clEnumValN(IntrinsicCostStrategy::IntrinsicCost, "intrinsic-cost", "Use TargetTransformInfo::getIntrinsicInstrCost"), clEnumValN(IntrinsicCostStrategy::TypeBasedIntrinsicCost, "type-based-intrinsic-cost", "Calculate the intrinsic cost based only on argument types")))
This file defines the DenseMap class.
This is the interface for a simple mod/ref and alias analysis over globals.
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static void eraseInstruction(Instruction &I, ICFLoopSafetyInfo &SafetyInfo, MemorySSAUpdater &MSSAU)
uint64_t IntrinsicInst * II
FunctionAnalysisManager FAM
This file contains the declarations for profiling metadata utility functions.
const SmallVectorImpl< MachineOperand > & Cond
Func getContext().diagnose(DiagnosticInfoUnsupported(Func
This file defines the scope_exit class, which executes user-defined cleanup logic at scope exit.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static SymbolRef::Type getType(const Symbol *Sym)
static bool isEquivBitcast(Value *X, Value *Y)
Helper to peek through bitcasts to the same value.
static bool isFreeConcat(ArrayRef< InstLane > Item, TTI::TargetCostKind CostKind, const TargetTransformInfo &TTI)
Detect concat of multiple values into a vector.
static void analyzeCostOfVecReduction(const IntrinsicInst &II, TTI::TargetCostKind CostKind, const TargetTransformInfo &TTI, InstructionCost &CostBeforeReduction, InstructionCost &CostAfterReduction)
static Value * generateNewInstTree(ArrayRef< InstLane > Item, Use *From, const DenseSet< std::pair< Value *, Use * > > &IdentityLeafs, const DenseSet< std::pair< Value *, Use * > > &SplatLeafs, const DenseSet< std::pair< Value *, Use * > > &ConcatLeafs, IRBuilderBase &Builder, InstructionWorklist &WorkList, const TargetTransformInfo *TTI)
static SmallVector< InstLane > generateInstLaneVectorFromOperand(ArrayRef< InstLane > Item, int Op)
static Value * createShiftShuffle(Value *Vec, unsigned OldIndex, unsigned NewIndex, IRBuilderBase &Builder)
Create a shuffle that translates (shifts) 1 element from the input vector to a new element location.
std::pair< Value *, int > InstLane
static bool isKnownNonPositive(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Used by foldReduceAddCmpZero to check if we can prove that a value is non-positive.
static Value * materializeScalarizedGEPIndex(Value *Idx, IntegerType *GEPIndexTy, IRBuilderBase &Builder)
Materialize an index for a scalarized GEP after profitability is known.
static Align computeAlignmentAfterScalarization(Align VectorAlignment, Type *ScalarType, Value *Idx, const DataLayout &DL)
The memory operation on a vector of ScalarType had alignment of VectorAlignment.
static bool feedsIntoVectorReduction(ShuffleVectorInst *SVI)
Returns true if this ShuffleVectorInst eventually feeds into a vector reduction intrinsic (e....
static cl::opt< bool > DisableVectorCombine("disable-vector-combine", cl::init(false), cl::Hidden, cl::desc("Disable all vector combine transforms"))
static bool canWidenLoad(LoadInst *Load, const TargetTransformInfo &TTI)
static const unsigned InvalidIndex
static IntegerType * getScalarizedGEPIndexInfo(VectorType *VecTy, Value *Idx, Type *PtrTy, const DataLayout &DL)
Return the GEP index type if the unsigned vector index Idx can be represented by an inbounds GEP.
static Value * translateExtract(ExtractElementInst *ExtElt, unsigned NewIndex, IRBuilderBase &Builder)
Given an extract element instruction with constant index operand, shuffle the source vector (shift th...
static ScalarizationResult canScalarizeAccess(VectorType *VecTy, Value *Idx, const SimplifyQuery &SQ)
Check if it is legal to scalarize a memory access to VecTy at index Idx.
static cl::opt< unsigned > MaxInstrsToScan("vector-combine-max-scan-instrs", cl::init(30), cl::Hidden, cl::desc("Max number of instructions to scan for vector combining."))
static cl::opt< bool > DisableBinopExtractShuffle("disable-binop-extract-shuffle", cl::init(false), cl::Hidden, cl::desc("Disable binop extract to shuffle transforms"))
static unsigned getAlignedNumElements(unsigned MaxIdx, FixedVectorType *LoadTy, const TargetTransformInfo &TTI, const DataLayout &DL)
Given the maximum shuffle index and load vector type, compute the number of elements for the shrunk l...
static InstLane lookThroughShuffles(Value *V, int Lane)
static bool isMemModifiedBetween(BasicBlock::iterator Begin, BasicBlock::iterator End, const MemoryLocation &Loc, AAResults &AA)
static constexpr int Concat[]
A manager for alias analyses.
Class for arbitrary precision integers.
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
uint64_t getZExtValue() const
Get zero extended value.
bool isAllOnes() const
Determine if all bits are set. This is true for zero-width values.
bool ugt(const APInt &RHS) const
Unsigned greater than comparison.
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
unsigned getBitWidth() const
Return the number of bits in the APInt.
static APInt getSignedMaxValue(unsigned numBits)
Gets maximum signed value of APInt for a specific bit width.
bool isNegative() const
Determine sign of this APInt.
unsigned countl_one() const
Count the number of leading one bits.
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
bool isOne() const
Determine if this is a value of 1.
static APInt getOneBitSet(unsigned numBits, unsigned BitNo)
Return an APInt with exactly one bit set in the result.
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & front() const
Get the first element.
size_t size() const
Get the array size.
A function analysis which provides an AssumptionCache.
A cache of @llvm.assume calls within a function.
InstListType::iterator iterator
Instruction iterators...
BinaryOps getOpcode() const
Represents analyses that only rely on functions' control flow.
Value * getArgOperand(unsigned i) const
void addParamAttrs(unsigned ArgNo, const AttrBuilder &B)
Adds attributes to the indicated argument.
static LLVM_ABI CastInst * Create(Instruction::CastOps, Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Provides a way to construct any of the CastInst subclasses using an opcode instead of the subclass's ...
static Type * makeCmpResultType(Type *opnd_type)
Create a result type for fcmp/icmp.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
bool isFPPredicate() const
static LLVM_ABI std::optional< CmpPredicate > getMatching(CmpPredicate A, CmpPredicate B)
Compares two CmpPredicates taking samesign into account and returns the canonicalized CmpPredicate if...
static LLVM_ABI Constant * getExtractElement(Constant *Vec, Constant *Idx, Type *OnlyIfReducedTy=nullptr)
static LLVM_ABI Constant * getBinOpIdentity(unsigned Opcode, Type *Ty, bool AllowRHSConstant=false, bool NSZ=false)
Return the identity constant for a binary opcode.
This is the shared class of boolean and integer constants.
const APInt & getValue() const
Return the constant as an APInt value reference.
This class represents a range of values.
LLVM_ABI ConstantRange urem(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned remainder operation of...
LLVM_ABI ConstantRange binaryAnd(const ConstantRange &Other) const
Return a new range representing the possible values resulting from a binary-and of a value in this ra...
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
static LLVM_ABI Constant * getSplat(ElementCount EC, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
static LLVM_ABI Constant * get(ArrayRef< Constant * > V)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
iterator find(const_arg_type_t< KeyT > Val)
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
Implements a dense probed hash-table based set.
Analysis pass which computes a DominatorTree.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
LLVM_ABI bool isReachableFromEntry(const Use &U) const
Provide an overload for a Use.
LLVM_ABI bool dominates(const BasicBlock *BB, const Use &U) const
Return true if the (end of the) basic block BB dominates the use U.
static constexpr ElementCount get(ScalarTy MinVal, bool Scalable)
Convenience struct for specifying and reasoning about fast-math flags.
bool noSignedZeros() const
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static FixedVectorType * getDoubleElementsVectorType(FixedVectorType *VTy)
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Predicate getSignedPredicate() const
For example, EQ->EQ, SLE->SLE, UGT->SGT, etc.
bool isEquality() const
Return true if this predicate is either EQ or NE.
Common base class shared among various IRBuilders.
LLVM_ABI CallInst * CreateIntrinsicWithoutFolding(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={})
Create a call to intrinsic ID with Args, mangled using OverloadTypes.
Value * CreateNUWMul(Value *LHS, Value *RHS, const Twine &Name="")
Value * CreateInsertElement(Type *VecTy, Value *NewElt, Value *Idx, const Twine &Name="")
Value * CreateExtractElement(Value *Vec, Value *Idx, const Twine &Name="")
LoadInst * CreateAlignedLoad(Type *Ty, Value *Ptr, MaybeAlign Align, const char *Name)
LLVM_ABI Value * CreateSelectFMF(Value *C, Value *True, Value *False, FMFSource FMFSource, const Twine &Name="", Instruction *MDFrom=nullptr)
LLVM_ABI Value * CreateVectorSplat(unsigned NumElts, Value *V, const Twine &Name="")
Return a vector value that contains.
Value * CreateExtractValue(Value *Agg, ArrayRef< unsigned > Idxs, const Twine &Name="")
ConstantInt * getTrue()
Get the constant value for i1 true.
LLVM_ABI Value * CreateSelect(Value *C, Value *True, Value *False, const Twine &Name="", Instruction *MDFrom=nullptr)
Value * CreateFreeze(Value *V, const Twine &Name="")
void SetCurrentDebugLocation(const DebugLoc &L)
Set location information used by debugging information.
Value * CreateLShr(Value *LHS, Value *RHS, const Twine &Name="", bool isExact=false)
Value * CreateCast(Instruction::CastOps Op, Value *V, Type *DestTy, const Twine &Name="", MDNode *FPMathTag=nullptr, FMFSource FMFSource={})
Value * CreateIsNotNeg(Value *Arg, const Twine &Name="")
Return a boolean value testing if Arg > -1.
Value * CreateInBoundsGEP(Type *Ty, Value *Ptr, ArrayRef< Value * > IdxList, const Twine &Name="")
Value * CreatePointerBitCastOrAddrSpaceCast(Value *V, Type *DestTy, const Twine &Name="")
ConstantInt * getInt64(uint64_t C)
Get a constant 64-bit value.
LLVM_ABI Value * CreateOrReduce(Value *Src)
Create a vector int OR reduction intrinsic of the source vector.
ConstantInt * getInt32(uint32_t C)
Get a constant 32-bit value.
Value * CreateCmp(CmpInst::Predicate Pred, Value *LHS, Value *RHS, const Twine &Name="", MDNode *FPMathTag=nullptr)
PHINode * CreatePHI(Type *Ty, unsigned NumReservedValues, const Twine &Name="")
InstTy * Insert(InstTy *I, const Twine &Name="") const
Insert and return the specified instruction.
Value * CreateIsNeg(Value *Arg, const Twine &Name="")
Return a boolean value testing if Arg < 0.
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
LoadInst * CreateLoad(Type *Ty, Value *Ptr, const char *Name)
Provided to resolve 'CreateLoad(Ty, Ptr, "...")' correctly, instead of converting the string to 'bool...
Value * CreateShl(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
LLVM_ABI Value * CreateNAryOp(unsigned Opc, ArrayRef< Value * > Ops, const Twine &Name="", MDNode *FPMathTag=nullptr)
Create either a UnaryOperator or BinaryOperator depending on Opc.
Value * CreateZExt(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNeg=false)
Value * CreateShuffleVector(Value *V1, Value *V2, Value *Mask, const Twine &Name="")
Value * CreateAnd(Value *LHS, Value *RHS, const Twine &Name="")
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
StoreInst * CreateStore(Value *Val, Value *Ptr, bool isVolatile=false)
Value * CreateTrunc(Value *V, Type *DestTy, const Twine &Name="", bool IsNUW=false, bool IsNSW=false)
PointerType * getPtrTy(unsigned AddrSpace=0)
Fetch the type representing a pointer.
Value * CreateBinOp(Instruction::BinaryOps Opc, Value *LHS, Value *RHS, const Twine &Name="", MDNode *FPMathTag=nullptr)
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
Value * CreateFNegFMF(Value *V, FMFSource FMFSource, const Twine &Name="", MDNode *FPMathTag=nullptr)
Value * CreateICmp(CmpInst::Predicate P, Value *LHS, Value *RHS, const Twine &Name="")
Value * CreateOr(Value *LHS, Value *RHS, const Twine &Name="", bool IsDisjoint=false)
IntegerType * getInt8Ty()
Fetch the type representing an 8-bit integer.
LLVM_ABI Value * CreateUnaryIntrinsic(Intrinsic::ID ID, Value *Op, FMFSource FMFSource={}, const Twine &Name="")
Create a call to intrinsic ID with 1 operand which is mangled on its type.
InstSimplifyFolder - Use InstructionSimplify to fold operations to existing values.
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
InstructionWorklist - This is the worklist management logic for InstCombine and other simplification ...
void push(Instruction *I)
Push the instruction onto the worklist stack.
LLVM_ABI void setHasNoUnsignedWrap(bool b=true)
Set or clear the nuw flag on this instruction, which must be an operator which supports this flag.
LLVM_ABI void copyIRFlags(const Value *V, bool IncludeWrapFlags=true)
Convenience method to copy supported exact, fast-math, and (optionally) wrapping flags from V to this...
LLVM_ABI void setHasNoSignedWrap(bool b=true)
Set or clear the nsw flag on this instruction, which must be an operator which supports this flag.
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI void andIRFlags(const Value *V)
Logical 'and' of any supported wrapping, exact, and fast-math flags of V and this instruction.
LLVM_ABI void setNonNeg(bool b=true)
Set or clear the nneg flag on this instruction, which must be a zext instruction.
LLVM_ABI bool comesBefore(const Instruction *Other) const
Given an instruction Other in the same basic block as this instruction, return true if this instructi...
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
LLVM_ABI AAMDNodes getAAMetadata() const
Returns the AA metadata for this instruction.
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
bool isIdempotent() const
Return true if the instruction is idempotent:
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
LLVM_ABI bool hasAllowReassoc() const LLVM_READONLY
Determine whether the allow-reassociation flag is set.
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
unsigned getBitWidth() const
Get the number of bits in this IntegerType.
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
An instruction for reading from memory.
unsigned getPointerAddressSpace() const
Returns the address space of the pointer operand.
void setAlignment(Align Align)
Type * getPointerOperandType() const
Align getAlign() const
Return the alignment of the access that is being performed.
Representation for a specific memory location.
static LLVM_ABI MemoryLocation get(const LoadInst *LI)
Return a location with information about the memory reference by the given instruction.
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
A set of analyses that are preserved following a run of a transformation pass.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
PreservedAnalyses & preserveSet()
Mark an analysis set as preserved.
const SDValue & getOperand(unsigned Num) const
bool contains(const_arg_type key) const
Check if the SetVector contains the given key.
bool empty() const
Determine if the SetVector is empty or not.
bool insert(const value_type &X)
Insert a new element into the SetVector.
This instruction constructs a fixed permutation of two input vectors.
int getMaskValue(unsigned Elt) const
Return the shuffle mask value of this instruction for the given element index.
VectorType * getType() const
Overload to return most specific vector type.
static LLVM_ABI void getShuffleMask(const Constant *Mask, SmallVectorImpl< int > &Result)
Convert the input shuffle mask operand to a vector of integers.
static LLVM_ABI bool isIdentityMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from exactly one source vector without lane crossin...
static void commuteShuffleMask(MutableArrayRef< int > Mask, unsigned InVecNumElts)
Change values in a shuffle permute mask assuming the two vector operands of length InVecNumElts have ...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
void assign(size_type NumElts, ValueParamT Elt)
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
void setAlignment(Align Align)
Analysis pass providing the TargetTransformInfo.
The instances of the Type class are immutable: once they are created, they are never changed.
LLVM_ABI unsigned getIntegerBitWidth() const
bool isPointerTy() const
True if this is an instance of PointerType.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntegerTy() const
True if this is an instance of IntegerType.
bool isFPOrFPVectorTy() const
Return true if this is a FP type or a vector of FP.
A Use represents the edge between a Value definition and its users.
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
const Value * stripAndAccumulateInBoundsConstantOffsets(const DataLayout &DL, APInt &Offset) const
This is a wrapper around stripAndAccumulateConstantOffsets with the in-bounds requirement set to fals...
LLVM_ABI bool hasOneUser() const
Return true if there is exactly one user of this value.
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
iterator_range< user_iterator > users()
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
unsigned getValueID() const
Return an ID for the concrete type of this object.
LLVM_ABI bool hasNUses(unsigned N) const
Return true if this Value has exactly N uses.
LLVM_ABI const Value * stripPointerCasts() const
Strip off pointer casts, all-zero GEPs and address space casts.
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
LLVM_ABI PreservedAnalyses run(Function &F, FunctionAnalysisManager &)
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
Type * getElementType() const
std::pair< iterator, bool > insert(const ValueT &V)
constexpr bool hasKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns true if there exists a value X where RHS.multiplyCoefficientBy(X) will result in a value whos...
constexpr ScalarTy getFixedValue() const
constexpr ScalarTy getKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns a value X where RHS.multiplyCoefficientBy(X) will result in a value whose quantity matches ou...
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
constexpr bool isZero() const
const ParentTy * getParent() const
self_iterator getIterator()
NodeTy * getNextNode()
Get the next node, or nullptr for the list tail.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
Abstract Attribute helper functions.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
const APInt & smin(const APInt &A, const APInt &B)
Determine the smaller of two APInts considered to be signed.
const APInt & smax(const APInt &A, const APInt &B)
Determine the larger of two APInts considered to be signed.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ BasicBlock
Various leaf nodes.
LLVM_ABI Intrinsic::ID getInterleaveIntrinsicID(unsigned Factor)
Returns the corresponding llvm.vector.interleaveN intrinsic for factor N.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SpecificConstantMatch, SrcTy, TargetOpcode::G_SUB > m_Neg(const SrcTy &&Src)
Matches a register negated by a G_SUB.
AllOnesConstantMatch m_AllOnes()
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_combine_and< Ty... > m_CombineAnd(const Ty &...Ps)
Combine pattern matchers matching all of Ps patterns.
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_BSwap(const Opnd0 &Op0)
auto m_Cmp()
Matches any compare instruction and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
auto m_BitReverse(const Opnd0 &Op0)
BinaryOp_match< LHS, RHS, Instruction::URem > m_URem(const LHS &L, const RHS &R)
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
DisjointOr_match< LHS, RHS > m_DisjointOr(const LHS &L, const RHS &R)
BinOpPred_match< LHS, RHS, is_right_shift_op > m_Shr(const LHS &L, const RHS &R)
Matches logical shift operations.
CmpClass_match< LHS, RHS, ICmpInst, true > m_c_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
Matches an ICmp with a predicate over LHS and RHS in either order.
TwoOps_match< Val_t, Idx_t, Instruction::ExtractElement > m_ExtractElt(const Val_t &Val, const Idx_t &Idx)
Matches ExtractElementInst.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
auto m_Value()
Match an arbitrary value and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
auto m_Constant()
Match an arbitrary Constant and ignore it.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
cst_pred_ty< is_non_zero_int > m_NonZeroInt()
Match a non-zero integer or a vector with all non-zero elements.
OneOps_match< OpTy, Instruction::Load > m_Load(const OpTy &Op)
Matches LoadInst.
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
OverflowingBinaryOp_match< LHS, RHS, Instruction::Shl, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWShl(const LHS &L, const RHS &R)
auto m_AnyIntrinsic()
Matches any intrinsic call and ignore it.
OverflowingBinaryOp_match< LHS, RHS, Instruction::Mul, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWMul(const LHS &L, const RHS &R)
BinOpPred_match< LHS, RHS, is_bitwiselogic_op, true > m_c_BitwiseLogic(const LHS &L, const RHS &R)
Matches bitwise logic operations in either order.
CastOperator_match< OpTy, Instruction::BitCast > m_BitCast(const OpTy &Op)
Matches BitCast.
match_combine_or< CastInst_match< OpTy, SExtInst >, NNegZExt_match< OpTy > > m_SExtLike(const OpTy &Op)
Match either "sext" or "zext nneg".
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
auto m_Deinterleave2(const Opnd &Op)
BinaryOp_match< LHS, RHS, Instruction::LShr > m_LShr(const LHS &L, const RHS &R)
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
auto m_Undef()
Match an arbitrary undef constant.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
BinaryOp_match< LHS, RHS, Instruction::Or, true > m_c_Or(const LHS &L, const RHS &R)
Matches an Or with LHS and RHS in either order.
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
@ Valid
The data is already valid.
initializer< Ty > init(const Ty &Val)
DXILDebugInfoMap run(Module &M)
@ User
could "use" a pointer
NodeAddr< PhiNode * > Phi
NodeAddr< UseNode * > Use
friend class Instruction
Iterator for Instructions in a `BasicBlock.
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
LLVM_ABI bool willNotFreeBetween(const Instruction *Assume, const Instruction *CtxI)
Returns true, if no instruction between Assume and CtxI may free (including through synchronization).
detail::zippy< detail::zip_shortest, T, U, Args... > zip(T &&t, U &&u, Args &&...args)
zip iterator for two or more iteratable types.
void stable_sort(R &&Range)
LLVM_ABI cl::opt< bool > ProfcheckDisableMetadataFixes
UnaryFunction for_each(R &&Range, UnaryFunction F)
Provide wrappers to std::for_each which take ranges instead of having to pass begin/end explicitly.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI Intrinsic::ID getMinMaxReductionIntrinsicOp(Intrinsic::ID RdxID)
Returns the min/max intrinsic used when expanding a min/max reduction.
LLVM_ABI bool RecursivelyDeleteTriviallyDeadInstructions(Value *V, const TargetLibraryInfo *TLI=nullptr, MemorySSAUpdater *MSSAU=nullptr, std::function< void(Value *)> AboutToDeleteCallback=std::function< void(Value *)>())
If the specified value is a trivially dead instruction, delete it.
RelativeUniformCounterPtr Values
LLVM_ABI SDValue peekThroughBitcasts(SDValue V)
Return the non-bitcasted source operand of V if it exists.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
LLVM_ABI Value * simplifyUnOp(unsigned Opcode, Value *Op, const SimplifyQuery &Q)
Given operand for a UnaryOperator, fold the result or return null.
scope_exit(Callable) -> scope_exit< Callable >
@ Load
The value being inserted comes from a load (InsertElement only).
auto map_to_vector(ContainerTy &&C, FuncTy &&F)
Map a range to a SmallVector with element types deduced from the mapping.
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
LLVM_ABI unsigned getArithmeticReductionInstruction(Intrinsic::ID RdxID)
Returns the arithmetic instruction opcode used when expanding a reduction.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
constexpr bool isUIntN(unsigned N, uint64_t x)
Checks if an unsigned integer fits into the given (dynamic) bit width.
LLVM_ABI Value * simplifyCall(CallBase *Call, Value *Callee, ArrayRef< Value * > Args, const SimplifyQuery &Q)
Given a callsite, callee, and arguments, fold the result or return null.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
LLVM_ABI bool mustSuppressSpeculation(const LoadInst &LI)
Return true if speculation of the given load must be suppressed to avoid ordering or interfering with...
LLVM_ABI bool widenShuffleMaskElts(int Scale, ArrayRef< int > Mask, SmallVectorImpl< int > &ScaledMask)
Try to transform a shuffle mask by replacing elements with the scaled index for an equivalent mask of...
LLVM_ABI bool isSafeToSpeculativelyExecute(const Instruction *I, const Instruction *CtxI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr, const TargetLibraryInfo *TLI=nullptr, bool UseVariableInfo=true, bool IgnoreUBImplyingAttrs=true)
Return true if the instruction does not have any effects besides calculating the result and does not ...
LLVM_ABI Instruction * propagateMetadata(Instruction *I, ArrayRef< Value * > VL)
Specifically, let Kinds = [MD_tbaa, MD_alias_scope, MD_noalias, MD_fpmath, MD_nontemporal,...
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
unsigned M1(unsigned Val)
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI bool isInstructionTriviallyDead(Instruction *I, const TargetLibraryInfo *TLI=nullptr)
Return true if the result produced by the instruction is not used, and the instruction will return.
LLVM_ABI bool isSplatValue(const Value *V, int Index=-1, unsigned Depth=0)
Return true if each element of the vector value V is poisoned or equal to every other non-poisoned el...
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
auto reverse(ContainerTy &&C)
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
bool isModSet(const ModRefInfo MRI)
void sort(IteratorTy Start, IteratorTy End)
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
LLVM_ABI bool programUndefinedIfPoison(const Instruction *Inst)
LLVM_ABI bool isSafeToLoadUnconditionally(Value *V, Align Alignment, const APInt &Size, const DataLayout &DL, Instruction *ScanFrom, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr, const TargetLibraryInfo *TLI=nullptr)
Return true if we know that executing a load from this value cannot trap.
LLVM_ABI unsigned getDeinterleaveIntrinsicFactor(Intrinsic::ID ID)
Returns the corresponding factor of llvm.vector.deinterleaveN intrinsics.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
LLVM_ABI void propagateIRFlags(Value *I, ArrayRef< Value * > VL, Value *OpValue=nullptr, bool IncludeWrapFlags=true)
Get the intersection (logical and) of all of the potential IR flags of each scalar operation (VL) tha...
MutableArrayRef(T &OneElt) -> MutableArrayRef< T >
constexpr int PoisonMaskElem
IRBuilder(LLVMContext &, FolderTy, InserterTy, MDNode *, ArrayRef< OperandBundleDef >) -> IRBuilder< FolderTy, InserterTy >
LLVM_ABI Value * simplifyBinOp(unsigned Opcode, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a BinaryOperator, fold the result or return null.
LLVM_ABI void narrowShuffleMaskElts(int Scale, ArrayRef< int > Mask, SmallVectorImpl< int > &ScaledMask)
Replace each shuffle mask index with the scaled sequential indices for an equivalent mask of narrowed...
LLVM_ABI Intrinsic::ID getReductionForBinop(Instruction::BinaryOps Opc)
Returns the reduction intrinsic id corresponding to the binary operation.
@ And
Bitwise or logical AND of integers.
LLVM_ABI bool isVectorIntrinsicWithScalarOpAtArg(Intrinsic::ID ID, unsigned ScalarOpdIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic has a scalar operand.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
DWARFExpression::Operation Op
unsigned M0(unsigned Val)
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI unsigned ComputeNumSignBits(const Value *Op, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Return the number of times the sign bit of the register is replicated into the other bits.
constexpr unsigned BitWidth
LLVM_ABI bool isGuaranteedToTransferExecutionToSuccessor(const Instruction *I)
Return true if this function can prove that the instruction I will always transfer execution to one o...
LLVM_ABI Constant * getLosslessInvCast(Constant *C, Type *InvCastTo, unsigned CastOp, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
Try to cast C to InvC losslessly, satisfying CastOp(InvC) equals C, or CastOp(InvC) is a refined valu...
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
constexpr bool isIntN(unsigned N, int64_t x)
Checks if an signed integer fits into the given (dynamic) bit width.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Align commonAlignment(Align A, uint64_t Offset)
Returns the alignment that satisfies both alignments.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
LLVM_ABI Value * simplifyCmpInst(CmpPredicate Predicate, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a CmpInst, fold the result or return null.
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
LLVM_ABI bool isGuaranteedNotToBePoison(const Value *V, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Returns true if V cannot be poison, but may be undef.
LLVM_ABI bool isKnownNonNegative(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Returns true if the give value is known to be non-negative.
LLVM_ABI bool isTriviallyVectorizable(Intrinsic::ID ID)
Identify if the intrinsic is trivially vectorizable.
LLVM_ABI Intrinsic::ID getMinMaxReductionIntrinsicID(Intrinsic::ID IID)
Returns the llvm.vector.reduce min/max intrinsic that corresponds to the intrinsic op.
LLVM_ABI ConstantRange computeConstantRange(const Value *V, bool ForSigned, const SimplifyQuery &SQ, unsigned Depth=0)
Determine the possible constant range of an integer or vector of integer value.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
LLVM_ABI AAMDNodes adjustForAccess(unsigned AccessSize)
Create a new AAMDNode for accessing AccessSize bytes of this AAMDNode.
This struct is a compact representation of a valid (non-zero power of two) alignment.
unsigned countMaxActiveBits() const
Returns the maximum number of bits needed to represent all possible unsigned values with these known ...
unsigned countMinLeadingZeros() const
Returns the minimum number of leading zero bits.
APInt getMaxValue() const
Return the maximal unsigned value possible given these KnownBits.
SimplifyQuery getWithInstruction(const Instruction *I) const