51 cl::desc(
"Use partial reduction intrinsics for "
52 "all supported unordered reductions."));
61 "should not try to widen irregular types");
76 auto IsConsecutiveAccess = [&](
VPValue *Addr,
Type *AccessTy) {
85 if (!VPBB->getParent())
88 auto EndIter = Term ? Term->getIterator() : VPBB->end();
93 VPValue *VPV = Ingredient.getVPSingleValue();
114 IsConsecutiveAccess(VPI->getOperand(0), VPI->getScalarType());
116 nullptr , IsConsecutive,
117 *VPI, Ingredient.getDebugLoc());
119 bool IsConsecutive = IsConsecutiveAccess(
120 VPI->getOperand(1), VPI->getOperand(0)->getScalarType());
122 *
Store, Ingredient.getOperand(1), Ingredient.getOperand(0),
123 nullptr , IsConsecutive, *VPI, Ingredient.getDebugLoc());
126 Ingredient.operands(), *VPI,
127 Ingredient.getDebugLoc(),
GEP);
139 if (VectorID == Intrinsic::experimental_noalias_scope_decl)
144 if (VectorID == Intrinsic::assume ||
145 VectorID == Intrinsic::lifetime_end ||
146 VectorID == Intrinsic::lifetime_start ||
147 VectorID == Intrinsic::sideeffect ||
148 VectorID == Intrinsic::pseudoprobe) {
153 const bool IsSingleScalar = VectorID != Intrinsic::assume &&
154 VectorID != Intrinsic::pseudoprobe;
158 Ingredient.getDebugLoc());
161 *CI, VectorID,
drop_end(Ingredient.operands()), CI->getType(),
162 VPIRFlags(*CI), *VPI, CI->getDebugLoc());
166 CI->getOpcode(), Ingredient.getOperand(0), CI->getType(), CI,
170 *VPI, Ingredient.getDebugLoc());
174 "inductions must be created earlier");
183 "Only recpies with zero or one defined values expected");
184 Ingredient.eraseFromParent();
195 const Loop *L =
nullptr;
200 if (
A->getOpcode() != Instruction::Store ||
201 B->getOpcode() != Instruction::Store)
214 const APInt *Distance;
220 Type *TyA =
A->getOperand(0)->getScalarType();
221 uint64_t SizeA =
DL.getTypeStoreSize(TyA);
222 Type *TyB =
B->getOperand(0)->getScalarType();
223 uint64_t SizeB =
DL.getTypeStoreSize(TyB);
228 uint64_t MaxStoreSize = std::max(SizeA, SizeB);
230 auto VFs =
B->getParent()->getPlan()->vectorFactors();
241 : ExcludeRecipes(ExcludeRecipes.begin(), ExcludeRecipes.end()),
242 GroupLeader(GroupLeader), PSE(&PSE), L(&L) {}
251 return ExcludeRecipes.contains(
Store) ||
252 (
Store && isNoAliasViaDistance(
Store, &GroupLeader));
265 std::optional<SinkStoreInfo> SinkInfo = {}) {
266 bool CheckReads = SinkInfo.has_value();
270 if (SinkInfo && SinkInfo->shouldSkip(R))
274 if (!
R.mayWriteToMemory() && !(CheckReads &&
R.mayReadFromMemory()))
299template <
unsigned Opcode>
304 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
305 "Only Load and Store opcodes supported");
306 constexpr bool IsLoad = (Opcode == Instruction::Load);
309 RecipesByAddressAndType;
314 if (!RepR || RepR->getOpcode() != Opcode || !FilterFn(RepR))
318 VPValue *Addr = RepR->getOperand(IsLoad ? 0 : 1);
322 RecipesByAddressAndType[{AddrSCEV, LoadStoreTy}].push_back(RepR);
327 for (
auto &Group :
Groups) {
342 auto InsertIfValidSinkCandidate = [ScalarVFOnly, &WorkList](
349 if (Candidate->getParent() == SinkTo ||
350 all_of(Candidate->operands(),
351 [](
VPValue *
Op) { return Op->isDefinedOutsideLoopRegions(); }) ||
363 WorkList.
insert({SinkTo, Candidate});
375 for (
auto &Recipe : *VPBB)
377 InsertIfValidSinkCandidate(VPBB,
Op);
381 for (
unsigned I = 0;
I != WorkList.
size(); ++
I) {
384 std::tie(SinkTo, SinkCandidate) = WorkList[
I];
389 auto UsersOutsideSinkTo =
391 return cast<VPRecipeBase>(U)->getParent() != SinkTo;
393 if (
any_of(UsersOutsideSinkTo, [SinkCandidate](
VPUser *U) {
394 return !U->usesFirstLaneOnly(SinkCandidate);
397 bool NeedsDuplicating = !UsersOutsideSinkTo.empty();
399 if (NeedsDuplicating) {
403 if (
auto *SinkCandidateRepR =
408 SinkCandidateRepR->getOpcode(), SinkCandidate->
operands(),
409 nullptr, *SinkCandidateRepR, *SinkCandidateRepR,
413 Clone = SinkCandidate->
clone();
423 InsertIfValidSinkCandidate(SinkTo,
Op);
432 if (EntryBB->getNumSuccessors() != 2)
437 if (!Succ0 || !Succ1)
440 if (Succ0->getNumSuccessors() + Succ1->getNumSuccessors() != 1)
442 if (Succ0->getSingleSuccessor() == Succ1)
444 if (Succ1->getSingleSuccessor() == Succ0)
461 if (!Region1->isReplicator())
463 auto *MiddleBasicBlock =
465 if (!MiddleBasicBlock || !MiddleBasicBlock->empty())
470 if (!Region2 || !Region2->isReplicator())
473 VPValue *Mask1 = Region1->getEntryBranchOnMask()->getOperand(0);
474 VPValue *Mask2 = Region2->getEntryBranchOnMask()->getOperand(0);
475 if (!Mask1 || Mask1 != Mask2)
478 assert(Mask1 && Mask2 &&
"both region must have conditions");
484 if (TransformedRegions.
contains(Region1))
491 if (!Then1 || !Then2)
499 std::optional<VPExecutionFrequency> Freq1 =
502 if (Freq1 && Freq2) {
503 if (Freq2->Freq < Freq1->Freq) {
506 Freq1.emplace(Freq1->Freq, Freq1->IsEstimated || Freq2->IsEstimated);
530 VPValue *Phi1ToMoveV = Phi1ToMove.getVPSingleValue();
536 if (Phi1ToMove.getVPSingleValue()->user_empty()) {
537 Phi1ToMove.eraseFromParent();
540 Phi1ToMove.moveBefore(*Merge2, Merge2->begin());
554 TransformedRegions.
insert(Region1);
557 return !TransformedRegions.
empty();
565 std::string RegionName = (
Twine(
"pred.") + Instr->getOpcodeName()).str();
566 assert(Instr->getParent() &&
"Predicated instruction not in any basic block");
567 auto *BlockInMask = PredRecipe->
getMask();
582 BOMRecipe->setExecutionFrequency(RecipeWithoutMask->getExecutionFrequency(),
584 RecipeWithoutMask->clearExecutionFrequency();
593 Region->setParent(ParentRegion);
599 RecipeWithoutMask->getDebugLoc());
600 Exiting->appendRecipe(PHIRecipe);
613 if (RepR->isPredicated())
632 if (ParentRegion && ParentRegion->
getExiting() == CurrentBlock)
644 if (!VPBB->getParent())
648 if (!PredVPBB || PredVPBB->getNumSuccessors() != 1 ||
657 R.moveBefore(*PredVPBB, PredVPBB->
end());
659 auto *ParentRegion = VPBB->getParent();
660 if (ParentRegion && ParentRegion->getExiting() == VPBB)
661 ParentRegion->setExiting(PredVPBB);
665 return !WorkList.
empty();
672 bool ShouldSimplify =
true;
673 while (ShouldSimplify) {
689 if (!
IV ||
IV->getTruncInst())
704 for (
auto *U : FindMyCast->
users()) {
706 if (UserCast && UserCast->getUnderlyingValue() == IRCast) {
707 FoundUserCast = UserCast;
714 FindMyCast = FoundUserCast;
716 if (FindMyCast !=
IV)
740 PhiR->replaceAllUsesWith(PhiR->getOperand(0));
742 PhiR->eraseFromParent();
799 Def->user_empty() || !Def->getUnderlyingValue() ||
800 (RepR && (RepR->isSingleScalar() || RepR->isPredicated())))
813 Def->getUnderlyingInstr()->getOpcode(), Def->operands(),
815 Def->getUnderlyingInstr());
816 Clone->insertAfter(Def);
817 Def->replaceAllUsesWith(Clone);
818 Def->eraseFromParent();
833 PtrIV->replaceAllUsesWith(PtrAdd);
840 if (HasOnlyVectorVFs &&
none_of(WideIV->users(), [WideIV](
VPUser *U) {
841 return U->usesScalars(WideIV);
850 WrapFlags = {
static_cast<bool>(WideIV->getNoWrapFlagsOrNone().HasNUW),
853 Plan, ID.getKind(), ID.getInductionOpcode(),
855 WideIV->getTruncInst(), WideIV->getStartValue(), WideIV->getStepValue(),
856 WideIV->getDebugLoc(), Builder, WrapFlags);
859 if (!HasOnlyVectorVFs) {
861 "plans containing a scalar VF cannot also include scalable VFs");
862 WideIV->replaceAllUsesWith(Steps);
865 WideIV->replaceUsesWithIf(Steps,
866 [WideIV, HasScalableVF](
VPUser &U,
unsigned) {
868 return U.usesFirstLaneOnly(WideIV);
869 return U.usesScalars(WideIV);
885 return (IntOrFpIV && IntOrFpIV->getTruncInst()) ? nullptr : WideIV;
890 if (!Def || Def->getNumOperands() != 2)
898 auto IsWideIVInc = [&]() {
899 auto &ID = WideIV->getInductionDescriptor();
902 VPValue *IVStep = WideIV->getStepValue();
903 switch (ID.getInductionOpcode()) {
904 case Instruction::Add:
906 case Instruction::FAdd:
908 case Instruction::FSub:
911 case Instruction::Sub: {
931 return IsWideIVInc() ? WideIV :
nullptr;
955 VPValue *FirstActiveLane =
B.createFirstActiveLane(Mask,
DL);
957 B.createScalarZExtOrTrunc(FirstActiveLane, CanonicalIVType,
DL);
958 VPValue *EndValue =
B.createAdd(CanonicalIV, FirstActiveLane,
DL);
963 if (Incoming != WideIV) {
965 EndValue =
B.createAdd(EndValue, One,
DL);
970 VPValue *Start = WideIV->getStartValue();
971 VPValue *Step = WideIV->getStepValue();
972 EndValue =
B.createDerivedIV(
974 Start, EndValue, Step);
988 if (WideIntOrFp && WideIntOrFp->getTruncInst())
998 Start, VectorTC, Step);
1030 assert(EndValue &&
"Must have computed the end value up front");
1035 if (Incoming != WideIV)
1047 auto *Zero = Plan.
getZero(StepTy);
1048 return B.createPtrAdd(EndValue,
B.createSub(Zero, Step),
1053 return B.createNaryOp(
1054 ID.getInductionBinOp()->getOpcode() == Instruction::FAdd
1056 : Instruction::FAdd,
1057 {EndValue, Step}, {ID.getInductionBinOp()->getFastMathFlags()});
1074 const SCEV *Start, *Step;
1092 VPValue *ExitCount = Builder.createOverflowingOp(
1095 return Builder.createDerivedIV(Kind,
nullptr, StartVPV, ExitCount,
1104 VPBuilder VectorPHBuilder(VectorPH, VectorPH->getFirstNonPhi());
1114 EndValues[WideIV] = EndValue;
1124 R.getVPSingleValue()->replaceAllUsesWith(EndValue);
1125 R.eraseFromParent();
1134 for (
auto [Idx, PredVPBB] :
enumerate(ExitVPBB->getPredecessors())) {
1136 if (PredVPBB == MiddleVPBB) {
1138 Plan, ExitIRI->getOperand(Idx), EndValues, PSE);
1141 Plan, ExitIRI->getOperand(Idx), PSE, ResumeTC, L);
1144 Plan, ExitIRI->getOperand(Idx), PSE);
1147 ExitIRI->setOperand(Idx, Escape);
1164 const auto &[V, Inserted] = SCEV2VPV.
try_emplace(ExpR->getSCEV(), ExpR);
1168 ExpR->replaceAllUsesWith(V->second);
1172 ExpR->eraseFromParent();
1202 return Plan.
getZero(Def->getScalarType());
1219 return Def->getOperand(1);
1259 return Plan.
getZero(Def->getScalarType());
1263 Def->getScalarType() ==
A->getScalarType())
1267 if (Def->getScalarType() ==
A->getScalarType())
1277 A->getScalarType() == Def->getScalarType())
1283 return Def->getOperand(0);
1289 return BuildVector->getOperand(BuildVector->getNumOperands() - 1);
1305 return BuildVector->getOperand(BuildVector->getNumOperands() - 2);
1311 return BuildVector->getOperand(Idx);
1315 if (Def->getNumOperands() == 1) {
1316 return Def->getOperand(0);
1320 return Phi->getOperand(0);
1326 if (Def->getNumOperands() == 1 &&
1332 A->getScalarType() == Def->getScalarType())
1360 return VPR->getOperand(0);
1366 return Steps->getOperand(0);
1377 Def->replaceAllUsesWith(V);
1386 RepR && RepR->isPredicated() && RepR->getOpcode() == Instruction::Store &&
1390 RepR->getUnderlyingInstr(), RepR->operandsWithoutMask(),
1391 RepR->isSingleScalar(),
nullptr, *RepR, *RepR,
1392 RepR->getDebugLoc());
1393 Unmasked->insertBefore(RepR);
1407 bool CanCreateNewRecipe =
1413 if (CanCreateNewRecipe &&
1416 return Builder.createLogicalAnd(
X,
Y);
1419 if (CanCreateNewRecipe &&
1424 (!Def->getOperand(0)->hasMoreThanOneUniqueUser() ||
1425 !Def->getOperand(1)->hasMoreThanOneUniqueUser()))
1426 return Builder.createLogicalAnd(
X, Builder.createOr(
Y, Z));
1429 if (CanCreateNewRecipe &&
1433 return Builder.createLogicalOr(Z,
Y);
1437 if (CanCreateNewRecipe &&
1439 return Builder.createNot(
C);
1443 Def->setOperand(0,
C);
1444 Def->setOperand(1,
Y);
1445 Def->setOperand(2,
X);
1450 if (CanCreateNewRecipe &&
1454 Y->getScalarType()->isIntegerTy(1))
1455 return Builder.createOr(
Y, Builder.createLogicalAnd(
X, Z));
1459 if (CanCreateNewRecipe &&
1465 return Builder.createSelect(Builder.createLogicalAnd(Mask0, Mask1),
X,
Y,
1466 Def->getDebugLoc());
1472 Type *TruncTy = Def->getScalarType();
1473 Type *XTy =
X->getScalarType();
1476 unsigned ExtOpcode =
1480 if (
auto *UnderlyingExt =
Y->getUnderlyingValue()) {
1482 Ext->setUnderlyingValue(UnderlyingExt);
1486 auto *Trunc = Builder.createWidenCast(Instruction::Trunc,
X, TruncTy);
1495 return Builder.createSub(Plan.
getZero(
X->getScalarType()),
X,
1496 Def->getDebugLoc(),
"", NW);
1499 if (CanCreateNewRecipe &&
1507 return Builder.createSub(
X,
Y, Def->getDebugLoc(),
"", NW);
1514 Def->getDebugLoc());
1521 MulR->hasNoSignedWrap() &&
1523 return Builder.createNaryOp(
1526 Def->getDebugLoc());
1531 return Builder.createNaryOp(
1544 return match(U, m_Not(m_Specific(Cmp))) ||
1545 (match(U, m_Select(m_Specific(Cmp), m_VPValue(),
1547 U->getOperand(1) != Cmp && U->getOperand(2) != Cmp);
1554 R->setOperand(1,
Y);
1555 R->setOperand(2,
X);
1559 R->replaceAllUsesWith(Cmp);
1564 if (!Cmp->getDebugLoc() && Def->getDebugLoc())
1565 Cmp->setDebugLoc(Def->getDebugLoc());
1578 if (
Op->getNumUsers() > 1 ||
1582 }
else if (!UnpairedCmp) {
1583 UnpairedCmp =
Op->getDefiningRecipe();
1587 UnpairedCmp =
nullptr;
1594 if (NewOps.
size() < Def->getNumOperands())
1601 if (CanCreateNewRecipe &&
1610 X->getScalarType() != Def->getScalarType())
1611 return Builder.createWidenCast(Instruction::Trunc,
X, Def->getScalarType());
1618 Def->getScalarType()->isIntegerTy(1)) {
1619 Def->setOperand(1, Plan.
getTrue());
1620 Def->setOperand(0,
Y);
1630 Def->replaceUsesWithIf(Def->getOperand(0), [Def](
VPUser &U,
unsigned) {
1631 return U.usesFirstLaneOnly(Def);
1641 "broadcast operand must be single-scalar");
1642 Def->setOperand(0, Z);
1647 Def->replaceUsesWithIf(
1648 X, [Def](
const VPUser &U,
unsigned) {
return U.usesScalars(Def); });
1660 return Builder.createNaryOp(Instruction::ExtractElement, {
X, LaneToExtract},
1661 Def->getDebugLoc());
1674 if (IVInc->getNumUsers() == 2) {
1679 if (Phi->getNumUsers() == 1 || (Phi->getNumUsers() == 2 && Inc)) {
1680 Def->replaceAllUsesWith(IVInc);
1682 Inc->replaceAllUsesWith(Phi);
1683 Phi->setOperand(0,
Y);
1693 Def->replaceUsesWithIf(StartV, [](
const VPUser &U,
unsigned Idx) {
1695 return PhiR && PhiR->isInLoop();
1712 Def->replaceAllUsesWith(New);
1713 Def->eraseFromParent();
1716 Def->eraseFromParent();
1735 R.getVPSingleValue()->replaceAllUsesWith(
X);
1751 while (!Worklist.
empty()) {
1760 R->replaceAllUsesWith(
1761 Builder.createLogicalAnd(HeaderMask, Builder.createLogicalAnd(
X,
Y)));
1765static std::optional<Instruction::BinaryOps>
1768 case Intrinsic::masked_udiv:
1769 return Instruction::UDiv;
1770 case Intrinsic::masked_sdiv:
1771 return Instruction::SDiv;
1772 case Intrinsic::masked_urem:
1773 return Instruction::URem;
1774 case Intrinsic::masked_srem:
1775 return Instruction::SRem;
1792 if (RepR && (RepR->isSingleScalar() || RepR->isPredicated()))
1796 if (RepR && RepR->getOpcode() == Instruction::Store &&
1799 RepOrWidenR->getUnderlyingInstr(), RepOrWidenR->operands(),
1800 true ,
nullptr , *RepR ,
1801 *RepR , RepR->getDebugLoc());
1802 Clone->insertBefore(RepOrWidenR);
1804 VPValue *ExtractOp = Clone->getOperand(0);
1810 Clone->setOperand(0, ExtractOp);
1811 RepR->eraseFromParent();
1823 VPValue *SafeDivisor = Builder.createSelect(
1824 IntrR->getOperand(2), IntrR->getOperand(1),
1826 VPValue *Clone = Builder.createNaryOp(
1827 *
Opc, {IntrR->getOperand(0), SafeDivisor},
1830 IntrR->eraseFromParent();
1839 auto IntroducesBCastOf = [](
const VPValue *
Op) {
1848 return !U->usesScalars(
Op);
1852 if (
any_of(RepOrWidenR->users(), IntroducesBCastOf(RepOrWidenR)) &&
1855 make_filter_range(Op->users(), not_equal_to(RepOrWidenR)),
1856 IntroducesBCastOf(Op)))
1860 bool LiveInNeedsBroadcast =
1861 isa<VPIRValue>(Op) && !isa<VPConstant>(Op);
1862 auto *OpR = dyn_cast<VPReplicateRecipe>(Op);
1863 return LiveInNeedsBroadcast || (OpR && OpR->isSingleScalar());
1870 RepOrWidenR->getUnderlyingInstr());
1871 Clone->insertBefore(RepOrWidenR);
1872 RepOrWidenR->replaceAllUsesWith(Clone);
1874 RepOrWidenR->eraseFromParent();
1910 if (Blend->isNormalized() || !
match(Blend->getMask(0),
m_False()))
1911 UniqueValues.
insert(Blend->getIncomingValue(0));
1912 for (
unsigned I = 1;
I != Blend->getNumIncomingValues(); ++
I)
1914 UniqueValues.
insert(Blend->getIncomingValue(
I));
1916 if (UniqueValues.
size() == 1) {
1917 Blend->replaceAllUsesWith(*UniqueValues.
begin());
1918 Blend->eraseFromParent();
1922 if (Blend->isNormalized())
1928 unsigned StartIndex = 0;
1929 for (
unsigned I = 0;
I != Blend->getNumIncomingValues(); ++
I) {
1941 OperandsWithMask.
push_back(Blend->getIncomingValue(StartIndex));
1943 for (
unsigned I = 0;
I != Blend->getNumIncomingValues(); ++
I) {
1944 if (
I == StartIndex)
1946 OperandsWithMask.
push_back(Blend->getIncomingValue(
I));
1947 OperandsWithMask.
push_back(Blend->getMask(
I));
1952 OperandsWithMask, *Blend, Blend->getDebugLoc());
1953 NewBlend->insertBefore(&R);
1955 VPValue *DeadMask = Blend->getMask(StartIndex);
1957 Blend->eraseFromParent();
1962 if (NewBlend->getNumOperands() == 3 &&
1964 VPValue *Inc0 = NewBlend->getOperand(0);
1965 VPValue *Inc1 = NewBlend->getOperand(1);
1966 VPValue *OldMask = NewBlend->getOperand(2);
1967 NewBlend->setOperand(0, Inc1);
1968 NewBlend->setOperand(1, Inc0);
1969 NewBlend->setOperand(2, NewMask);
1996 APInt MaxVal = AlignedTC - 1;
1999 unsigned NewBitWidth =
2005 bool MadeChange =
false;
2030 "canonical IV is not expected to have a truncation");
2035 NewWideIV->insertBefore(WideIV);
2042 Cmp->replaceAllUsesWith(
2043 VPBuilder(Cmp).createICmp(Cmp->getPredicate(), NewWideIV, NewBTC));
2057 return any_of(
Cond->getDefiningRecipe()->operands(), [&Plan, BestVF, BestUF,
2059 return isConditionTrueViaVFAndUF(C, Plan, BestVF, BestUF, PSE);
2073 const SCEV *VectorTripCount =
2078 "Trip count SCEV must be computable");
2093 bool MadeChange =
false;
2101 for (
VPBasicBlock *VPBB : {PreheaderVPBB, ExitingVPBB}) {
2110 Builder.setInsertPoint(Extract);
2113 Start = Builder.createAdd(
2118 Extract->eraseFromParent();
2133 auto *Term = &ExitingVPBB->
back();
2139 bool MatchedCanIVInc =
2145 if (MatchedCanIVInc ||
2153 const SCEV *VectorTripCount =
2159 "Trip count SCEV must be computable");
2178 Term->setOperand(1, Plan.
getTrue());
2183 {}, Term->getDebugLoc());
2185 Term->eraseFromParent();
2193 assert(Plan.
hasVF(BestVF) &&
"BestVF is not available in Plan");
2194 assert(Plan.
hasUF(BestUF) &&
"BestUF is not available in Plan");
2213 RecurKind RK = PhiR->getRecurrenceKind();
2220 RecWithFlags->dropPoisonGeneratingFlags();
2226struct VPCSEDenseMapInfo :
public DenseMapInfo<VPSingleDefRecipe *> {
2235 return GEP->getSourceElementType();
2238 .Case<VPVectorPointerRecipe, VPWidenGEPRecipe>(
2239 [](
auto *
I) {
return I->getSourceElementType(); })
2240 .
Default([](
auto *) {
return nullptr; });
2244 static bool canHandle(
const VPSingleDefRecipe *Def) {
2253 if (!
C || (!
C->first && (
C->second == Instruction::InsertValue ||
2254 C->second == Instruction::ExtractValue)))
2260 if (
Def->mayWriteToMemory())
2262 return !
Def->mayReadFromMemory() ||
2267 static unsigned getHashValue(
const VPSingleDefRecipe *Def) {
2270 getGEPSourceElementType(Def),
Def->getScalarType(),
2273 if (RFlags->hasPredicate())
2276 return hash_combine(Result, SIVSteps->getInductionOpcode());
2285 static bool isEqual(
const VPSingleDefRecipe *L,
const VPSingleDefRecipe *R) {
2286 if (
L->getVPRecipeID() !=
R->getVPRecipeID() ||
2289 getGEPSourceElementType(L) != getGEPSourceElementType(R) ||
2291 !
equal(
L->operands(),
R->operands()))
2295 "must have valid opcode info for both recipes");
2297 if (LFlags->hasPredicate() &&
2298 LFlags->getPredicate() !=
2302 if (LSIV->getInductionOpcode() !=
2317 const VPRegionBlock *RegionL =
L->getRegion();
2318 const VPRegionBlock *RegionR =
R->getRegion();
2321 L->getParent() !=
R->getParent())
2323 return L->getScalarType() ==
R->getScalarType();
2342 if (R.mayWriteToMemory())
2345 if (!Def || !VPCSEDenseMapInfo::canHandle(Def))
2348 auto [It, Inserted] =
2349 (IsLoad ? LoadCSEMap : CSEMap).try_emplace(Def, Def);
2354 if (!VPDT.
dominates(V->getParent(), VPBB))
2359 if (EarlierLoad->getAlign() <
Load->getAlign()) {
2366 EarlierLoad->intersect(*
Load);
2371 Def->replaceAllUsesWith(V);
2382 bool Sinking =
false) {
2411 "Expected vector prehader's successor to be the vector loop region");
2419 return !Op->isDefinedOutsideLoopRegions();
2422 R.moveBefore(*Preheader, Preheader->
end());
2442 assert(!RepR->isPredicated() &&
2443 "Expected prior transformation of predicated replicates to "
2444 "replicate regions");
2449 if (!RepR->isSingleScalar())
2453 if (RepR->getOpcode() == Instruction::Store &&
2454 !RepR->getOperand(1)->isDefinedOutsideLoopRegions())
2459 assert((!R.mayWriteToMemory() ||
2460 (RepR && RepR->getOpcode() == Instruction::Store &&
2461 RepR->getOperand(1)->isDefinedOutsideLoopRegions())) &&
2462 "The only recipes that may write to memory are expected to be "
2463 "stores with invariant pointer-operand");
2473 if (
any_of(Def->users(), [&SinkBB, &LoopRegion](
VPUser *U) {
2474 auto *UserR = cast<VPRecipeBase>(U);
2475 VPBasicBlock *Parent = UserR->getParent();
2477 if (SinkBB && SinkBB != Parent)
2482 return UserR->isPhi() || Parent->getEnclosingLoopRegion() ||
2483 Parent->getSinglePredecessor() != LoopRegion;
2493 "Defining block must dominate sink block");
2518 VPValue *ResultVPV = R.getVPSingleValue();
2520 unsigned NewResSizeInBits = MinBWs.
lookup(UI);
2521 if (!NewResSizeInBits)
2534 (void)OldResSizeInBits;
2542 VPW->dropPoisonGeneratingFlags();
2544 assert((OldResSizeInBits != NewResSizeInBits ||
2546 "Only ICmps should not need extending the result.");
2552 if (OldResSizeInBits != NewResSizeInBits) {
2554 Instruction::ZExt, ResultVPV, OldResTy);
2556 Ext->setOperand(0, ResultVPV);
2566 unsigned OpSizeInBits =
Op->getScalarType()->getScalarSizeInBits();
2567 if (OpSizeInBits == NewResSizeInBits)
2569 assert(OpSizeInBits > NewResSizeInBits &&
"nothing to truncate");
2570 auto [ProcessedIter, Inserted] = ProcessedTruncs.
try_emplace(
Op);
2576 Builder.setInsertPoint(&R);
2577 ProcessedIter->second =
2578 Builder.createWidenCast(Instruction::Trunc,
Op, NewResTy);
2580 Op = ProcessedIter->second;
2584 NWR->insertBefore(&R);
2588 VPValue *Replacement = NWR->getVPSingleValue();
2589 if (OldResSizeInBits != NewResSizeInBits)
2595 R.eraseFromParent();
2601 std::optional<VPDominatorTree> VPDT;
2609 bool SimplifiedPhi =
false;
2619 assert(VPBB->getNumSuccessors() == 2 &&
2620 "Two successors expected for BranchOnCond");
2621 unsigned RemovedIdx;
2632 "There must be a single edge between VPBB and its successor");
2637 SimplifiedPhi =
true;
2641 if (!PhiR || PhiR->getNumIncoming() != 1)
2643 PhiR->replaceAllUsesWith(PhiR->getOperand(0));
2644 PhiR->eraseFromParent();
2649 VPBB->back().eraseFromParent();
2661 if (Reachable.contains(
B))
2672 for (
VPValue *Def : R.definedValues())
2673 Def->replaceAllUsesWith(&Tmp);
2674 R.eraseFromParent();
2678 return SimplifiedPhi;
2704 auto GetSimplifiedLiveInViaSCEV = [&](
VPValue *VPV) ->
VPValue * {
2713 if (
VPValue *SimplifiedLiveIn = GetSimplifiedLiveInViaSCEV(LiveIn))
2714 LiveIn->replaceAllUsesWith(SimplifiedLiveIn);
2725 "expected to run before loop regions are created");
2727 auto CanUseVersionedStride = [&VPDT, Header = Header, &Plan](
VPUser &U,
2734 return VPDT.
dominates(Header, R->getParent());
2738 Value *StrideV = Stride->getValue();
2739 const APInt *StrideConst;
2746 CanUseVersionedStride);
2760 CanUseVersionedStride);
2762 RewriteMap[StrideV] = StrideExpr;
2769 const SCEV *ScevExpr = ExpSCEV->getSCEV();
2772 if (NewSCEV != ScevExpr) {
2774 ExpSCEV->replaceAllUsesWith(NewExp);
2785 auto CollectPoisonGeneratingInstrsInBackwardSlice([&](
VPRecipeBase *Root) {
2790 while (!Worklist.
empty()) {
2793 if (!Visited.
insert(CurRec).second)
2815 RecWithFlags->isDisjoint()) {
2818 Builder.createAdd(
A,
B, RecWithFlags->getDebugLoc());
2819 New->setUnderlyingValue(RecWithFlags->getUnderlyingValue());
2820 RecWithFlags->replaceAllUsesWith(New);
2821 RecWithFlags->eraseFromParent();
2824 RecWithFlags->dropPoisonGeneratingFlags();
2829 assert((!Instr || !Instr->hasPoisonGeneratingFlags()) &&
2830 "found instruction with poison generating flags not covered by "
2831 "VPRecipeWithIRFlags");
2836 if (
VPRecipeBase *OpDef = Operand->getDefiningRecipe())
2858 VPRecipeBase *AddrDef = WidenRec->getAddr()->getDefiningRecipe();
2859 if (AddrDef && WidenRec->isConsecutive() && WidenRec->getMask() &&
2860 match(WidenRec->getMask(), m_UnlessHdrMask))
2861 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2863 VPRecipeBase *AddrDef = InterleaveRec->getAddr()->getDefiningRecipe();
2864 if (AddrDef && InterleaveRec->getMask() &&
2865 match(InterleaveRec->getMask(), m_UnlessHdrMask))
2866 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2876 const bool &EpilogueAllowed) {
2877 if (InterleaveGroups.empty())
2888 IRMemberToRecipe[&MemR->getIngredient()] = MemR;
2895 for (
const auto *IG : InterleaveGroups) {
2898 for (
auto *Member : IG->members())
2900 StartMember = Member;
2908 for (
unsigned I = 0;
I < IG->getFactor(); ++
I) {
2914 StoredValues.
push_back(StoreR->getStoredValue());
2921 bool NeedsMaskForGaps =
2922 (IG->requiresScalarEpilogue() && !EpilogueAllowed) ||
2923 (!StoredValues.
empty() && !IG->isFull());
2926 auto *InsertPos = IRMemberToRecipe.
lookup(IRInsertPos);
2930 "Dead member in non-load group?");
2935 InsertPos->getAsRecipe()))
2936 InsertPos = MemberR;
2937 IRInsertPos = &InsertPos->getIngredient();
2947 VPValue *Addr = Start->getAddr();
2949 if (IG->getIndex(StartMember) != 0 ||
2957 assert(IG->getIndex(IRInsertPos) != 0 &&
2958 "index of insert position shouldn't be zero");
2962 IG->getIndex(IRInsertPos),
2966 Addr =
B.createNoWrapPtrAdd(InsertPos->getAddr(), OffsetVPV, NW);
2972 if (IG->isReverse()) {
2975 -(int64_t)IG->getFactor(), NW, InsertPosR->
getDebugLoc());
2976 ReversePtr->insertBefore(InsertPosR);
2980 IG, Addr, StoredValues, InsertPos->getMask(), NeedsMaskForGaps,
2982 VPIG->insertBefore(InsertPosR);
2985 for (
unsigned i = 0; i < IG->getFactor(); ++i)
2988 if (!Member->getType()->isVoidTy()) {
3059 VPValue *UncountableCondition =
nullptr;
3066 Worklist.
push_back(UncountableCondition);
3067 while (!Worklist.
empty()) {
3071 if (V->isDefinedOutsideLoopRegions())
3077 if (V->getNumUsers() > 1)
3107 if (Recipes.
empty() ||
3111 return UncountableCondition;
3167 for (
auto &Exit : Exits) {
3168 if (Exit.EarlyExitingVPBB == LatchVPBB)
3172 cast<VPIRPhi>(&R)->removeIncomingValueFor(Exit.EarlyExitingVPBB);
3173 Exit.EarlyExitingVPBB->getTerminator()->eraseFromParent();
3199 assert(
Load &&
"Couldn't find exactly one load");
3202 "Uncountable exit condition load is conditional.");
3216 DL.getTypeStoreSize(
Load->getScalarType()).getFixedValue());
3240 while (InsertIt != HeaderVPBB->
end() &&
3242 erase(ConditionRecipes, &*InsertIt);
3245 for (
auto *Recipe :
reverse(ConditionRecipes))
3246 Recipe->moveBefore(*HeaderVPBB, InsertIt);
3250 VPBuilder MaskBuilder(HeaderVPBB, InsertIt);
3252 Type *IVScalarTy =
IV->getScalarType();
3258 "uncountable.exit.mask");
3263 if (R.mayReadOrWriteMemory() && &R !=
Load) {
3265 if (!VPDT.
dominates(R.getParent(), LatchVPBB))
3275 "Expected BranchOnCond terminator for MiddleVPBB");
3286 auto Phis = ScalarPH->
phis();
3296 "Continuing from different IV");
3318 VPBuilder LatchBuilder(LatchVPBB->getTerminator());
3320 for (
auto [EarlyExitingVPBB, ExitBlock] :
3324 VPValue *CondOfEarlyExitingVPBB;
3325 [[maybe_unused]]
bool Matched =
3326 match(EarlyExitingVPBB->getTerminator(),
3328 assert(Matched &&
"Terminator must be BranchOnCond");
3332 VPBuilder EarlyExitingBuilder(EarlyExitingVPBB->getTerminator());
3333 auto *CondToEarlyExit = EarlyExitingBuilder.
createNaryOp(
3335 TrueSucc == ExitBlock
3336 ? CondOfEarlyExitingVPBB
3337 : EarlyExitingBuilder.
createNot(CondOfEarlyExitingVPBB));
3343 "exit condition must dominate the latch");
3351 assert(!Exits.
empty() &&
"must have at least one early exit");
3358 for (
const auto &[Num, VPB] :
enumerate(RPOT))
3361 return RPOIdx[
A.EarlyExitingVPBB] < RPOIdx[
B.EarlyExitingVPBB];
3367 for (
unsigned I = 0;
I + 1 < Exits.
size(); ++
I)
3368 for (
unsigned J =
I + 1; J < Exits.
size(); ++J)
3370 Exits[
I].EarlyExitingVPBB) &&
3371 "RPO sort must place dominating exits before dominated ones");
3377 VPValue *Combined = Exits[0].CondToExit;
3390 "Unexpected terminator");
3391 VPValue *IsLatchExitTaken = LatchExitingBranch->getOperand(0);
3392 DebugLoc LatchDL = LatchExitingBranch->getDebugLoc();
3393 LatchExitingBranch->eraseFromParent();
3396 {IsAnyExitTaken, IsLatchExitTaken}, LatchDL);
3397 LatchVPBB->clearSuccessors();
3402 LatchVPBB->setSuccessors({MiddleVPBB, MiddleVPBB, HeaderVPBB});
3403 MiddleVPBB->clearPredecessors();
3404 MiddleVPBB->setPredecessors({LatchVPBB, LatchVPBB});
3406 Plan, Exits, HeaderVPBB, LatchVPBB, MiddleVPBB, TheLoop, PSE, DT, AC);
3411 for (
unsigned Idx = 0; Idx != Exits.
size(); ++Idx) {
3415 VectorEarlyExitVPBBs[Idx] = VectorEarlyExitVPBB;
3423 Exits.
size() == 1 ? VectorEarlyExitVPBBs[0]
3426 LatchVPBB->setSuccessors({DispatchVPBB, MiddleVPBB, HeaderVPBB});
3458 for (
auto [Exit, VectorEarlyExitVPBB] :
3459 zip_equal(Exits, VectorEarlyExitVPBBs)) {
3460 auto &[EarlyExitingVPBB, EarlyExitVPBB,
_] = Exit;
3472 ExitIRI->getIncomingValueForBlock(EarlyExitingVPBB);
3473 VPValue *NewIncoming = IncomingVal;
3475 VPBuilder EarlyExitBuilder(VectorEarlyExitVPBB);
3480 ExitIRI->removeIncomingValueFor(EarlyExitingVPBB);
3481 ExitIRI->addIncoming(NewIncoming);
3484 EarlyExitingVPBB->getTerminator()->eraseFromParent();
3518 bool IsLastDispatch = (
I + 2 == Exits.
size());
3520 IsLastDispatch ? VectorEarlyExitVPBBs.
back()
3526 VectorEarlyExitVPBBs[
I]->setPredecessors({CurrentBB});
3529 CurrentBB = FalseBB;
3544 VPValue *VecOp = Red->getVecOp();
3547 if (Red->isPartialReduction())
3551 auto IsExtendedRedValidAndClampRange =
3564 "getExtendedReductionCost only supports integer types");
3565 ExtRedCost = Ctx.TTI.getExtendedReductionCost(
3566 Opcode, ExtOpc == Instruction::CastOps::ZExt, RedTy, SrcVecTy,
3567 Red->getFastMathFlagsOrNone(),
CostKind);
3568 return ExtRedCost.
isValid() && ExtRedCost < ExtCost + RedCost;
3576 IsExtendedRedValidAndClampRange(
3597 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3598 Opcode != Instruction::FAdd)
3602 if (Red->isPartialReduction())
3608 auto IsMulAccValidAndClampRange =
3620 (Ext0->getOpcode() != Ext1->getOpcode() ||
3621 Ext0->getOpcode() == Instruction::CastOps::FPExt))
3625 !Ext0 || Ext0->getOpcode() == Instruction::CastOps::ZExt;
3627 MulAccCost = Ctx.TTI.getMulAccReductionCost(IsZExt, Opcode, RedTy,
3634 ExtCost += Ext0->computeCost(VF, Ctx);
3636 ExtCost += Ext1->computeCost(VF, Ctx);
3638 ExtCost += OuterExt->computeCost(VF, Ctx);
3640 return MulAccCost.
isValid() &&
3641 MulAccCost < ExtCost + MulCost + RedCost;
3646 VPValue *VecOp = Red->getVecOp();
3684 Builder.createWidenCast(Instruction::CastOps::Trunc, ValB, NarrowTy);
3686 ValB = ExtB = Builder.createWidenCast(ExtOpc, Trunc, WideTy);
3687 Mul->setOperand(1, ExtB);
3697 ExtendAndReplaceConstantOp(RecipeA, RecipeB,
B,
Mul);
3702 IsMulAccValidAndClampRange(
Mul, RecipeA, RecipeB,
nullptr)) {
3709 if (!
Sub && IsMulAccValidAndClampRange(
Mul,
nullptr,
nullptr,
nullptr))
3726 ExtendAndReplaceConstantOp(Ext0, Ext1,
B,
Mul);
3735 (Ext->getOpcode() == Ext0->getOpcode() || Ext0 == Ext1) &&
3736 Ext0->getOpcode() == Ext1->getOpcode() &&
3737 IsMulAccValidAndClampRange(
Mul, Ext0, Ext1, Ext) &&
Mul->hasOneUse()) {
3739 Ext0->getOpcode(), Ext0->getOperand(0), Ext->getScalarType(),
nullptr,
3740 *Ext0, *Ext0, Ext0->getDebugLoc());
3741 NewExt0->insertBefore(Ext0);
3746 Ext->getScalarType(),
nullptr, *Ext1,
3747 *Ext1, Ext1->getDebugLoc());
3750 auto *NewMul =
Mul->cloneWithOperands({NewExt0, NewExt1});
3751 NewMul->insertBefore(
Mul);
3752 Ext->replaceAllUsesWith(NewMul);
3753 Ext->eraseFromParent();
3754 Mul->eraseFromParent();
3768 if (Red->isPartialReduction())
3772 auto IP = std::next(Red->getIterator());
3773 auto *VPBB = Red->getParent();
3783 Red->replaceAllUsesWith(AbstractR);
3806 return CommonMetadata;
3809template <
unsigned Opcode>
3814 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
3815 "Only Load and Store opcodes supported");
3816 [[maybe_unused]]
constexpr bool IsLoad = (Opcode == Instruction::Load);
3823 for (
auto Recipes :
Groups) {
3824 if (Recipes.size() < 2)
3829 "Expected all recipes in group to have the same load-store type");
3836 VPValue *MaskI = RecipeI->getMask();
3842 bool HasComplementaryMask =
false;
3847 VPValue *MaskJ = RecipeJ->getMask();
3856 if (HasComplementaryMask) {
3857 assert(Group.
size() >= 2 &&
"must have at least 2 entries");
3867template <
typename InstType>
3885 for (
auto &Group :
Groups) {
3905 return R->isSingleScalar() == IsSingleScalar;
3907 "all members in group must agree on IsSingleScalar");
3912 LoadWithMinAlign->getUnderlyingInstr(), {EarliestLoad->getOperand(0)},
3913 IsSingleScalar,
nullptr, *EarliestLoad, CommonMetadata);
3915 UnpredicatedLoad->insertBefore(EarliestLoad);
3919 Load->replaceAllUsesWith(UnpredicatedLoad);
3920 Load->eraseFromParent();
3929 if (!StoreLoc || !StoreLoc->AATags.Scope)
3936 SinkStoreInfo SinkInfo(StoresToSink, *StoresToSink[0], PSE, L);
3948 for (
auto &Group :
Groups) {
3961 VPValue *SelectedValue = Group[0]->getOperand(0);
3964 bool IsSingleScalar = Group[0]->isSingleScalar();
3965 for (
unsigned I = 1;
I < Group.size(); ++
I) {
3966 assert(IsSingleScalar == Group[
I]->isSingleScalar() &&
3967 "all members in group must agree on IsSingleScalar");
3968 VPValue *Mask = Group[
I]->getMask();
3970 SelectedValue = Builder.createSelect(
3973 Value->getScalarType()));
3981 StoreWithMinAlign->getUnderlyingInstr(),
3982 {SelectedValue, LastStore->getOperand(1)}, IsSingleScalar,
3983 nullptr, *LastStore, CommonMetadata);
3984 UnpredicatedStore->insertBefore(*InsertBB, LastStore->
getIterator());
3988 Store->eraseFromParent();
4003 VPValue *OpV,
unsigned Idx,
bool IsScalable) {
4008 if (Member0Op == OpV)
4018 return !IsScalable && !W->getMask() && W->isConsecutive() &&
4021 return IR->getInterleaveGroup()->isFull() &&
IR->getVPValue(Idx) == OpV;
4036 if (R->getScalarType() != WideMember0->getScalarType())
4038 if (R->hasPredicate() && R->getPredicate() != WideMember0->getPredicate())
4042 for (
unsigned Idx = 0; Idx != WideMember0->getNumOperands(); ++Idx) {
4045 OpsI.
push_back(
Op->getDefiningRecipe()->getOperand(Idx));
4050 if (
any_of(
enumerate(OpsI), [WideMember0, Idx, IsScalable](
const auto &
P) {
4051 const auto &[OpIdx, OpV] =
P;
4052 return !
canNarrowLoad(WideMember0, Idx, OpV, OpIdx, IsScalable);
4063static std::optional<ElementCount>
4067 if (!InterleaveR || InterleaveR->
getMask())
4068 return std::nullopt;
4070 Type *GroupElementTy =
nullptr;
4074 return Op->getScalarType() == GroupElementTy;
4076 return std::nullopt;
4080 return Op->getScalarType() == GroupElementTy;
4082 return std::nullopt;
4086 if (IG->getFactor() != IG->getNumMembers())
4087 return std::nullopt;
4093 assert(
Size.isScalable() == VF.isScalable() &&
4094 "if Size is scalable, VF must be scalable and vice versa");
4095 return Size.getKnownMinValue();
4099 unsigned MinVal = VF.getKnownMinValue();
4101 if (IG->getFactor() == MinVal && GroupSize == GetVectorBitWidthForVF(VF))
4104 return std::nullopt;
4112 return RepR && RepR->isSingleScalar();
4126 if (V->isDefinedOutsideLoopRegions()) {
4129 return M->isDefinedOutsideLoopRegions() &&
4130 M->getScalarType() == V->getScalarType();
4132 "expected distinct loop-invariant values of matching scalar type");
4147 for (
unsigned Idx = 0,
E = WideMember0->getNumOperands(); Idx !=
E; ++Idx) {
4149 for (
VPValue *Member : Members)
4150 OpsI.
push_back(Member->getDefiningRecipe()->getOperand(Idx));
4151 WideMember0->setOperand(
4160 auto *LI =
cast<LoadInst>(LoadGroup->getInterleaveGroup()->getInsertPos());
4162 *LI, LoadGroup->getAddr(), LoadGroup->getMask(),
true,
4163 *LoadGroup, LoadGroup->getDebugLoc());
4169 assert(RepR->isSingleScalar() && RepR->getOpcode() == Instruction::Load &&
4170 "must be a single scalar load");
4171 NarrowedOps.
insert(RepR);
4176 VPValue *PtrOp = WideLoad->getAddr();
4178 PtrOp = VecPtr->getOperand(0);
4183 nullptr, {}, *WideLoad);
4184 N->insertBefore(WideLoad);
4189std::unique_ptr<VPlan>
4209 "unexpected branch-on-count");
4212 std::optional<ElementCount> VFToOptimize;
4226 if (R.mayWriteToMemory() && !InterleaveR)
4232 return any_of(V->users(), [&](VPUser *U) {
4233 auto *UR = cast<VPRecipeBase>(U);
4234 return UR->getParent()->getParent() != VectorLoop;
4251 std::optional<ElementCount> NarrowedVF =
4253 if (!NarrowedVF || (VFToOptimize && NarrowedVF != VFToOptimize))
4255 VFToOptimize = NarrowedVF;
4258 if (InterleaveR->getStoredValues().empty())
4263 auto *Member0 = InterleaveR->getStoredValues()[0];
4273 VPRecipeBase *DefR = Op.value()->getDefiningRecipe();
4276 auto *IR = dyn_cast<VPInterleaveRecipe>(DefR);
4277 return IR && IR->getInterleaveGroup()->isFull() &&
4278 IR->getVPValue(Op.index()) == Op.value();
4287 VFToOptimize->isScalable()))
4292 if (StoreGroups.empty())
4296 bool RequiresScalarEpilogue =
4307 std::unique_ptr<VPlan> NewPlan;
4309 NewPlan = std::unique_ptr<VPlan>(Plan.
duplicate());
4310 Plan.
setVF(*VFToOptimize);
4311 NewPlan->removeVF(*VFToOptimize);
4318 for (
auto *StoreGroup : StoreGroups) {
4320 NarrowedOps, Preheader);
4326 StoreGroup->getDebugLoc());
4333 Type *CanIVTy = VectorLoop->getCanonicalIVType();
4339 if (VFToOptimize->isScalable()) {
4342 Step = PHBuilder.createOverflowingOp(Instruction::Mul, {VScale,
UF},
4350 materializeVectorTripCount(Plan, VectorPH,
false,
4351 RequiresScalarEpilogue, Step);
4356 removeDeadRecipes(Plan);
4359 "All VPVectorPointerRecipes should have been removed");
4379 "Cannot handle loops with uncountable early exits");
4386 assert(RecurSplice &&
"expected FirstOrderRecurrenceSplice");
4393 if (
any_of(RecurSplice->users(),
4394 [](
VPUser *U) { return !cast<VPRecipeBase>(U)->getRegion(); }) &&
4475 {},
"vector.recur.extract.for.phi");
4478 ExitPhi->replaceUsesOfWith(ExtractR, PenultimateElement);
4492 VPValue *WidenIVCandidate = BinOp->getOperand(0);
4493 VPValue *InvariantCandidate = BinOp->getOperand(1);
4495 std::swap(WidenIVCandidate, InvariantCandidate);
4509 auto *ClonedOp = BinOp->
clone();
4510 if (ClonedOp->getOperand(0) == WidenIV) {
4511 ClonedOp->setOperand(0, ScalarIV);
4513 assert(ClonedOp->getOperand(1) == WidenIV &&
"one operand must be WideIV");
4514 ClonedOp->setOperand(1, ScalarIV);
4528 return std::nullopt;
4533 return std::nullopt;
4545 auto CheckSentinel = [&SE](
const SCEV *IVSCEV,
4546 bool UseMax) -> std::optional<APSInt> {
4548 for (
bool Signed : {
true,
false}) {
4557 return std::nullopt;
4565 PhiR->getRecurrenceKind()))
4574 VPValue *BackedgeVal = PhiR->getBackedgeValue();
4588 !
match(FindLastSelect,
4597 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression, PSE,
4602 "IVOfExpressionToSink not being an AddRec must imply "
4603 "FindLastExpression not being an AddRec.");
4612 bool UseMax = *StepDirection;
4613 std::optional<APSInt> SentinelVal = CheckSentinel(IVSCEV, UseMax);
4614 bool UseSigned = SentinelVal && SentinelVal->isSigned();
4621 if (IVOfExpressionToSink) {
4622 const SCEV *FindLastExpressionSCEV =
4624 if (std::optional<bool> NewUseMax =
4626 if (
auto NewSentinel =
4627 CheckSentinel(FindLastExpressionSCEV, *NewUseMax)) {
4630 SentinelVal = *NewSentinel;
4631 UseSigned = NewSentinel->isSigned();
4632 UseMax = *NewUseMax;
4633 IVSCEV = FindLastExpressionSCEV;
4634 IVOfExpressionToSink =
nullptr;
4644 if (AR->hasNoSignedWrap())
4646 else if (AR->hasNoUnsignedWrap())
4656 VPValue *NewFindLastSelect = BackedgeVal;
4658 if (!SentinelVal || IVOfExpressionToSink) {
4661 DebugLoc DL = FindLastSelect->getDefiningRecipe()->getDebugLoc();
4662 VPBuilder LoopBuilder(FindLastSelect->getDefiningRecipe());
4663 if (
match(FindLastSelect,
4665 SelectCond = LoopBuilder.
createNot(SelectCond);
4672 if (SelectCond !=
Cond || IVOfExpressionToSink) {
4675 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression,
4684 VPIRFlags Flags(MinMaxKind,
false,
false,
4690 NewFindLastSelect, Flags, ExitDL);
4693 VPValue *VectorRegionExitingVal = ReducedIV;
4694 if (IVOfExpressionToSink)
4695 VectorRegionExitingVal =
4697 ReducedIV, IVOfExpressionToSink);
4700 VPValue *StartVPV = PhiR->getStartValue();
4707 NewRdxResult = MiddleBuilder.
createSelect(Cmp, VectorRegionExitingVal,
4717 AnyOfPhi->insertAfter(PhiR);
4724 OrVal, VectorRegionExitingVal, StartVPV, ExitDL);
4737 PhiR->hasUsesOutsideReductionChain());
4738 NewPhiR->insertBefore(PhiR);
4739 PhiR->replaceAllUsesWith(NewPhiR);
4740 PhiR->eraseFromParent();
4747struct ReductionExtend {
4748 Type *SrcType =
nullptr;
4749 ExtendKind Kind = ExtendKind::PR_None;
4755struct ExtendedReductionOperand {
4759 ReductionExtend ExtendA, ExtendB;
4767struct VPPartialReductionChain {
4770 VPWidenRecipe *ReductionBinOp =
nullptr;
4772 ExtendedReductionOperand ExtendedOp;
4779 unsigned AccumulatorOpIdx;
4780 unsigned ScaleFactor;
4783 VPBlendRecipe *Blend =
nullptr;
4788static std::optional<unsigned>
4792 "Expected a non-normalized blend with two incoming values");
4798 return std::nullopt;
4799 return FirstIncomingHasOneUse ? 0 : 1;
4811 if (!
Op->hasOneUse() ||
4817 auto *Trunc = Builder.createWidenCast(Instruction::CastOps::Trunc,
4818 Op->getOperand(1), NarrowTy);
4820 Op->setOperand(1, Builder.createWidenCast(ExtOpc, Trunc, WideTy));
4829 auto *
Sub =
Op->getOperand(0)->getDefiningRecipe();
4831 assert(Ext->getOpcode() ==
4833 "Expected both the LHS and RHS extends to be the same");
4834 bool IsSigned = Ext->getOpcode() == Instruction::SExt;
4837 auto *FreezeX = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
X}));
4838 auto *FreezeY = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
Y}));
4839 auto *
Max = Builder.insert(
4841 {FreezeX, FreezeY}, SrcTy));
4842 auto *Min = Builder.insert(
4844 {FreezeX, FreezeY}, SrcTy));
4845 auto *AbsDiff = Builder.insert(
4848 return Builder.createWidenCast(Instruction::CastOps::ZExt, AbsDiff,
4849 Op->getScalarType());
4861 if (!
Mul->hasOneUse() ||
4862 (Ext->getOpcode() != MulLHS->getOpcode() && MulLHS != MulRHS) ||
4863 MulLHS->getOpcode() != MulRHS->getOpcode())
4866 auto *NewLHS = Builder.createWidenCast(
4867 MulLHS->getOpcode(), MulLHS->getOperand(0), Ext->getScalarType());
4868 auto *NewRHS = MulLHS == MulRHS
4870 : Builder.createWidenCast(MulRHS->getOpcode(),
4871 MulRHS->getOperand(0),
4872 Ext->getScalarType());
4873 auto *NewMul =
Mul->cloneWithOperands({NewLHS, NewRHS});
4874 Builder.insert(NewMul);
4875 Op->replaceAllUsesWith(NewMul);
4876 Op->eraseFromParent();
4877 Mul->eraseFromParent();
4886 VPValue *VecOp = Red->getVecOp();
4940static void transformToPartialReduction(
const VPPartialReductionChain &Chain,
4948 WidenRecipe->
getOperand(1 - Chain.AccumulatorOpIdx));
4951 ExtendedOp = optimizeExtendsForPartialReduction(ExtendedOp);
4967 if ((WidenRecipe->
getOpcode() == Instruction::Sub &&
4969 (WidenRecipe->
getOpcode() == Instruction::FSub &&
4974 if (WidenRecipe->
getOpcode() == Instruction::FSub) {
4986 Builder.insert(NegRecipe);
4987 ExtendedOp = NegRecipe;
5002 std::optional<unsigned> BlendReductionIdx =
5003 getBlendReductionUpdateValueIdx(Chain.Blend);
5004 assert(BlendReductionIdx &&
5006 "Expected blend to contain the reduction update");
5023 assert((!ExitValue || IsLastInChain) &&
5024 "if we found ExitValue, it must match RdxPhi's backedge value");
5035 PartialRed->insertBefore(WidenRecipe);
5045 E->insertBefore(WidenRecipe);
5046 PartialRed->replaceAllUsesWith(
E);
5059 auto *NewScaleFactor = Plan.
getConstantInt(32, Chain.ScaleFactor);
5060 StartInst->setOperand(2, NewScaleFactor);
5068 VPValue *OldStartValue = StartInst->getOperand(0);
5069 StartInst->setOperand(0, StartInst->getOperand(1));
5073 assert(RdxResult &&
"Could not find reduction result");
5076 unsigned SubOpc = Chain.RK ==
RecurKind::FSub ? Instruction::BinaryOps::FSub
5077 : Instruction::BinaryOps::Sub;
5083 [&NewResult](
VPUser &U,
unsigned Idx) {
return &
U != NewResult; });
5089 const VPPartialReductionChain &Link,
5092 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5093 std::optional<unsigned> BinOpc = std::nullopt;
5095 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5096 BinOpc = ExtendedOp.ExtendsUser->
getOpcode();
5098 std::optional<llvm::FastMathFlags>
Flags;
5102 auto GetLinkOpcode = [&Link]() ->
unsigned {
5105 return Instruction::Add;
5107 return Instruction::FAdd;
5109 return Link.ReductionBinOp->
getOpcode();
5114 GetLinkOpcode(), ExtendedOp.ExtendA.SrcType, ExtendedOp.ExtendB.SrcType,
5115 RdxType, VF, ExtendedOp.ExtendA.Kind, ExtendedOp.ExtendB.Kind, BinOpc,
5136static std::optional<ExtendedReductionOperand>
5139 "Op should be operand of UpdateR");
5147 if (
Op->hasOneUse() &&
5156 Type *RHSInputType =
Y->getScalarType();
5157 if (LHSInputType != RHSInputType ||
5158 LHSExt->getOpcode() != RHSExt->getOpcode())
5159 return std::nullopt;
5162 return ExtendedReductionOperand{
5164 {LHSInputType, getPartialReductionExtendKind(LHSExt)},
5168 std::optional<TTI::PartialReductionExtendKind> OuterExtKind;
5171 VPValue *CastSource = CastRecipe->getOperand(0);
5172 OuterExtKind = getPartialReductionExtendKind(CastRecipe);
5182 return ExtendedReductionOperand{
5189 if (!
Op->hasOneUse())
5190 return std::nullopt;
5195 return std::nullopt;
5205 return std::nullopt;
5209 ExtendKind LHSExtendKind = getPartialReductionExtendKind(LHSCast);
5212 const APInt *RHSConst =
nullptr;
5218 return std::nullopt;
5222 if (Cast && OuterExtKind &&
5223 getPartialReductionExtendKind(Cast) != OuterExtKind)
5224 return std::nullopt;
5226 Type *RHSInputType = LHSInputType;
5227 ExtendKind RHSExtendKind = LHSExtendKind;
5230 RHSExtendKind = getPartialReductionExtendKind(RHSCast);
5233 return ExtendedReductionOperand{
5234 MulOp, {LHSInputType, LHSExtendKind}, {RHSInputType, RHSExtendKind}};
5241static std::optional<SmallVector<VPPartialReductionChain>>
5248 return std::nullopt;
5258 VPValue *CurrentValue = ExitValue;
5259 while (CurrentValue != RedPhiR) {
5261 std::optional<unsigned> BlendReductionIdx;
5265 return std::nullopt;
5267 BlendReductionIdx = getBlendReductionUpdateValueIdx(Blend);
5268 if (!BlendReductionIdx)
5269 return std::nullopt;
5276 return std::nullopt;
5283 std::optional<ExtendedReductionOperand> ExtendedOp =
5284 matchExtendedReductionOperand(UpdateR,
Op);
5286 ExtendedOp = matchExtendedReductionOperand(UpdateR, PrevValue);
5288 return std::nullopt;
5296 return std::nullopt;
5298 Type *ExtSrcType = ExtendedOp->ExtendA.SrcType;
5301 return std::nullopt;
5303 VPPartialReductionChain Link(
5304 {UpdateR, *ExtendedOp, RK,
5309 CurrentValue = PrevValue;
5314 std::reverse(Chain.
begin(), Chain.
end());
5334 if (
auto Chains = getScaledReductions(RedPhiR))
5335 ChainsByPhi.
try_emplace(RedPhiR, std::move(*Chains));
5346 for (
auto *Rdx : UnorderedReductions) {
5362 ? std::make_optional(Rdx->getFastMathFlagsOrNone())
5366 Backedge->getOpcode(), ScalarTy,
nullptr,
5368 std::nullopt, CostCtx.
CostKind, FMF);
5369 return PRCost <= CurrentCost;
5375 Rdx->getRecurrenceKind(), Rdx->getFastMathFlagsOrNone(),
5376 Backedge->getUnderlyingInstr(), Rdx, OtherOp,
nullptr,
5379 Partial->insertBefore(Backedge);
5380 Backedge->replaceAllUsesWith(Partial);
5381 Backedge->eraseFromParent();
5384 if (ChainsByPhi.
empty())
5392 for (
const auto &[
_, Chains] : ChainsByPhi)
5393 for (
const VPPartialReductionChain &Chain : Chains) {
5394 PartialReductionOps.
insert(Chain.ExtendedOp.ExtendsUser);
5396 PartialReductionBlends.
insert(Chain.Blend);
5397 ScaledReductionMap[Chain.ReductionBinOp] = Chain.ScaleFactor;
5403 auto ExtendUsersValid = [&](
VPValue *Ext) {
5405 return PartialReductionOps.contains(cast<VPRecipeBase>(U));
5409 auto IsProfitablePartialReductionChainForVF =
5416 for (
const VPPartialReductionChain &Link : Chain) {
5417 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5418 InstructionCost LinkCost = getPartialReductionLinkCost(CostCtx, Link, VF);
5422 PartialCost += LinkCost;
5423 RegularCost += Link.ReductionBinOp->
computeCost(VF, CostCtx);
5425 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5426 RegularCost += ExtendedOp.ExtendsUser->
computeCost(VF, CostCtx);
5429 RegularCost += Extend->computeCost(VF, CostCtx);
5431 return PartialCost.
isValid() && PartialCost < RegularCost;
5439 for (
auto &[RedPhiR, Chains] : ChainsByPhi) {
5440 for (
const VPPartialReductionChain &Chain : Chains) {
5441 if (!
all_of(Chain.ExtendedOp.ExtendsUser->operands(), ExtendUsersValid)) {
5445 auto UseIsValid = [&, RedPhiR = RedPhiR](
VPUser *U) {
5447 return PhiR == RedPhiR;
5451 return Blend == Chain.Blend || PartialReductionBlends.
contains(Blend);
5453 return Chain.ScaleFactor == ScaledReductionMap.
lookup_or(R, 0) ||
5459 if (!
all_of(Chain.ReductionBinOp->users(), UseIsValid)) {
5468 auto *RepR = dyn_cast<VPReplicateRecipe>(U);
5469 return RepR && RepR->getOpcode() == Instruction::Store;
5480 return IsProfitablePartialReductionChainForVF(Chains, VF);
5486 for (
auto &[Phi, Chains] : ChainsByPhi)
5487 for (
const VPPartialReductionChain &Chain : Chains)
5488 transformToPartialReduction(Chain, Plan, Phi);
5503 if (VPI && VPI->getUnderlyingValue() &&
5514 auto ProcessSubset = [&](
VPlan &,
auto ProcessVPInst) {
5517 if (!ProcessVPInst(VPI))
5526 assert(New->getParent() &&
"New recipe must have been inserted");
5527 if (VPI->
getOpcode() == Instruction::Load)
5536 return ReplaceWith(VPI,
VPBuilder(VPI).insert(
5543 "lowerMemoryIdioms", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5545 VPI, FinalRedStoresBuilder))
5554 return ReplaceWith(VPI,
VPBuilder(VPI).insert(Histogram));
5567 "scalarizeMemOpsWithIrregularTypes", ProcessSubset, Plan,
5571 return Scalarize(VPI);
5578 "makeVPlanMemOpDecision", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5580 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5590 const SCEV *PtrSCEV =
5592 bool IsSingleScalarLoad =
5598 I, Ptr, IsSingleScalarLoad,
5607 "widenConsecutiveMemOps", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5609 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5613 std::optional<int64_t> Stride =
5615 if (Stride != 1 && Stride != -1)
5646 return ReplaceWith(VPI,
Load);
5655 auto *StoreR = Builder.createWidenStore(
5658 return ReplaceWith(VPI, StoreR);
5665 return ReplaceWith(VPI, Recipe);
5667 return Scalarize(VPI);
5690 if (VPI->mayHaveSideEffects())
5694 if (VPI->isMasked() && !VPI->isSafeToSpeculativelyExecute())
5699 if (VPI->getOpcode() == Instruction::Add &&
5708 VPI->getOpcode(), VPI->operandsWithoutMask(),
nullptr, *VPI,
5709 *VPI, VPI->getDebugLoc(),
I);
5710 Recipe->insertBefore(VPI);
5711 VPI->replaceAllUsesWith(Recipe);
5712 VPI->eraseFromParent();
5722 switch (Param.ParamKind) {
5723 case VFParamKind::Vector:
5724 case VFParamKind::GlobalPredicate:
5726 case VFParamKind::OMP_Uniform:
5727 return SE->isSCEVable(Args[Param.ParamPos]->getScalarType()) &&
5728 SE->isLoopInvariant(
5729 vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5731 case VFParamKind::OMP_Linear:
5732 return match(vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5733 m_scev_AffineAddRec(
5734 m_SCEV(), m_scev_SpecificSInt(Param.LinearStepOrPos),
5735 m_SpecificLoop(L)));
5752 const auto *It =
find_if(Mappings, [&](
const VFInfo &Info) {
5753 return Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()) &&
5756 if (It == Mappings.end())
5763struct CallWideningDecision {
5764 enum class KindTy { Scalarize,
Intrinsic, VectorVariant };
5765 CallWideningDecision(KindTy Kind,
Function *Variant =
nullptr)
5788 return CallWideningDecision::KindTy::Scalarize;
5798 return CallWideningDecision::KindTy::Scalarize;
5802 false, VF, CostCtx);
5817 return CallWideningDecision::KindTy::Intrinsic;
5821 if (VecFunc && ScalarCost >= VecCallCost)
5822 return {CallWideningDecision::KindTy::VectorVariant, VecFunc};
5824 return CallWideningDecision::KindTy::Scalarize;
5834 if (!VPI || !VPI->getUnderlyingValue() ||
5835 VPI->getOpcode() != Instruction::Call)
5840 VPI->op_begin() + CI->arg_size());
5842 CallWideningDecision Decision =
5851 switch (Decision.Kind) {
5852 case CallWideningDecision::KindTy::Intrinsic: {
5856 *VPI, VPI->getDebugLoc());
5859 case CallWideningDecision::KindTy::VectorVariant: {
5863 VPValue *Mask = VPI->isMasked() ? VPI->getMask() : Plan.
getTrue();
5864 Ops.push_back(Mask);
5866 Ops.push_back(VPI->getOperand(VPI->getNumOperandsWithoutMask() - 1));
5868 *VPI, VPI->getDebugLoc());
5871 case CallWideningDecision::KindTy::Scalarize:
5877 VPI->replaceAllUsesWith(Replacement);
5878 VPI->eraseFromParent();
5900 if (!MemR || MemR->isConsecutive())
5903 VPValue *Ptr = MemR->getAddr();
5915 VPValue *StoredValue =
nullptr;
5919 StoredValue = StoreR->getStoredValue();
5921 IntrinID = Intrinsic::experimental_vp_strided_store;
5925 IntrinID = Intrinsic::experimental_vp_strided_load;
5928 Align Alignment = MemR->getAlign();
5931 if (!Ctx.TTI.isLegalStridedLoadStore(VectorTy, Alignment))
5936 IntrinID, VectorTy, MemR->isMasked(), Alignment, Ctx);
5937 return StridedLoadStoreCost < CurrentCost;
5948 Ctx.invalidateWideningDecision(&MemR->getIngredient(), VF);
5953 I32VF = Builder.createScalarZExtOrTrunc(
5967 "Stride type from SCEV must match the index type");
5968 VPValue *CanIV = Builder.createScalarZExtOrTrunc(
5971 auto *
Offset = Builder.createOverflowingOp(
5972 Instruction::Mul, {CanIV, StrideInBytes},
5973 {AddRecPtr->hasNoUnsignedWrap(),
false});
5977 VPValue *BasePtr = Builder.createNoWrapPtrAdd(StartVPV,
Offset, NWFlags);
5980 VPValue *NewPtr = Builder.createVectorPointer(
5984 VPValue *Mask = MemR->getMask();
5989 Ops.push_back(StoredValue);
5990 Ops.append({NewPtr, StrideInBytes, Mask, I32VF});
5992 auto *StridedR = Builder.createWidenMemIntrinsic(
5995 *MemR, R.getDebugLoc());
5998 R.eraseFromParent();
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static bool isEqual(const Function &Caller, const Function &Callee)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static cl::opt< IntrinsicCostStrategy > IntrinsicCost("intrinsic-cost-strategy", cl::desc("Costing strategy for intrinsic instructions"), cl::init(IntrinsicCostStrategy::InstructionCost), cl::values(clEnumValN(IntrinsicCostStrategy::InstructionCost, "instruction-cost", "Use TargetTransformInfo::getInstructionCost"), clEnumValN(IntrinsicCostStrategy::IntrinsicCost, "intrinsic-cost", "Use TargetTransformInfo::getIntrinsicInstrCost"), clEnumValN(IntrinsicCostStrategy::TypeBasedIntrinsicCost, "type-based-intrinsic-cost", "Calculate the intrinsic cost based only on argument types")))
iv Induction Variable Users
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
Legalize the Machine IR a function s Machine IR
This file provides utility analysis objects describing memory locations.
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
This file builds on the ADT/GraphTraits.h file to build a generic graph post order iterator.
const SmallVectorImpl< MachineOperand > & Cond
This is the interface for a metadata-based scoped no-alias analysis.
This file implements a set that has insertion order iteration characteristics.
This file defines the SmallPtrSet class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
This file implements the TypeSwitch template, which mimics a switch() statement whose cases are type ...
This file implements dominator tree analysis for a single level of a VPlan's H-CFG.
This file contains the declarations of different VPlan-related auxiliary helpers.
This file contains the declarations of the Vectorization Plan base classes:
static const X86InstrFMA3Group Groups[]
static const uint32_t IV[8]
Helper for extra no-alias checks via known-safe recipe and SCEV.
SinkStoreInfo(ArrayRef< VPReplicateRecipe * > ExcludeRecipes, VPReplicateRecipe &GroupLeader, PredicatedScalarEvolution &PSE, const Loop &L)
SinkStoreInfo(VPReplicateRecipe &GroupLeader)
bool shouldSkip(VPRecipeBase &R) const
Return true if R should be skipped during alias checking, either because it's in the exclude set or b...
Class for arbitrary precision integers.
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
unsigned getActiveBits() const
Compute the number of active bits in the value.
APInt abs() const
Get the absolute value.
unsigned getBitWidth() const
Return the number of bits in the APInt.
int32_t exactLogBase2() const
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
An arbitrary precision integer that knows its signedness.
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
@ NoAlias
The two locations do not alias at all.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & back() const
Get the last element.
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
const T & front() const
Get the first element.
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
const Function * getParent() const
Return the enclosing method, or null if none.
bool isNoBuiltin() const
Return true if the call should not be treated as a call to a builtin.
This class represents a function call, abstracting a target machine's calling convention.
@ ICMP_ULT
unsigned less than
@ ICMP_ULE
unsigned less or equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
This class represents a range of values.
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
A parsed version of the target data layout string in and methods for querying it.
LLVM_ABI IntegerType * getIndexType(LLVMContext &C, unsigned AddressSpace) const
Returns the type of a GEP index in AddressSpace.
static DebugLoc getUnknown()
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
ValueT lookup_or(const_arg_type_t< KeyT > Val, U &&Default) const
bool dominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
dominates - Returns true iff A dominates B.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
static constexpr ElementCount getScalable(ScalarTy MinVal)
constexpr bool isScalar() const
Exactly one element.
Convenience struct for specifying and reasoning about fast-math flags.
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags noUnsignedWrap()
bool hasNoUnsignedWrap() const
GEPNoWrapFlags withoutNoUnsignedWrap() const
static GEPNoWrapFlags none()
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
A struct for saving information about induction variables.
InductionKind
This enum represents the kinds of inductions that we support.
@ IK_PtrInduction
Pointer induction var. Step = C.
@ IK_IntInduction
Integer induction variable. Step = C.
static InstructionCost getInvalid(CostType Val=0)
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
The group of interleaved loads/stores sharing the same stride and close to each other.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
static bool getDecisionAndClampRange(const std::function< bool(ElementCount)> &Predicate, VFRange &Range)
Test a Predicate on a Range of VF's.
Represents a single loop in the control flow graph.
This class implements a map that also provides access to all stored values in a deterministic order.
ValueT lookup(const KeyT &Key) const
std::pair< iterator, bool > try_emplace(const KeyT &Key, Ts &&...Args)
Representation for a specific memory location.
Function * getFunction(StringRef Name) const
Look up the specified function in the module symbol table.
Post-order traversal of a graph.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
ScalarEvolution * getSE() const
Returns the ScalarEvolution analysis used.
LLVM_ABI const SCEV * getSCEV(Value *V)
Returns the SCEV expression of V, in the context of the current SCEV predicate.
static LLVM_ABI unsigned getOpcode(RecurKind Kind)
Returns the opcode corresponding to the RecurrenceKind.
unsigned getOpcode() const
static bool isFindLastRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
RegionT * getParent() const
Get the parent of the Region.
This class represents a constant integer value.
ConstantInt * getValue() const
static const SCEV * rewrite(const SCEV *Scev, ScalarEvolution &SE, ValueToSCEVMapTy &Map)
This means that we are dealing with an entirely unknown SCEV value, and only represent it as its LLVM...
This class represents an analyzed expression in the program.
Type * getType() const
Return the LLVM type of this SCEV expression.
The main scalar evolution driver.
const DataLayout & getDataLayout() const
Return the DataLayout associated with the module this SCEV instance is operating on.
LLVM_ABI const SCEV * getNegativeSCEV(const SCEV *V, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
Return the SCEV object corresponding to -V.
LLVM_ABI bool isKnownNegative(const SCEV *S)
Test if the given expression is known to be negative.
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
LLVM_ABI const SCEV * getMinusSCEV(SCEVUse LHS, SCEVUse RHS, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap, unsigned Depth=0)
Return LHS-RHS.
ConstantRange getSignedRange(const SCEV *S)
Determine the signed range for a particular SCEV.
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
LLVM_ABI bool isKnownPositive(const SCEV *S)
Test if the given expression is known to be positive.
LLVM_ABI const SCEV * getElementCount(Type *Ty, ElementCount EC, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
ConstantRange getUnsignedRange(const SCEV *S)
Determine the unsigned range for a particular SCEV.
LLVM_ABI bool isKnownPredicate(CmpPredicate Pred, SCEVUse LHS, SCEVUse RHS)
Test if the given expression is known to satisfy the condition described by Pred, LHS,...
static LLVM_ABI AliasResult alias(const MemoryLocation &LocA, const MemoryLocation &LocB)
A vector that has set insertion semantics.
size_type size() const
Determine the number of elements in the SetVector.
bool insert(const value_type &X)
Insert a new element into the SetVector.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Provides information about what library functions are available for the current target.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
This class implements a switch-like dispatch statement for a value of 'T' using dyn_cast functionalit...
TypeSwitch< T, ResultT > & Case(CallableT &&caseFn)
Add a case on the given type.
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isPointerTy() const
True if this is an instance of PointerType.
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntOrPtrTy() const
Return true if this is an integer type or a pointer type.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static SmallVector< VFInfo, 8 > getMappings(const CallInst &CI)
Retrieve all the VFInfo instances associated to the CallInst CI.
bool isLegalMaskedLoadOrStore(bool IsLoad, Type *ScalarTy, Align Alignment, unsigned AddressSpace) const
Returns true if the target machine supports a masked load (if IsLoad) or masked store of scalar type ...
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
void appendRecipe(VPRecipeBase *Recipe)
Augment the existing recipes of a VPBasicBlock with an additional Recipe as the last recipe.
iterator begin()
Recipe iterator methods.
iterator_range< iterator > phis()
Returns an iterator range over the PHI-like recipes in the block.
iterator getFirstNonPhi()
Return the position of the first non-phi node recipe in the block.
VPBasicBlock * splitAt(iterator SplitAt)
Split current block at SplitAt by inserting a new block between the current block and its successors ...
const VPRecipeBase & front() const
VPRecipeBase * getTerminator()
If the block has multiple successors, return the branch recipe terminating the block.
const VPRecipeBase & back() const
A recipe for vectorizing a phi-node as a sequence of mask-based select instructions.
VPValue * getIncomingValue(unsigned Idx) const
Return incoming value number Idx.
VPValue * getMask(unsigned Idx) const
Return mask number Idx.
unsigned getNumIncomingValues() const
Return the number of incoming values, taking into account when normalized the first incoming value wi...
void setMask(unsigned Idx, VPValue *V)
Set mask number Idx to V.
bool isNormalized() const
A normalized blend is one that has an odd number of operands, whereby the first operand does not have...
VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
void setSuccessors(ArrayRef< VPBlockBase * > NewSuccs)
Set each VPBasicBlock in NewSuccss as successor of this VPBlockBase.
VPRegionBlock * getParent()
const VPBasicBlock * getExitingBasicBlock() const
size_t getNumSuccessors() const
void setPredecessors(ArrayRef< VPBlockBase * > NewPreds)
Set each VPBasicBlock in NewPreds as predecessor of this VPBlockBase.
const VPBlocksTy & getPredecessors() const
VPBlockBase * getSinglePredecessor() const
const VPBasicBlock * getEntryBasicBlock() const
VPBlockBase * getSingleSuccessor() const
const VPBlocksTy & getSuccessors() const
static auto blocksAs(T &&Range)
Return an iterator range over Range with each block cast to BlockTy.
static void insertOnEdge(VPBlockBase *From, VPBlockBase *To, VPBlockBase *BlockPtr)
Inserts BlockPtr on the edge between From and To.
static bool isLatch(const VPBlockBase *VPB, const VPDominatorTree &VPDT)
Returns true if VPB is a loop latch, using isHeader().
static VPBasicBlock * getPlainCFGMiddleBlock(const VPlan &Plan)
Returns the middle block of Plan in plain CFG form (before regions are formed).
static void insertTwoBlocksAfter(VPBlockBase *IfTrue, VPBlockBase *IfFalse, VPBlockBase *BlockPtr)
Insert disconnected VPBlockBases IfTrue and IfFalse after BlockPtr.
static void connectBlocks(VPBlockBase *From, VPBlockBase *To, unsigned PredIdx=-1u, unsigned SuccIdx=-1u)
Connect VPBlockBases From and To bi-directionally.
static void disconnectBlocks(VPBlockBase *From, VPBlockBase *To)
Disconnect VPBlockBases From and To bi-directionally.
static auto blocksOnly(T &&Range)
Return an iterator range over Range which only includes BlockTy blocks.
static std::pair< VPBasicBlock *, VPBasicBlock * > getPlainCFGHeaderAndLatch(const VPlan &Plan)
Returns the header and latch of the outermost loop of Plan in plain CFG form (before regions are form...
static void transferSuccessors(VPBlockBase *Old, VPBlockBase *New)
Transfer successors from Old to New. New must have no successors.
static SmallVector< VPBasicBlock * > blocksInSingleSuccessorChainBetween(VPBasicBlock *FirstBB, VPBasicBlock *LastBB)
Returns the blocks between FirstBB and LastBB, where FirstBB to LastBB forms a single-sucessor chain.
A recipe for generating conditional branches on the bits of a mask.
VPlan-based builder utility analogous to IRBuilder.
VPInstruction * createFirstActiveLane(ArrayRef< VPValue * > Masks, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenStoreRecipe * createWidenStore(StoreInst &Store, VPValue *Addr, VPValue *StoredVal, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Store, storing StoredVal to Addr with Mask (may be null).
VPInstruction * createAdd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", VPRecipeWithIRFlags::WrapFlagsTy WrapFlags={false, false})
VPInstruction * createOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createLogicalOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenLoadRecipe * createWidenLoad(LoadInst &Load, VPValue *Addr, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Load, loading from Addr with Mask (may be null).
VPInstruction * createNot(VPValue *Operand, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createAnyOfReduction(VPValue *ChainOp, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown())
Create an AnyOf reduction pattern: or-reduce ChainOp, freeze the result, then select between TrueVal ...
void setInsertPoint(const VPInsertPoint &IP)
Set the current insert point.
VPInstruction * createLogicalAnd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createScalarCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy, DebugLoc DL, std::optional< VPIRFlags > Flags=std::nullopt, const VPIRMetadata &Metadata={})
VPValue * createScalarZExtOrTrunc(VPValue *Op, Type *ResultTy, DebugLoc DL)
static VPBuilder getToInsertAfter(VPRecipeBase *R)
Create a VPBuilder to insert after R.
VPDerivedIVRecipe * createDerivedIV(InductionDescriptor::InductionKind Kind, FPMathOperator *FPBinOp, VPValue *Start, VPValue *Current, VPValue *Step, const VPIRFlags::WrapFlagsTy &Flags={})
Convert Current to Start + Current * Step.
VPWidenCastRecipe * createWidenCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy)
VPInstruction * createICmp(CmpInst::Predicate Pred, VPValue *A, VPValue *B, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
Create a new ICmp VPInstruction with predicate Pred and operands A and B.
VPInstruction * createSelect(VPValue *Cond, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", std::optional< VPIRFlags > Flags=std::nullopt)
Create a select of TrueVal and FalseVal based on Cond, using the default flags for the result type,...
VPInstruction * createNaryOp(unsigned Opcode, ArrayRef< VPValue * > Operands, Instruction *Inst=nullptr, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
Create an N-ary operation with Opcode, Operands and set Inst as its underlying Instruction.
static VPSingleDefRecipe * createSingleScalarOp(unsigned Opcode, ArrayRef< VPValue * > Operands, VPValue *Mask, const VPIRFlags &Flags, const VPIRMetadata &Metadata, DebugLoc DL, Instruction *UV)
Create a single-scalar recipe with Opcode and Operands without inserting it.
unsigned getNumDefinedValues() const
Returns the number of values defined by the VPDef.
VPValue * getVPSingleValue()
Returns the only VPValue defined by the VPDef.
VPValue * getVPValue(unsigned I)
Returns the VPValue with index I defined by the VPDef.
ArrayRef< VPRecipeValue * > definedValues()
Returns an ArrayRef of the values defined by the VPDef.
Template specialization of the standard LLVM dominator tree utility for VPBlockBases.
bool properlyDominates(const VPRecipeBase *A, const VPRecipeBase *B) const
A recipe to combine multiple recipes into a single 'expression' recipe, which should be considered a ...
A recipe representing a sequence of load -> update -> store as part of a histogram operation.
A special type of VPBasicBlock that wraps an existing IR basic block.
Class to record and manage LLVM IR flags.
static VPIRFlags getDefaultFlags(unsigned Opcode, Type *ResultTy=nullptr)
Returns default flags for Opcode and scalar ResultTy for opcodes that support it, asserts otherwise.
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
This is a concrete Recipe that models a single VPlan-level instruction.
unsigned getNumOperandsWithoutMask() const
Returns the number of operands, excluding the mask if the VPInstruction is masked.
@ ExtractLane
Extracts a single lane (first operand) from a set of vector operands.
@ ExtractPenultimateElement
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
@ BuildVector
Creates a fixed-width vector containing all operands.
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
unsigned getOpcode() const
VPValue * getMask() const
Returns the mask for the VPInstruction.
const InterleaveGroup< Instruction > * getInterleaveGroup() const
VPValue * getMask() const
Return the mask used by this recipe.
ArrayRef< VPValue * > getStoredValues() const
Return the VPValues stored by this interleave group.
VPInterleaveRecipe is a recipe for transforming an interleave group of load or stores into one wide l...
VPPredInstPHIRecipe is a recipe for generating the phi nodes needed when control converges back from ...
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
VPRegionBlock * getRegion()
VPBasicBlock * getParent()
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
void insertAfter(VPRecipeBase *InsertPos)
Insert an unlinked Recipe into a basic block immediately after the specified Recipe.
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
Helper class to create VPRecipies from IR instructions.
VPHistogramRecipe * widenIfHistogram(VPInstruction *VPI)
If VPI represents a histogram operation (as determined by LoopVectorizationLegality) make that safe f...
bool prefersVectorizedAddressing() const
Returns true if the target prefers vectorized addressing.
VPRecipeBase * tryToWidenMemory(VPInstruction *VPI, VFRange &Range)
Check if the load or store instruction VPI should widened for Range.Start and potentially masked.
bool replaceWithFinalIfReductionStore(VPInstruction *VPI, VPBuilder &FinalRedStoresBuilder)
If VPI is a store of a reduction into an invariant address, delete it.
VPSingleDefRecipe * handleReplication(VPInstruction *VPI, VFRange &Range)
Build a replicating or single-scalar recipe for VPI.
bool isPredicatedInst(Instruction *I) const
Returns true if I needs to be predicated (i.e.
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
A recipe for handling reduction phis.
bool isOrdered() const
Returns true, if the phi is part of an ordered reduction.
void setVFScaleFactor(unsigned ScaleFactor)
Set the VFScaleFactor for this reduction phi.
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
bool isInLoop() const
Returns true if the phi is part of an in-loop reduction.
RecurKind getRecurrenceKind() const
Returns the recurrence kind of the reduction.
A recipe to represent inloop, ordered or partial reduction operations.
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
const VPBlockBase * getEntry() const
bool isReplicator() const
An indicator whether this region is to generate multiple replicated instances of output IR correspond...
void setExiting(VPBlockBase *ExitingBlock)
Set ExitingBlock as the exiting VPBlockBase of this VPRegionBlock.
Type * getCanonicalIVType() const
Return the type of the canonical IV for loop regions.
VPRegionValue * getCanonicalIV()
Return the canonical induction variable of the region, null for replicating regions.
const VPBlockBase * getExiting() const
VPRegionValue * getHeaderMask() const
Return the header mask of the region, or null if not set.
VPReplicateRecipe replicates a given instruction producing multiple scalar copies of the original sca...
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
static InstructionCost computeCallCost(Function *CalledFn, Type *ResultTy, ArrayRef< const VPValue * > ArgOps, bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx)
Return the cost of scalarizing a call to CalledFn with argument operands ArgOps for a given VF.
operand_range operandsWithoutMask()
Return the recipe's operands, excluding the mask of a predicated recipe.
bool isPredicated() const
VPValue * getMask()
Return the mask of a predicated VPReplicateRecipe.
Lightweight SCEV-to-VPlan expander.
VPValue * expand(const SCEV *S)
Expand S into recipes and live-ins using the builder.
A recipe for handling phi nodes of integer and floating-point inductions, producing their scalar valu...
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
VPSingleDefRecipe * clone() override=0
Clone the current recipe.
A symbolic live-in VPValue, used for values like vector trip count, VF, and VFxUF.
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
void setOperand(unsigned I, VPValue *New)
unsigned getNumOperands() const
VPValue * getOperand(unsigned N) const
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
bool isDefinedOutsideLoopRegions() const
Returns true if the VPValue is defined outside any loop.
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
bool hasMoreThanOneUniqueUser() const
Returns true if the value has more than one unique user.
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
VPUser * getSingleUser()
Return the single user of this value, or nullptr if there is not exactly one user.
void replaceAllUsesWith(VPValue *New)
void replaceUsesWithIf(VPValue *New, llvm::function_ref< bool(VPUser &U, unsigned Idx)> ShouldReplace)
Go through the uses list for this VPValue and make each use point to New if the callback ShouldReplac...
A recipe to compute a pointer to the last element of each part of a widened memory access for widened...
A recipe for widening Call instructions using library calls.
static InstructionCost computeCallCost(Function *Variant, VPCostContext &Ctx)
Return the cost of widening a call using the vector function Variant.
VPWidenCastRecipe is a recipe to create vector cast instructions.
Instruction::CastOps getOpcode() const
A recipe for handling GEP instructions.
Base class for widened induction (VPWidenIntOrFpInductionRecipe and VPWidenPointerInductionRecipe),...
PHINode * getPHINode() const
Returns the underlying PHINode if one exists, or null otherwise.
VPValue * getStepValue()
Returns the step value of the induction.
const InductionDescriptor & getInductionDescriptor() const
Returns the induction descriptor for the recipe.
A recipe for handling phi nodes of integer and floating-point inductions, producing their vector valu...
TruncInst * getTruncInst()
Returns the first defined value as TruncInst, if it is one or nullptr otherwise.
A recipe for widening vector intrinsics.
static InstructionCost computeCallCost(Intrinsic::ID ID, ArrayRef< const VPValue * > Operands, const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx)
Compute the cost of a vector intrinsic with ID and Operands.
static InstructionCost computeMemIntrinsicCost(Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment, VPCostContext &Ctx)
Helper function for computing the cost of vector memory intrinsic.
A common mixin class for widening memory operations.
virtual VPRecipeBase * getAsRecipe()=0
Return a VPRecipeBase* to the current object.
A recipe for widened phis.
VPWidenRecipe is a recipe for producing a widened instruction using the opcode and operands of the re...
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenRecipe.
VPWidenRecipe * clone() override
Clone the current recipe.
unsigned getOpcode() const
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
VPIRValue * getLiveIn(Value *V) const
Return the live-in VPIRValue for V, if there is one or nullptr otherwise.
bool hasVF(ElementCount VF) const
const DataLayout & getDataLayout() const
LLVMContext & getContext() const
VPBasicBlock * getEntry()
bool hasScalableVF() const
VPValue * getTripCount() const
The trip count of the original loop.
VPValue * getOrCreateBackedgeTakenCount()
The backedge taken count of the original loop.
iterator_range< SmallSetVector< ElementCount, 2 >::iterator > vectorFactors() const
Returns an iterator range over all VFs of the plan.
VPIRValue * getFalse()
Return a VPIRValue wrapping i1 false.
VPSymbolicValue & getVFxUF()
Returns VF * UF of the vector loop region.
VPIRValue * getAllOnesValue(Type *Ty)
Return a VPIRValue wrapping the AllOnes value of type Ty.
VPRegionBlock * createReplicateRegion(VPBlockBase *Entry, VPBlockBase *Exiting, const std::string &Name="")
Create a new replicate region with Entry, Exiting and Name.
auto getLiveIns() const
Return the list of live-in VPValues available in the VPlan.
bool hasUF(unsigned UF) const
ArrayRef< VPIRBasicBlock * > getExitBlocks() const
Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of the original scalar loop.
VPSymbolicValue & getVectorTripCount()
The vector trip count.
VPValue * getBackedgeTakenCount() const
VPIRValue * getOrAddLiveIn(Value *V)
Gets the live-in VPIRValue for V or adds a new live-in (if none exists yet) for V.
VPIRValue * getZero(Type *Ty)
Return a VPIRValue wrapping the null value of type Ty.
void setVF(ElementCount VF)
bool isUnrolled() const
Returns true if the VPlan already has been unrolled, i.e.
LLVM_ABI_FOR_TEST VPRegionBlock * getVectorLoopRegion()
Returns the VPRegionBlock of the vector loop.
unsigned getConcreteUF() const
Returns the concrete UF of the plan, after unrolling.
void resetTripCount(VPValue *NewTripCount)
Resets the trip count for the VPlan.
VPBasicBlock * getMiddleBlock()
Returns the 'middle' block of the plan, that is the block that selects whether to execute the scalar ...
VPBasicBlock * createVPBasicBlock(const Twine &Name, VPRecipeBase *Recipe=nullptr)
Create a new VPBasicBlock with Name and containing Recipe if present.
VPIRValue * getTrue()
Return a VPIRValue wrapping i1 true.
VPBasicBlock * getVectorPreheader() const
Returns the preheader of the vector loop region, if one exists, or null otherwise.
VPSymbolicValue & getUF()
Returns the UF of the vector loop region.
bool hasScalarVFOnly() const
VPBasicBlock * getScalarPreheader() const
Return the VPBasicBlock for the preheader of the scalar loop.
bool hasTailFolded() const
Returns true if the vector loop region is tail-folded.
VPSymbolicValue & getVF()
Returns the VF of the vector loop region.
LLVM_ABI_FOR_TEST VPlan * duplicate()
Clone the current VPlan, update all VPValues of the new VPlan and cloned recipes to refer to the clon...
VPIRValue * getConstantInt(Type *Ty, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given type and value.
LLVM Value Representation.
iterator_range< user_iterator > users()
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
constexpr bool hasKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns true if there exists a value X where RHS*X will result in a value whose quantity matches our ...
constexpr ScalarTy getFixedValue() const
constexpr ScalarTy getKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns a value X where RHS*X will result in a value whose quantity matches our own.
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
An efficient, type-erasing, non-owning reference to a callable.
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
std::variant< std::monostate, Loc::Single, Loc::Multi, Loc::MMI, Loc::EntryValue > Variant
Alias for the std::variant specialization base class of DbgVariable.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
AllOnesConstantMatch m_AllOnes()
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_unless< Pattern > m_Unless(const Pattern &P)
Match if the inner matcher does NOT match.
match_isa< To... > m_Isa()
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
auto m_Cmp()
Matches any compare instruction and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::URem > m_URem(const LHS &L, const RHS &R)
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
SpecificCmpClass_match< LHS, RHS, CmpInst > m_SpecificCmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::UDiv > m_UDiv(const LHS &L, const RHS &R)
SelectLike_match< CondTy, LTy, RTy > m_SelectLike(const CondTy &C, const LTy &TrueC, const RTy &FalseC)
Matches a value that behaves like a boolean-controlled select, i.e.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
CastOperator_match< OpTy, Instruction::BitCast > m_BitCast(const OpTy &Op)
Matches BitCast.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinaryOp_match< LHS, RHS, Instruction::FAdd, true > m_c_FAdd(const LHS &L, const RHS &R)
Matches FAdd with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And, true > m_c_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R with LHS and RHS in either order.
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Sub > m_Sub(const LHS &L, const RHS &R)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
bind_cst_ty m_scev_APInt(const APInt *&C)
Match an SCEV constant and bind it to an APInt.
specificloop_ty m_SpecificLoop(const Loop *L)
bool match(const SCEV *S, const Pattern &P)
SCEVAffineAddRec_match< Op0_t, Op1_t, match_isa< const Loop > > m_scev_AffineAddRec(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > > m_ExtractLastLaneOfLastPart(const Op0_t &Op0)
AllRecipe_commutative_match< Instruction::And, Op0_t, Op1_t > m_c_BinaryAnd(const Op0_t &Op0, const Op1_t &Op1)
Match a binary AND operation.
AllRecipe_match< Instruction::Or, Op0_t, Op1_t > m_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
Match a binary OR operation.
VPInstruction_match< VPInstruction::AnyOf > m_AnyOf()
AllRecipe_commutative_match< Instruction::Or, Op0_t, Op1_t > m_c_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ComputeReductionResult, Op0_t > m_ComputeReductionResult(const Op0_t &Op0)
auto m_WidenAnyExtend(const Op0_t &Op0)
match_bind< VPIRValue > m_VPIRValue(VPIRValue *&V)
Match a VPIRValue.
VPInstruction_match< VPInstruction::WideActiveLaneMask, Op0_t, Op1_t, Op2_t > m_WideActiveLaneMask(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
auto m_VPPhi(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::BranchOnTwoConds > m_BranchOnTwoConds()
AllRecipe_match< Opcode, Op0_t, Op1_t > m_Binary(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::LastActiveLane, Op0_t > m_LastActiveLane(const Op0_t &Op0)
auto m_WidenIntrinsic(const T &...Ops)
canonical_widen_iv_match m_CanonicalWidenIV()
VPInstruction_match< VPInstruction::ExitingIVValue, Op0_t > m_ExitingIVValue(const Op0_t &Op0)
VPInstruction_match< Instruction::ExtractElement, Op0_t, Op1_t > m_ExtractElement(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, Op0_t > m_ExtractLastLane(const Op0_t &Op0)
int_pred_ty< is_zero_int, 1 > m_False()
match_bind< VPSingleDefRecipe > m_VPSingleDefRecipe(VPSingleDefRecipe *&V)
Match a VPSingleDefRecipe, capturing if we match.
VPInstruction_match< VPInstruction::BranchOnCount > m_BranchOnCount()
auto m_GetElementPtr(const Op0_t &Op0, const Op1_t &Op1)
auto m_VPValue()
Match an arbitrary VPValue and ignore it.
VPInstruction_match< VPInstruction::ExtractVectorForPart, Op0_t, Op1_t > m_ExtractVectorForPart(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > m_ExtractLastPart(const Op0_t &Op0)
VPRecipeBase * findUserOf(VPValue *V, const MatchT &P)
If V is used by a recipe matching pattern P, return it.
VPInstruction_match< VPInstruction::Broadcast, Op0_t > m_Broadcast(const Op0_t &Op0)
header_mask_match m_HeaderMask()
VPInstruction_match< VPInstruction::BuildVector > m_BuildVector()
BuildVector is matches only its opcode, w/o matching its operands as the number of operands is not fi...
VPInstruction_match< VPInstruction::ExtractPenultimateElement, Op0_t > m_ExtractPenultimateElement(const Op0_t &Op0)
match_bind< VPInstruction > m_VPInstruction(VPInstruction *&V)
Match a VPInstruction, capturing if we match.
VPInstruction_match< VPInstruction::FirstActiveLane, Op0_t > m_FirstActiveLane(const Op0_t &Op0)
int_pred_ty< is_one, 1 > m_True()
auto m_DerivedIV(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
VPInstruction_match< VPInstruction::BranchOnCond > m_BranchOnCond()
VPInstruction_match< VPInstruction::ExtractLane, Op0_t, Op1_t > m_ExtractLane(const Op0_t &Op0, const Op1_t &Op1)
auto m_AnyNeg(const Op0_t &Op0)
VPInstruction_match< VPInstruction::Reverse, Op0_t > m_Reverse(const Op0_t &Op0)
initializer< Ty > init(const Ty &Val)
NodeAddr< DefNode * > Def
bool isSingleScalar(const VPValue *VPV)
Returns true if VPV is a single scalar, either because it produces the same value for all lanes or on...
VPValue * getOrCreateVPValueForSCEVExpr(VPlan &Plan, const SCEV *Expr)
Get or create a VPValue that corresponds to the expansion of Expr.
bool cannotHoistOrSinkRecipe(const VPRecipeBase &R, bool Sinking=false)
Return true if we do not know how to (mechanically) hoist or sink R.
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
VPInstruction * findComputeReductionResult(VPReductionPHIRecipe *PhiR)
Find the ComputeReductionResult recipe for PhiR, looking through selects inserted for predicated redu...
VPInstruction * findCanonicalIVIncrement(VPlan &Plan)
Find the canonical IV increment of Plan's vector loop region.
std::optional< MemoryLocation > getMemoryLocation(const VPRecipeBase &R)
Return a MemoryLocation for R with noalias metadata populated from R, if the recipe is supported and ...
bool onlyFirstLaneUsed(const VPValue *Def)
Returns true if only the first lane of Def is used.
VPIRValue * tryToFoldLiveIns(VPSingleDefRecipe &R, ArrayRef< VPValue * > Operands, const DataLayout &DL)
Try to fold R using InstSimplifyFolder.
SmallVector< std::pair< VPBasicBlock *, VPIRBasicBlock * > > getEarlyExits(const VPlan &Plan, const VPBlockBase *MiddleVPBB)
Returns the (early exiting block, exit block) pairs of Plan, i.e.
void recursivelyDeleteDeadRecipes(VPValue *V)
Recursively delete V and any of its operands that become dead.
bool doesGeneratePerAllLanes(const VPRecipeBase *R)
Returns true if R produces scalar values for all VF lanes.
bool isDeadRecipe(VPRecipeBase &R)
Returns true if R is dead, i.e.
VPRecipeBase * findRecipe(VPValue *Start, PredT Pred)
Search Start's users for a recipe satisfying Pred, looking through recipes with definitions.
bool isUniformAcrossVFsAndUFs(const VPValue *V)
Checks if V is uniform across all VF lanes and UF parts.
bool isUsedByLoadStoreAddress(const VPValue *V)
Returns true if V is used as part of the address of another load or store.
std::optional< std::pair< bool, unsigned > > getOpcodeOrIntrinsicID(const VPValue *V)
Get the instruction opcode or intrinsic ID for the recipe defining V.
VPValue * scalarizeVPWidenPointerInduction(VPWidenPointerInductionRecipe *PtrIV, VPlan &Plan, VPBuilder &Builder)
Scalarize a VPWidenPointerInductionRecipe by replacing it with a PtrAdd (IndStart,...
const SCEV * getSCEVExprForVPValue(const VPValue *V, PredicatedScalarEvolution &PSE, const Loop *L=nullptr)
Return the SCEV expression for V.
void pullOutPermutations(VPlan &Plan, Match_t Perm, Builder Build)
Removes the permutation pattern Perm from any elementwise operations in the plan, by constructing a n...
SmallVector< VPUser * > collectUsersRecursively(VPValue *V)
Collect all users of V, looking through recipes that define other values.
VPScalarIVStepsRecipe * createScalarIVSteps(VPlan &Plan, InductionDescriptor::InductionKind Kind, Instruction::BinaryOps InductionOpcode, FPMathOperator *FPBinOp, Instruction *TruncI, VPValue *StartV, VPValue *Step, DebugLoc DL, VPBuilder &Builder, const VPIRFlags::WrapFlagsTy &Flags={})
Create a scalar-iv-steps recipe over Plan's canonical IV for an induction of Kind with InductionOpcod...
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
SmallVector< VPBasicBlock * > vp_rpo_plain_cfg_loop_body(VPBasicBlock *Header)
Returns the VPBasicBlocks forming the loop body of a plain (pre-region) VPlan in reverse post-order s...
void stable_sort(R &&Range)
auto min_element(R &&Range)
Provide wrappers to std::min_element which take ranges instead of having to pass begin/end explicitly...
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
unsigned getLoadStoreAddressSpace(const Value *I)
A helper function that returns the address space of the pointer operand of load or store instruction.
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
LLVM_ABI Intrinsic::ID getVectorIntrinsicIDForCall(const CallInst *CI, const TargetLibraryInfo *TLI)
Returns intrinsic ID for call.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
ReductionStyle getReductionStyle(bool InLoop, bool Ordered, unsigned ScaleFactor)
DenseMap< const Value *, const SCEV * > ValueToSCEVMapTy
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
constexpr from_range_t from_range
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
auto cast_or_null(const Y &Val)
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
iterator_range< df_iterator< VPBlockShallowTraversalWrapper< VPBlockBase * > > > vp_depth_first_shallow(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order.
constexpr auto bind_back(FnT &&Fn, BindArgsT &&...BindArgs)
C++23 bind_back.
bool isa_and_nonnull(const Y &Val)
iterator_range< df_iterator< VPBlockDeepTraversalWrapper< VPBlockBase * > > > vp_depth_first_deep(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order while traversing t...
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
bool operator==(const AddressRangeValuePair &LHS, const AddressRangeValuePair &RHS)
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
auto dyn_cast_or_null(const Y &Val)
void erase(Container &C, ValueType V)
Wrapper function to remove a value from a container:
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
auto reverse(ContainerTy &&C)
constexpr size_t range_size(R &&Range)
Returns the size of the Range, i.e., the number of elements.
void sort(IteratorTy Start, IteratorTy End)
DenseMap< Value *, const SCEVUnknown * > SymbolicStrideMap
Maps a pointer to its symbolic (non-constant) stride.
bool hasIrregularType(Type *Ty, const DataLayout &DL)
A helper function that returns true if the given type is irregular.
UncountableExitStyle
Different methods of handling early exits.
@ ReadOnly
No side effects to worry about, so we can process any uncountable exits in the loop and branch either...
@ MaskedHandleExitInScalarLoop
All memory operations other than the load(s) required to determine whether an uncountable exit occurr...
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
iterator_range< filter_iterator< detail::IterOfRange< RangeT >, PredicateT > > make_filter_range(RangeT &&Range, PredicateT Pred)
Convenience function that takes a range of elements and a predicate, and return a new filter_iterator...
bool canConstantBeExtended(const APInt *C, Type *NarrowType, TTI::PartialReductionExtendKind ExtKind)
Check if a constant CI can be safely treated as having been extended from a narrower type with the gi...
T * find_singleton(R &&Range, Predicate P, bool AllowRepeats=false)
Return the single value in Range that satisfies P(<member of Range> *, AllowRepeats)->T * returning n...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
RecurKind
These are the kinds of recurrences that we support.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ FindIV
FindIV reduction with select(icmp(),x,y) where one of (x,y) is a loop induction variable (increasing ...
@ Or
Bitwise or logical OR of integers.
@ Mul
Product of integers.
@ FSub
Subtraction of floats.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
LLVM_ABI Value * getRecurrenceIdentity(RecurKind K, Type *Tp, FastMathFlags FMF)
Given information about an recurrence kind, return the identity for the @llvm.vector....
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
DWARFExpression::Operation Op
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
ArrayRef(const T &OneElt) -> ArrayRef< T >
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
hash_code hash_combine(const Ts &...args)
Combine values into a single hash_code.
LLVM_ABI std::optional< int64_t > getStrideFromAddRec(const SCEVAddRecExpr *AR, const Loop *Lp, Type *AccessTy, Value *Ptr, PredicatedScalarEvolution &PSE)
If AR is an affine AddRec for Lp with a constant step, return the step in units of AccessTy's allocat...
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
LLVM_ABI bool isDereferenceableAndAlignedInLoop(LoadInst *LI, Loop *L, ScalarEvolution &SE, DominatorTree &DT, AssumptionCache *AC=nullptr, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Return true if we can prove that the given load (which is assumed to be within the specified loop) wo...
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
hash_code hash_combine_range(InputIteratorT first, InputIteratorT last)
Compute a hash_code for a sequence of values.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
VPBasicBlock * EarlyExitingVPBB
VPIRBasicBlock * EarlyExitVPBB
This struct is a compact representation of a valid (non-zero power of two) alignment.
An information struct used to provide DenseMap with the various necessary components for a given valu...
This reduction is unordered with the partial result scaled down by some factor.
Holds the VFShape for a specific scalar to vector function mapping.
Encapsulates information needed to describe a parameter.
A range of powers-of-2 vectorization factors with fixed start and adjustable end.
Struct to hold various analysis needed for cost computations.
const VFSelectionContext & Config
static bool isFreeScalarIntrinsic(Intrinsic::ID ID)
Returns true if ID is a pseudo intrinsic that is dropped via scalarization rather than widened.
bool isMaskRequired(Instruction *I) const
Forwards to LoopVectorizationCostModel::isMaskRequired.
PredicatedScalarEvolution & PSE
bool willBeScalarized(Instruction *I, ElementCount VF) const
Returns true if I is known to be scalarized at VF.
TargetTransformInfo::TargetCostKind CostKind
const TargetLibraryInfo & TLI
const TargetTransformInfo & TTI
A VPValue representing a live-in from the input IR or a constant.
Type * getType() const
Returns the type of the underlying IR value.
A recipe for widening load operations, using the address to load from and an optional mask.
A recipe for widening store operations, using the stored value, the address to store to and an option...