59 cl::desc(
"Use partial reduction intrinsics for "
60 "all supported unordered reductions."));
68 auto IsConsecutiveAccess = [&](
VPValue *Addr,
Type *AccessTy) {
77 if (!VPBB->getParent())
80 auto EndIter = Term ? Term->getIterator() : VPBB->end();
85 VPValue *VPV = Ingredient.getVPSingleValue();
106 IsConsecutiveAccess(VPI->getOperand(0), VPI->getScalarType());
108 nullptr , IsConsecutive,
109 *VPI, Ingredient.getDebugLoc());
111 bool IsConsecutive = IsConsecutiveAccess(
112 VPI->getOperand(1), VPI->getOperand(0)->getScalarType());
114 *
Store, Ingredient.getOperand(1), Ingredient.getOperand(0),
115 nullptr , IsConsecutive, *VPI, Ingredient.getDebugLoc());
118 Ingredient.operands(), *VPI,
119 Ingredient.getDebugLoc(),
GEP);
131 if (VectorID == Intrinsic::experimental_noalias_scope_decl)
136 if (VectorID == Intrinsic::assume ||
137 VectorID == Intrinsic::lifetime_end ||
138 VectorID == Intrinsic::lifetime_start ||
139 VectorID == Intrinsic::sideeffect ||
140 VectorID == Intrinsic::pseudoprobe) {
145 const bool IsSingleScalar = VectorID != Intrinsic::assume &&
146 VectorID != Intrinsic::pseudoprobe;
150 Ingredient.getDebugLoc());
153 *CI, VectorID,
drop_end(Ingredient.operands()), CI->getType(),
154 VPIRFlags(*CI), *VPI, CI->getDebugLoc());
158 CI->getOpcode(), Ingredient.getOperand(0), CI->getType(), CI,
162 *VPI, Ingredient.getDebugLoc());
166 "inductions must be created earlier");
175 "Only recpies with zero or one defined values expected");
176 Ingredient.eraseFromParent();
187 const Loop *L =
nullptr;
192 if (
A->getOpcode() != Instruction::Store ||
193 B->getOpcode() != Instruction::Store)
206 const APInt *Distance;
212 Type *TyA =
A->getOperand(0)->getScalarType();
213 uint64_t SizeA =
DL.getTypeStoreSize(TyA);
214 Type *TyB =
B->getOperand(0)->getScalarType();
215 uint64_t SizeB =
DL.getTypeStoreSize(TyB);
220 uint64_t MaxStoreSize = std::max(SizeA, SizeB);
222 auto VFs =
B->getParent()->getPlan()->vectorFactors();
233 : ExcludeRecipes(ExcludeRecipes.begin(), ExcludeRecipes.end()),
234 GroupLeader(GroupLeader), PSE(&PSE), L(&L) {}
243 return ExcludeRecipes.contains(
Store) ||
244 (
Store && isNoAliasViaDistance(
Store, &GroupLeader));
257 std::optional<SinkStoreInfo> SinkInfo = {}) {
258 bool CheckReads = SinkInfo.has_value();
262 if (SinkInfo && SinkInfo->shouldSkip(R))
266 if (!
R.mayWriteToMemory() && !(CheckReads &&
R.mayReadFromMemory()))
291template <
unsigned Opcode>
296 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
297 "Only Load and Store opcodes supported");
298 constexpr bool IsLoad = (Opcode == Instruction::Load);
301 RecipesByAddressAndType;
305 if (RepR.getOpcode() != Opcode || !FilterFn(&RepR))
309 VPValue *Addr = RepR.getOperand(IsLoad ? 0 : 1);
313 RecipesByAddressAndType[{AddrSCEV, LoadStoreTy}].push_back(&RepR);
318 for (
auto &Group :
Groups) {
333 auto InsertIfValidSinkCandidate = [ScalarVFOnly, &WorkList](
340 if (Candidate->getParent() == SinkTo ||
341 all_of(Candidate->operands(),
342 [](
VPValue *
Op) { return Op->isDefinedOutsideLoopRegions(); }) ||
354 WorkList.
insert({SinkTo, Candidate});
366 for (
auto &Recipe : *VPBB)
368 InsertIfValidSinkCandidate(VPBB,
Op);
372 for (
unsigned I = 0;
I != WorkList.
size(); ++
I) {
375 std::tie(SinkTo, SinkCandidate) = WorkList[
I];
380 auto UsersOutsideSinkTo =
382 return cast<VPRecipeBase>(U)->getParent() != SinkTo;
384 if (
any_of(UsersOutsideSinkTo, [SinkCandidate](
VPUser *U) {
385 return !U->usesFirstLaneOnly(SinkCandidate);
388 bool NeedsDuplicating = !UsersOutsideSinkTo.empty();
390 if (NeedsDuplicating) {
394 if (
auto *SinkCandidateRepR =
399 SinkCandidateRepR->getOpcode(), SinkCandidate->
operands(),
400 nullptr, *SinkCandidateRepR, *SinkCandidateRepR,
404 Clone = SinkCandidate->
clone();
414 InsertIfValidSinkCandidate(SinkTo,
Op);
423 if (EntryBB->getNumSuccessors() != 2)
428 if (!Succ0 || !Succ1)
431 if (Succ0->getNumSuccessors() + Succ1->getNumSuccessors() != 1)
433 if (Succ0->getSingleSuccessor() == Succ1)
435 if (Succ1->getSingleSuccessor() == Succ0)
452 if (!Region1->isReplicator())
454 auto *MiddleBasicBlock =
456 if (!MiddleBasicBlock || !MiddleBasicBlock->empty())
461 if (!Region2 || !Region2->isReplicator())
464 VPValue *Mask1 = Region1->getEntryBranchOnMask()->getOperand(0);
465 VPValue *Mask2 = Region2->getEntryBranchOnMask()->getOperand(0);
466 if (!Mask1 || Mask1 != Mask2)
469 assert(Mask1 && Mask2 &&
"both region must have conditions");
475 if (TransformedRegions.
contains(Region1))
482 if (!Then1 || !Then2)
490 std::optional<VPExecutionFrequency> Freq1 =
493 if (Freq1 && Freq2) {
494 if (Freq2->Freq < Freq1->Freq) {
497 Freq1.emplace(Freq1->Freq, Freq1->IsEstimated || Freq2->IsEstimated);
521 VPValue *Phi1ToMoveV = Phi1ToMove.getVPSingleValue();
527 if (Phi1ToMove.getVPSingleValue()->user_empty()) {
528 Phi1ToMove.eraseFromParent();
531 Phi1ToMove.moveBefore(*Merge2, Merge2->begin());
545 TransformedRegions.
insert(Region1);
548 return !TransformedRegions.
empty();
556 std::string RegionName = (
Twine(
"pred.") + Instr->getOpcodeName()).str();
557 assert(Instr->getParent() &&
"Predicated instruction not in any basic block");
558 auto *BlockInMask = PredRecipe->
getMask();
573 BOMRecipe->setExecutionFrequency(RecipeWithoutMask->getExecutionFrequency(),
575 RecipeWithoutMask->clearExecutionFrequency();
584 Region->setParent(ParentRegion);
590 RecipeWithoutMask->getDebugLoc());
591 Exiting->appendRecipe(PHIRecipe);
603 if (RepR.isPredicated())
621 if (ParentRegion && ParentRegion->
getExiting() == CurrentBlock)
633 if (!VPBB->getParent())
637 if (!PredVPBB || PredVPBB->getNumSuccessors() != 1 ||
646 R.moveBefore(*PredVPBB, PredVPBB->
end());
648 auto *ParentRegion = VPBB->getParent();
649 if (ParentRegion && ParentRegion->getExiting() == VPBB)
650 ParentRegion->setExiting(PredVPBB);
654 return !WorkList.
empty();
661 bool ShouldSimplify =
true;
662 while (ShouldSimplify) {
679 if (
IV.getTruncInst())
694 for (
auto *U : FindMyCast->
users()) {
696 if (UserCast && UserCast->getUnderlyingValue() == IRCast) {
697 FoundUserCast = UserCast;
704 FindMyCast = FoundUserCast;
706 if (FindMyCast != &
IV)
731 PhiR->replaceAllUsesWith(PhiR->getOperand(0));
733 PhiR->eraseFromParent();
799 Def->user_empty() || !Def->getUnderlyingValue() ||
800 (RepR && (RepR->isSingleScalar() || RepR->isPredicated())))
813 Def->getUnderlyingInstr()->getOpcode(), Def->operands(),
815 Def->getUnderlyingInstr());
816 Clone->insertAfter(Def);
817 Def->replaceAllUsesWith(Clone);
818 Def->eraseFromParent();
833 PtrIV->replaceAllUsesWith(PtrAdd);
840 if (HasOnlyVectorVFs &&
none_of(WideIV->users(), [WideIV](
VPUser *U) {
841 return U->usesScalars(WideIV);
850 WrapFlags = {
static_cast<bool>(WideIV->getNoWrapFlagsOrNone().HasNUW),
853 Plan, ID.getKind(), ID.getInductionOpcode(),
855 WideIV->getTruncInst(), WideIV->getStartValue(), WideIV->getStepValue(),
856 WideIV->getDebugLoc(), Builder, WrapFlags);
859 if (!HasOnlyVectorVFs) {
861 "plans containing a scalar VF cannot also include scalable VFs");
862 WideIV->replaceAllUsesWith(Steps);
865 WideIV->replaceUsesWithIf(Steps,
866 [WideIV, HasScalableVF](
VPUser &U,
unsigned) {
868 return U.usesFirstLaneOnly(WideIV);
869 return U.usesScalars(WideIV);
885 return (IntOrFpIV && IntOrFpIV->getTruncInst()) ? nullptr : WideIV;
890 if (!Def || Def->getNumOperands() != 2)
898 auto IsWideIVInc = [&]() {
899 auto &ID = WideIV->getInductionDescriptor();
902 VPValue *IVStep = WideIV->getStepValue();
903 switch (ID.getInductionOpcode()) {
904 case Instruction::Add:
906 case Instruction::FAdd:
908 case Instruction::FSub:
911 case Instruction::Sub: {
931 return IsWideIVInc() ? WideIV :
nullptr;
955 VPValue *FirstActiveLane =
B.createFirstActiveLane(Mask,
DL);
957 B.createScalarZExtOrTrunc(FirstActiveLane, CanonicalIVType,
DL);
958 VPValue *EndValue =
B.createAdd(CanonicalIV, FirstActiveLane,
DL);
963 if (Incoming != WideIV) {
965 EndValue =
B.createAdd(EndValue, One,
DL);
970 VPValue *Start = WideIV->getStartValue();
971 VPValue *Step = WideIV->getStepValue();
972 EndValue =
B.createDerivedIV(
974 Start, EndValue, Step);
988 if (WideIntOrFp && WideIntOrFp->getTruncInst())
998 Start, VectorTC, Step);
1030 assert(EndValue &&
"Must have computed the end value up front");
1035 if (Incoming != WideIV)
1047 auto *Zero = Plan.
getZero(StepTy);
1048 return B.createPtrAdd(EndValue,
B.createSub(Zero, Step),
1053 return B.createNaryOp(
1054 ID.getInductionBinOp()->getOpcode() == Instruction::FAdd
1056 : Instruction::FAdd,
1057 {EndValue, Step}, {ID.getInductionBinOp()->getFastMathFlags()});
1074 const SCEV *Start, *Step;
1092 VPValue *ExitCount = Builder.createOverflowingOp(
1095 return Builder.createDerivedIV(Kind,
nullptr, StartVPV, ExitCount,
1104 VPBuilder VectorPHBuilder(VectorPH, VectorPH->getFirstNonPhi());
1111 &WideIV, VectorPHBuilder, ResumeTC))
1112 EndValues[&WideIV] = EndValue;
1122 R.getVPSingleValue()->replaceAllUsesWith(EndValue);
1123 R.eraseFromParent();
1132 for (
auto [Idx, PredVPBB] :
enumerate(ExitVPBB->getPredecessors())) {
1134 if (PredVPBB == MiddleVPBB) {
1136 Plan, ExitIRI->getOperand(Idx), EndValues, PSE);
1139 Plan, ExitIRI->getOperand(Idx), PSE, ResumeTC, L);
1142 Plan, ExitIRI->getOperand(Idx), PSE);
1145 ExitIRI->setOperand(Idx, Escape);
1159 const auto &[V, Inserted] = SCEV2VPV.
try_emplace(ExpR.getSCEV(), &ExpR);
1163 ExpR.replaceAllUsesWith(V->second);
1167 ExpR.eraseFromParent();
1197 return Plan.
getZero(Def->getScalarType());
1214 return Def->getOperand(1);
1254 return Plan.
getZero(Def->getScalarType());
1258 Def->getScalarType() ==
A->getScalarType())
1268 if (Def->getScalarType() ==
A->getScalarType())
1278 A->getScalarType() == Def->getScalarType())
1284 return Def->getOperand(0);
1290 return BuildVector->getOperand(BuildVector->getNumOperands() - 1);
1306 return BuildVector->getOperand(BuildVector->getNumOperands() - 2);
1312 return BuildVector->getOperand(Idx);
1316 if (Def->getNumOperands() == 1) {
1317 return Def->getOperand(0);
1321 return Phi->getOperand(0);
1327 if (Def->getNumOperands() == 1 &&
1333 A->getScalarType() == Def->getScalarType())
1361 return VPR->getOperand(0);
1367 return Steps->getOperand(0);
1385 Def->replaceAllUsesWith(V);
1394 RepR && RepR->isPredicated() && RepR->getOpcode() == Instruction::Store &&
1398 RepR->getUnderlyingInstr(), RepR->operandsWithoutMask(),
1399 RepR->isSingleScalar(),
nullptr, *RepR, *RepR,
1400 RepR->getDebugLoc());
1401 Unmasked->insertBefore(RepR);
1415 bool CanCreateNewRecipe =
1421 if (CanCreateNewRecipe &&
1424 return Builder.createLogicalAnd(
X,
Y);
1427 if (CanCreateNewRecipe &&
1432 (!Def->getOperand(0)->hasMoreThanOneUniqueUser() ||
1433 !Def->getOperand(1)->hasMoreThanOneUniqueUser()))
1434 return Builder.createLogicalAnd(
X, Builder.createOr(
Y, Z));
1437 if (CanCreateNewRecipe &&
1441 return Builder.createLogicalOr(Z,
Y);
1445 if (CanCreateNewRecipe &&
1447 return Builder.createNot(
C);
1451 Def->setOperand(0,
C);
1452 Def->setOperand(1,
Y);
1453 Def->setOperand(2,
X);
1458 if (CanCreateNewRecipe &&
1462 Y->getScalarType()->isIntegerTy(1))
1463 return Builder.createOr(
Y, Builder.createLogicalAnd(
X, Z));
1467 if (CanCreateNewRecipe &&
1473 return Builder.createSelect(Builder.createLogicalAnd(Mask0, Mask1),
X,
Y,
1474 Def->getDebugLoc());
1480 Type *TruncTy = Def->getScalarType();
1481 Type *XTy =
X->getScalarType();
1484 unsigned ExtOpcode =
1488 if (
auto *UnderlyingExt =
Y->getUnderlyingValue()) {
1490 Ext->setUnderlyingValue(UnderlyingExt);
1494 auto *Trunc = Builder.createWidenCast(Instruction::Trunc,
X, TruncTy);
1503 return Builder.createSub(Plan.
getZero(
X->getScalarType()),
X,
1504 Def->getDebugLoc(),
"", NW);
1507 if (CanCreateNewRecipe &&
1515 return Builder.createSub(
X,
Y, Def->getDebugLoc(),
"", NW);
1522 Def->getDebugLoc());
1529 MulR->hasNoSignedWrap() &&
1531 return Builder.createNaryOp(
1534 Def->getDebugLoc());
1539 return Builder.createNaryOp(
1552 return match(U, m_Not(m_Specific(Cmp))) ||
1553 (match(U, m_Select(m_Specific(Cmp), m_VPValue(),
1555 U->getOperand(1) != Cmp && U->getOperand(2) != Cmp);
1562 R->setOperand(1,
Y);
1563 R->setOperand(2,
X);
1567 R->replaceAllUsesWith(Cmp);
1572 if (!Cmp->getDebugLoc() && Def->getDebugLoc())
1573 Cmp->setDebugLoc(Def->getDebugLoc());
1586 if (
Op->getNumUsers() > 1 ||
1590 }
else if (!UnpairedCmp) {
1591 UnpairedCmp =
Op->getDefiningRecipe();
1595 UnpairedCmp =
nullptr;
1602 if (NewOps.
size() < Def->getNumOperands())
1609 if (CanCreateNewRecipe &&
1618 X->getScalarType() != Def->getScalarType())
1619 return Builder.createWidenCast(Instruction::Trunc,
X, Def->getScalarType());
1626 Def->getScalarType()->isIntegerTy(1)) {
1627 Def->setOperand(1, Plan.
getTrue());
1628 Def->setOperand(0,
Y);
1638 Def->replaceUsesWithIf(Def->getOperand(0), [Def](
VPUser &U,
unsigned) {
1639 return U.usesFirstLaneOnly(Def);
1649 "broadcast operand must be single-scalar");
1650 Def->setOperand(0, Z);
1655 Def->replaceUsesWithIf(
1656 X, [Def](
const VPUser &U,
unsigned) {
return U.usesScalars(Def); });
1668 return Builder.createNaryOp(Instruction::ExtractElement, {
X, LaneToExtract},
1669 Def->getDebugLoc());
1681 IVInc->getNumUsers() == 2) {
1687 if ((Phi->getNumUsers() == 1 || (Phi->getNumUsers() == 2 && Inc)) &&
1689 Def->replaceAllUsesWith(IVInc);
1691 Inc->replaceAllUsesWith(Phi);
1692 Phi->setOperand(0,
Y);
1701 Def->replaceUsesWithIf(StartV, [](
const VPUser &U,
unsigned Idx) {
1703 return PhiR && PhiR->isInLoop();
1720 [[maybe_unused]]
unsigned InitWorklistSize = Worklist.
size();
1722 while (!Worklist.
empty()) {
1723 assert(Worklist.
size() < InitWorklistSize * 2 &&
1724 "Worklist is growing large, possible cycle?");
1731 Def->replaceAllUsesWith(New);
1732 Def->eraseFromParent();
1737 Def->eraseFromParent();
1755 R.getVPSingleValue()->replaceAllUsesWith(
X);
1771 while (!Worklist.
empty()) {
1780 R->replaceAllUsesWith(
1781 Builder.createLogicalAnd(HeaderMask, Builder.createLogicalAnd(
X,
Y)));
1785static std::optional<Instruction::BinaryOps>
1788 case Intrinsic::masked_udiv:
1789 return Instruction::UDiv;
1790 case Intrinsic::masked_sdiv:
1791 return Instruction::SDiv;
1792 case Intrinsic::masked_urem:
1793 return Instruction::URem;
1794 case Intrinsic::masked_srem:
1795 return Instruction::SRem;
1812 if (RepR && (RepR->isSingleScalar() || RepR->isPredicated()))
1816 if (RepR && RepR->getOpcode() == Instruction::Store &&
1819 RepOrWidenR->getUnderlyingInstr(), RepOrWidenR->operands(),
1820 true ,
nullptr , *RepR ,
1821 *RepR , RepR->getDebugLoc());
1822 Clone->insertBefore(RepOrWidenR);
1824 VPValue *ExtractOp = Clone->getOperand(0);
1830 Clone->setOperand(0, ExtractOp);
1831 RepR->eraseFromParent();
1843 VPValue *SafeDivisor = Builder.createSelect(
1844 IntrR->getOperand(2), IntrR->getOperand(1),
1846 VPValue *Clone = Builder.createNaryOp(
1847 *
Opc, {IntrR->getOperand(0), SafeDivisor},
1850 IntrR->eraseFromParent();
1859 auto IntroducesBCastOf = [](
const VPValue *
Op) {
1868 return !U->usesScalars(
Op);
1872 if (
any_of(RepOrWidenR->users(), IntroducesBCastOf(RepOrWidenR)) &&
1875 make_filter_range(Op->users(), not_equal_to(RepOrWidenR)),
1876 IntroducesBCastOf(Op)))
1880 bool LiveInNeedsBroadcast =
1881 isa<VPIRValue>(Op) && !isa<VPConstant>(Op);
1882 auto *OpR = dyn_cast<VPReplicateRecipe>(Op);
1883 return LiveInNeedsBroadcast || (OpR && OpR->isSingleScalar());
1891 Clone->insertBefore(RepOrWidenR);
1892 RepOrWidenR->replaceAllUsesWith(Clone);
1894 RepOrWidenR->eraseFromParent();
1927 if (Blend.isNormalized() || !
match(Blend.getMask(0),
m_False()))
1928 UniqueValues.
insert(Blend.getIncomingValue(0));
1929 for (
unsigned I = 1;
I != Blend.getNumIncomingValues(); ++
I)
1931 UniqueValues.
insert(Blend.getIncomingValue(
I));
1933 if (UniqueValues.
size() == 1) {
1934 Blend.replaceAllUsesWith(*UniqueValues.
begin());
1935 Blend.eraseFromParent();
1939 if (Blend.isNormalized())
1945 unsigned StartIndex = 0;
1946 for (
unsigned I = 0;
I != Blend.getNumIncomingValues(); ++
I) {
1958 OperandsWithMask.
push_back(Blend.getIncomingValue(StartIndex));
1960 for (
unsigned I = 0;
I != Blend.getNumIncomingValues(); ++
I) {
1961 if (
I == StartIndex)
1963 OperandsWithMask.
push_back(Blend.getIncomingValue(
I));
1964 OperandsWithMask.
push_back(Blend.getMask(
I));
1969 OperandsWithMask, Blend, Blend.getDebugLoc());
1970 NewBlend->insertBefore(&Blend);
1972 VPValue *DeadMask = Blend.getMask(StartIndex);
1974 Blend.eraseFromParent();
1979 if (NewBlend->getNumOperands() == 3 &&
1981 VPValue *Inc0 = NewBlend->getOperand(0);
1982 VPValue *Inc1 = NewBlend->getOperand(1);
1983 VPValue *OldMask = NewBlend->getOperand(2);
1984 NewBlend->setOperand(0, Inc1);
1985 NewBlend->setOperand(1, Inc0);
1986 NewBlend->setOperand(2, NewMask);
2013 APInt MaxVal = AlignedTC - 1;
2016 unsigned NewBitWidth =
2022 bool MadeChange =
false;
2047 "canonical IV is not expected to have a truncation");
2052 NewWideIV->insertBefore(WideIV);
2059 Cmp->replaceAllUsesWith(
2060 VPBuilder(Cmp).createICmp(Cmp->getPredicate(), NewWideIV, NewBTC));
2074 return any_of(
Cond->getDefiningRecipe()->operands(), [&Plan, BestVF, BestUF,
2076 return isConditionTrueViaVFAndUF(C, Plan, BestVF, BestUF, PSE);
2090 const SCEV *VectorTripCount =
2095 "Trip count SCEV must be computable");
2110 bool MadeChange =
false;
2118 for (
VPBasicBlock *VPBB : {PreheaderVPBB, ExitingVPBB}) {
2127 Builder.setInsertPoint(Extract);
2130 Start = Builder.createAdd(
2135 Extract->eraseFromParent();
2150 auto *Term = &ExitingVPBB->
back();
2156 bool MatchedCanIVInc =
2162 if (MatchedCanIVInc ||
2170 const SCEV *VectorTripCount =
2176 "Trip count SCEV must be computable");
2195 Term->setOperand(1, Plan.
getTrue());
2200 {}, Term->getDebugLoc());
2202 Term->eraseFromParent();
2210 assert(Plan.
hasVF(BestVF) &&
"BestVF is not available in Plan");
2211 assert(Plan.
hasUF(BestUF) &&
"BestUF is not available in Plan");
2227 RecurKind RK = PhiR.getRecurrenceKind();
2234 RecWithFlags->dropPoisonGeneratingFlags();
2240struct VPCSEDenseMapInfo :
public DenseMapInfo<VPSingleDefRecipe *> {
2249 return GEP->getSourceElementType();
2252 .Case<VPVectorPointerRecipe, VPWidenGEPRecipe>(
2253 [](
auto *
I) {
return I->getSourceElementType(); })
2254 .
Default([](
auto *) {
return nullptr; });
2258 static bool canHandle(
const VPSingleDefRecipe *Def) {
2267 if (!
C || (!
C->first && (
C->second == Instruction::InsertValue ||
2268 C->second == Instruction::ExtractValue)))
2274 if (
Def->mayWriteToMemory())
2276 return !
Def->mayReadFromMemory() ||
2281 static unsigned getHashValue(
const VPSingleDefRecipe *Def) {
2284 getGEPSourceElementType(Def),
Def->getScalarType(),
2287 if (RFlags->hasPredicate())
2290 return hash_combine(Result, SIVSteps->getInductionOpcode());
2299 static bool isEqual(
const VPSingleDefRecipe *L,
const VPSingleDefRecipe *R) {
2300 if (
L->getVPRecipeID() !=
R->getVPRecipeID() ||
2303 getGEPSourceElementType(L) != getGEPSourceElementType(R) ||
2305 !
equal(
L->operands(),
R->operands()))
2309 "must have valid opcode info for both recipes");
2311 if (LFlags->hasPredicate() &&
2312 LFlags->getPredicate() !=
2316 if (LSIV->getInductionOpcode() !=
2331 const VPRegionBlock *RegionL =
L->getRegion();
2332 const VPRegionBlock *RegionR =
R->getRegion();
2335 L->getParent() !=
R->getParent())
2337 return L->getScalarType() ==
R->getScalarType();
2356 if (R.mayWriteToMemory())
2359 if (!Def || !VPCSEDenseMapInfo::canHandle(Def))
2362 auto [It, Inserted] =
2363 (IsLoad ? LoadCSEMap : CSEMap).try_emplace(Def, Def);
2368 if (!VPDT.
dominates(V->getParent(), VPBB))
2373 if (EarlierLoad->getAlign() <
Load->getAlign()) {
2380 EarlierLoad->intersect(*
Load);
2385 Def->replaceAllUsesWith(V);
2396 bool Sinking =
false) {
2425 "Expected vector prehader's successor to be the vector loop region");
2433 return !Op->isDefinedOutsideLoopRegions();
2436 R.moveBefore(*Preheader, Preheader->
end());
2456 assert(!RepR->isPredicated() &&
2457 "Expected prior transformation of predicated replicates to "
2458 "replicate regions");
2463 if (!RepR->isSingleScalar())
2467 if (RepR->getOpcode() == Instruction::Store &&
2468 !RepR->getOperand(1)->isDefinedOutsideLoopRegions())
2473 assert((!R.mayWriteToMemory() ||
2474 (RepR && RepR->getOpcode() == Instruction::Store &&
2475 RepR->getOperand(1)->isDefinedOutsideLoopRegions())) &&
2476 "The only recipes that may write to memory are expected to be "
2477 "stores with invariant pointer-operand");
2487 if (
any_of(Def->users(), [&SinkBB, &LoopRegion](
VPUser *U) {
2488 auto *UserR = cast<VPRecipeBase>(U);
2489 VPBasicBlock *Parent = UserR->getParent();
2491 if (SinkBB && SinkBB != Parent)
2496 return UserR->isPhi() || Parent->getEnclosingLoopRegion() ||
2497 Parent->getSinglePredecessor() != LoopRegion;
2507 "Defining block must dominate sink block");
2532 VPValue *ResultVPV = R.getVPSingleValue();
2534 unsigned NewResSizeInBits = MinBWs.
lookup(UI);
2535 if (!NewResSizeInBits)
2548 (void)OldResSizeInBits;
2556 VPW->dropPoisonGeneratingFlags();
2558 assert((OldResSizeInBits != NewResSizeInBits ||
2560 "Only ICmps should not need extending the result.");
2573 unsigned OpSizeInBits =
Op->getScalarType()->getScalarSizeInBits();
2574 if (OpSizeInBits == NewResSizeInBits)
2576 assert(OpSizeInBits > NewResSizeInBits &&
"nothing to truncate");
2577 auto [ProcessedIter, Inserted] = ProcessedTruncs.
try_emplace(
Op);
2583 Builder.setInsertPoint(&R);
2584 ProcessedIter->second =
2585 Builder.createWidenCast(Instruction::Trunc,
Op, NewResTy);
2587 Op = ProcessedIter->second;
2591 NWR->insertBefore(&R);
2596 VPValue *Replacement = NWR->getVPSingleValue();
2603 R.eraseFromParent();
2609 std::optional<VPDominatorTree> VPDT;
2617 bool SimplifiedPhi =
false;
2627 assert(VPBB->getNumSuccessors() == 2 &&
2628 "Two successors expected for BranchOnCond");
2629 unsigned RemovedIdx;
2640 "There must be a single edge between VPBB and its successor");
2645 SimplifiedPhi =
true;
2649 if (!PhiR || PhiR->getNumIncoming() != 1)
2651 PhiR->replaceAllUsesWith(PhiR->getOperand(0));
2652 PhiR->eraseFromParent();
2657 VPBB->back().eraseFromParent();
2669 if (Reachable.contains(
B))
2680 for (
VPValue *Def : R.definedValues())
2681 Def->replaceAllUsesWith(&Tmp);
2682 R.eraseFromParent();
2686 return SimplifiedPhi;
2712 auto GetSimplifiedLiveInViaSCEV = [&](
VPValue *VPV) ->
VPValue * {
2721 if (
VPValue *SimplifiedLiveIn = GetSimplifiedLiveInViaSCEV(LiveIn))
2722 LiveIn->replaceAllUsesWith(SimplifiedLiveIn);
2733 "expected to run before loop regions are created");
2735 auto CanUseVersionedStride = [&VPDT, Header = Header, &Plan](
VPUser &U,
2742 return VPDT.
dominates(Header, R->getParent());
2746 Value *StrideV = Stride->getValue();
2747 const APInt *StrideConst;
2754 CanUseVersionedStride);
2768 CanUseVersionedStride);
2770 RewriteMap[StrideV] = StrideExpr;
2775 const SCEV *ScevExpr = ExpSCEV.getSCEV();
2778 if (NewSCEV != ScevExpr) {
2780 ExpSCEV.replaceAllUsesWith(NewExp);
2791 auto CollectPoisonGeneratingInstrsInBackwardSlice([&](
VPRecipeBase *Root) {
2796 while (!Worklist.
empty()) {
2799 if (!Visited.
insert(CurRec).second)
2821 RecWithFlags->isDisjoint()) {
2824 Builder.createAdd(
A,
B, RecWithFlags->getDebugLoc());
2825 New->setUnderlyingValue(RecWithFlags->getUnderlyingValue());
2826 RecWithFlags->replaceAllUsesWith(New);
2827 RecWithFlags->eraseFromParent();
2830 RecWithFlags->dropPoisonGeneratingFlags();
2835 assert((!Instr || !Instr->hasPoisonGeneratingFlags()) &&
2836 "found instruction with poison generating flags not covered by "
2837 "VPRecipeWithIRFlags");
2842 if (
VPRecipeBase *OpDef = Operand->getDefiningRecipe())
2864 VPRecipeBase *AddrDef = WidenRec->getAddr()->getDefiningRecipe();
2865 if (AddrDef && WidenRec->isConsecutive() && WidenRec->getMask() &&
2866 match(WidenRec->getMask(), m_UnlessHdrMask))
2867 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2869 VPRecipeBase *AddrDef = InterleaveRec->getAddr()->getDefiningRecipe();
2870 if (AddrDef && InterleaveRec->getMask() &&
2871 match(InterleaveRec->getMask(), m_UnlessHdrMask))
2872 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2882 const bool &EpilogueAllowed) {
2883 if (InterleaveGroups.empty())
2894 IRMemberToRecipe[&MemR->getIngredient()] = MemR;
2901 for (
const auto *IG : InterleaveGroups) {
2904 for (
auto *Member : IG->members())
2906 StartMember = Member;
2914 for (
unsigned I = 0;
I < IG->getFactor(); ++
I) {
2920 StoredValues.
push_back(StoreR->getStoredValue());
2927 bool NeedsMaskForGaps =
2928 (IG->requiresScalarEpilogue() && !EpilogueAllowed) ||
2929 (!StoredValues.
empty() && !IG->isFull());
2932 auto *InsertPos = IRMemberToRecipe.
lookup(IRInsertPos);
2936 "Dead member in non-load group?");
2941 InsertPos->getAsRecipe()))
2942 InsertPos = MemberR;
2943 IRInsertPos = &InsertPos->getIngredient();
2953 VPValue *Addr = Start->getAddr();
2955 if (IG->getIndex(StartMember) != 0 ||
2963 assert(IG->getIndex(IRInsertPos) != 0 &&
2964 "index of insert position shouldn't be zero");
2968 IG->getIndex(IRInsertPos),
2972 Addr =
B.createNoWrapPtrAdd(InsertPos->getAddr(), OffsetVPV, NW);
2978 if (IG->isReverse()) {
2981 -(int64_t)IG->getFactor(), NW, InsertPosR->
getDebugLoc());
2982 ReversePtr->insertBefore(InsertPosR);
2986 IG, Addr, StoredValues, InsertPos->getMask(), NeedsMaskForGaps,
2988 VPIG->insertBefore(InsertPosR);
2991 for (
unsigned i = 0; i < IG->getFactor(); ++i)
2994 if (!Member->getType()->isVoidTy()) {
3066 VPValue *UncountableCondition =
nullptr;
3073 Worklist.
push_back(UncountableCondition);
3074 while (!Worklist.
empty()) {
3078 if (V->isDefinedOutsideLoopRegions())
3084 if (V->getNumUsers() > 1)
3116 if (Recipes.
empty() ||
3120 return UncountableCondition;
3177 for (
auto &Exit : Exits) {
3178 if (Exit.EarlyExitingVPBB == LatchVPBB)
3182 cast<VPIRPhi>(&R)->removeIncomingValueFor(Exit.EarlyExitingVPBB);
3183 Exit.EarlyExitingVPBB->getTerminator()->eraseFromParent();
3209 assert(
Load &&
"Couldn't find exactly one load");
3212 "Uncountable exit condition load is conditional.");
3226 DL.getTypeStoreSize(
Load->getScalarType()).getFixedValue());
3250 while (InsertIt != HeaderVPBB->
end() &&
3252 erase(ConditionRecipes, &*InsertIt);
3255 for (
auto *Recipe :
reverse(ConditionRecipes))
3256 Recipe->moveBefore(*HeaderVPBB, InsertIt);
3260 VPBuilder MaskBuilder(HeaderVPBB, InsertIt);
3262 Type *IVScalarTy =
IV->getScalarType();
3268 "uncountable.exit.mask");
3273 if (R.mayReadOrWriteMemory() && &R !=
Load) {
3275 if (!VPDT.
dominates(R.getParent(), LatchVPBB))
3285 "Expected BranchOnCond terminator for MiddleVPBB");
3296 auto Phis = ScalarPH->
phis();
3306 "Continuing from different IV");
3328 VPBuilder LatchBuilder(LatchVPBB->getTerminator());
3330 for (
auto [EarlyExitingVPBB, ExitBlock] :
3334 VPValue *CondOfEarlyExitingVPBB;
3335 [[maybe_unused]]
bool Matched =
3336 match(EarlyExitingVPBB->getTerminator(),
3338 assert(Matched &&
"Terminator must be BranchOnCond");
3342 VPBuilder EarlyExitingBuilder(EarlyExitingVPBB->getTerminator());
3343 auto *CondToEarlyExit = EarlyExitingBuilder.
createNaryOp(
3345 TrueSucc == ExitBlock
3346 ? CondOfEarlyExitingVPBB
3347 : EarlyExitingBuilder.
createNot(CondOfEarlyExitingVPBB));
3353 "exit condition must dominate the latch");
3361 assert(!Exits.
empty() &&
"must have at least one early exit");
3368 for (
const auto &[Num, VPB] :
enumerate(RPOT))
3371 return RPOIdx[
A.EarlyExitingVPBB] < RPOIdx[
B.EarlyExitingVPBB];
3377 for (
unsigned I = 0;
I + 1 < Exits.
size(); ++
I)
3378 for (
unsigned J =
I + 1; J < Exits.
size(); ++J)
3380 Exits[
I].EarlyExitingVPBB) &&
3381 "RPO sort must place dominating exits before dominated ones");
3387 VPValue *Combined = Exits[0].CondToExit;
3407 "Unexpected terminator");
3408 VPValue *IsLatchExitTaken = LatchExitingBranch->getOperand(0);
3409 DebugLoc LatchDL = LatchExitingBranch->getDebugLoc();
3410 LatchExitingBranch->eraseFromParent();
3413 {IsAnyExitTaken, IsLatchExitTaken}, LatchDL);
3414 LatchVPBB->clearSuccessors();
3419 LatchVPBB->setSuccessors({MiddleVPBB, MiddleVPBB, HeaderVPBB});
3420 MiddleVPBB->clearPredecessors();
3421 MiddleVPBB->setPredecessors({LatchVPBB, LatchVPBB});
3423 Plan, Exits, HeaderVPBB, LatchVPBB, MiddleVPBB, TheLoop, PSE, DT, AC);
3428 for (
unsigned Idx = 0; Idx != Exits.
size(); ++Idx) {
3432 VectorEarlyExitVPBBs[Idx] = VectorEarlyExitVPBB;
3440 Exits.
size() == 1 ? VectorEarlyExitVPBBs[0]
3443 LatchVPBB->setSuccessors({DispatchVPBB, MiddleVPBB, HeaderVPBB});
3475 for (
auto [Exit, VectorEarlyExitVPBB] :
3476 zip_equal(Exits, VectorEarlyExitVPBBs)) {
3477 auto &[EarlyExitingVPBB, EarlyExitVPBB,
_] = Exit;
3489 ExitIRI->getIncomingValueForBlock(EarlyExitingVPBB);
3490 VPValue *NewIncoming = IncomingVal;
3492 VPBuilder EarlyExitBuilder(VectorEarlyExitVPBB);
3497 ExitIRI->removeIncomingValueFor(EarlyExitingVPBB);
3498 ExitIRI->addIncoming(NewIncoming);
3501 EarlyExitingVPBB->getTerminator()->eraseFromParent();
3535 bool IsLastDispatch = (
I + 2 == Exits.
size());
3537 IsLastDispatch ? VectorEarlyExitVPBBs.
back()
3543 VectorEarlyExitVPBBs[
I]->setPredecessors({CurrentBB});
3546 CurrentBB = FalseBB;
3561 VPValue *VecOp = Red->getVecOp();
3564 if (Red->isPartialReduction())
3568 auto IsExtendedRedValidAndClampRange =
3581 "getExtendedReductionCost only supports integer types");
3582 ExtRedCost = Ctx.TTI.getExtendedReductionCost(
3583 Opcode, ExtOpc == Instruction::CastOps::ZExt, RedTy, SrcVecTy,
3584 Red->getFastMathFlagsOrNone(),
CostKind);
3585 return ExtRedCost.
isValid() && ExtRedCost < ExtCost + RedCost;
3593 IsExtendedRedValidAndClampRange(
3614 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3615 Opcode != Instruction::FAdd)
3619 if (Red->isPartialReduction())
3625 auto IsMulAccValidAndClampRange =
3637 (Ext0->getOpcode() != Ext1->getOpcode() ||
3638 Ext0->getOpcode() == Instruction::CastOps::FPExt))
3642 !Ext0 || Ext0->getOpcode() == Instruction::CastOps::ZExt;
3644 MulAccCost = Ctx.TTI.getMulAccReductionCost(IsZExt, Opcode, RedTy,
3651 ExtCost += Ext0->computeCost(VF, Ctx);
3653 ExtCost += Ext1->computeCost(VF, Ctx);
3655 ExtCost += OuterExt->computeCost(VF, Ctx);
3657 return MulAccCost.
isValid() &&
3658 MulAccCost < ExtCost + MulCost + RedCost;
3663 VPValue *VecOp = Red->getVecOp();
3701 Builder.createWidenCast(Instruction::CastOps::Trunc, ValB, NarrowTy);
3703 ValB = ExtB = Builder.createWidenCast(ExtOpc, Trunc, WideTy);
3704 Mul->setOperand(1, ExtB);
3714 ExtendAndReplaceConstantOp(RecipeA, RecipeB,
B,
Mul);
3719 IsMulAccValidAndClampRange(
Mul, RecipeA, RecipeB,
nullptr)) {
3726 if (!
Sub && IsMulAccValidAndClampRange(
Mul,
nullptr,
nullptr,
nullptr))
3743 ExtendAndReplaceConstantOp(Ext0, Ext1,
B,
Mul);
3752 (Ext->getOpcode() == Ext0->getOpcode() || Ext0 == Ext1) &&
3753 Ext0->getOpcode() == Ext1->getOpcode() &&
3754 IsMulAccValidAndClampRange(
Mul, Ext0, Ext1, Ext) &&
Mul->hasOneUse()) {
3756 Ext0->getOpcode(), Ext0->getOperand(0), Ext->getScalarType(),
nullptr,
3757 *Ext0, *Ext0, Ext0->getDebugLoc());
3758 NewExt0->insertBefore(Ext0);
3763 Ext->getScalarType(),
nullptr, *Ext1,
3764 *Ext1, Ext1->getDebugLoc());
3767 auto *NewMul =
Mul->cloneWithOperands({NewExt0, NewExt1});
3768 NewMul->insertBefore(
Mul);
3769 Ext->replaceAllUsesWith(NewMul);
3770 Ext->eraseFromParent();
3771 Mul->eraseFromParent();
3785 if (Red->isPartialReduction())
3789 auto IP = std::next(Red->getIterator());
3790 auto *VPBB = Red->getParent();
3800 Red->replaceAllUsesWith(AbstractR);
3822 return CommonMetadata;
3825template <
unsigned Opcode>
3830 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
3831 "Only Load and Store opcodes supported");
3832 [[maybe_unused]]
constexpr bool IsLoad = (Opcode == Instruction::Load);
3839 for (
auto Recipes :
Groups) {
3840 if (Recipes.size() < 2)
3845 "Expected all recipes in group to have the same load-store type");
3852 VPValue *MaskI = RecipeI->getMask();
3858 bool HasComplementaryMask =
false;
3863 VPValue *MaskJ = RecipeJ->getMask();
3872 if (HasComplementaryMask) {
3873 assert(Group.
size() >= 2 &&
"must have at least 2 entries");
3883template <
typename InstType>
3901 for (
auto &Group :
Groups) {
3921 return R->isSingleScalar() == IsSingleScalar;
3923 "all members in group must agree on IsSingleScalar");
3928 LoadWithMinAlign->getUnderlyingInstr(), {EarliestLoad->getOperand(0)},
3929 IsSingleScalar,
nullptr, *EarliestLoad, CommonMetadata);
3931 UnpredicatedLoad->insertBefore(EarliestLoad);
3935 Load->replaceAllUsesWith(UnpredicatedLoad);
3936 Load->eraseFromParent();
3945 if (!StoreLoc || !StoreLoc->AATags.Scope)
3952 SinkStoreInfo SinkInfo(StoresToSink, *StoresToSink[0], PSE, L);
3964 for (
auto &Group :
Groups) {
3977 VPValue *SelectedValue = Group[0]->getOperand(0);
3980 bool IsSingleScalar = Group[0]->isSingleScalar();
3981 for (
unsigned I = 1;
I < Group.size(); ++
I) {
3982 assert(IsSingleScalar == Group[
I]->isSingleScalar() &&
3983 "all members in group must agree on IsSingleScalar");
3984 VPValue *Mask = Group[
I]->getMask();
3986 SelectedValue = Builder.createSelect(
3989 Value->getScalarType()));
3997 StoreWithMinAlign->getUnderlyingInstr(),
3998 {SelectedValue, LastStore->getOperand(1)}, IsSingleScalar,
3999 nullptr, *LastStore, CommonMetadata);
4000 UnpredicatedStore->insertBefore(*InsertBB, LastStore->
getIterator());
4004 Store->eraseFromParent();
4019 VPValue *OpV,
unsigned Idx,
bool IsScalable) {
4024 if (Member0Op == OpV)
4034 return !IsScalable && !W->getMask() && W->isConsecutive() &&
4037 return IR->getInterleaveGroup()->isFull() &&
IR->getVPValue(Idx) == OpV;
4052 if (R->getScalarType() != WideMember0->getScalarType())
4054 if (R->hasPredicate() && R->getPredicate() != WideMember0->getPredicate())
4058 for (
unsigned Idx = 0; Idx != WideMember0->getNumOperands(); ++Idx) {
4061 OpsI.
push_back(
Op->getDefiningRecipe()->getOperand(Idx));
4066 if (
any_of(
enumerate(OpsI), [WideMember0, Idx, IsScalable](
const auto &
P) {
4067 const auto &[OpIdx, OpV] =
P;
4068 return !
canNarrowLoad(WideMember0, Idx, OpV, OpIdx, IsScalable);
4079static std::optional<ElementCount>
4083 if (!InterleaveR || InterleaveR->
getMask())
4084 return std::nullopt;
4086 Type *GroupElementTy =
nullptr;
4090 return Op->getScalarType() == GroupElementTy;
4092 return std::nullopt;
4096 return Op->getScalarType() == GroupElementTy;
4098 return std::nullopt;
4102 if (IG->getFactor() != IG->getNumMembers())
4103 return std::nullopt;
4109 assert(
Size.isScalable() == VF.isScalable() &&
4110 "if Size is scalable, VF must be scalable and vice versa");
4111 return Size.getKnownMinValue();
4115 unsigned MinVal = VF.getKnownMinValue();
4117 if (IG->getFactor() == MinVal && GroupSize == GetVectorBitWidthForVF(VF))
4120 return std::nullopt;
4128 return RepR && RepR->isSingleScalar();
4142 if (V->isDefinedOutsideLoopRegions()) {
4145 return M->isDefinedOutsideLoopRegions() &&
4146 M->getScalarType() == V->getScalarType();
4148 "expected distinct loop-invariant values of matching scalar type");
4163 for (
unsigned Idx = 0,
E = WideMember0->getNumOperands(); Idx !=
E; ++Idx) {
4165 for (
VPValue *Member : Members)
4166 OpsI.
push_back(Member->getDefiningRecipe()->getOperand(Idx));
4167 WideMember0->setOperand(
4176 auto *LI =
cast<LoadInst>(LoadGroup->getInterleaveGroup()->getInsertPos());
4178 *LI, LoadGroup->getAddr(), LoadGroup->getMask(),
true,
4179 *LoadGroup, LoadGroup->getDebugLoc());
4185 assert(RepR->isSingleScalar() && RepR->getOpcode() == Instruction::Load &&
4186 "must be a single scalar load");
4187 NarrowedOps.
insert(RepR);
4192 VPValue *PtrOp = WideLoad->getAddr();
4194 PtrOp = VecPtr->getOperand(0);
4199 nullptr, {}, *WideLoad);
4200 N->insertBefore(WideLoad);
4205std::unique_ptr<VPlan>
4225 "unexpected branch-on-count");
4228 std::optional<ElementCount> VFToOptimize;
4242 if (R.mayWriteToMemory() && !InterleaveR)
4248 return any_of(V->users(), [&](VPUser *U) {
4249 auto *UR = cast<VPRecipeBase>(U);
4250 return UR->getParent()->getParent() != VectorLoop;
4267 std::optional<ElementCount> NarrowedVF =
4269 if (!NarrowedVF || (VFToOptimize && NarrowedVF != VFToOptimize))
4271 VFToOptimize = NarrowedVF;
4274 if (InterleaveR->getStoredValues().empty())
4279 auto *Member0 = InterleaveR->getStoredValues()[0];
4289 VPRecipeBase *DefR = Op.value()->getDefiningRecipe();
4292 auto *IR = dyn_cast<VPInterleaveRecipe>(DefR);
4293 return IR && IR->getInterleaveGroup()->isFull() &&
4294 IR->getVPValue(Op.index()) == Op.value();
4303 VFToOptimize->isScalable()))
4308 if (StoreGroups.empty())
4312 bool RequiresScalarEpilogue =
4323 std::unique_ptr<VPlan> NewPlan;
4325 NewPlan = std::unique_ptr<VPlan>(Plan.
duplicate());
4326 Plan.
setVF(*VFToOptimize);
4327 NewPlan->removeVF(*VFToOptimize);
4334 for (
auto *StoreGroup : StoreGroups) {
4336 NarrowedOps, Preheader);
4342 StoreGroup->getDebugLoc());
4349 Type *CanIVTy = VectorLoop->getCanonicalIVType();
4355 if (VFToOptimize->isScalable()) {
4358 Step = PHBuilder.createOverflowingOp(Instruction::Mul, {VScale,
UF},
4366 materializeVectorTripCount(Plan, VectorPH,
false,
4367 RequiresScalarEpilogue, Step);
4372 removeDeadRecipes(Plan);
4375 "All VPVectorPointerRecipes should have been removed");
4393 "Cannot handle loops with uncountable early exits");
4400 assert(RecurSplice &&
"expected FirstOrderRecurrenceSplice");
4407 if (
any_of(RecurSplice->users(),
4408 [](
VPUser *U) { return !cast<VPRecipeBase>(U)->getRegion(); }) &&
4489 {},
"vector.recur.extract.for.phi");
4492 ExitPhi->replaceUsesOfWith(ExtractR, PenultimateElement);
4506 VPValue *WidenIVCandidate = BinOp->getOperand(0);
4507 VPValue *InvariantCandidate = BinOp->getOperand(1);
4509 std::swap(WidenIVCandidate, InvariantCandidate);
4523 auto *ClonedOp = BinOp->
clone();
4524 if (ClonedOp->getOperand(0) == WidenIV) {
4525 ClonedOp->setOperand(0, ScalarIV);
4527 assert(ClonedOp->getOperand(1) == WidenIV &&
"one operand must be WideIV");
4528 ClonedOp->setOperand(1, ScalarIV);
4542 return std::nullopt;
4547 return std::nullopt;
4559 auto CheckSentinel = [&SE](
const SCEV *IVSCEV,
4560 bool UseMax) -> std::optional<APSInt> {
4562 for (
bool Signed : {
true,
false}) {
4571 return std::nullopt;
4579 PhiR->getRecurrenceKind()))
4588 VPValue *BackedgeVal = PhiR->getBackedgeValue();
4602 !
match(FindLastSelect,
4611 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression, PSE,
4616 "IVOfExpressionToSink not being an AddRec must imply "
4617 "FindLastExpression not being an AddRec.");
4626 bool UseMax = *StepDirection;
4627 std::optional<APSInt> SentinelVal = CheckSentinel(IVSCEV, UseMax);
4628 bool UseSigned = SentinelVal && SentinelVal->isSigned();
4635 if (IVOfExpressionToSink) {
4636 const SCEV *FindLastExpressionSCEV =
4638 if (std::optional<bool> NewUseMax =
4640 if (
auto NewSentinel =
4641 CheckSentinel(FindLastExpressionSCEV, *NewUseMax)) {
4644 SentinelVal = *NewSentinel;
4645 UseSigned = NewSentinel->isSigned();
4646 UseMax = *NewUseMax;
4647 IVSCEV = FindLastExpressionSCEV;
4648 IVOfExpressionToSink =
nullptr;
4658 if (AR->hasNoSignedWrap())
4660 else if (AR->hasNoUnsignedWrap())
4670 VPValue *NewFindLastSelect = BackedgeVal;
4672 if (!SentinelVal || IVOfExpressionToSink) {
4675 DebugLoc DL = FindLastSelect->getDefiningRecipe()->getDebugLoc();
4676 VPBuilder LoopBuilder(FindLastSelect->getDefiningRecipe());
4677 if (
match(FindLastSelect,
4679 SelectCond = LoopBuilder.
createNot(SelectCond);
4686 if (SelectCond !=
Cond || IVOfExpressionToSink) {
4689 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression,
4698 VPIRFlags Flags(MinMaxKind,
false,
false,
4704 NewFindLastSelect, Flags, ExitDL);
4707 VPValue *VectorRegionExitingVal = ReducedIV;
4708 if (IVOfExpressionToSink)
4709 VectorRegionExitingVal =
4711 ReducedIV, IVOfExpressionToSink);
4714 VPValue *StartVPV = PhiR->getStartValue();
4721 NewRdxResult = MiddleBuilder.
createSelect(Cmp, VectorRegionExitingVal,
4731 AnyOfPhi->insertAfter(PhiR);
4738 OrVal, VectorRegionExitingVal, StartVPV, ExitDL);
4751 PhiR->hasUsesOutsideReductionChain());
4752 NewPhiR->insertBefore(PhiR);
4753 PhiR->replaceAllUsesWith(NewPhiR);
4754 PhiR->eraseFromParent();
4761struct ReductionExtend {
4762 Type *SrcType =
nullptr;
4763 ExtendKind Kind = ExtendKind::PR_None;
4769struct ExtendedReductionOperand {
4773 ReductionExtend ExtendA, ExtendB;
4781struct VPPartialReductionChain {
4784 VPWidenRecipe *ReductionBinOp =
nullptr;
4786 ExtendedReductionOperand ExtendedOp;
4793 unsigned AccumulatorOpIdx;
4794 unsigned ScaleFactor;
4797 VPBlendRecipe *Blend =
nullptr;
4802static std::optional<unsigned>
4806 "Expected a non-normalized blend with two incoming values");
4812 return std::nullopt;
4813 return FirstIncomingHasOneUse ? 0 : 1;
4825 if (!
Op->hasOneUse() ||
4831 auto *Trunc = Builder.createWidenCast(Instruction::CastOps::Trunc,
4832 Op->getOperand(1), NarrowTy);
4834 Op->setOperand(1, Builder.createWidenCast(ExtOpc, Trunc, WideTy));
4843 auto *
Sub =
Op->getOperand(0)->getDefiningRecipe();
4845 assert(Ext->getOpcode() ==
4847 "Expected both the LHS and RHS extends to be the same");
4848 bool IsSigned = Ext->getOpcode() == Instruction::SExt;
4851 auto *FreezeX = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
X}));
4852 auto *FreezeY = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
Y}));
4853 auto *
Max = Builder.insert(
4855 {FreezeX, FreezeY}, SrcTy));
4856 auto *Min = Builder.insert(
4858 {FreezeX, FreezeY}, SrcTy));
4859 auto *AbsDiff = Builder.insert(
4862 return Builder.createWidenCast(Instruction::CastOps::ZExt, AbsDiff,
4863 Op->getScalarType());
4875 if (!
Mul->hasOneUse() ||
4876 (Ext->getOpcode() != MulLHS->getOpcode() && MulLHS != MulRHS) ||
4877 MulLHS->getOpcode() != MulRHS->getOpcode())
4880 auto *NewLHS = Builder.createWidenCast(
4881 MulLHS->getOpcode(), MulLHS->getOperand(0), Ext->getScalarType());
4882 auto *NewRHS = MulLHS == MulRHS
4884 : Builder.createWidenCast(MulRHS->getOpcode(),
4885 MulRHS->getOperand(0),
4886 Ext->getScalarType());
4887 auto *NewMul =
Mul->cloneWithOperands({NewLHS, NewRHS});
4888 Builder.insert(NewMul);
4889 Op->replaceAllUsesWith(NewMul);
4890 Op->eraseFromParent();
4891 Mul->eraseFromParent();
4900 VPValue *VecOp = Red->getVecOp();
4954static void transformToPartialReduction(
const VPPartialReductionChain &Chain,
4962 WidenRecipe->
getOperand(1 - Chain.AccumulatorOpIdx));
4965 ExtendedOp = optimizeExtendsForPartialReduction(ExtendedOp);
4981 if ((WidenRecipe->
getOpcode() == Instruction::Sub &&
4983 (WidenRecipe->
getOpcode() == Instruction::FSub &&
4988 if (WidenRecipe->
getOpcode() == Instruction::FSub) {
5000 Builder.insert(NegRecipe);
5001 ExtendedOp = NegRecipe;
5016 std::optional<unsigned> BlendReductionIdx =
5017 getBlendReductionUpdateValueIdx(Chain.Blend);
5018 assert(BlendReductionIdx &&
5020 "Expected blend to contain the reduction update");
5037 assert((!ExitValue || IsLastInChain) &&
5038 "if we found ExitValue, it must match RdxPhi's backedge value");
5049 PartialRed->insertBefore(WidenRecipe);
5059 E->insertBefore(WidenRecipe);
5060 PartialRed->replaceAllUsesWith(
E);
5073 auto *NewScaleFactor = Plan.
getConstantInt(32, Chain.ScaleFactor);
5074 StartInst->setOperand(2, NewScaleFactor);
5082 VPValue *OldStartValue = StartInst->getOperand(0);
5083 StartInst->setOperand(0, StartInst->getOperand(1));
5087 assert(RdxResult &&
"Could not find reduction result");
5090 unsigned SubOpc = Chain.RK ==
RecurKind::FSub ? Instruction::BinaryOps::FSub
5091 : Instruction::BinaryOps::Sub;
5097 [&NewResult](
VPUser &U,
unsigned Idx) {
return &
U != NewResult; });
5103 const VPPartialReductionChain &Link,
5106 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5107 std::optional<unsigned> BinOpc = std::nullopt;
5109 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5110 BinOpc = ExtendedOp.ExtendsUser->
getOpcode();
5112 std::optional<llvm::FastMathFlags>
Flags;
5116 auto GetLinkOpcode = [&Link]() ->
unsigned {
5119 return Instruction::Add;
5121 return Instruction::FAdd;
5123 return Link.ReductionBinOp->
getOpcode();
5128 GetLinkOpcode(), ExtendedOp.ExtendA.SrcType, ExtendedOp.ExtendB.SrcType,
5129 RdxType, VF, ExtendedOp.ExtendA.Kind, ExtendedOp.ExtendB.Kind, BinOpc,
5150static std::optional<ExtendedReductionOperand>
5153 "Op should be operand of UpdateR");
5161 if (
Op->hasOneUse() &&
5170 Type *RHSInputType =
Y->getScalarType();
5171 if (LHSInputType != RHSInputType ||
5172 LHSExt->getOpcode() != RHSExt->getOpcode())
5173 return std::nullopt;
5176 return ExtendedReductionOperand{
5178 {LHSInputType, getPartialReductionExtendKind(LHSExt)},
5182 std::optional<TTI::PartialReductionExtendKind> OuterExtKind;
5185 VPValue *CastSource = CastRecipe->getOperand(0);
5186 OuterExtKind = getPartialReductionExtendKind(CastRecipe);
5196 return ExtendedReductionOperand{
5203 if (!
Op->hasOneUse())
5204 return std::nullopt;
5209 return std::nullopt;
5219 return std::nullopt;
5223 ExtendKind LHSExtendKind = getPartialReductionExtendKind(LHSCast);
5226 const APInt *RHSConst =
nullptr;
5232 return std::nullopt;
5236 if (Cast && OuterExtKind &&
5237 getPartialReductionExtendKind(Cast) != OuterExtKind)
5238 return std::nullopt;
5240 Type *RHSInputType = LHSInputType;
5241 ExtendKind RHSExtendKind = LHSExtendKind;
5244 RHSExtendKind = getPartialReductionExtendKind(RHSCast);
5247 return ExtendedReductionOperand{
5248 MulOp, {LHSInputType, LHSExtendKind}, {RHSInputType, RHSExtendKind}};
5255static std::optional<SmallVector<VPPartialReductionChain>>
5262 return std::nullopt;
5272 VPValue *CurrentValue = ExitValue;
5273 while (CurrentValue != RedPhiR) {
5275 std::optional<unsigned> BlendReductionIdx;
5279 return std::nullopt;
5281 BlendReductionIdx = getBlendReductionUpdateValueIdx(Blend);
5282 if (!BlendReductionIdx)
5283 return std::nullopt;
5290 return std::nullopt;
5297 std::optional<ExtendedReductionOperand> ExtendedOp =
5298 matchExtendedReductionOperand(UpdateR,
Op);
5300 ExtendedOp = matchExtendedReductionOperand(UpdateR, PrevValue);
5302 return std::nullopt;
5310 return std::nullopt;
5312 Type *ExtSrcType = ExtendedOp->ExtendA.SrcType;
5315 return std::nullopt;
5317 VPPartialReductionChain Link(
5318 {UpdateR, *ExtendedOp, RK,
5323 CurrentValue = PrevValue;
5328 std::reverse(Chain.
begin(), Chain.
end());
5345 if (
auto Chains = getScaledReductions(&RedPhiR))
5346 ChainsByPhi.
try_emplace(&RedPhiR, std::move(*Chains));
5351 UnorderedReductions.
push_back(&RedPhiR);
5357 for (
auto *Rdx : UnorderedReductions) {
5373 ? std::make_optional(Rdx->getFastMathFlagsOrNone())
5377 Backedge->getOpcode(), ScalarTy,
nullptr,
5379 std::nullopt, CostCtx.
CostKind, FMF);
5380 return PRCost <= CurrentCost;
5386 Rdx->getRecurrenceKind(), Rdx->getFastMathFlagsOrNone(),
5387 Backedge->getUnderlyingInstr(), Rdx, OtherOp,
nullptr,
5390 Partial->insertBefore(Backedge);
5391 Backedge->replaceAllUsesWith(Partial);
5392 Backedge->eraseFromParent();
5395 if (ChainsByPhi.
empty())
5403 for (
const auto &[
_, Chains] : ChainsByPhi)
5404 for (
const VPPartialReductionChain &Chain : Chains) {
5405 PartialReductionOps.
insert(Chain.ExtendedOp.ExtendsUser);
5407 PartialReductionBlends.
insert(Chain.Blend);
5408 ScaledReductionMap[Chain.ReductionBinOp] = Chain.ScaleFactor;
5414 auto ExtendUsersValid = [&](
VPValue *Ext) {
5416 return PartialReductionOps.contains(cast<VPRecipeBase>(U));
5420 auto IsProfitablePartialReductionChainForVF =
5427 for (
const VPPartialReductionChain &Link : Chain) {
5428 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5429 InstructionCost LinkCost = getPartialReductionLinkCost(CostCtx, Link, VF);
5433 PartialCost += LinkCost;
5434 RegularCost += Link.ReductionBinOp->
computeCost(VF, CostCtx);
5436 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5437 RegularCost += ExtendedOp.ExtendsUser->
computeCost(VF, CostCtx);
5440 RegularCost += Extend->computeCost(VF, CostCtx);
5442 return PartialCost.
isValid() && PartialCost < RegularCost;
5450 for (
auto &[RedPhiR, Chains] : ChainsByPhi) {
5451 for (
const VPPartialReductionChain &Chain : Chains) {
5452 if (!
all_of(Chain.ExtendedOp.ExtendsUser->operands(), ExtendUsersValid)) {
5456 auto UseIsValid = [&, RedPhiR = RedPhiR](
VPUser *U) {
5458 return PhiR == RedPhiR;
5462 return Blend == Chain.Blend || PartialReductionBlends.
contains(Blend);
5464 return Chain.ScaleFactor == ScaledReductionMap.
lookup_or(R, 0) ||
5470 if (!
all_of(Chain.ReductionBinOp->users(), UseIsValid)) {
5479 auto *RepR = dyn_cast<VPReplicateRecipe>(U);
5480 return RepR && RepR->getOpcode() == Instruction::Store;
5491 return IsProfitablePartialReductionChainForVF(Chains, VF);
5497 for (
auto &[Phi, Chains] : ChainsByPhi)
5498 for (
const VPPartialReductionChain &Chain : Chains)
5499 transformToPartialReduction(Chain, Plan, Phi);
5513 if (VPI.getUnderlyingValue() &&
5524 auto ProcessSubset = [&](
VPlan &,
auto ProcessVPInst) {
5527 if (!ProcessVPInst(VPI))
5536 assert(New->getParent() &&
"New recipe must have been inserted");
5537 if (VPI->
getOpcode() == Instruction::Load)
5546 return ReplaceWith(VPI,
VPBuilder(VPI).insert(
5553 "lowerMemoryIdioms", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5555 VPI, FinalRedStoresBuilder))
5564 return ReplaceWith(VPI,
VPBuilder(VPI).insert(Histogram));
5577 "scalarizeMemOpsWithIrregularTypes", ProcessSubset, Plan,
5581 return Scalarize(VPI);
5588 "makeVPlanMemOpDecision", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5590 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5600 const SCEV *PtrSCEV =
5602 bool IsSingleScalarLoad =
5608 I, Ptr, IsSingleScalarLoad,
5617 "widenConsecutiveMemOps", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5619 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5623 std::optional<int64_t> Stride =
5625 if (Stride != 1 && Stride != -1)
5656 return ReplaceWith(VPI,
Load);
5665 auto *StoreR = Builder.createWidenStore(
5668 return ReplaceWith(VPI, StoreR);
5675 return ReplaceWith(VPI, Recipe);
5677 return Scalarize(VPI);
5697 if (VPI.mayHaveSideEffects())
5701 if (VPI.isMasked() && !VPI.isSafeToSpeculativelyExecute())
5706 if (VPI.getOpcode() == Instruction::Add &&
5715 VPI.getOpcode(), VPI.operandsWithoutMask(),
nullptr, VPI,
5716 VPI, VPI.getDebugLoc(),
I);
5717 Recipe->insertBefore(&VPI);
5718 VPI.replaceAllUsesWith(Recipe);
5719 VPI.eraseFromParent();
5729 switch (Param.ParamKind) {
5730 case VFParamKind::Vector:
5731 case VFParamKind::GlobalPredicate:
5733 case VFParamKind::OMP_Uniform:
5734 return SE->isSCEVable(Args[Param.ParamPos]->getScalarType()) &&
5735 SE->isLoopInvariant(
5736 vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5738 case VFParamKind::OMP_Linear:
5739 return match(vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5740 m_scev_AffineAddRec(
5741 m_SCEV(), m_scev_SpecificSInt(Param.LinearStepOrPos),
5742 m_SpecificLoop(L)));
5759 const auto *It =
find_if(Mappings, [&](
const VFInfo &Info) {
5760 return Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()) &&
5763 if (It == Mappings.end())
5770struct CallWideningDecision {
5771 enum class KindTy { Scalarize,
Intrinsic, VectorVariant };
5772 CallWideningDecision(KindTy Kind,
Function *Variant =
nullptr)
5795 return CallWideningDecision::KindTy::Scalarize;
5805 return CallWideningDecision::KindTy::Scalarize;
5809 false, VF, CostCtx);
5824 return CallWideningDecision::KindTy::Intrinsic;
5828 if (VecFunc && ScalarCost >= VecCallCost)
5829 return {CallWideningDecision::KindTy::VectorVariant, VecFunc};
5831 return CallWideningDecision::KindTy::Scalarize;
5841 if (!VPI.getUnderlyingValue() || VPI.getOpcode() != Instruction::Call)
5846 VPI.op_begin() + CI->arg_size());
5848 CallWideningDecision Decision =
5857 switch (Decision.Kind) {
5858 case CallWideningDecision::KindTy::Intrinsic: {
5862 VPI, VPI.getDebugLoc());
5865 case CallWideningDecision::KindTy::VectorVariant: {
5870 Ops.push_back(Mask);
5872 Ops.push_back(VPI.getOperand(VPI.getNumOperandsWithoutMask() - 1));
5877 case CallWideningDecision::KindTy::Scalarize:
5883 VPI.replaceAllUsesWith(Replacement);
5884 VPI.eraseFromParent();
5906 if (!MemR || MemR->isConsecutive())
5909 VPValue *Ptr = MemR->getAddr();
5921 VPValue *StoredValue =
nullptr;
5925 StoredValue = StoreR->getStoredValue();
5927 IntrinID = Intrinsic::experimental_vp_strided_store;
5931 IntrinID = Intrinsic::experimental_vp_strided_load;
5934 Align Alignment = MemR->getAlign();
5937 if (!Ctx.TTI.isLegalStridedLoadStore(VectorTy, Alignment))
5942 IntrinID, VectorTy, MemR->isMasked(), Alignment, Ctx);
5943 return StridedLoadStoreCost < CurrentCost;
5954 Ctx.invalidateWideningDecision(&MemR->getIngredient(), VF);
5959 I32VF = Builder.createScalarZExtOrTrunc(
5973 "Stride type from SCEV must match the index type");
5974 VPValue *CanIV = Builder.createScalarZExtOrTrunc(
5977 auto *
Offset = Builder.createOverflowingOp(
5978 Instruction::Mul, {CanIV, StrideInBytes},
5979 {AddRecPtr->hasNoUnsignedWrap(),
false});
5983 VPValue *BasePtr = Builder.createNoWrapPtrAdd(StartVPV,
Offset, NWFlags);
5986 VPValue *NewPtr = Builder.createVectorPointer(
5990 VPValue *Mask = MemR->getMask();
5995 Ops.push_back(StoredValue);
5996 Ops.append({NewPtr, StrideInBytes, Mask, I32VF});
5998 auto *StridedR = Builder.createWidenMemIntrinsic(
6001 *MemR, R.getDebugLoc());
6004 R.eraseFromParent();
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static bool isEqual(const Function &Caller, const Function &Callee)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static cl::opt< IntrinsicCostStrategy > IntrinsicCost("intrinsic-cost-strategy", cl::desc("Costing strategy for intrinsic instructions"), cl::init(IntrinsicCostStrategy::InstructionCost), cl::values(clEnumValN(IntrinsicCostStrategy::InstructionCost, "instruction-cost", "Use TargetTransformInfo::getInstructionCost"), clEnumValN(IntrinsicCostStrategy::IntrinsicCost, "intrinsic-cost", "Use TargetTransformInfo::getIntrinsicInstrCost"), clEnumValN(IntrinsicCostStrategy::TypeBasedIntrinsicCost, "type-based-intrinsic-cost", "Calculate the intrinsic cost based only on argument types")))
iv Induction Variable Users
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
Legalize the Machine IR a function s Machine IR
This file provides utility analysis objects describing memory locations.
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
This file builds on the ADT/GraphTraits.h file to build a generic graph post order iterator.
const SmallVectorImpl< MachineOperand > & Cond
This is the interface for a metadata-based scoped no-alias analysis.
This file implements a set that has insertion order iteration characteristics.
This file defines the SmallPtrSet class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
This file implements the TypeSwitch template, which mimics a switch() statement whose cases are type ...
This file implements dominator tree analysis for a single level of a VPlan's H-CFG.
This file contains the declarations of different VPlan-related auxiliary helpers.
This file contains the declarations of the Vectorization Plan base classes:
static const X86InstrFMA3Group Groups[]
static const uint32_t IV[8]
Helper for extra no-alias checks via known-safe recipe and SCEV.
SinkStoreInfo(ArrayRef< VPReplicateRecipe * > ExcludeRecipes, VPReplicateRecipe &GroupLeader, PredicatedScalarEvolution &PSE, const Loop &L)
SinkStoreInfo(VPReplicateRecipe &GroupLeader)
bool shouldSkip(VPRecipeBase &R) const
Return true if R should be skipped during alias checking, either because it's in the exclude set or b...
Class for arbitrary precision integers.
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
unsigned getActiveBits() const
Compute the number of active bits in the value.
APInt abs() const
Get the absolute value.
unsigned getBitWidth() const
Return the number of bits in the APInt.
int32_t exactLogBase2() const
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
An arbitrary precision integer that knows its signedness.
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
@ NoAlias
The two locations do not alias at all.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & back() const
Get the last element.
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
const T & front() const
Get the first element.
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
const Function * getParent() const
Return the enclosing method, or null if none.
bool isNoBuiltin() const
Return true if the call should not be treated as a call to a builtin.
This class represents a function call, abstracting a target machine's calling convention.
@ ICMP_ULT
unsigned less than
@ ICMP_ULE
unsigned less or equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
This class represents a range of values.
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
A parsed version of the target data layout string in and methods for querying it.
LLVM_ABI IntegerType * getIndexType(LLVMContext &C, unsigned AddressSpace) const
Returns the type of a GEP index in AddressSpace.
static DebugLoc getUnknown()
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
ValueT lookup_or(const_arg_type_t< KeyT > Val, U &&Default) const
bool dominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
dominates - Returns true iff A dominates B.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
static constexpr ElementCount getScalable(ScalarTy MinVal)
constexpr bool isScalar() const
Exactly one element.
Convenience struct for specifying and reasoning about fast-math flags.
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags noUnsignedWrap()
bool hasNoUnsignedWrap() const
GEPNoWrapFlags withoutNoUnsignedWrap() const
static GEPNoWrapFlags none()
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
A struct for saving information about induction variables.
InductionKind
This enum represents the kinds of inductions that we support.
@ IK_PtrInduction
Pointer induction var. Step = C.
@ IK_IntInduction
Integer induction variable. Step = C.
static InstructionCost getInvalid(CostType Val=0)
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
The group of interleaved loads/stores sharing the same stride and close to each other.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
static bool getDecisionAndClampRange(const std::function< bool(ElementCount)> &Predicate, VFRange &Range)
Test a Predicate on a Range of VF's.
Represents a single loop in the control flow graph.
This class implements a map that also provides access to all stored values in a deterministic order.
ValueT lookup(const KeyT &Key) const
std::pair< iterator, bool > try_emplace(const KeyT &Key, Ts &&...Args)
Representation for a specific memory location.
Function * getFunction(StringRef Name) const
Look up the specified function in the module symbol table.
Post-order traversal of a graph.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
ScalarEvolution * getSE() const
Returns the ScalarEvolution analysis used.
LLVM_ABI const SCEV * getSCEV(Value *V)
Returns the SCEV expression of V, in the context of the current SCEV predicate.
static LLVM_ABI unsigned getOpcode(RecurKind Kind)
Returns the opcode corresponding to the RecurrenceKind.
unsigned getOpcode() const
static bool isFindLastRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
RegionT * getParent() const
Get the parent of the Region.
This class represents a constant integer value.
ConstantInt * getValue() const
static const SCEV * rewrite(const SCEV *Scev, ScalarEvolution &SE, ValueToSCEVMapTy &Map)
This means that we are dealing with an entirely unknown SCEV value, and only represent it as its LLVM...
This class represents an analyzed expression in the program.
Type * getType() const
Return the LLVM type of this SCEV expression.
The main scalar evolution driver.
const DataLayout & getDataLayout() const
Return the DataLayout associated with the module this SCEV instance is operating on.
LLVM_ABI const SCEV * getNegativeSCEV(const SCEV *V, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
Return the SCEV object corresponding to -V.
LLVM_ABI bool isKnownNegative(const SCEV *S)
Test if the given expression is known to be negative.
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
LLVM_ABI const SCEV * getMinusSCEV(SCEVUse LHS, SCEVUse RHS, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap, unsigned Depth=0)
Return LHS-RHS.
ConstantRange getSignedRange(const SCEV *S)
Determine the signed range for a particular SCEV.
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
LLVM_ABI bool isKnownPositive(const SCEV *S)
Test if the given expression is known to be positive.
LLVM_ABI const SCEV * getElementCount(Type *Ty, ElementCount EC, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
ConstantRange getUnsignedRange(const SCEV *S)
Determine the unsigned range for a particular SCEV.
LLVM_ABI bool isKnownPredicate(CmpPredicate Pred, SCEVUse LHS, SCEVUse RHS)
Test if the given expression is known to satisfy the condition described by Pred, LHS,...
static LLVM_ABI AliasResult alias(const MemoryLocation &LocA, const MemoryLocation &LocB)
A vector that has set insertion semantics.
size_type size() const
Determine the number of elements in the SetVector.
bool insert(const value_type &X)
Insert a new element into the SetVector.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Provides information about what library functions are available for the current target.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
This class implements a switch-like dispatch statement for a value of 'T' using dyn_cast functionalit...
TypeSwitch< T, ResultT > & Case(CallableT &&caseFn)
Add a case on the given type.
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isPointerTy() const
True if this is an instance of PointerType.
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntOrPtrTy() const
Return true if this is an integer type or a pointer type.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static SmallVector< VFInfo, 8 > getMappings(const CallInst &CI)
Retrieve all the VFInfo instances associated to the CallInst CI.
bool isLegalMaskedLoadOrStore(bool IsLoad, Type *ScalarTy, Align Alignment, unsigned AddressSpace) const
Returns true if the target machine supports a masked load (if IsLoad) or masked store of scalar type ...
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
void appendRecipe(VPRecipeBase *Recipe)
Augment the existing recipes of a VPBasicBlock with an additional Recipe as the last recipe.
iterator begin()
Recipe iterator methods.
iterator_range< iterator > phis()
Returns an iterator range over the PHI-like recipes in the block.
iterator getFirstNonPhi()
Return the position of the first non-phi node recipe in the block.
VPBasicBlock * splitAt(iterator SplitAt)
Split current block at SplitAt by inserting a new block between the current block and its successors ...
const VPRecipeBase & front() const
VPRecipeBase * getTerminator()
If the block has multiple successors, return the branch recipe terminating the block.
const VPRecipeBase & back() const
A recipe for vectorizing a phi-node as a sequence of mask-based select instructions.
VPValue * getIncomingValue(unsigned Idx) const
Return incoming value number Idx.
VPValue * getMask(unsigned Idx) const
Return mask number Idx.
unsigned getNumIncomingValues() const
Return the number of incoming values, taking into account when normalized the first incoming value wi...
void setMask(unsigned Idx, VPValue *V)
Set mask number Idx to V.
bool isNormalized() const
A normalized blend is one that has an odd number of operands, whereby the first operand does not have...
VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
void setSuccessors(ArrayRef< VPBlockBase * > NewSuccs)
Set each VPBasicBlock in NewSuccss as successor of this VPBlockBase.
VPRegionBlock * getParent()
const VPBasicBlock * getExitingBasicBlock() const
size_t getNumSuccessors() const
void setPredecessors(ArrayRef< VPBlockBase * > NewPreds)
Set each VPBasicBlock in NewPreds as predecessor of this VPBlockBase.
const VPBlocksTy & getPredecessors() const
VPBlockBase * getSinglePredecessor() const
const VPBasicBlock * getEntryBasicBlock() const
VPBlockBase * getSingleSuccessor() const
const VPBlocksTy & getSuccessors() const
static auto blocksAs(T &&Range)
Return an iterator range over Range with each block cast to BlockTy.
static void insertOnEdge(VPBlockBase *From, VPBlockBase *To, VPBlockBase *BlockPtr)
Inserts BlockPtr on the edge between From and To.
static bool isLatch(const VPBlockBase *VPB, const VPDominatorTree &VPDT)
Returns true if VPB is a loop latch, using isHeader().
static VPBasicBlock * getPlainCFGMiddleBlock(const VPlan &Plan)
Returns the middle block of Plan in plain CFG form (before regions are formed).
static void insertTwoBlocksAfter(VPBlockBase *IfTrue, VPBlockBase *IfFalse, VPBlockBase *BlockPtr)
Insert disconnected VPBlockBases IfTrue and IfFalse after BlockPtr.
static void connectBlocks(VPBlockBase *From, VPBlockBase *To, unsigned PredIdx=-1u, unsigned SuccIdx=-1u)
Connect VPBlockBases From and To bi-directionally.
static void disconnectBlocks(VPBlockBase *From, VPBlockBase *To)
Disconnect VPBlockBases From and To bi-directionally.
static auto blocksOnly(T &&Range)
Return an iterator range over Range which only includes BlockTy blocks.
static std::pair< VPBasicBlock *, VPBasicBlock * > getPlainCFGHeaderAndLatch(const VPlan &Plan)
Returns the header and latch of the outermost loop of Plan in plain CFG form (before regions are form...
static void transferSuccessors(VPBlockBase *Old, VPBlockBase *New)
Transfer successors from Old to New. New must have no successors.
static SmallVector< VPBasicBlock * > blocksInSingleSuccessorChainBetween(VPBasicBlock *FirstBB, VPBasicBlock *LastBB)
Returns the blocks between FirstBB and LastBB, where FirstBB to LastBB forms a single-sucessor chain.
A recipe for generating conditional branches on the bits of a mask.
VPlan-based builder utility analogous to IRBuilder.
VPInstruction * createFirstActiveLane(ArrayRef< VPValue * > Masks, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenStoreRecipe * createWidenStore(StoreInst &Store, VPValue *Addr, VPValue *StoredVal, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Store, storing StoredVal to Addr with Mask (may be null).
VPInstruction * createAdd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", VPRecipeWithIRFlags::WrapFlagsTy WrapFlags={false, false})
VPInstruction * createOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createLogicalOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenLoadRecipe * createWidenLoad(LoadInst &Load, VPValue *Addr, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Load, loading from Addr with Mask (may be null).
VPInstruction * createNot(VPValue *Operand, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createAnyOfReduction(VPValue *ChainOp, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown())
Create an AnyOf reduction pattern: or-reduce ChainOp, freeze the result, then select between TrueVal ...
void setInsertPoint(const VPInsertPoint &IP)
Set the current insert point.
VPInstruction * createLogicalAnd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createScalarCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy, DebugLoc DL, std::optional< VPIRFlags > Flags=std::nullopt, const VPIRMetadata &Metadata={})
VPInstruction * createFreeze(VPValue *Op, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPValue * createScalarZExtOrTrunc(VPValue *Op, Type *ResultTy, DebugLoc DL)
static VPBuilder getToInsertAfter(VPRecipeBase *R)
Create a VPBuilder to insert after R.
VPDerivedIVRecipe * createDerivedIV(InductionDescriptor::InductionKind Kind, FPMathOperator *FPBinOp, VPValue *Start, VPValue *Current, VPValue *Step, const VPIRFlags::WrapFlagsTy &Flags={})
Convert Current to Start + Current * Step.
VPWidenCastRecipe * createWidenCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy)
VPInstruction * createICmp(CmpInst::Predicate Pred, VPValue *A, VPValue *B, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
Create a new ICmp VPInstruction with predicate Pred and operands A and B.
VPInstruction * createSelect(VPValue *Cond, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", std::optional< VPIRFlags > Flags=std::nullopt)
Create a select of TrueVal and FalseVal based on Cond, using the default flags for the result type,...
VPInstruction * createNaryOp(unsigned Opcode, ArrayRef< VPValue * > Operands, Instruction *Inst=nullptr, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
Create an N-ary operation with Opcode, Operands and set Inst as its underlying Instruction.
static VPSingleDefRecipe * createSingleScalarOp(unsigned Opcode, ArrayRef< VPValue * > Operands, VPValue *Mask, const VPIRFlags &Flags, const VPIRMetadata &Metadata, DebugLoc DL, Instruction *UV)
Create a single-scalar recipe with Opcode and Operands without inserting it.
unsigned getNumDefinedValues() const
Returns the number of values defined by the VPDef.
VPValue * getVPSingleValue()
Returns the only VPValue defined by the VPDef.
VPValue * getVPValue(unsigned I)
Returns the VPValue with index I defined by the VPDef.
ArrayRef< VPRecipeValue * > definedValues()
Returns an ArrayRef of the values defined by the VPDef.
Template specialization of the standard LLVM dominator tree utility for VPBlockBases.
bool properlyDominates(const VPRecipeBase *A, const VPRecipeBase *B) const
Recipe to expand a SCEV expression.
A recipe to combine multiple recipes into a single 'expression' recipe, which should be considered a ...
A recipe representing a sequence of load -> update -> store as part of a histogram operation.
A special type of VPBasicBlock that wraps an existing IR basic block.
Class to record and manage LLVM IR flags.
static VPIRFlags getDefaultFlags(unsigned Opcode, Type *ResultTy=nullptr)
Returns default flags for Opcode and scalar ResultTy for opcodes that support it, asserts otherwise.
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
This is a concrete Recipe that models a single VPlan-level instruction.
unsigned getNumOperandsWithoutMask() const
Returns the number of operands, excluding the mask if the VPInstruction is masked.
@ ExtractLane
Extracts a single lane (first operand) from a set of vector operands.
@ ExtractPenultimateElement
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
@ BuildVector
Creates a fixed-width vector containing all operands.
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
unsigned getOpcode() const
VPValue * getMask() const
Returns the mask for the VPInstruction.
const InterleaveGroup< Instruction > * getInterleaveGroup() const
VPValue * getMask() const
Return the mask used by this recipe.
ArrayRef< VPValue * > getStoredValues() const
Return the VPValues stored by this interleave group.
VPInterleaveRecipe is a recipe for transforming an interleave group of load or stores into one wide l...
VPPredInstPHIRecipe is a recipe for generating the phi nodes needed when control converges back from ...
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
VPRegionBlock * getRegion()
VPBasicBlock * getParent()
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
void insertAfter(VPRecipeBase *InsertPos)
Insert an unlinked Recipe into a basic block immediately after the specified Recipe.
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
Helper class to create VPRecipies from IR instructions.
VPHistogramRecipe * widenIfHistogram(VPInstruction *VPI)
If VPI represents a histogram operation (as determined by LoopVectorizationLegality) make that safe f...
bool prefersVectorizedAddressing() const
Returns true if the target prefers vectorized addressing.
VPRecipeBase * tryToWidenMemory(VPInstruction *VPI, VFRange &Range)
Check if the load or store instruction VPI should widened for Range.Start and potentially masked.
bool replaceWithFinalIfReductionStore(VPInstruction *VPI, VPBuilder &FinalRedStoresBuilder)
If VPI is a store of a reduction into an invariant address, delete it.
VPSingleDefRecipe * handleReplication(VPInstruction *VPI, VFRange &Range)
Build a replicating or single-scalar recipe for VPI.
bool isPredicatedInst(Instruction *I) const
Returns true if I needs to be predicated (i.e.
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
A recipe for handling reduction phis.
bool isOrdered() const
Returns true, if the phi is part of an ordered reduction.
void setVFScaleFactor(unsigned ScaleFactor)
Set the VFScaleFactor for this reduction phi.
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
bool isInLoop() const
Returns true if the phi is part of an in-loop reduction.
RecurKind getRecurrenceKind() const
Returns the recurrence kind of the reduction.
A recipe to represent inloop, ordered or partial reduction operations.
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
const VPBlockBase * getEntry() const
bool isReplicator() const
An indicator whether this region is to generate multiple replicated instances of output IR correspond...
void setExiting(VPBlockBase *ExitingBlock)
Set ExitingBlock as the exiting VPBlockBase of this VPRegionBlock.
Type * getCanonicalIVType() const
Return the type of the canonical IV for loop regions.
VPRegionValue * getCanonicalIV()
Return the canonical induction variable of the region, null for replicating regions.
const VPBlockBase * getExiting() const
VPRegionValue * getHeaderMask() const
Return the header mask of the region, or null if not set.
VPReplicateRecipe replicates a given instruction producing multiple scalar copies of the original sca...
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
static InstructionCost computeCallCost(Function *CalledFn, Type *ResultTy, ArrayRef< const VPValue * > ArgOps, bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx)
Return the cost of scalarizing a call to CalledFn with argument operands ArgOps for a given VF.
operand_range operandsWithoutMask()
Return the recipe's operands, excluding the mask of a predicated recipe.
bool isPredicated() const
VPValue * getMask()
Return the mask of a predicated VPReplicateRecipe.
Lightweight SCEV-to-VPlan expander.
VPValue * expand(const SCEV *S)
Expand S into recipes and live-ins using the builder.
A recipe for handling phi nodes of integer and floating-point inductions, producing their scalar valu...
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
VPSingleDefRecipe * clone() override=0
Clone the current recipe.
A symbolic live-in VPValue, used for values like vector trip count, VF, and VFxUF.
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
void setOperand(unsigned I, VPValue *New)
unsigned getNumOperands() const
VPValue * getOperand(unsigned N) const
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
bool isDefinedOutsideLoopRegions() const
Returns true if the VPValue is defined outside any loop.
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
bool hasMoreThanOneUniqueUser() const
Returns true if the value has more than one unique user.
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
VPUser * getSingleUser()
Return the single user of this value, or nullptr if there is not exactly one user.
void replaceAllUsesWith(VPValue *New)
void replaceUsesWithIf(VPValue *New, llvm::function_ref< bool(VPUser &U, unsigned Idx)> ShouldReplace)
Go through the uses list for this VPValue and make each use point to New if the callback ShouldReplac...
A recipe to compute a pointer to the last element of each part of a widened memory access for widened...
A recipe for widening Call instructions using library calls.
static InstructionCost computeCallCost(Function *Variant, VPCostContext &Ctx)
Return the cost of widening a call using the vector function Variant.
VPWidenCastRecipe is a recipe to create vector cast instructions.
Instruction::CastOps getOpcode() const
A recipe for handling GEP instructions.
Base class for widened induction (VPWidenIntOrFpInductionRecipe and VPWidenPointerInductionRecipe),...
PHINode * getPHINode() const
Returns the underlying PHINode if one exists, or null otherwise.
VPValue * getStepValue()
Returns the step value of the induction.
const InductionDescriptor & getInductionDescriptor() const
Returns the induction descriptor for the recipe.
A recipe for handling phi nodes of integer and floating-point inductions, producing their vector valu...
TruncInst * getTruncInst()
Returns the first defined value as TruncInst, if it is one or nullptr otherwise.
A recipe for widening vector intrinsics.
static InstructionCost computeCallCost(Intrinsic::ID ID, ArrayRef< const VPValue * > Operands, const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx)
Compute the cost of a vector intrinsic with ID and Operands.
static InstructionCost computeMemIntrinsicCost(Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment, VPCostContext &Ctx)
Helper function for computing the cost of vector memory intrinsic.
A common mixin class for widening memory operations.
virtual VPRecipeBase * getAsRecipe()=0
Return a VPRecipeBase* to the current object.
A recipe for widened phis.
VPWidenRecipe is a recipe for producing a widened instruction using the opcode and operands of the re...
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenRecipe.
VPWidenRecipe * clone() override
Clone the current recipe.
unsigned getOpcode() const
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
VPIRValue * getLiveIn(Value *V) const
Return the live-in VPIRValue for V, if there is one or nullptr otherwise.
bool hasVF(ElementCount VF) const
const DataLayout & getDataLayout() const
LLVMContext & getContext() const
VPBasicBlock * getEntry()
bool hasScalableVF() const
VPValue * getTripCount() const
The trip count of the original loop.
VPValue * getOrCreateBackedgeTakenCount()
The backedge taken count of the original loop.
iterator_range< SmallSetVector< ElementCount, 2 >::iterator > vectorFactors() const
Returns an iterator range over all VFs of the plan.
VPIRValue * getFalse()
Return a VPIRValue wrapping i1 false.
VPSymbolicValue & getVFxUF()
Returns VF * UF of the vector loop region.
VPIRValue * getAllOnesValue(Type *Ty)
Return a VPIRValue wrapping the AllOnes value of type Ty.
VPRegionBlock * createReplicateRegion(VPBlockBase *Entry, VPBlockBase *Exiting, const std::string &Name="")
Create a new replicate region with Entry, Exiting and Name.
auto getLiveIns() const
Return the list of live-in VPValues available in the VPlan.
bool hasUF(unsigned UF) const
ArrayRef< VPIRBasicBlock * > getExitBlocks() const
Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of the original scalar loop.
VPSymbolicValue & getVectorTripCount()
The vector trip count.
VPValue * getBackedgeTakenCount() const
VPIRValue * getOrAddLiveIn(Value *V)
Gets the live-in VPIRValue for V or adds a new live-in (if none exists yet) for V.
VPIRValue * getZero(Type *Ty)
Return a VPIRValue wrapping the null value of type Ty.
void setVF(ElementCount VF)
bool isUnrolled() const
Returns true if the VPlan already has been unrolled, i.e.
LLVM_ABI_FOR_TEST VPRegionBlock * getVectorLoopRegion()
Returns the VPRegionBlock of the vector loop.
unsigned getConcreteUF() const
Returns the concrete UF of the plan, after unrolling.
void resetTripCount(VPValue *NewTripCount)
Resets the trip count for the VPlan.
VPBasicBlock * getMiddleBlock()
Returns the 'middle' block of the plan, that is the block that selects whether to execute the scalar ...
VPBasicBlock * createVPBasicBlock(const Twine &Name, VPRecipeBase *Recipe=nullptr)
Create a new VPBasicBlock with Name and containing Recipe if present.
VPIRValue * getTrue()
Return a VPIRValue wrapping i1 true.
VPBasicBlock * getVectorPreheader() const
Returns the preheader of the vector loop region, if one exists, or null otherwise.
VPSymbolicValue & getUF()
Returns the UF of the vector loop region.
bool hasScalarVFOnly() const
VPBasicBlock * getScalarPreheader() const
Return the VPBasicBlock for the preheader of the scalar loop.
bool hasTailFolded() const
Returns true if the vector loop region is tail-folded.
VPSymbolicValue & getVF()
Returns the VF of the vector loop region.
LLVM_ABI_FOR_TEST VPlan * duplicate()
Clone the current VPlan, update all VPValues of the new VPlan and cloned recipes to refer to the clon...
VPIRValue * getConstantInt(Type *Ty, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given type and value.
LLVM Value Representation.
iterator_range< user_iterator > users()
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
constexpr bool hasKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns true if there exists a value X where RHS*X will result in a value whose quantity matches our ...
constexpr ScalarTy getFixedValue() const
constexpr ScalarTy getKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns a value X where RHS*X will result in a value whose quantity matches our own.
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
An efficient, type-erasing, non-owning reference to a callable.
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
std::variant< std::monostate, Loc::Single, Loc::Multi, Loc::MMI, Loc::EntryValue > Variant
Alias for the std::variant specialization base class of DbgVariable.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
AllOnesConstantMatch m_AllOnes()
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_unless< Pattern > m_Unless(const Pattern &P)
Match if the inner matcher does NOT match.
match_isa< To... > m_Isa()
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
auto m_Cmp()
Matches any compare instruction and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::AShr > m_AShr(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::URem > m_URem(const LHS &L, const RHS &R)
OneOps_match< OpTy, Instruction::Freeze > m_Freeze(const OpTy &Op)
Matches FreezeInst.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
SpecificCmpClass_match< LHS, RHS, CmpInst > m_SpecificCmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::UDiv > m_UDiv(const LHS &L, const RHS &R)
SelectLike_match< CondTy, LTy, RTy > m_SelectLike(const CondTy &C, const LTy &TrueC, const RTy &FalseC)
Matches a value that behaves like a boolean-controlled select, i.e.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
CastOperator_match< OpTy, Instruction::BitCast > m_BitCast(const OpTy &Op)
Matches BitCast.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
BinaryOp_match< LHS, RHS, Instruction::LShr > m_LShr(const LHS &L, const RHS &R)
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinaryOp_match< LHS, RHS, Instruction::FAdd, true > m_c_FAdd(const LHS &L, const RHS &R)
Matches FAdd with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And, true > m_c_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Sub > m_Sub(const LHS &L, const RHS &R)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
bind_cst_ty m_scev_APInt(const APInt *&C)
Match an SCEV constant and bind it to an APInt.
specificloop_ty m_SpecificLoop(const Loop *L)
bool match(const SCEV *S, const Pattern &P)
SCEVAffineAddRec_match< Op0_t, Op1_t, match_isa< const Loop > > m_scev_AffineAddRec(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > > m_ExtractLastLaneOfLastPart(const Op0_t &Op0)
AllRecipe_commutative_match< Instruction::And, Op0_t, Op1_t > m_c_BinaryAnd(const Op0_t &Op0, const Op1_t &Op1)
Match a binary AND operation.
AllRecipe_match< Instruction::Or, Op0_t, Op1_t > m_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
Match a binary OR operation.
VPInstruction_match< VPInstruction::AnyOf > m_AnyOf()
AllRecipe_commutative_match< Instruction::Or, Op0_t, Op1_t > m_c_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ComputeReductionResult, Op0_t > m_ComputeReductionResult(const Op0_t &Op0)
auto m_WidenAnyExtend(const Op0_t &Op0)
match_bind< VPIRValue > m_VPIRValue(VPIRValue *&V)
Match a VPIRValue.
VPInstruction_match< VPInstruction::WideActiveLaneMask, Op0_t, Op1_t, Op2_t > m_WideActiveLaneMask(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
auto m_VPPhi(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::BranchOnTwoConds > m_BranchOnTwoConds()
AllRecipe_match< Opcode, Op0_t, Op1_t > m_Binary(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::LastActiveLane, Op0_t > m_LastActiveLane(const Op0_t &Op0)
auto m_WidenIntrinsic(const T &...Ops)
canonical_widen_iv_match m_CanonicalWidenIV()
VPInstruction_match< VPInstruction::ExitingIVValue, Op0_t > m_ExitingIVValue(const Op0_t &Op0)
VPInstruction_match< Instruction::ExtractElement, Op0_t, Op1_t > m_ExtractElement(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, Op0_t > m_ExtractLastLane(const Op0_t &Op0)
int_pred_ty< is_zero_int, 1 > m_False()
match_bind< VPSingleDefRecipe > m_VPSingleDefRecipe(VPSingleDefRecipe *&V)
Match a VPSingleDefRecipe, capturing if we match.
VPInstruction_match< VPInstruction::BranchOnCount > m_BranchOnCount()
auto m_GetElementPtr(const Op0_t &Op0, const Op1_t &Op1)
auto m_VPValue()
Match an arbitrary VPValue and ignore it.
VPInstruction_match< VPInstruction::ExtractVectorForPart, Op0_t, Op1_t > m_ExtractVectorForPart(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > m_ExtractLastPart(const Op0_t &Op0)
VPRecipeBase * findUserOf(VPValue *V, const MatchT &P)
If V is used by a recipe matching pattern P, return it.
VPInstruction_match< VPInstruction::Broadcast, Op0_t > m_Broadcast(const Op0_t &Op0)
header_mask_match m_HeaderMask()
VPInstruction_match< VPInstruction::BuildVector > m_BuildVector()
BuildVector is matches only its opcode, w/o matching its operands as the number of operands is not fi...
VPInstruction_match< VPInstruction::ExtractPenultimateElement, Op0_t > m_ExtractPenultimateElement(const Op0_t &Op0)
match_bind< VPInstruction > m_VPInstruction(VPInstruction *&V)
Match a VPInstruction, capturing if we match.
VPInstruction_match< VPInstruction::FirstActiveLane, Op0_t > m_FirstActiveLane(const Op0_t &Op0)
int_pred_ty< is_one, 1 > m_True()
auto m_DerivedIV(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
VPInstruction_match< VPInstruction::BranchOnCond > m_BranchOnCond()
VPInstruction_match< VPInstruction::ExtractLane, Op0_t, Op1_t > m_ExtractLane(const Op0_t &Op0, const Op1_t &Op1)
auto m_AnyNeg(const Op0_t &Op0)
VPInstruction_match< VPInstruction::Reverse, Op0_t > m_Reverse(const Op0_t &Op0)
initializer< Ty > init(const Ty &Val)
NodeAddr< DefNode * > Def
bool isSingleScalar(const VPValue *VPV)
Returns true if VPV is a single scalar, either because it produces the same value for all lanes or on...
VPValue * getOrCreateVPValueForSCEVExpr(VPlan &Plan, const SCEV *Expr)
Get or create a VPValue that corresponds to the expansion of Expr.
bool cannotHoistOrSinkRecipe(const VPRecipeBase &R, bool Sinking=false)
Return true if we do not know how to (mechanically) hoist or sink R.
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
std::optional< int64_t > getConstantStride(VPValue *Addr, Type *AccessTy, PredicatedScalarEvolution &PSE, const Loop *L)
If the pointer operand Addr of a memory access is an affine AddRec w.r.t.
VPInstruction * findComputeReductionResult(VPReductionPHIRecipe *PhiR)
Find the ComputeReductionResult recipe for PhiR, looking through selects inserted for predicated redu...
VPInstruction * findCanonicalIVIncrement(VPlan &Plan)
Find the canonical IV increment of Plan's vector loop region.
std::optional< MemoryLocation > getMemoryLocation(const VPRecipeBase &R)
Return a MemoryLocation for R with noalias metadata populated from R, if the recipe is supported and ...
bool onlyFirstLaneUsed(const VPValue *Def)
Returns true if only the first lane of Def is used.
VPIRValue * tryToFoldLiveIns(VPSingleDefRecipe &R, ArrayRef< VPValue * > Operands, const DataLayout &DL)
Try to fold R using InstSimplifyFolder.
SmallVector< std::pair< VPBasicBlock *, VPIRBasicBlock * > > getEarlyExits(const VPlan &Plan, const VPBlockBase *MiddleVPBB)
Returns the (early exiting block, exit block) pairs of Plan, i.e.
void recursivelyDeleteDeadRecipes(VPValue *V)
Recursively delete V and any of its operands that become dead.
bool doesGeneratePerAllLanes(const VPRecipeBase *R)
Returns true if R produces scalar values for all VF lanes.
bool isDeadRecipe(VPRecipeBase &R)
Returns true if R is dead, i.e.
VPRecipeBase * findRecipe(VPValue *Start, PredT Pred)
Search Start's users for a recipe satisfying Pred, looking through recipes with definitions.
bool isUniformAcrossVFsAndUFs(const VPValue *V)
Checks if V is uniform across all VF lanes and UF parts.
bool isUsedByLoadStoreAddress(const VPValue *V)
Returns true if V is used as part of the address of another load or store.
std::optional< std::pair< bool, unsigned > > getOpcodeOrIntrinsicID(const VPValue *V)
Get the instruction opcode or intrinsic ID for the recipe defining V.
VPValue * scalarizeVPWidenPointerInduction(VPWidenPointerInductionRecipe *PtrIV, VPlan &Plan, VPBuilder &Builder)
Scalarize a VPWidenPointerInductionRecipe by replacing it with a PtrAdd (IndStart,...
const SCEV * getSCEVExprForVPValue(const VPValue *V, PredicatedScalarEvolution &PSE, const Loop *L=nullptr)
Return the SCEV expression for V.
void pullOutPermutations(VPlan &Plan, Match_t Perm, Builder Build)
Removes the permutation pattern Perm from any elementwise operations in the plan, by constructing a n...
SmallVector< VPUser * > collectUsersRecursively(VPValue *V)
Collect all users of V, looking through recipes that define other values.
VPScalarIVStepsRecipe * createScalarIVSteps(VPlan &Plan, InductionDescriptor::InductionKind Kind, Instruction::BinaryOps InductionOpcode, FPMathOperator *FPBinOp, Instruction *TruncI, VPValue *StartV, VPValue *Step, DebugLoc DL, VPBuilder &Builder, const VPIRFlags::WrapFlagsTy &Flags={})
Create a scalar-iv-steps recipe over Plan's canonical IV for an induction of Kind with InductionOpcod...
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
SmallVector< VPBasicBlock * > vp_rpo_plain_cfg_loop_body(VPBasicBlock *Header)
Returns the VPBasicBlocks forming the loop body of a plain (pre-region) VPlan in reverse post-order s...
void stable_sort(R &&Range)
auto min_element(R &&Range)
Provide wrappers to std::min_element which take ranges instead of having to pass begin/end explicitly...
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
unsigned getLoadStoreAddressSpace(const Value *I)
A helper function that returns the address space of the pointer operand of load or store instruction.
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
LLVM_ABI Intrinsic::ID getVectorIntrinsicIDForCall(const CallInst *CI, const TargetLibraryInfo *TLI)
Returns intrinsic ID for call.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
ReductionStyle getReductionStyle(bool InLoop, bool Ordered, unsigned ScaleFactor)
DenseMap< const Value *, const SCEV * > ValueToSCEVMapTy
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
constexpr from_range_t from_range
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
auto cast_or_null(const Y &Val)
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
iterator_range< df_iterator< VPBlockShallowTraversalWrapper< VPBlockBase * > > > vp_depth_first_shallow(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order.
constexpr auto bind_back(FnT &&Fn, BindArgsT &&...BindArgs)
C++23 bind_back.
bool isa_and_nonnull(const Y &Val)
iterator_range< df_iterator< VPBlockDeepTraversalWrapper< VPBlockBase * > > > vp_depth_first_deep(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order while traversing t...
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
bool operator==(const AddressRangeValuePair &LHS, const AddressRangeValuePair &RHS)
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
auto make_isa_range(RangeT &&Range)
Return a range over Range containing only elements for which isa<T> holds, casting each of them to T.
auto dyn_cast_or_null(const Y &Val)
void erase(Container &C, ValueType V)
Wrapper function to remove a value from a container:
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
auto reverse(ContainerTy &&C)
constexpr size_t range_size(R &&Range)
Returns the size of the Range, i.e., the number of elements.
void sort(IteratorTy Start, IteratorTy End)
DenseMap< Value *, const SCEVUnknown * > SymbolicStrideMap
Maps a pointer to its symbolic (non-constant) stride.
bool hasIrregularType(Type *Ty, const DataLayout &DL)
A helper function that returns true if the given type is irregular.
UncountableExitStyle
Different methods of handling early exits.
@ ReadOnly
No side effects to worry about, so we can process any uncountable exits in the loop and branch either...
@ MaskedHandleExitInScalarLoop
All memory operations other than the load(s) required to determine whether an uncountable exit occurr...
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
iterator_range< filter_iterator< detail::IterOfRange< RangeT >, PredicateT > > make_filter_range(RangeT &&Range, PredicateT Pred)
Convenience function that takes a range of elements and a predicate, and return a new filter_iterator...
bool canConstantBeExtended(const APInt *C, Type *NarrowType, TTI::PartialReductionExtendKind ExtKind)
Check if a constant CI can be safely treated as having been extended from a narrower type with the gi...
T * find_singleton(R &&Range, Predicate P, bool AllowRepeats=false)
Return the single value in Range that satisfies P(<member of Range> *, AllowRepeats)->T * returning n...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
RecurKind
These are the kinds of recurrences that we support.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ FindIV
FindIV reduction with select(icmp(),x,y) where one of (x,y) is a loop induction variable (increasing ...
@ Or
Bitwise or logical OR of integers.
@ Mul
Product of integers.
@ FSub
Subtraction of floats.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
LLVM_ABI Value * getRecurrenceIdentity(RecurKind K, Type *Tp, FastMathFlags FMF)
Given information about an recurrence kind, return the identity for the @llvm.vector....
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
DWARFExpression::Operation Op
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
ArrayRef(const T &OneElt) -> ArrayRef< T >
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
hash_code hash_combine(const Ts &...args)
Combine values into a single hash_code.
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
LLVM_ABI bool isDereferenceableAndAlignedInLoop(LoadInst *LI, Loop *L, ScalarEvolution &SE, DominatorTree &DT, AssumptionCache *AC=nullptr, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Return true if we can prove that the given load (which is assumed to be within the specified loop) wo...
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
hash_code hash_combine_range(InputIteratorT first, InputIteratorT last)
Compute a hash_code for a sequence of values.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
VPBasicBlock * EarlyExitingVPBB
VPIRBasicBlock * EarlyExitVPBB
This struct is a compact representation of a valid (non-zero power of two) alignment.
An information struct used to provide DenseMap with the various necessary components for a given valu...
This reduction is unordered with the partial result scaled down by some factor.
Holds the VFShape for a specific scalar to vector function mapping.
Encapsulates information needed to describe a parameter.
A range of powers-of-2 vectorization factors with fixed start and adjustable end.
Struct to hold various analysis needed for cost computations.
const VFSelectionContext & Config
static bool isFreeScalarIntrinsic(Intrinsic::ID ID)
Returns true if ID is a pseudo intrinsic that is dropped via scalarization rather than widened.
bool isMaskRequired(Instruction *I) const
Forwards to LoopVectorizationCostModel::isMaskRequired.
PredicatedScalarEvolution & PSE
bool willBeScalarized(Instruction *I, ElementCount VF) const
Returns true if I is known to be scalarized at VF.
TargetTransformInfo::TargetCostKind CostKind
const TargetLibraryInfo & TLI
const TargetTransformInfo & TTI
A recipe for handling first-order recurrence phis.
A VPValue representing a live-in from the input IR or a constant.
Type * getType() const
Returns the type of the underlying IR value.
A recipe for widening load operations, using the address to load from and an optional mask.
A recipe for widening store operations, using the stored value, the address to store to and an option...