102#define DEBUG_TYPE "sroa"
104STATISTIC(NumAllocasAnalyzed,
"Number of allocas analyzed for replacement");
105STATISTIC(NumAllocaPartitions,
"Number of alloca partitions formed");
106STATISTIC(MaxPartitionsPerAlloca,
"Maximum number of partitions per alloca");
107STATISTIC(NumAllocaPartitionUses,
"Number of alloca partition uses rewritten");
108STATISTIC(MaxUsesPerAllocaPartition,
"Maximum number of uses of a partition");
109STATISTIC(NumNewAllocas,
"Number of new, smaller allocas introduced");
110STATISTIC(NumPromoted,
"Number of allocas promoted to SSA values");
111STATISTIC(NumLoadsSpeculated,
"Number of loads speculated to allow promotion");
113 "Number of loads rewritten into predicated loads to allow promotion");
116 "Number of stores rewritten into predicated loads to allow promotion");
118STATISTIC(NumVectorized,
"Number of vectorized aggregates");
129class AllocaSliceRewriter;
133class SelectHandSpeculativity {
134 unsigned char Storage = 0;
138 SelectHandSpeculativity() =
default;
139 SelectHandSpeculativity &setAsSpeculatable(
bool isTrueVal);
140 bool isSpeculatable(
bool isTrueVal)
const;
141 bool areAllSpeculatable()
const;
142 bool areAnySpeculatable()
const;
143 bool areNoneSpeculatable()
const;
145 explicit operator intptr_t()
const {
return static_cast<intptr_t
>(Storage); }
146 explicit SelectHandSpeculativity(intptr_t Storage_) : Storage(Storage_) {}
148static_assert(
sizeof(SelectHandSpeculativity) ==
sizeof(
unsigned char));
150using PossiblySpeculatableLoad =
153using RewriteableMemOp =
154 std::variant<PossiblySpeculatableLoad, UnspeculatableStore>;
176 LLVMContext *
const C;
177 DomTreeUpdater *
const DTU;
178 AssumptionCache *
const AC;
179 const bool PreserveCFG;
180 const bool AggregateToVector;
189 SmallSetVector<AllocaInst *, 16> Worklist;
204 SmallSetVector<AllocaInst *, 16> PostPromotionWorklist;
207 SetVector<AllocaInst *, SmallVector<AllocaInst *>,
208 SmallPtrSet<AllocaInst *, 16>, 16>
216 SmallSetVector<PHINode *, 8> SpeculatablePHIs;
220 SmallMapVector<SelectInst *, RewriteableMemOps, 8> SelectsToRewrite;
238 static std::optional<RewriteableMemOps>
239 isSafeSelectToSpeculate(SelectInst &SI,
bool PreserveCFG);
242 SROA(LLVMContext *C, DomTreeUpdater *DTU, AssumptionCache *AC,
244 : C(C), DTU(DTU), AC(AC),
245 PreserveCFG(
Options.
CFG == SROAOptions::PreserveCFG),
246 AggregateToVector(
Options.AggregateToVector) {}
249 std::pair<
bool ,
bool > runSROA(Function &
F);
252 friend class AllocaSliceRewriter;
254 bool presplitLoadsAndStores(AllocaInst &AI, AllocaSlices &AS);
255 std::pair<AllocaInst *, uint64_t>
256 rewritePartition(AllocaInst &AI, AllocaSlices &AS, Partition &
P);
257 bool splitAlloca(AllocaInst &AI, AllocaSlices &AS);
258 bool propagateStoredValuesToLoads(AllocaInst &AI, AllocaSlices &AS);
259 std::pair<
bool ,
bool > runOnAlloca(AllocaInst &AI);
260 void clobberUse(Use &U);
261 bool deleteDeadInstructions(SmallPtrSetImpl<AllocaInst *> &DeletedAllocas);
262 bool promoteAllocas();
276enum FragCalcResult { UseFrag, UseNoFrag,
Skip };
280 uint64_t NewStorageSliceOffsetInBits,
282 std::optional<DIExpression::FragmentInfo> StorageFragment,
283 std::optional<DIExpression::FragmentInfo> CurrentFragment,
287 if (StorageFragment) {
289 std::min(NewStorageSliceSizeInBits, StorageFragment->SizeInBits);
291 NewStorageSliceOffsetInBits + StorageFragment->OffsetInBits;
293 Target.SizeInBits = NewStorageSliceSizeInBits;
294 Target.OffsetInBits = NewStorageSliceOffsetInBits;
300 if (!CurrentFragment) {
301 if (
auto Size = Variable->getSizeInBits()) {
304 if (
Target == CurrentFragment)
311 if (!CurrentFragment || *CurrentFragment ==
Target)
317 if (
Target.startInBits() < CurrentFragment->startInBits() ||
318 Target.endInBits() > CurrentFragment->endInBits())
357 if (DVRAssignMarkerRange.empty())
363 LLVM_DEBUG(
dbgs() <<
" OldAllocaOffsetInBits: " << OldAllocaOffsetInBits
365 LLVM_DEBUG(
dbgs() <<
" SliceSizeInBits: " << SliceSizeInBits <<
"\n");
377 DVR->getExpression()->getFragmentInfo();
390 auto *Expr = DbgAssign->getExpression();
391 bool SetKillLocation =
false;
394 std::optional<DIExpression::FragmentInfo> BaseFragment;
397 if (R == BaseFragments.
end())
399 BaseFragment = R->second;
401 std::optional<DIExpression::FragmentInfo> CurrentFragment =
402 Expr->getFragmentInfo();
405 DbgAssign->getVariable(), OldAllocaOffsetInBits, SliceSizeInBits,
406 BaseFragment, CurrentFragment, NewFragment);
410 if (Result == UseFrag && !(NewFragment == CurrentFragment)) {
411 if (CurrentFragment) {
416 NewFragment.
OffsetInBits -= CurrentFragment->OffsetInBits;
429 SetKillLocation =
true;
437 Inst->
setMetadata(LLVMContext::MD_DIAssignID, NewID);
444 Inst, NewValue, DbgAssign->getVariable(), Expr, Dest,
448 NewAssign = DbgAssign;
467 Value && (DbgAssign->hasArgList() ||
468 !DbgAssign->getExpression()->isSingleLocationExpression());
485 if (NewAssign != DbgAssign) {
486 NewAssign->
moveBefore(DbgAssign->getIterator());
489 LLVM_DEBUG(
dbgs() <<
"Created new assign: " << *NewAssign <<
"\n");
492 for_each(DVRAssignMarkerRange, MigrateDbgAssign);
502 Twine getNameWithPrefix(
const Twine &Name)
const {
507 void SetNamePrefix(
const Twine &
P) { Prefix =
P.str(); }
509 void InsertHelper(Instruction *
I,
const Twine &Name,
527 uint64_t BeginOffset = 0;
530 uint64_t EndOffset = 0;
534 PointerIntPair<Use *, 1, bool> UseAndIsSplittable;
539 Slice(uint64_t BeginOffset, uint64_t EndOffset, Use *U,
bool IsSplittable)
540 : BeginOffset(BeginOffset), EndOffset(EndOffset),
541 UseAndIsSplittable(
U, IsSplittable) {}
543 uint64_t beginOffset()
const {
return BeginOffset; }
544 uint64_t endOffset()
const {
return EndOffset; }
546 bool isSplittable()
const {
return UseAndIsSplittable.getInt(); }
547 void makeUnsplittable() { UseAndIsSplittable.setInt(
false); }
549 Use *getUse()
const {
return UseAndIsSplittable.getPointer(); }
551 bool isDead()
const {
return getUse() ==
nullptr; }
552 void kill() { UseAndIsSplittable.setPointer(
nullptr); }
561 if (beginOffset() <
RHS.beginOffset())
563 if (beginOffset() >
RHS.beginOffset())
565 if (isSplittable() !=
RHS.isSplittable())
566 return !isSplittable();
567 if (endOffset() >
RHS.endOffset())
573 [[maybe_unused]]
friend bool operator<(
const Slice &
LHS, uint64_t RHSOffset) {
574 return LHS.beginOffset() < RHSOffset;
576 [[maybe_unused]]
friend bool operator<(uint64_t LHSOffset,
const Slice &
RHS) {
577 return LHSOffset <
RHS.beginOffset();
581 return isSplittable() ==
RHS.isSplittable() &&
582 beginOffset() ==
RHS.beginOffset() && endOffset() ==
RHS.endOffset();
597 AllocaSlices(
const DataLayout &
DL, AllocaInst &AI);
603 bool isEscaped()
const {
return PointerEscapingInstr; }
604 bool isEscapedReadOnly()
const {
return PointerEscapingInstrReadOnly; }
609 using range = iterator_range<iterator>;
611 iterator
begin() {
return Slices.begin(); }
612 iterator
end() {
return Slices.end(); }
615 using const_range = iterator_range<const_iterator>;
617 const_iterator
begin()
const {
return Slices.begin(); }
618 const_iterator
end()
const {
return Slices.end(); }
622 void erase(iterator Start, iterator Stop) { Slices.erase(Start, Stop); }
630 int OldSize = Slices.size();
631 Slices.append(NewSlices.
begin(), NewSlices.
end());
632 auto SliceI = Slices.begin() + OldSize;
633 std::stable_sort(SliceI, Slices.end());
634 std::inplace_merge(Slices.begin(), SliceI, Slices.end());
647 return DeadUseIfPromotable;
658#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
659 void print(raw_ostream &OS, const_iterator
I, StringRef Indent =
" ")
const;
660 void printSlice(raw_ostream &OS, const_iterator
I,
661 StringRef Indent =
" ")
const;
662 void printUse(raw_ostream &OS, const_iterator
I,
663 StringRef Indent =
" ")
const;
664 void print(raw_ostream &OS)
const;
665 void dump(const_iterator
I)
const;
670 template <
typename DerivedT,
typename RetT =
void>
class BuilderBase;
673 friend class AllocaSlices::SliceBuilder;
675#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
703 SmallVector<Instruction *, 8> DeadUsers;
730 friend class AllocaSlices;
731 friend class AllocaSlices::partition_iterator;
733 using iterator = AllocaSlices::iterator;
737 uint64_t BeginOffset = 0, EndOffset = 0;
747 Partition(iterator SI) : SI(SI), SJ(SI) {}
753 uint64_t beginOffset()
const {
return BeginOffset; }
758 uint64_t endOffset()
const {
return EndOffset; }
763 uint64_t
size()
const {
764 assert(BeginOffset < EndOffset &&
"Partitions must span some bytes!");
765 return EndOffset - BeginOffset;
770 bool empty()
const {
return SI == SJ; }
781 iterator
begin()
const {
return SI; }
782 iterator
end()
const {
return SJ; }
814 AllocaSlices::iterator SE;
818 uint64_t MaxSplitSliceEndOffset = 0;
822 partition_iterator(AllocaSlices::iterator
SI, AllocaSlices::iterator SE)
834 assert((
P.SI != SE || !
P.SplitTails.empty()) &&
835 "Cannot advance past the end of the slices!");
838 if (!
P.SplitTails.empty()) {
839 if (
P.EndOffset >= MaxSplitSliceEndOffset) {
841 P.SplitTails.clear();
842 MaxSplitSliceEndOffset = 0;
848 [&](Slice *S) { return S->endOffset() <= P.EndOffset; });
851 return S->endOffset() == MaxSplitSliceEndOffset;
853 "Could not find the current max split slice offset!");
856 return S->endOffset() <= MaxSplitSliceEndOffset;
858 "Max split slice end offset is not actually the max!");
865 assert(P.SplitTails.empty() &&
"Failed to clear the split slices!");
875 if (S.isSplittable() && S.endOffset() > P.EndOffset) {
876 P.SplitTails.push_back(&S);
877 MaxSplitSliceEndOffset =
878 std::max(S.endOffset(), MaxSplitSliceEndOffset);
886 P.BeginOffset = P.EndOffset;
887 P.EndOffset = MaxSplitSliceEndOffset;
894 if (!P.SplitTails.empty() && P.SI->beginOffset() != P.EndOffset &&
895 !P.SI->isSplittable()) {
896 P.BeginOffset = P.EndOffset;
897 P.EndOffset = P.SI->beginOffset();
907 P.BeginOffset = P.SplitTails.empty() ? P.SI->beginOffset() : P.EndOffset;
908 P.EndOffset = P.SI->endOffset();
913 if (!P.SI->isSplittable()) {
916 assert(P.BeginOffset == P.SI->beginOffset());
920 while (P.SJ != SE && P.SJ->beginOffset() < P.EndOffset) {
921 if (!P.SJ->isSplittable())
922 P.EndOffset = std::max(P.EndOffset, P.SJ->endOffset());
934 assert(P.SI->isSplittable() &&
"Forming a splittable partition!");
937 while (P.SJ != SE && P.SJ->beginOffset() < P.EndOffset &&
938 P.SJ->isSplittable()) {
939 P.EndOffset = std::max(P.EndOffset, P.SJ->endOffset());
946 if (P.SJ != SE && P.SJ->beginOffset() < P.EndOffset) {
947 assert(!P.SJ->isSplittable());
948 P.EndOffset = P.SJ->beginOffset();
955 "End iterators don't match between compared partition iterators!");
962 if (P.SI == RHS.P.SI && P.SplitTails.empty() == RHS.P.SplitTails.empty()) {
963 assert(P.SJ == RHS.P.SJ &&
964 "Same set of slices formed two different sized partitions!");
965 assert(P.SplitTails.size() == RHS.P.SplitTails.size() &&
966 "Same slice position with differently sized non-empty split "
989 return make_range(partition_iterator(begin(), end()),
990 partition_iterator(end(), end()));
998 return SI.getOperand(1 + CI->isZero());
999 if (
SI.getOperand(1) ==
SI.getOperand(2))
1000 return SI.getOperand(1);
1009 return PN->hasConstantValue();
1040 if (VisitedDeadInsts.
insert(&
I).second)
1045 bool IsSplittable =
false) {
1051 <<
" which has zero size or starts outside of the "
1052 << AllocSize <<
" byte alloca:\n"
1053 <<
" alloca: " << AS.AI <<
"\n"
1054 <<
" use: " <<
I <<
"\n");
1055 return markAsDead(
I);
1058 uint64_t BeginOffset =
Offset.getZExtValue();
1059 uint64_t EndOffset = BeginOffset +
Size;
1067 assert(AllocSize >= BeginOffset);
1068 if (
Size > AllocSize - BeginOffset) {
1070 <<
Offset <<
" to remain within the " << AllocSize
1071 <<
" byte alloca:\n"
1072 <<
" alloca: " << AS.AI <<
"\n"
1073 <<
" use: " <<
I <<
"\n");
1074 EndOffset = AllocSize;
1077 AS.Slices.push_back(Slice(BeginOffset, EndOffset, U, IsSplittable));
1080 void visitBitCastInst(BitCastInst &BC) {
1082 return markAsDead(BC);
1084 return Base::visitBitCastInst(BC);
1087 void visitAddrSpaceCastInst(AddrSpaceCastInst &ASC) {
1089 return markAsDead(ASC);
1091 return Base::visitAddrSpaceCastInst(ASC);
1094 void visitGetElementPtrInst(GetElementPtrInst &GEPI) {
1096 return markAsDead(GEPI);
1098 return Base::visitGetElementPtrInst(GEPI);
1101 void handleLoadOrStore(
Type *Ty, Instruction &
I,
const APInt &
Offset,
1102 uint64_t
Size,
bool IsVolatile) {
1112 void visitLoadInst(LoadInst &LI) {
1114 "All simple FCA loads should have been pre-split");
1119 return PI.setEscapedReadOnly(&LI);
1122 if (
Size.isScalable()) {
1125 return PI.setAborted(&LI);
1134 void visitStoreInst(StoreInst &SI) {
1135 Value *ValOp =
SI.getValueOperand();
1137 return PI.setEscapedAndAborted(&SI);
1139 return PI.setAborted(&SI);
1141 TypeSize StoreSize =
DL.getTypeStoreSize(ValOp->
getType());
1143 unsigned VScale =
SI.getFunction()->getVScaleValue();
1145 return PI.setAborted(&SI);
1161 <<
Offset <<
" which extends past the end of the "
1162 << AllocSize <<
" byte alloca:\n"
1163 <<
" alloca: " << AS.AI <<
"\n"
1164 <<
" use: " << SI <<
"\n");
1165 return markAsDead(SI);
1169 "All simple FCA stores should have been pre-split");
1173 void visitMemSetInst(MemSetInst &
II) {
1174 assert(
II.getRawDest() == *U &&
"Pointer use is not the destination?");
1177 (IsOffsetKnown &&
Offset.uge(AllocSize)))
1179 return markAsDead(
II);
1182 return PI.setAborted(&
II);
1186 : AllocSize -
Offset.getLimitedValue(),
1190 void visitMemTransferInst(MemTransferInst &
II) {
1194 return markAsDead(
II);
1198 if (VisitedDeadInsts.
count(&
II))
1202 return PI.setAborted(&
II);
1209 if (
Offset.uge(AllocSize)) {
1210 auto MTPI = MemTransferSliceMap.
find(&
II);
1211 if (MTPI != MemTransferSliceMap.
end())
1212 AS.Slices[MTPI->second].kill();
1213 return markAsDead(
II);
1216 uint64_t RawOffset =
Offset.getLimitedValue();
1217 uint64_t
Size =
Length ?
Length->getLimitedValue() : AllocSize - RawOffset;
1221 if (*U ==
II.getRawDest() && *U ==
II.getRawSource()) {
1223 if (!
II.isVolatile())
1224 return markAsDead(
II);
1232 SmallDenseMap<Instruction *, unsigned>::iterator MTPI;
1233 std::tie(MTPI, Inserted) =
1234 MemTransferSliceMap.
insert(std::make_pair(&
II, AS.Slices.size()));
1235 unsigned PrevIdx = MTPI->second;
1237 Slice &PrevP = AS.Slices[PrevIdx];
1241 if (!
II.isVolatile() && PrevP.beginOffset() == RawOffset) {
1243 return markAsDead(
II);
1248 PrevP.makeUnsplittable();
1255 assert(AS.Slices[PrevIdx].getUse()->getUser() == &
II &&
1256 "Map index doesn't point back to a slice with this user.");
1262 void visitIntrinsicInst(IntrinsicInst &
II) {
1263 if (
II.isDroppable()) {
1264 AS.DeadUseIfPromotable.push_back(U);
1269 return PI.setAborted(&
II);
1271 if (
II.isLifetimeStartOrEnd()) {
1272 insertUse(
II,
Offset, AllocSize,
true);
1276 Base::visitIntrinsicInst(
II);
1279 Instruction *hasUnsafePHIOrSelectUse(Instruction *Root, uint64_t &
Size) {
1284 SmallPtrSet<Instruction *, 4> Visited;
1294 std::tie(UsedI,
I) =
Uses.pop_back_val();
1297 TypeSize LoadSize =
DL.getTypeStoreSize(LI->
getType());
1309 TypeSize StoreSize =
DL.getTypeStoreSize(
Op->getType());
1319 if (!
GEP->hasAllZeroIndices())
1326 for (User *U :
I->users())
1329 }
while (!
Uses.empty());
1334 void visitPHINodeOrSelectInst(Instruction &
I) {
1337 return markAsDead(
I);
1343 return PI.setAborted(&
I);
1361 AS.DeadOperands.push_back(U);
1367 return PI.setAborted(&
I);
1370 uint64_t &
Size = PHIOrSelectSizes[&
I];
1373 if (Instruction *UnsafeI = hasUnsafePHIOrSelectUse(&
I,
Size))
1374 return PI.setAborted(UnsafeI);
1383 if (
Offset.uge(AllocSize)) {
1384 AS.DeadOperands.push_back(U);
1391 void visitPHINode(PHINode &PN) { visitPHINodeOrSelectInst(PN); }
1393 void visitSelectInst(SelectInst &SI) { visitPHINodeOrSelectInst(SI); }
1396 void visitInstruction(Instruction &
I) { PI.setAborted(&
I); }
1398 void visitCallBase(CallBase &CB) {
1404 PI.setEscapedReadOnly(&CB);
1408 Base::visitCallBase(CB);
1412AllocaSlices::AllocaSlices(
const DataLayout &
DL, AllocaInst &AI)
1414#
if !defined(
NDEBUG) || defined(LLVM_ENABLE_DUMP)
1417 PointerEscapingInstr(nullptr), PointerEscapingInstrReadOnly(nullptr) {
1419 SliceBuilder::PtrInfo PtrI =
PB.visitPtr(AI);
1420 if (PtrI.isEscaped() || PtrI.isAborted()) {
1423 PointerEscapingInstr = PtrI.getEscapingInst() ? PtrI.getEscapingInst()
1424 : PtrI.getAbortingInst();
1425 assert(PointerEscapingInstr &&
"Did not track a bad instruction");
1428 PointerEscapingInstrReadOnly = PtrI.getEscapedReadOnlyInst();
1430 llvm::erase_if(Slices, [](
const Slice &S) {
return S.isDead(); });
1437#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1439void AllocaSlices::print(raw_ostream &OS, const_iterator
I,
1440 StringRef Indent)
const {
1441 printSlice(OS,
I, Indent);
1443 printUse(OS,
I, Indent);
1446void AllocaSlices::printSlice(raw_ostream &OS, const_iterator
I,
1447 StringRef Indent)
const {
1448 OS << Indent <<
"[" <<
I->beginOffset() <<
"," <<
I->endOffset() <<
")"
1449 <<
" slice #" << (
I -
begin())
1450 << (
I->isSplittable() ?
" (splittable)" :
"");
1453void AllocaSlices::printUse(raw_ostream &OS, const_iterator
I,
1454 StringRef Indent)
const {
1455 OS << Indent <<
" used by: " << *
I->getUse()->getUser() <<
"\n";
1458void AllocaSlices::print(raw_ostream &OS)
const {
1459 if (PointerEscapingInstr) {
1460 OS <<
"Can't analyze slices for alloca: " << AI <<
"\n"
1461 <<
" A pointer to this alloca escaped by:\n"
1462 <<
" " << *PointerEscapingInstr <<
"\n";
1466 if (PointerEscapingInstrReadOnly)
1467 OS <<
"Escapes into ReadOnly: " << *PointerEscapingInstrReadOnly <<
"\n";
1469 OS <<
"Slices of alloca: " << AI <<
"\n";
1483static std::pair<Type *, IntegerType *>
1487 bool TyIsCommon =
true;
1492 for (AllocaSlices::const_iterator
I =
B;
I !=
E; ++
I) {
1493 Use *U =
I->getUse();
1496 if (
I->beginOffset() !=
B->beginOffset() ||
I->endOffset() != EndOffset)
1499 Type *UserTy =
nullptr;
1503 UserTy =
SI->getValueOperand()->getType();
1511 if (UserITy->getBitWidth() % 8 != 0 ||
1512 UserITy->getBitWidth() / 8 > (EndOffset -
B->beginOffset()))
1517 if (!ITy || ITy->
getBitWidth() < UserITy->getBitWidth())
1523 if (!UserTy || (Ty && Ty != UserTy))
1529 return {TyIsCommon ? Ty :
nullptr, ITy};
1560 Type *LoadType =
nullptr;
1573 if (LoadType != LI->
getType())
1582 if (BBI->mayWriteToMemory())
1585 MaxAlign = std::max(MaxAlign, LI->
getAlign());
1592 APInt(APWidth,
DL.getTypeStoreSize(LoadType).getFixedValue());
1629 IRB.SetInsertPoint(&PN);
1631 PN.
getName() +
".sroa.speculated");
1661 IRB.SetInsertPoint(TI);
1664 LoadTy, InVal, Alignment,
1665 (PN.
getName() +
".sroa.speculate.load." + Pred->getName()));
1666 ++NumLoadsSpeculated;
1668 Load->setAAMetadata(AATags);
1670 InjectedLoads[Pred] =
Load;
1677SelectHandSpeculativity &
1678SelectHandSpeculativity::setAsSpeculatable(
bool isTrueVal) {
1686bool SelectHandSpeculativity::isSpeculatable(
bool isTrueVal)
const {
1691bool SelectHandSpeculativity::areAllSpeculatable()
const {
1692 return isSpeculatable(
true) &&
1693 isSpeculatable(
false);
1696bool SelectHandSpeculativity::areAnySpeculatable()
const {
1697 return isSpeculatable(
true) ||
1698 isSpeculatable(
false);
1700bool SelectHandSpeculativity::areNoneSpeculatable()
const {
1701 return !areAnySpeculatable();
1704static SelectHandSpeculativity
1707 SelectHandSpeculativity
Spec;
1713 Spec.setAsSpeculatable(
Value ==
SI.getTrueValue());
1714 else if (PreserveCFG)
1720std::optional<RewriteableMemOps>
1721SROA::isSafeSelectToSpeculate(SelectInst &SI,
bool PreserveCFG) {
1722 RewriteableMemOps
Ops;
1724 for (User *U :
SI.users()) {
1732 if (
Store->isVolatile() || PreserveCFG)
1745 PossiblySpeculatableLoad
Load(LI);
1755 SelectHandSpeculativity Spec =
1757 if (PreserveCFG && !Spec.areAllSpeculatable())
1771 Value *TV =
SI.getTrueValue();
1772 Value *FV =
SI.getFalseValue();
1777 IRB.SetInsertPoint(&LI);
1781 LI.
getName() +
".sroa.speculate.load.true");
1784 LI.
getName() +
".sroa.speculate.load.false");
1785 NumLoadsSpeculated += 2;
1797 Value *V = IRB.CreateSelect(
SI.getCondition(), TL, FL,
1798 LI.
getName() +
".sroa.speculated",
1805template <
typename T>
1807 SelectHandSpeculativity
Spec,
1814 if (
Spec.areNoneSpeculatable())
1816 SI.getMetadata(LLVMContext::MD_prof), &DTU);
1819 SI.getMetadata(LLVMContext::MD_prof), &DTU,
1821 if (
Spec.isSpeculatable(
true))
1827 Tail->setName(Head->
getName() +
".cont");
1832 bool IsThen = SuccBB == HeadBI->getSuccessor(0);
1833 int SuccIdx = IsThen ? 0 : 1;
1834 auto *NewMemOpBB = SuccBB == Tail ? Head : SuccBB;
1835 auto &CondMemOp =
cast<T>(*
I.clone());
1836 if (NewMemOpBB != Head) {
1837 NewMemOpBB->setName(Head->
getName() + (IsThen ?
".then" :
".else"));
1839 ++NumLoadsPredicated;
1841 ++NumStoresPredicated;
1843 CondMemOp.dropUBImplyingAttrsAndMetadata();
1844 ++NumLoadsSpeculated;
1846 CondMemOp.insertBefore(NewMemOpBB->getTerminator()->getIterator());
1847 Value *Ptr =
SI.getOperand(1 + SuccIdx);
1848 CondMemOp.setOperand(
I.getPointerOperandIndex(), Ptr);
1850 CondMemOp.setName(
I.getName() + (IsThen ?
".then" :
".else") +
".val");
1858 I.replaceAllUsesWith(PN);
1863 SelectHandSpeculativity
Spec,
1874 const RewriteableMemOps &
Ops,
1876 bool CFGChanged =
false;
1879 for (
const RewriteableMemOp &
Op :
Ops) {
1880 SelectHandSpeculativity
Spec;
1882 if (
auto *
const *US = std::get_if<UnspeculatableStore>(&
Op)) {
1885 auto PSL = std::get<PossiblySpeculatableLoad>(
Op);
1886 I = PSL.getPointer();
1887 Spec = PSL.getInt();
1889 if (
Spec.areAllSpeculatable()) {
1892 assert(DTU &&
"Should not get here when not allowed to modify the CFG!");
1896 I->eraseFromParent();
1901 SI.eraseFromParent();
1909 const Twine &NamePrefix) {
1911 Ptr = IRB.CreateInBoundsPtrAdd(Ptr, IRB.getInt(
Offset),
1912 NamePrefix +
"sroa_idx");
1913 return IRB.CreatePointerBitCastOrAddrSpaceCast(Ptr,
PointerTy,
1914 NamePrefix +
"sroa_cast");
1929 unsigned VScale = 0) {
1939 "We can't have the same bitwidth for different int types");
1943 TypeSize NewSize =
DL.getTypeSizeInBits(NewTy);
1944 TypeSize OldSize =
DL.getTypeSizeInBits(OldTy);
1971 if (NewSize != OldSize)
1987 return OldAS == NewAS ||
1988 (!
DL.isNonIntegralAddressSpace(OldAS) &&
1989 !
DL.isNonIntegralAddressSpace(NewAS) &&
1990 DL.getPointerSize(OldAS) ==
DL.getPointerSize(NewAS));
1996 return !
DL.isNonIntegralPointerType(NewTy);
2000 if (!
DL.isNonIntegralPointerType(OldTy))
2023 std::max(S.beginOffset(),
P.beginOffset()) -
P.beginOffset();
2024 uint64_t BeginIndex = BeginOffset / ElementSize;
2025 if (BeginIndex * ElementSize != BeginOffset ||
2028 uint64_t EndOffset = std::min(S.endOffset(),
P.endOffset()) -
P.beginOffset();
2029 uint64_t EndIndex = EndOffset / ElementSize;
2030 if (EndIndex * ElementSize != EndOffset ||
2034 assert(EndIndex > BeginIndex &&
"Empty vector!");
2035 uint64_t NumElements = EndIndex - BeginIndex;
2036 Type *SliceTy = (NumElements == 1)
2037 ? Ty->getElementType()
2043 Use *U = S.getUse();
2046 if (
MI->isVolatile())
2048 if (!S.isSplittable())
2051 if (!
II->isLifetimeStartOrEnd() && !
II->isDroppable())
2058 if (LTy->isStructTy())
2060 if (
P.beginOffset() > S.beginOffset() ||
P.endOffset() < S.endOffset()) {
2061 assert(LTy->isIntegerTy());
2067 if (
SI->isVolatile())
2069 Type *STy =
SI->getValueOperand()->getType();
2073 if (
P.beginOffset() > S.beginOffset() ||
P.endOffset() < S.endOffset()) {
2093 bool HaveCommonEltTy,
Type *CommonEltTy,
2094 bool HaveVecPtrTy,
bool HaveCommonVecPtrTy,
2095 VectorType *CommonVecPtrTy,
unsigned VScale) {
2097 if (CandidateTys.
empty())
2104 if (HaveVecPtrTy && !HaveCommonVecPtrTy)
2108 if (!HaveCommonEltTy && HaveVecPtrTy) {
2110 CandidateTys.
clear();
2112 }
else if (!HaveCommonEltTy && !HaveVecPtrTy) {
2115 if (!VTy->getElementType()->isIntegerTy())
2117 VTy->getContext(), VTy->getScalarSizeInBits())));
2124 assert(
DL.getTypeSizeInBits(RHSTy).getFixedValue() ==
2125 DL.getTypeSizeInBits(LHSTy).getFixedValue() &&
2126 "Cannot have vector types of different sizes!");
2127 assert(RHSTy->getElementType()->isIntegerTy() &&
2128 "All non-integer types eliminated!");
2129 assert(LHSTy->getElementType()->isIntegerTy() &&
2130 "All non-integer types eliminated!");
2136 assert(
DL.getTypeSizeInBits(RHSTy).getFixedValue() ==
2137 DL.getTypeSizeInBits(LHSTy).getFixedValue() &&
2138 "Cannot have vector types of different sizes!");
2139 assert(RHSTy->getElementType()->isIntegerTy() &&
2140 "All non-integer types eliminated!");
2141 assert(LHSTy->getElementType()->isIntegerTy() &&
2142 "All non-integer types eliminated!");
2146 llvm::sort(CandidateTys, RankVectorTypesComp);
2147 CandidateTys.erase(
llvm::unique(CandidateTys, RankVectorTypesEq),
2148 CandidateTys.end());
2154 assert(VTy->getElementType() == CommonEltTy &&
2155 "Unaccounted for element type!");
2156 assert(VTy == CandidateTys[0] &&
2157 "Different vector types with the same element type!");
2160 CandidateTys.resize(1);
2167 std::numeric_limits<unsigned short>::max();
2173 DL.getTypeSizeInBits(VTy->getElementType()).getFixedValue();
2177 if (ElementSize % 8)
2179 assert((
DL.getTypeSizeInBits(VTy).getFixedValue() % 8) == 0 &&
2180 "vector size not a multiple of element size?");
2183 for (
const Slice &S :
P)
2187 for (
const Slice *S :
P.splitSliceTails())
2193 return VTy != CandidateTys.
end() ? *VTy :
nullptr;
2200 bool &HaveCommonEltTy,
Type *&CommonEltTy,
bool &HaveVecPtrTy,
2201 bool &HaveCommonVecPtrTy,
VectorType *&CommonVecPtrTy,
unsigned VScale) {
2203 CandidateTysCopy.
size() ? CandidateTysCopy[0] :
nullptr;
2206 for (
Type *Ty : OtherTys) {
2209 unsigned TypeSize =
DL.getTypeSizeInBits(Ty).getFixedValue();
2212 for (
VectorType *
const VTy : CandidateTysCopy) {
2214 assert(CandidateTysCopy[0] == OriginalElt &&
"Different Element");
2215 unsigned VectorSize =
DL.getTypeSizeInBits(VTy).getFixedValue();
2216 unsigned ElementSize =
2217 DL.getTypeSizeInBits(VTy->getElementType()).getFixedValue();
2221 CheckCandidateType(NewVTy);
2227 P,
DL, CandidateTys, HaveCommonEltTy, CommonEltTy, HaveVecPtrTy,
2228 HaveCommonVecPtrTy, CommonVecPtrTy, VScale);
2247 Type *CommonEltTy =
nullptr;
2249 bool HaveVecPtrTy =
false;
2250 bool HaveCommonEltTy =
true;
2251 bool HaveCommonVecPtrTy =
true;
2252 auto CheckCandidateType = [&](
Type *Ty) {
2255 if (!CandidateTys.
empty()) {
2257 if (
DL.getTypeSizeInBits(VTy).getFixedValue() !=
2258 DL.getTypeSizeInBits(V).getFixedValue()) {
2259 CandidateTys.
clear();
2264 Type *EltTy = VTy->getElementType();
2267 CommonEltTy = EltTy;
2268 else if (CommonEltTy != EltTy)
2269 HaveCommonEltTy =
false;
2272 HaveVecPtrTy =
true;
2273 if (!CommonVecPtrTy)
2274 CommonVecPtrTy = VTy;
2275 else if (CommonVecPtrTy != VTy)
2276 HaveCommonVecPtrTy =
false;
2282 for (
const Slice &S :
P) {
2287 Ty =
SI->getValueOperand()->getType();
2291 auto CandTy = Ty->getScalarType();
2292 if (CandTy->isPointerTy() && (S.beginOffset() !=
P.beginOffset() ||
2293 S.endOffset() !=
P.endOffset())) {
2300 if (S.beginOffset() ==
P.beginOffset() && S.endOffset() ==
P.endOffset())
2301 CheckCandidateType(Ty);
2306 LoadStoreTys, CandidateTysCopy, CheckCandidateType,
P,
DL,
2307 CandidateTys, HaveCommonEltTy, CommonEltTy, HaveVecPtrTy,
2308 HaveCommonVecPtrTy, CommonVecPtrTy, VScale))
2311 CandidateTys.
clear();
2313 DeferredTys, CandidateTysCopy, CheckCandidateType,
P,
DL, CandidateTys,
2314 HaveCommonEltTy, CommonEltTy, HaveVecPtrTy, HaveCommonVecPtrTy,
2315 CommonVecPtrTy, VScale);
2326 bool &WholeAllocaOp) {
2329 uint64_t RelBegin = S.beginOffset() - AllocBeginOffset;
2330 uint64_t RelEnd = S.endOffset() - AllocBeginOffset;
2332 Use *U = S.getUse();
2339 if (
II->isLifetimeStartOrEnd() ||
II->isDroppable())
2357 if (S.beginOffset() < AllocBeginOffset)
2363 WholeAllocaOp =
true;
2365 if (ITy->getBitWidth() <
DL.getTypeStoreSizeInBits(ITy).getFixedValue())
2367 }
else if (RelBegin != 0 || RelEnd !=
Size ||
2374 Type *ValueTy =
SI->getValueOperand()->getType();
2375 if (
SI->isVolatile())
2378 TypeSize StoreSize =
DL.getTypeStoreSize(ValueTy);
2383 if (S.beginOffset() < AllocBeginOffset)
2389 WholeAllocaOp =
true;
2391 if (ITy->getBitWidth() <
DL.getTypeStoreSizeInBits(ITy).getFixedValue())
2393 }
else if (RelBegin != 0 || RelEnd !=
Size ||
2402 if (!S.isSplittable())
2419 uint64_t SizeInBits =
DL.getTypeSizeInBits(AllocaTy).getFixedValue();
2425 if (SizeInBits !=
DL.getTypeStoreSizeInBits(AllocaTy).getFixedValue())
2443 bool WholeAllocaOp =
P.empty() &&
DL.isLegalInteger(SizeInBits);
2445 for (
const Slice &S :
P)
2450 for (
const Slice *S :
P.splitSliceTails())
2455 return WholeAllocaOp;
2460 const Twine &Name) {
2464 DL.getTypeStoreSize(IntTy).getFixedValue() &&
2465 "Element extends past full value");
2467 if (
DL.isBigEndian())
2468 ShAmt = 8 * (
DL.getTypeStoreSize(IntTy).getFixedValue() -
2469 DL.getTypeStoreSize(Ty).getFixedValue() -
Offset);
2471 V = IRB.CreateLShr(V, ShAmt, Name +
".shift");
2474 assert(Ty->getBitWidth() <= IntTy->getBitWidth() &&
2475 "Cannot extract to a larger integer!");
2477 V = IRB.CreateTrunc(V, Ty, Name +
".trunc");
2487 assert(Ty->getBitWidth() <= IntTy->getBitWidth() &&
2488 "Cannot insert a larger integer!");
2491 V = IRB.CreateZExt(V, IntTy, Name +
".ext");
2495 DL.getTypeStoreSize(IntTy).getFixedValue() &&
2496 "Element store outside of alloca store");
2498 if (
DL.isBigEndian())
2499 ShAmt = 8 * (
DL.getTypeStoreSize(IntTy).getFixedValue() -
2500 DL.getTypeStoreSize(Ty).getFixedValue() -
Offset);
2502 V = IRB.CreateShl(V, ShAmt, Name +
".shift");
2506 if (ShAmt || Ty->getBitWidth() < IntTy->getBitWidth()) {
2507 APInt Mask = ~Ty->getMask().zext(IntTy->getBitWidth()).shl(ShAmt);
2508 Old = IRB.CreateAnd(Old, Mask, Name +
".mask");
2510 V = IRB.CreateOr(Old, V, Name +
".insert");
2517 unsigned EndIndex,
const Twine &Name) {
2519 unsigned NumElements = EndIndex - BeginIndex;
2520 assert(NumElements <= VecTy->getNumElements() &&
"Too many elements!");
2522 if (NumElements == VecTy->getNumElements())
2525 if (NumElements == 1) {
2526 V = IRB.CreateExtractElement(V, BeginIndex, Name +
".extract");
2532 V = IRB.CreateShuffleVector(V, Mask, Name +
".extract");
2538 unsigned BeginIndex,
const Twine &Name) {
2540 assert(VecTy &&
"Can only insert a vector into a vector");
2545 V = IRB.CreateInsertElement(Old, V, BeginIndex, Name +
".insert");
2553 assert(NumSubElements <= NumElements &&
"Too many elements!");
2554 if (NumSubElements == NumElements) {
2555 assert(V->getType() == VecTy &&
"Vector type mismatch");
2558 unsigned EndIndex = BeginIndex + NumSubElements;
2565 Mask.reserve(NumElements);
2566 for (
unsigned Idx = 0; Idx != NumElements; ++Idx)
2567 if (Idx >= BeginIndex && Idx < EndIndex)
2568 Mask.push_back(Idx - BeginIndex);
2571 V = IRB.CreateShuffleVector(V, Mask, Name +
".expand");
2575 for (
unsigned Idx = 0; Idx != NumElements; ++Idx)
2576 if (Idx >= BeginIndex && Idx < EndIndex)
2577 Mask.push_back(Idx);
2579 Mask.push_back(Idx + NumElements);
2580 V = IRB.CreateShuffleVector(V, Old, Mask, Name +
"blend");
2619 const char *DebugName) {
2620 Type *EltType = VecType->getElementType();
2621 if (EltType != NewAIEltTy) {
2623 unsigned TotalBits =
2624 VecType->getNumElements() *
DL.getTypeSizeInBits(EltType);
2625 unsigned NewNumElts = TotalBits /
DL.getTypeSizeInBits(NewAIEltTy);
2628 V = Builder.CreateBitCast(V, NewVecType);
2629 VecType = NewVecType;
2630 LLVM_DEBUG(
dbgs() <<
" bitcast " << DebugName <<
": " << *V <<
"\n");
2634 BitcastIfNeeded(V0, VecType0,
"V0");
2635 BitcastIfNeeded(
V1, VecType1,
"V1");
2637 unsigned NumElts0 = VecType0->getNumElements();
2638 unsigned NumElts1 = VecType1->getNumElements();
2642 if (NumElts0 == NumElts1) {
2643 for (
unsigned i = 0; i < NumElts0 + NumElts1; ++i)
2644 ShuffleMask.push_back(i);
2648 unsigned SmallSize = std::min(NumElts0, NumElts1);
2649 unsigned LargeSize = std::max(NumElts0, NumElts1);
2650 bool IsV0Smaller = NumElts0 < NumElts1;
2651 Value *&ExtendedVec = IsV0Smaller ? V0 :
V1;
2653 for (
unsigned i = 0; i < SmallSize; ++i)
2655 for (
unsigned i = SmallSize; i < LargeSize; ++i)
2657 ExtendedVec = Builder.CreateShuffleVector(
2659 LLVM_DEBUG(
dbgs() <<
" shufflevector: " << *ExtendedVec <<
"\n");
2660 for (
unsigned i = 0; i < NumElts0; ++i)
2661 ShuffleMask.push_back(i);
2662 for (
unsigned i = 0; i < NumElts1; ++i)
2663 ShuffleMask.push_back(LargeSize + i);
2666 return Builder.CreateShuffleVector(V0,
V1, ShuffleMask);
2677class AllocaSliceRewriter :
public InstVisitor<AllocaSliceRewriter, bool> {
2679 friend class InstVisitor<AllocaSliceRewriter, bool>;
2681 using Base = InstVisitor<AllocaSliceRewriter, bool>;
2683 const DataLayout &
DL;
2686 AllocaInst &OldAI, &NewAI;
2687 const uint64_t NewAllocaBeginOffset, NewAllocaEndOffset;
2707 uint64_t ElementSize;
2711 uint64_t BeginOffset = 0;
2712 uint64_t EndOffset = 0;
2716 uint64_t NewBeginOffset = 0, NewEndOffset = 0;
2718 uint64_t SliceSize = 0;
2719 bool IsSplittable =
false;
2720 bool IsSplit =
false;
2721 Use *OldUse =
nullptr;
2725 SmallSetVector<PHINode *, 8> &PHIUsers;
2726 SmallSetVector<SelectInst *, 8> &SelectUsers;
2734 Value *getPtrToNewAI(
unsigned AddrSpace,
bool IsVolatile) {
2738 Type *AccessTy = IRB.getPtrTy(AddrSpace);
2739 return IRB.CreateAddrSpaceCast(&NewAI, AccessTy);
2743 AllocaSliceRewriter(
const DataLayout &
DL, AllocaSlices &AS, SROA &
Pass,
2744 AllocaInst &OldAI, AllocaInst &NewAI,
Type *NewAllocaTy,
2745 uint64_t NewAllocaBeginOffset,
2746 uint64_t NewAllocaEndOffset,
bool IsIntegerPromotable,
2747 VectorType *PromotableVecTy,
2748 SmallSetVector<PHINode *, 8> &PHIUsers,
2749 SmallSetVector<SelectInst *, 8> &SelectUsers)
2750 :
DL(
DL), AS(AS),
Pass(
Pass), OldAI(OldAI), NewAI(NewAI),
2751 NewAllocaBeginOffset(NewAllocaBeginOffset),
2752 NewAllocaEndOffset(NewAllocaEndOffset), NewAllocaTy(NewAllocaTy),
2753 IntTy(IsIntegerPromotable
2756 DL.getTypeSizeInBits(NewAllocaTy).getFixedValue())
2758 VecTy(PromotableVecTy),
2759 ElementTy(VecTy ? VecTy->getElementType() : nullptr),
2760 ElementSize(VecTy ?
DL.getTypeSizeInBits(ElementTy).getFixedValue() / 8
2762 PHIUsers(PHIUsers), SelectUsers(SelectUsers),
2765 assert((
DL.getTypeSizeInBits(ElementTy).getFixedValue() % 8) == 0 &&
2766 "Only multiple-of-8 sized vector elements are viable");
2769 assert((!IntTy && !VecTy) || (IntTy && !VecTy) || (!IntTy && VecTy));
2772 bool visit(AllocaSlices::const_iterator
I) {
2773 bool CanSROA =
true;
2774 BeginOffset =
I->beginOffset();
2775 EndOffset =
I->endOffset();
2776 IsSplittable =
I->isSplittable();
2778 BeginOffset < NewAllocaBeginOffset || EndOffset > NewAllocaEndOffset;
2779 LLVM_DEBUG(
dbgs() <<
" rewriting " << (IsSplit ?
"split " :
""));
2784 assert(BeginOffset < NewAllocaEndOffset);
2785 assert(EndOffset > NewAllocaBeginOffset);
2786 NewBeginOffset = std::max(BeginOffset, NewAllocaBeginOffset);
2787 NewEndOffset = std::min(EndOffset, NewAllocaEndOffset);
2789 SliceSize = NewEndOffset - NewBeginOffset;
2790 LLVM_DEBUG(
dbgs() <<
" Begin:(" << BeginOffset <<
", " << EndOffset
2791 <<
") NewBegin:(" << NewBeginOffset <<
", "
2792 << NewEndOffset <<
") NewAllocaBegin:("
2793 << NewAllocaBeginOffset <<
", " << NewAllocaEndOffset
2795 assert(IsSplit || NewBeginOffset == BeginOffset);
2796 OldUse =
I->getUse();
2800 IRB.SetInsertPoint(OldUserI);
2801 IRB.SetCurrentDebugLocation(OldUserI->
getDebugLoc());
2803 if (!IRB.getContext().shouldDiscardValueNames())
2804 IRB.getInserter().SetNamePrefix(Twine(NewAI.
getName()) +
"." +
2805 Twine(BeginOffset) +
".");
2867 std::optional<SmallVector<Value *, 4>>
2868 rewriteTreeStructuredMerge(Partition &
P) {
2870 if (
P.splitSliceTails().size() > 0)
2871 return std::nullopt;
2876 uint64_t BeginOffset;
2879 StoreInfo(StoreInst *SI, uint64_t Begin, uint64_t End,
Value *Val)
2880 :
Store(
SI), BeginOffset(Begin), EndOffset(End), StoredValue(Val) {}
2884 uint64_t BeginOffset;
2890 LoadInst *FullLoad =
nullptr;
2891 StoreInst *InitStore =
nullptr;
2895 Type *AllocatedEltTy =
2899 unsigned AllocatedEltTySize =
DL.getTypeSizeInBits(AllocatedEltTy);
2906 auto IsTypeValidForTreeStructuredMerge = [&](
Type *Ty) ->
bool {
2908 return FixedVecTy &&
2909 DL.getTypeSizeInBits(FixedVecTy->getElementType()) % 8 == 0 &&
2910 !FixedVecTy->getElementType()->isPointerTy();
2913 for (Slice &S :
P) {
2917 bool IsFullWidth = (S.beginOffset() == NewAllocaBeginOffset &&
2918 S.endOffset() == NewAllocaEndOffset);
2922 !IsTypeValidForTreeStructuredMerge(LI->
getType()))
2923 return std::nullopt;
2928 return std::nullopt;
2932 LoadInfos.
push_back({LI, S.beginOffset(), S.endOffset()});
2944 if (!
SI->isSimple() || !IsTypeValidForTreeStructuredMerge(
2945 SI->getValueOperand()->getType()))
2946 return std::nullopt;
2948 unsigned NumElts = StVecTy->getNumElements();
2949 unsigned EltSize =
DL.getTypeSizeInBits(StVecTy->getElementType());
2950 if (NumElts * EltSize % AllocatedEltTySize != 0)
2951 return std::nullopt;
2956 return std::nullopt;
2959 StoreInfos.
emplace_back(SI, S.beginOffset(), S.endOffset(),
2960 SI->getValueOperand());
2965 return std::nullopt;
2972 if (StoreInfos.
size() < 2)
2973 return std::nullopt;
2981 bool IsRMWPattern = InitStore && VecTy && !LoadInfos.
empty();
2982 bool IsStoresOnlyPattern = !InitStore && FullLoad && LoadInfos.
empty();
2983 if (!IsRMWPattern && !IsStoresOnlyPattern)
2984 return std::nullopt;
2988 BasicBlock *StoreBB = StoreInfos[0].Store->getParent();
2989 for (
auto &Info : StoreInfos)
2990 if (
Info.Store->getParent() != StoreBB)
2991 return std::nullopt;
2993 SmallVector<Value *, 4> DeletedValues;
3000 auto TreeMerge = [&](SmallVectorImpl<Value *> &Vals,
3003 while (Vals.
size() > 1) {
3004 SmallVector<Value *, 8>
Next;
3005 for (
unsigned I = 0,
E = Vals.
size();
I + 1 <
E;
I += 2) {
3011 if (Vals.
size() % 2 == 1)
3013 Vals = std::move(
Next);
3022 auto ReplaceFullLoad = [&](LoadInst *LoadToReplace,
Value *Merged) {
3024 Value *NewLoad = LoadBuilder.CreateAlignedLoad(
3025 Merged->getType(), &NewAI, getSliceAlign(),
3027 LoadToReplace->
getName() +
".sroa.new.load");
3029 NewLoad = LoadBuilder.CreateBitCast(NewLoad, LoadToReplace->
getType());
3034 if (IsStoresOnlyPattern) {
3037 llvm::sort(StoreInfos, [](
const StoreInfo &
A,
const StoreInfo &
B) {
3038 return A.BeginOffset <
B.BeginOffset;
3043 uint64_t Expected = NewAllocaBeginOffset;
3044 for (
auto &Info : StoreInfos) {
3045 if (
Info.BeginOffset != Expected)
3046 return std::nullopt;
3047 Expected =
Info.EndOffset;
3050 if (Expected != NewAllocaEndOffset)
3051 return std::nullopt;
3061 if (LoadBB == StoreBB) {
3062 for (
auto &Info : StoreInfos)
3063 if (!
Info.Store->comesBefore(FullLoad))
3064 return std::nullopt;
3068 dbgs() <<
"Tree structured merge rewrite (stores-only):\n";
3069 dbgs() <<
" Load: " << *FullLoad <<
"\n Ordered stores:\n";
3070 for (
auto [
I, Info] :
enumerate(StoreInfos)) {
3071 dbgs() <<
" [" <<
I <<
"] Range[" <<
Info.BeginOffset <<
", "
3072 <<
Info.EndOffset <<
") \tStore: " << *
Info.Store
3073 <<
"\tValue: " << *
Info.StoredValue <<
"\n";
3086 SmallVector<Value *, 8> Vals;
3087 for (
const auto &Info : StoreInfos) {
3092 Value *Merged = TreeMerge(Vals, Builder);
3093 Builder.CreateAlignedStore(Merged, &NewAI, getSliceAlign());
3096 ReplaceFullLoad(FullLoad, Merged);
3097 return DeletedValues;
3105 return std::nullopt;
3106 if (
any_of(LoadInfos, [&](
const LoadInfo &
I) {
3107 return I.Load->getParent() != StoreBB;
3109 return std::nullopt;
3121 uint64_t BeginOffset, EndOffset;
3125 Accesses.reserve(LoadInfos.
size() + StoreInfos.size());
3126 for (
const auto &L : LoadInfos)
3127 Accesses.push_back({
L.Load,
L.BeginOffset,
L.EndOffset,
false});
3128 for (
const auto &S : StoreInfos)
3129 Accesses.push_back({S.Store, S.BeginOffset, S.EndOffset,
true});
3131 return A.Inst->comesBefore(
B.Inst);
3139 return std::nullopt;
3145 if (FullLoad && FullLoad->
getParent() == StoreBB &&
3146 !
Accesses.back().Inst->comesBefore(FullLoad))
3147 return std::nullopt;
3158 using SliceRange = std::pair<uint64_t, uint64_t>;
3162 SortedRanges.
emplace_back(Acc.BeginOffset, Acc.EndOffset);
3166 uint64_t Expected = NewAllocaBeginOffset;
3167 for (
auto &
Range : SortedRanges) {
3168 if (
Range.first != Expected)
3169 return std::nullopt;
3170 Expected =
Range.second;
3172 if (Expected != NewAllocaEndOffset)
3173 return std::nullopt;
3176 dbgs() <<
"Tree structured merge rewrite (RMW):\n";
3177 dbgs() <<
" Init store: " << *InitStore <<
"\n";
3179 dbgs() <<
" Final load: " << *FullLoad <<
"\n";
3180 dbgs() <<
" Slice ranges (" << SortedRanges.size() <<
"):\n";
3181 for (
auto &
Range : SortedRanges)
3192 if (InitVec->
getType() != NewAllocaTy)
3193 InitVec = IRB.CreateBitCast(InitVec, NewAllocaTy,
"init.cast");
3194 DenseMap<SliceRange, Value *> SliceValues;
3195 for (
auto &
Range : SortedRanges) {
3196 unsigned BeginIdx = getIndex(
Range.first);
3197 unsigned EndIdx = getIndex(
Range.second);
3198 SliceValues[
Range] = IRB.CreateShuffleVector(
3214 SliceRange
Range{Acc.BeginOffset, Acc.EndOffset};
3217 if (
V->getType() != Acc.Inst->getType()) {
3219 V = IRB.CreateBitCast(V, Acc.Inst->getType());
3221 Acc.Inst->replaceAllUsesWith(V);
3238 SmallVector<Value *, 8> Vals;
3239 for (
auto &
Range : SortedRanges)
3241 Value *Merged = TreeMerge(Vals, Builder);
3242 Builder.CreateAlignedStore(Merged, &NewAI, getSliceAlign());
3247 ReplaceFullLoad(FullLoad, Merged);
3249 return DeletedValues;
3257 bool visitInstruction(Instruction &
I) {
3265 assert(IsSplit || BeginOffset == NewBeginOffset);
3266 uint64_t
Offset = NewBeginOffset - NewAllocaBeginOffset;
3268 StringRef OldName = OldPtr->
getName();
3270 size_t LastSROAPrefix = OldName.
rfind(
".sroa.");
3272 OldName = OldName.
substr(LastSROAPrefix + strlen(
".sroa."));
3277 OldName = OldName.
substr(IndexEnd + 1);
3281 OldName = OldName.
substr(OffsetEnd + 1);
3285 OldName = OldName.
substr(0, OldName.
find(
".sroa_"));
3297 Align getSliceAlign() {
3299 NewBeginOffset - NewAllocaBeginOffset);
3302 unsigned getIndex(uint64_t
Offset) {
3303 assert(VecTy &&
"Can only call getIndex when rewriting a vector");
3304 uint64_t RelOffset =
Offset - NewAllocaBeginOffset;
3305 assert(RelOffset / ElementSize < UINT32_MAX &&
"Index out of bounds");
3306 uint32_t
Index = RelOffset / ElementSize;
3307 assert(Index * ElementSize == RelOffset);
3311 void deleteIfTriviallyDead(
Value *V) {
3314 Pass.DeadInsts.push_back(
I);
3317 Value *rewriteVectorizedLoadInst(LoadInst &LI) {
3318 unsigned BeginIndex = getIndex(NewBeginOffset);
3319 unsigned EndIndex = getIndex(NewEndOffset);
3320 assert(EndIndex > BeginIndex &&
"Empty vector!");
3323 IRB.CreateAlignedLoad(NewAllocaTy, &NewAI, NewAI.
getAlign(),
"load");
3325 Load->copyMetadata(LI, {LLVMContext::MD_mem_parallel_loop_access,
3326 LLVMContext::MD_access_group});
3330 Value *rewriteIntegerLoad(LoadInst &LI) {
3331 assert(IntTy &&
"We cannot insert an integer to the alloca");
3334 IRB.CreateAlignedLoad(NewAllocaTy, &NewAI, NewAI.
getAlign(),
"load");
3335 V = IRB.CreateBitPreservingCastChain(
DL, V, IntTy);
3336 assert(NewBeginOffset >= NewAllocaBeginOffset &&
"Out of bounds offset");
3337 uint64_t
Offset = NewBeginOffset - NewAllocaBeginOffset;
3338 if (
Offset > 0 || NewEndOffset < NewAllocaEndOffset) {
3339 IntegerType *ExtractTy = Type::getIntNTy(LI.
getContext(), SliceSize * 8);
3348 "Can only handle an extract for an overly wide load");
3350 V = IRB.CreateZExt(V, LI.
getType());
3354 bool visitLoadInst(LoadInst &LI) {
3363 Type *TargetTy = IsSplit ? Type::getIntNTy(LI.
getContext(), SliceSize * 8)
3365 bool IsPtrAdjusted =
false;
3368 V = rewriteVectorizedLoadInst(LI);
3370 V = rewriteIntegerLoad(LI);
3371 }
else if (NewBeginOffset == NewAllocaBeginOffset &&
3372 NewEndOffset == NewAllocaEndOffset &&
3375 DL.getTypeStoreSize(TargetTy).getFixedValue() > SliceSize &&
3378 getPtrToNewAI(LI.getPointerAddressSpace(), LI.isVolatile());
3379 LoadInst *NewLI = IRB.CreateAlignedLoad(
3380 NewAllocaTy, NewPtr, NewAI.getAlign(), LI.isVolatile(), LI.getName());
3381 if (LI.isVolatile())
3382 NewLI->setAtomic(LI.getOrdering(), LI.getSyncScopeID());
3383 if (NewLI->isAtomic())
3384 NewLI->setAlignment(LI.getAlign());
3389 copyMetadataForLoad(*NewLI, LI);
3393 NewLI->setAAMetadata(AATags.adjustForAccess(
3394 NewBeginOffset - BeginOffset, NewLI->getType(), DL));
3402 if (auto *AITy = dyn_cast<IntegerType>(NewAllocaTy))
3403 if (auto *TITy = dyn_cast<IntegerType>(TargetTy))
3404 if (AITy->getBitWidth() < TITy->getBitWidth()) {
3405 V = IRB.CreateZExt(V, TITy,
"load.ext");
3406 if (DL.isBigEndian())
3407 V = IRB.CreateShl(V, TITy->getBitWidth() - AITy->getBitWidth(),
3411 Type *LTy = IRB.getPtrTy(AS);
3413 IRB.CreateAlignedLoad(TargetTy, getNewAllocaSlicePtr(IRB, LTy),
3418 NewBeginOffset - BeginOffset, NewLI->
getType(),
DL));
3422 NewLI->
copyMetadata(LI, {LLVMContext::MD_mem_parallel_loop_access,
3423 LLVMContext::MD_access_group});
3426 IsPtrAdjusted =
true;
3428 V = IRB.CreateBitPreservingCastChain(
DL, V, TargetTy);
3433 "Only integer type loads and stores are split");
3434 assert(SliceSize <
DL.getTypeStoreSize(LI.
getType()).getFixedValue() &&
3435 "Split load isn't smaller than original load");
3437 "Non-byte-multiple bit width");
3443 LIIt.setHeadBit(
true);
3444 IRB.SetInsertPoint(LI.
getParent(), LIIt);
3449 Value *Placeholder =
3455 Placeholder->replaceAllUsesWith(&LI);
3456 Placeholder->deleteValue();
3461 Pass.DeadInsts.push_back(&LI);
3462 deleteIfTriviallyDead(OldOp);
3467 bool rewriteVectorizedStoreInst(
Value *V, StoreInst &SI,
Value *OldOp,
3472 if (
V->getType() != VecTy) {
3473 unsigned BeginIndex = getIndex(NewBeginOffset);
3474 unsigned EndIndex = getIndex(NewEndOffset);
3475 assert(EndIndex > BeginIndex &&
"Empty vector!");
3476 unsigned NumElements = EndIndex - BeginIndex;
3478 "Too many elements!");
3479 Type *SliceTy = (NumElements == 1)
3481 : FixedVectorType::
get(ElementTy, NumElements);
3482 if (
V->getType() != SliceTy)
3483 V = IRB.CreateBitPreservingCastChain(
DL, V, SliceTy);
3487 IRB.CreateAlignedLoad(NewAllocaTy, &NewAI, NewAI.
getAlign(),
"load");
3490 StoreInst *
Store = IRB.CreateAlignedStore(V, &NewAI, NewAI.
getAlign());
3491 Store->copyMetadata(SI, {LLVMContext::MD_mem_parallel_loop_access,
3492 LLVMContext::MD_access_group});
3496 Pass.DeadInsts.push_back(&SI);
3505 bool rewriteIntegerStore(
Value *V, StoreInst &SI, AAMDNodes AATags) {
3506 assert(IntTy &&
"We cannot extract an integer from the alloca");
3508 if (
DL.getTypeSizeInBits(
V->getType()).getFixedValue() !=
3510 Value *Old = IRB.CreateAlignedLoad(NewAllocaTy, &NewAI, NewAI.
getAlign(),
3512 Old = IRB.CreateBitPreservingCastChain(
DL, Old, IntTy);
3513 assert(BeginOffset >= NewAllocaBeginOffset &&
"Out of bounds offset");
3514 uint64_t
Offset = BeginOffset - NewAllocaBeginOffset;
3517 V = IRB.CreateBitPreservingCastChain(
DL, V, NewAllocaTy);
3518 StoreInst *
Store = IRB.CreateAlignedStore(V, &NewAI, NewAI.
getAlign());
3519 Store->copyMetadata(SI, {LLVMContext::MD_mem_parallel_loop_access,
3520 LLVMContext::MD_access_group});
3527 Store->getValueOperand(),
DL);
3529 Pass.DeadInsts.push_back(&SI);
3534 bool visitStoreInst(StoreInst &SI) {
3536 Value *OldOp =
SI.getOperand(1);
3539 AAMDNodes AATags =
SI.getAAMetadata();
3544 if (
V->getType()->isPointerTy())
3546 Pass.PostPromotionWorklist.insert(AI);
3548 TypeSize StoreSize =
DL.getTypeStoreSize(
V->getType());
3551 assert(
V->getType()->isIntegerTy() &&
3552 "Only integer type loads and stores are split");
3553 assert(
DL.typeSizeEqualsStoreSize(
V->getType()) &&
3554 "Non-byte-multiple bit width");
3555 IntegerType *NarrowTy = Type::getIntNTy(
SI.getContext(), SliceSize * 8);
3561 return rewriteVectorizedStoreInst(V, SI, OldOp, AATags);
3562 if (IntTy &&
V->getType()->isIntegerTy())
3563 return rewriteIntegerStore(V, SI, AATags);
3566 if (NewBeginOffset == NewAllocaBeginOffset &&
3567 NewEndOffset == NewAllocaEndOffset &&
3569 V = IRB.CreateBitPreservingCastChain(
DL, V, NewAllocaTy);
3571 getPtrToNewAI(
SI.getPointerAddressSpace(),
SI.isVolatile());
3574 IRB.CreateAlignedStore(V, NewPtr, NewAI.
getAlign(),
SI.isVolatile());
3576 unsigned AS =
SI.getPointerAddressSpace();
3577 Value *NewPtr = getNewAllocaSlicePtr(IRB, IRB.getPtrTy(AS));
3579 IRB.CreateAlignedStore(V, NewPtr, getSliceAlign(),
SI.isVolatile());
3581 NewSI->
copyMetadata(SI, {LLVMContext::MD_mem_parallel_loop_access,
3582 LLVMContext::MD_access_group});
3586 if (
SI.isVolatile())
3595 Pass.DeadInsts.push_back(&SI);
3596 deleteIfTriviallyDead(OldOp);
3614 assert(
Size > 0 &&
"Expected a positive number of bytes.");
3622 IRB.CreateZExt(V, SplatIntTy,
"zext"),
3632 V = IRB.CreateVectorSplat(NumElements, V,
"vsplat");
3637 bool visitMemSetInst(MemSetInst &
II) {
3641 AAMDNodes AATags =
II.getAAMetadata();
3647 assert(NewBeginOffset == BeginOffset);
3648 II.setDest(getNewAllocaSlicePtr(IRB, OldPtr->
getType()));
3649 II.setDestAlignment(getSliceAlign());
3654 "AT: Unexpected link to non-const GEP");
3655 deleteIfTriviallyDead(OldPtr);
3660 Pass.DeadInsts.push_back(&
II);
3664 const bool CanContinue = [&]() {
3667 if (BeginOffset > NewAllocaBeginOffset || EndOffset < NewAllocaEndOffset)
3671 const uint64_t
Len =
C->getLimitedValue();
3672 if (Len > std::numeric_limits<unsigned>::max())
3674 auto *Int8Ty = IntegerType::getInt8Ty(NewAI.
getContext());
3677 DL.isLegalInteger(
DL.getTypeSizeInBits(ScalarTy).getFixedValue());
3683 Type *SizeTy =
II.getLength()->getType();
3684 unsigned Sz = NewEndOffset - NewBeginOffset;
3687 getNewAllocaSlicePtr(IRB, OldPtr->
getType()),
II.getValue(),
Size,
3688 MaybeAlign(getSliceAlign()),
II.isVolatile()));
3694 New,
New->getRawDest(),
nullptr,
DL);
3709 assert(ElementTy == ScalarTy);
3711 unsigned BeginIndex = getIndex(NewBeginOffset);
3712 unsigned EndIndex = getIndex(NewEndOffset);
3713 assert(EndIndex > BeginIndex &&
"Empty vector!");
3714 unsigned NumElements = EndIndex - BeginIndex;
3716 "Too many elements!");
3719 II.getValue(),
DL.getTypeSizeInBits(ElementTy).getFixedValue() / 8);
3720 Splat = IRB.CreateBitPreservingCastChain(
DL,
Splat, ElementTy);
3721 if (NumElements > 1)
3724 Value *Old = IRB.CreateAlignedLoad(NewAllocaTy, &NewAI, NewAI.
getAlign(),
3732 uint64_t
Size = NewEndOffset - NewBeginOffset;
3733 V = getIntegerSplat(
II.getValue(),
Size);
3735 if (IntTy && (NewBeginOffset != NewAllocaBeginOffset ||
3736 NewEndOffset != NewAllocaEndOffset)) {
3737 Value *Old = IRB.CreateAlignedLoad(NewAllocaTy, &NewAI,
3739 Old = IRB.CreateBitPreservingCastChain(
DL, Old, IntTy);
3740 uint64_t
Offset = NewBeginOffset - NewAllocaBeginOffset;
3743 assert(
V->getType() == IntTy &&
3744 "Wrong type for an alloca wide integer!");
3746 V = IRB.CreateBitPreservingCastChain(
DL, V, NewAllocaTy);
3749 assert(NewBeginOffset == NewAllocaBeginOffset);
3750 assert(NewEndOffset == NewAllocaEndOffset);
3752 V = getIntegerSplat(
II.getValue(),
3753 DL.getTypeSizeInBits(ScalarTy).getFixedValue() / 8);
3758 V = IRB.CreateBitPreservingCastChain(
DL, V, NewAllocaTy);
3761 Value *NewPtr = getPtrToNewAI(
II.getDestAddressSpace(),
II.isVolatile());
3763 IRB.CreateAlignedStore(V, NewPtr, NewAI.
getAlign(),
II.isVolatile());
3764 New->copyMetadata(
II, {LLVMContext::MD_mem_parallel_loop_access,
3765 LLVMContext::MD_access_group});
3771 New,
New->getPointerOperand(), V,
DL);
3774 return !
II.isVolatile();
3777 bool visitMemTransferInst(MemTransferInst &
II) {
3783 AAMDNodes AATags =
II.getAAMetadata();
3785 bool IsDest = &
II.getRawDestUse() == OldUse;
3786 assert((IsDest &&
II.getRawDest() == OldPtr) ||
3787 (!IsDest &&
II.getRawSource() == OldPtr));
3789 Align SliceAlign = getSliceAlign();
3797 if (!IsSplittable) {
3798 Value *AdjustedPtr = getNewAllocaSlicePtr(IRB, OldPtr->
getType());
3803 DbgAssign->getAddress() ==
II.getDest())
3804 DbgAssign->replaceVariableLocationOp(
II.getDest(), AdjustedPtr);
3806 II.setDest(AdjustedPtr);
3807 II.setDestAlignment(SliceAlign);
3809 II.setSource(AdjustedPtr);
3810 II.setSourceAlignment(SliceAlign);
3814 deleteIfTriviallyDead(OldPtr);
3827 (BeginOffset > NewAllocaBeginOffset || EndOffset < NewAllocaEndOffset ||
3828 SliceSize !=
DL.getTypeStoreSize(NewAllocaTy).getFixedValue() ||
3829 !
DL.typeSizeEqualsStoreSize(NewAllocaTy) ||
3835 if (EmitMemCpy && &OldAI == &NewAI) {
3837 assert(NewBeginOffset == BeginOffset);
3840 if (NewEndOffset != EndOffset)
3841 II.setLength(NewEndOffset - NewBeginOffset);
3845 Pass.DeadInsts.push_back(&
II);
3849 Value *OtherPtr = IsDest ?
II.getRawSource() :
II.getRawDest();
3850 if (AllocaInst *AI =
3852 assert(AI != &OldAI && AI != &NewAI &&
3853 "Splittable transfers cannot reach the same alloca on both ends.");
3854 Pass.Worklist.insert(AI);
3861 unsigned OffsetWidth =
DL.getIndexSizeInBits(OtherAS);
3862 APInt OtherOffset(OffsetWidth, NewBeginOffset - BeginOffset);
3864 (IsDest ?
II.getSourceAlign() :
II.getDestAlign()).valueOrOne();
3866 commonAlignment(OtherAlign, OtherOffset.zextOrTrunc(64).getZExtValue());
3874 Value *OurPtr = getNewAllocaSlicePtr(IRB, OldPtr->
getType());
3875 Type *SizeTy =
II.getLength()->getType();
3876 Constant *
Size = ConstantInt::get(SizeTy, NewEndOffset - NewBeginOffset);
3878 Value *DestPtr, *SrcPtr;
3879 MaybeAlign DestAlign, SrcAlign;
3883 DestAlign = SliceAlign;
3885 SrcAlign = OtherAlign;
3888 DestAlign = OtherAlign;
3890 SrcAlign = SliceAlign;
3892 CallInst *
New = IRB.CreateMemCpy(DestPtr, DestAlign, SrcPtr, SrcAlign,
3895 New->setAAMetadata(AATags.
shift(NewBeginOffset - BeginOffset));
3900 &
II, New, DestPtr,
nullptr,
DL);
3905 SliceSize * 8, &
II, New, DestPtr,
nullptr,
DL);
3911 bool IsWholeAlloca = NewBeginOffset == NewAllocaBeginOffset &&
3912 NewEndOffset == NewAllocaEndOffset;
3913 uint64_t
Size = NewEndOffset - NewBeginOffset;
3914 unsigned BeginIndex = VecTy ? getIndex(NewBeginOffset) : 0;
3915 unsigned EndIndex = VecTy ? getIndex(NewEndOffset) : 0;
3916 unsigned NumElements = EndIndex - BeginIndex;
3917 IntegerType *SubIntTy =
3918 IntTy ? Type::getIntNTy(IntTy->
getContext(),
Size * 8) : nullptr;
3923 if (VecTy && !IsWholeAlloca) {
3924 if (NumElements == 1)
3925 OtherTy = VecTy->getElementType();
3928 }
else if (IntTy && !IsWholeAlloca) {
3931 OtherTy = NewAllocaTy;
3936 MaybeAlign SrcAlign = OtherAlign;
3937 MaybeAlign DstAlign = SliceAlign;
3945 DstPtr = getPtrToNewAI(
II.getDestAddressSpace(),
II.isVolatile());
3949 SrcPtr = getPtrToNewAI(
II.getSourceAddressSpace(),
II.isVolatile());
3953 if (VecTy && !IsWholeAlloca && !IsDest) {
3955 IRB.CreateAlignedLoad(NewAllocaTy, &NewAI, NewAI.
getAlign(),
"load");
3957 }
else if (IntTy && !IsWholeAlloca && !IsDest) {
3959 IRB.CreateAlignedLoad(NewAllocaTy, &NewAI, NewAI.
getAlign(),
"load");
3960 Src = IRB.CreateBitPreservingCastChain(
DL, Src, IntTy);
3961 uint64_t
Offset = NewBeginOffset - NewAllocaBeginOffset;
3964 LoadInst *
Load = IRB.CreateAlignedLoad(OtherTy, SrcPtr, SrcAlign,
3965 II.isVolatile(),
"copyload");
3966 Load->copyMetadata(
II, {LLVMContext::MD_mem_parallel_loop_access,
3967 LLVMContext::MD_access_group});
3974 if (VecTy && !IsWholeAlloca && IsDest) {
3975 Value *Old = IRB.CreateAlignedLoad(NewAllocaTy, &NewAI, NewAI.
getAlign(),
3978 }
else if (IntTy && !IsWholeAlloca && IsDest) {
3979 Value *Old = IRB.CreateAlignedLoad(NewAllocaTy, &NewAI, NewAI.
getAlign(),
3981 Old = IRB.CreateBitPreservingCastChain(
DL, Old, IntTy);
3982 uint64_t
Offset = NewBeginOffset - NewAllocaBeginOffset;
3984 Src = IRB.CreateBitPreservingCastChain(
DL, Src, NewAllocaTy);
3988 IRB.CreateAlignedStore(Src, DstPtr, DstAlign,
II.isVolatile()));
3989 Store->copyMetadata(
II, {LLVMContext::MD_mem_parallel_loop_access,
3990 LLVMContext::MD_access_group});
3993 Src->getType(),
DL));
4008 return !
II.isVolatile();
4011 bool visitIntrinsicInst(IntrinsicInst &
II) {
4012 assert((
II.isLifetimeStartOrEnd() ||
II.isDroppable()) &&
4013 "Unexpected intrinsic!");
4017 Pass.DeadInsts.push_back(&
II);
4019 if (
II.isDroppable()) {
4020 assert(
II.getIntrinsicID() == Intrinsic::assume &&
"Expected assume");
4026 assert(
II.getArgOperand(0) == OldPtr);
4030 if (
II.getIntrinsicID() == Intrinsic::lifetime_start)
4031 New = IRB.CreateLifetimeStart(Ptr);
4033 New = IRB.CreateLifetimeEnd(Ptr);
4041 void fixLoadStoreAlign(Instruction &Root) {
4045 SmallPtrSet<Instruction *, 4> Visited;
4046 SmallVector<Instruction *, 4>
Uses;
4048 Uses.push_back(&Root);
4057 SI->setAlignment(std::min(
SI->getAlign(), getSliceAlign()));
4064 for (User *U :
I->users())
4067 }
while (!
Uses.empty());
4070 bool visitPHINode(PHINode &PN) {
4072 assert(BeginOffset >= NewAllocaBeginOffset &&
"PHIs are unsplittable");
4073 assert(EndOffset <= NewAllocaEndOffset &&
"PHIs are unsplittable");
4079 IRBuilderBase::InsertPointGuard Guard(IRB);
4082 OldPtr->
getParent()->getFirstInsertionPt());
4084 IRB.SetInsertPoint(OldPtr);
4085 IRB.SetCurrentDebugLocation(OldPtr->
getDebugLoc());
4087 Value *NewPtr = getNewAllocaSlicePtr(IRB, OldPtr->
getType());
4092 deleteIfTriviallyDead(OldPtr);
4095 fixLoadStoreAlign(PN);
4104 bool visitSelectInst(SelectInst &SI) {
4106 assert((
SI.getTrueValue() == OldPtr ||
SI.getFalseValue() == OldPtr) &&
4107 "Pointer isn't an operand!");
4108 assert(BeginOffset >= NewAllocaBeginOffset &&
"Selects are unsplittable");
4109 assert(EndOffset <= NewAllocaEndOffset &&
"Selects are unsplittable");
4111 Value *NewPtr = getNewAllocaSlicePtr(IRB, OldPtr->
getType());
4113 if (
SI.getOperand(1) == OldPtr)
4114 SI.setOperand(1, NewPtr);
4115 if (
SI.getOperand(2) == OldPtr)
4116 SI.setOperand(2, NewPtr);
4119 deleteIfTriviallyDead(OldPtr);
4122 fixLoadStoreAlign(SI);
4137class AggLoadStoreRewriter :
public InstVisitor<AggLoadStoreRewriter, bool> {
4139 friend class InstVisitor<AggLoadStoreRewriter, bool>;
4145 SmallPtrSet<User *, 8> Visited;
4152 const DataLayout &
DL;
4157 AggLoadStoreRewriter(
const DataLayout &
DL, IRBuilderTy &IRB)
4158 :
DL(
DL), IRB(IRB) {}
4162 bool rewrite(Instruction &
I) {
4166 while (!
Queue.empty()) {
4167 U =
Queue.pop_back_val();
4176 void enqueueUsers(Instruction &
I) {
4177 for (Use &U :
I.uses())
4178 if (Visited.
insert(
U.getUser()).second)
4179 Queue.push_back(&U);
4183 bool visitInstruction(Instruction &
I) {
return false; }
4186 template <
typename Derived>
class OpSplitter {
4193 SmallVector<unsigned, 4> Indices;
4197 SmallVector<Value *, 4> GEPIndices;
4211 const DataLayout &
DL;
4215 OpSplitter(Instruction *InsertionPoint,
Value *Ptr,
Type *BaseTy,
4216 Align BaseAlign,
const DataLayout &
DL, IRBuilderTy &IRB)
4217 : IRB(IRB), GEPIndices(1, IRB.getInt32(0)), Ptr(Ptr), BaseTy(BaseTy),
4218 BaseAlign(BaseAlign),
DL(
DL) {
4219 IRB.SetInsertPoint(InsertionPoint);
4236 void emitSplitOps(
Type *Ty,
Value *&Agg,
const Twine &Name) {
4238 unsigned Offset =
DL.getIndexedOffsetInType(BaseTy, GEPIndices);
4239 return static_cast<Derived *
>(
this)->emitFunc(
4244 unsigned OldSize = Indices.
size();
4246 for (
unsigned Idx = 0,
Size = ATy->getNumElements(); Idx !=
Size;
4248 assert(Indices.
size() == OldSize &&
"Did not return to the old size");
4250 GEPIndices.
push_back(IRB.getInt32(Idx));
4251 emitSplitOps(ATy->getElementType(), Agg, Name +
"." + Twine(Idx));
4259 unsigned OldSize = Indices.
size();
4261 for (
unsigned Idx = 0,
Size = STy->getNumElements(); Idx !=
Size;
4263 assert(Indices.
size() == OldSize &&
"Did not return to the old size");
4265 GEPIndices.
push_back(IRB.getInt32(Idx));
4266 emitSplitOps(STy->getElementType(Idx), Agg, Name +
"." + Twine(Idx));
4277 struct LoadOpSplitter :
public OpSplitter<LoadOpSplitter> {
4281 SmallVector<Value *, 4> Components;
4286 LoadOpSplitter(Instruction *InsertionPoint,
Value *Ptr,
Type *BaseTy,
4287 AAMDNodes AATags, Align BaseAlign,
const DataLayout &
DL,
4289 : OpSplitter<LoadOpSplitter>(InsertionPoint, Ptr, BaseTy, BaseAlign,
DL,
4295 void emitFunc(
Type *Ty,
Value *&Agg, Align Alignment,
const Twine &Name) {
4299 IRB.CreateInBoundsGEP(BaseTy, Ptr, GEPIndices, Name +
".gep");
4301 IRB.CreateAlignedLoad(Ty,
GEP, Alignment, Name +
".load");
4307 Load->setAAMetadata(
4313 Agg = IRB.CreateInsertValue(Agg,
Load, Indices, Name +
".insert");
4318 void recordFakeUses(LoadInst &LI) {
4319 for (Use &U : LI.
uses())
4321 if (
II->getIntrinsicID() == Intrinsic::fake_use)
4327 void emitFakeUses() {
4328 for (Instruction *
I : FakeUses) {
4329 IRB.SetInsertPoint(
I);
4330 for (
auto *V : Components)
4331 IRB.CreateIntrinsic(Intrinsic::fake_use, {
V});
4332 I->eraseFromParent();
4337 bool visitLoadInst(LoadInst &LI) {
4346 Splitter.recordFakeUses(LI);
4349 Splitter.emitFakeUses();
4356 struct StoreOpSplitter :
public OpSplitter<StoreOpSplitter> {
4357 StoreOpSplitter(Instruction *InsertionPoint,
Value *Ptr,
Type *BaseTy,
4358 AAMDNodes AATags, StoreInst *AggStore, Align BaseAlign,
4359 const DataLayout &
DL, IRBuilderTy &IRB)
4360 : OpSplitter<StoreOpSplitter>(InsertionPoint, Ptr, BaseTy, BaseAlign,
4362 AATags(AATags), AggStore(AggStore) {}
4364 StoreInst *AggStore;
4367 void emitFunc(
Type *Ty,
Value *&Agg, Align Alignment,
const Twine &Name) {
4373 Value *ExtractValue =
4374 IRB.CreateExtractValue(Agg, Indices, Name +
".extract");
4375 Value *InBoundsGEP =
4376 IRB.CreateInBoundsGEP(BaseTy, Ptr, GEPIndices, Name +
".gep");
4378 IRB.CreateAlignedStore(ExtractValue, InBoundsGEP, Alignment);
4394 uint64_t SizeInBits =
4395 DL.getTypeSizeInBits(
Store->getValueOperand()->getType());
4397 SizeInBits, AggStore,
Store,
4398 Store->getPointerOperand(),
Store->getValueOperand(),
4402 "AT: unexpected debug.assign linked to store through "
4409 bool visitStoreInst(StoreInst &SI) {
4410 if (!
SI.isSimple() ||
SI.getPointerOperand() != *U)
4413 if (
V->getType()->isSingleValueType())
4418 StoreOpSplitter Splitter(&SI, *U,
V->getType(),
SI.getAAMetadata(), &SI,
4420 Splitter.emitSplitOps(
V->getType(), V,
V->getName() +
".fca");
4425 SI.eraseFromParent();
4429 bool visitBitCastInst(BitCastInst &BC) {
4434 bool visitAddrSpaceCastInst(AddrSpaceCastInst &ASC) {
4444 bool unfoldGEPSelect(GetElementPtrInst &GEPI) {
4463 if (!ZI->getSrcTy()->isIntegerTy(1))
4476 dbgs() <<
" original: " << *Sel <<
"\n";
4477 dbgs() <<
" " << GEPI <<
"\n";);
4479 auto GetNewOps = [&](
Value *SelOp) {
4492 Cond =
SI->getCondition();
4493 True =
SI->getTrueValue();
4494 False =
SI->getFalseValue();
4498 Cond = Sel->getOperand(0);
4499 True = ConstantInt::get(Sel->getType(), 1);
4500 False = ConstantInt::get(Sel->getType(), 0);
4505 IRB.SetInsertPoint(&GEPI);
4509 Value *NTrue = IRB.CreateGEP(Ty, TrueOps[0],
ArrayRef(TrueOps).drop_front(),
4510 True->
getName() +
".sroa.gep", NW);
4513 IRB.CreateGEP(Ty, FalseOps[0],
ArrayRef(FalseOps).drop_front(),
4514 False->
getName() +
".sroa.gep", NW);
4516 Value *NSel = MDFrom
4517 ? IRB.CreateSelect(
Cond, NTrue, NFalse,
4518 Sel->getName() +
".sroa.sel", MDFrom)
4519 : IRB.CreateSelectWithUnknownProfile(
4521 Sel->getName() +
".sroa.sel");
4522 Visited.
erase(&GEPI);
4527 enqueueUsers(*NSelI);
4530 dbgs() <<
" " << *NFalse <<
"\n";
4531 dbgs() <<
" " << *NSel <<
"\n";);
4540 bool unfoldGEPPhi(GetElementPtrInst &GEPI) {
4545 auto IsInvalidPointerOperand = [](
Value *
V) {
4549 return !AI->isStaticAlloca();
4553 if (
any_of(
Phi->operands(), IsInvalidPointerOperand))
4568 [](
Value *V) { return isa<ConstantInt>(V); }))
4581 dbgs() <<
" original: " << *
Phi <<
"\n";
4582 dbgs() <<
" " << GEPI <<
"\n";);
4584 auto GetNewOps = [&](
Value *PhiOp) {
4594 IRB.SetInsertPoint(Phi);
4595 PHINode *NewPhi = IRB.CreatePHI(GEPI.
getType(),
Phi->getNumIncomingValues(),
4596 Phi->getName() +
".sroa.phi");
4602 for (
unsigned I = 0,
E =
Phi->getNumIncomingValues();
I !=
E; ++
I) {
4611 IRB.CreateGEP(SourceTy, NewOps[0],
ArrayRef(NewOps).drop_front(),
4617 Visited.
erase(&GEPI);
4621 enqueueUsers(*NewPhi);
4627 dbgs() <<
"\n " << *NewPhi <<
'\n');
4632 bool visitGetElementPtrInst(GetElementPtrInst &GEPI) {
4633 if (unfoldGEPSelect(GEPI))
4636 if (unfoldGEPPhi(GEPI))
4643 bool visitPHINode(PHINode &PN) {
4648 bool visitSelectInst(SelectInst &SI) {
4662 if (Ty->isSingleValueType())
4665 uint64_t AllocSize =
DL.getTypeAllocSize(Ty).getFixedValue();
4670 InnerTy = ArrTy->getElementType();
4674 InnerTy = STy->getElementType(Index);
4679 if (AllocSize >
DL.getTypeAllocSize(InnerTy).getFixedValue() ||
4680 TypeSize >
DL.getTypeSizeInBits(InnerTy).getFixedValue())
4701 if (
Offset == 0 &&
DL.getTypeAllocSize(Ty).getFixedValue() ==
Size)
4703 if (
Offset >
DL.getTypeAllocSize(Ty).getFixedValue() ||
4704 (
DL.getTypeAllocSize(Ty).getFixedValue() -
Offset) <
Size)
4711 ElementTy = AT->getElementType();
4712 TyNumElements = AT->getNumElements();
4717 ElementTy = VT->getElementType();
4718 TyNumElements = VT->getNumElements();
4720 uint64_t ElementSize =
DL.getTypeAllocSize(ElementTy).getFixedValue();
4722 if (NumSkippedElements >= TyNumElements)
4724 Offset -= NumSkippedElements * ElementSize;
4736 if (
Size == ElementSize)
4740 if (NumElements * ElementSize !=
Size)
4764 uint64_t ElementSize =
DL.getTypeAllocSize(ElementTy).getFixedValue();
4765 if (
Offset >= ElementSize)
4776 if (
Size == ElementSize)
4783 if (Index == EndIndex)
4793 assert(Index < EndIndex);
4832bool SROA::presplitLoadsAndStores(AllocaInst &AI, AllocaSlices &AS) {
4846 struct SplitOffsets {
4848 std::vector<uint64_t> Splits;
4850 SmallDenseMap<Instruction *, SplitOffsets, 8> SplitOffsetsMap;
4863 SmallPtrSet<LoadInst *, 8> UnsplittableLoads;
4865 LLVM_DEBUG(
dbgs() <<
" Searching for candidate loads and stores\n");
4866 for (
auto &
P : AS.partitions()) {
4867 for (Slice &S :
P) {
4869 if (!S.isSplittable() || S.endOffset() <=
P.endOffset()) {
4874 UnsplittableLoads.
insert(LI);
4877 UnsplittableLoads.
insert(LI);
4880 assert(
P.endOffset() > S.beginOffset() &&
4881 "Empty or backwards partition!");
4890 auto IsLoadSimplyStored = [](LoadInst *LI) {
4891 for (User *LU : LI->
users()) {
4893 if (!SI || !
SI->isSimple())
4898 if (!IsLoadSimplyStored(LI)) {
4899 UnsplittableLoads.
insert(LI);
4905 if (S.getUse() != &
SI->getOperandUse(
SI->getPointerOperandIndex()))
4909 if (!StoredLoad || !StoredLoad->isSimple())
4911 assert(!
SI->isVolatile() &&
"Cannot split volatile stores!");
4921 auto &
Offsets = SplitOffsetsMap[
I];
4923 "Should not have splits the first time we see an instruction!");
4925 Offsets.Splits.push_back(
P.endOffset() - S.beginOffset());
4930 for (Slice *S :
P.splitSliceTails()) {
4931 auto SplitOffsetsMapI =
4933 if (SplitOffsetsMapI == SplitOffsetsMap.
end())
4935 auto &
Offsets = SplitOffsetsMapI->second;
4939 "Cannot have an empty set of splits on the second partition!");
4941 P.beginOffset() -
Offsets.S->beginOffset() &&
4942 "Previous split does not end where this one begins!");
4946 if (S->endOffset() >
P.endOffset())
4955 llvm::erase_if(Stores, [&UnsplittableLoads, &SplitOffsetsMap](StoreInst *SI) {
4961 if (UnsplittableLoads.
count(LI))
4964 auto LoadOffsetsI = SplitOffsetsMap.
find(LI);
4965 if (LoadOffsetsI == SplitOffsetsMap.
end())
4967 auto &LoadOffsets = LoadOffsetsI->second;
4970 auto &StoreOffsets = SplitOffsetsMap[
SI];
4975 if (LoadOffsets.Splits == StoreOffsets.Splits)
4979 <<
" " << *LI <<
"\n"
4980 <<
" " << *SI <<
"\n");
4986 UnsplittableLoads.
insert(LI);
4995 return UnsplittableLoads.
count(LI);
5000 return UnsplittableLoads.
count(LI);
5010 IRBuilderTy IRB(&AI);
5017 SmallPtrSet<AllocaInst *, 4> ResplitPromotableAllocas;
5027 SmallDenseMap<LoadInst *, std::vector<LoadInst *>, 1> SplitLoadsMap;
5028 std::vector<LoadInst *> SplitLoads;
5029 const DataLayout &
DL = AI.getDataLayout();
5030 for (LoadInst *LI : Loads) {
5033 auto &
Offsets = SplitOffsetsMap[LI];
5034 unsigned SliceSize =
Offsets.S->endOffset() -
Offsets.S->beginOffset();
5036 "Load must have type size equal to store size");
5038 "Load must be >= slice size");
5040 uint64_t BaseOffset =
Offsets.S->beginOffset();
5041 assert(BaseOffset + SliceSize > BaseOffset &&
5042 "Cannot represent alloca access size using 64-bit integers!");
5045 IRB.SetInsertPoint(LI);
5049 uint64_t PartOffset = 0, PartSize =
Offsets.Splits.front();
5052 auto *PartTy = Type::getIntNTy(LI->
getContext(), PartSize * 8);
5055 LoadInst *PLoad = IRB.CreateAlignedLoad(
5058 APInt(
DL.getIndexSizeInBits(AS), PartOffset),
5059 PartPtrTy,
BasePtr->getName() +
"."),
5062 PLoad->
copyMetadata(*LI, {LLVMContext::MD_mem_parallel_loop_access,
5063 LLVMContext::MD_access_group});
5067 SplitLoads.push_back(PLoad);
5071 Slice(BaseOffset + PartOffset, BaseOffset + PartOffset + PartSize,
5075 <<
", " << NewSlices.
back().endOffset()
5076 <<
"): " << *PLoad <<
"\n");
5083 PartOffset =
Offsets.Splits[Idx];
5085 PartSize = (Idx <
Size ?
Offsets.Splits[Idx] : SliceSize) - PartOffset;
5091 bool DeferredStores =
false;
5092 for (User *LU : LI->
users()) {
5094 if (!Stores.
empty() && SplitOffsetsMap.
count(SI)) {
5095 DeferredStores =
true;
5101 Value *StoreBasePtr =
SI->getPointerOperand();
5102 IRB.SetInsertPoint(SI);
5103 AAMDNodes AATags =
SI->getAAMetadata();
5105 LLVM_DEBUG(
dbgs() <<
" Splitting store of load: " << *SI <<
"\n");
5107 for (
int Idx = 0,
Size = SplitLoads.size(); Idx <
Size; ++Idx) {
5108 LoadInst *PLoad = SplitLoads[Idx];
5109 uint64_t PartOffset = Idx == 0 ? 0 :
Offsets.Splits[Idx - 1];
5110 auto *PartPtrTy =
SI->getPointerOperandType();
5112 auto AS =
SI->getPointerAddressSpace();
5113 StoreInst *PStore = IRB.CreateAlignedStore(
5116 APInt(
DL.getIndexSizeInBits(AS), PartOffset),
5117 PartPtrTy, StoreBasePtr->
getName() +
"."),
5120 PStore->
copyMetadata(*SI, {LLVMContext::MD_mem_parallel_loop_access,
5121 LLVMContext::MD_access_group,
5122 LLVMContext::MD_DIAssignID});
5127 LLVM_DEBUG(
dbgs() <<
" +" << PartOffset <<
":" << *PStore <<
"\n");
5135 ResplitPromotableAllocas.
insert(OtherAI);
5136 Worklist.insert(OtherAI);
5139 Worklist.insert(OtherAI);
5143 DeadInsts.push_back(SI);
5148 SplitLoadsMap.
insert(std::make_pair(LI, std::move(SplitLoads)));
5151 DeadInsts.push_back(LI);
5160 for (StoreInst *SI : Stores) {
5165 assert(StoreSize > 0 &&
"Cannot have a zero-sized integer store!");
5169 "Slice size should always match load size exactly!");
5170 uint64_t BaseOffset =
Offsets.S->beginOffset();
5171 assert(BaseOffset + StoreSize > BaseOffset &&
5172 "Cannot represent alloca access size using 64-bit integers!");
5180 auto SplitLoadsMapI = SplitLoadsMap.
find(LI);
5181 std::vector<LoadInst *> *SplitLoads =
nullptr;
5182 if (SplitLoadsMapI != SplitLoadsMap.
end()) {
5183 SplitLoads = &SplitLoadsMapI->second;
5185 "Too few split loads for the number of splits in the store!");
5190 uint64_t PartOffset = 0, PartSize =
Offsets.Splits.front();
5193 auto *PartTy = Type::getIntNTy(Ty->
getContext(), PartSize * 8);
5195 auto *StorePartPtrTy =
SI->getPointerOperandType();
5200 PLoad = (*SplitLoads)[Idx];
5202 IRB.SetInsertPoint(LI);
5204 PLoad = IRB.CreateAlignedLoad(
5207 APInt(
DL.getIndexSizeInBits(AS), PartOffset),
5208 LoadPartPtrTy, LoadBasePtr->
getName() +
"."),
5211 PLoad->
copyMetadata(*LI, {LLVMContext::MD_mem_parallel_loop_access,
5212 LLVMContext::MD_access_group});
5216 IRB.SetInsertPoint(SI);
5217 auto AS =
SI->getPointerAddressSpace();
5218 StoreInst *PStore = IRB.CreateAlignedStore(
5221 APInt(
DL.getIndexSizeInBits(AS), PartOffset),
5222 StorePartPtrTy, StoreBasePtr->
getName() +
"."),
5225 PStore->
copyMetadata(*SI, {LLVMContext::MD_mem_parallel_loop_access,
5226 LLVMContext::MD_access_group});
5230 Slice(BaseOffset + PartOffset, BaseOffset + PartOffset + PartSize,
5234 <<
", " << NewSlices.
back().endOffset()
5235 <<
"): " << *PStore <<
"\n");
5245 PartOffset =
Offsets.Splits[Idx];
5247 PartSize = (Idx <
Size ?
Offsets.Splits[Idx] : StoreSize) - PartOffset;
5257 assert(OtherAI != &AI &&
"We can't re-split our own alloca!");
5258 ResplitPromotableAllocas.
insert(OtherAI);
5259 Worklist.insert(OtherAI);
5262 assert(OtherAI != &AI &&
"We can't re-split our own alloca!");
5263 Worklist.insert(OtherAI);
5278 DeadInsts.push_back(LI);
5280 DeadInsts.push_back(SI);
5289 AS.insert(NewSlices);
5293 for (
auto I = AS.begin(),
E = AS.end();
I !=
E; ++
I)
5299 PromotableAllocas.set_subtract(ResplitPromotableAllocas);
5336 bool IsIntegralPointerTy =
5337 EltTy->
isPointerTy() && !
DL.isNonIntegralPointerType(EltTy);
5339 !IsIntegralPointerTy)
5346 if (
DL.getTypeSizeInBits(EltTy) !=
DL.getTypeAllocSizeInBits(EltTy))
5350 TypeSize StructSize =
DL.getStructLayout(STy)->getSizeInBytes();
5351 TypeSize VectorSize =
DL.getTypeStoreSize(VTy);
5354 if (StructSize != VectorSize)
5357 auto IsIgnorableOrMemIntrinsicSlice = [](
const Slice &S) {
5360 auto *U = S.getUse();
5364 User *Usr = U->getUser();
5371 for (
const Slice &S :
P)
5372 if (!IsIgnorableOrMemIntrinsicSlice(S))
5375 for (
const Slice *S :
P.splitSliceTails())
5376 if (!IsIgnorableOrMemIntrinsicSlice(*S))
5393static std::tuple<Type *, bool, VectorType *>
5397 VectorType *SelectedVecTy,
bool SelectedIntWidening) {
5399 dbgs() <<
"selectPartitionType path=" << Path
5404 dbgs() <<
"<unnamed>";
5405 dbgs() <<
" partition=[" <<
P.beginOffset() <<
"," <<
P.endOffset()
5406 <<
") size=" <<
P.size();
5408 dbgs() <<
" alloc-size=" << AllocSize->getKnownMinValue();
5410 dbgs() <<
" chosen=" << *SelectedTy;
5412 dbgs() <<
" vec=" << *SelectedVecTy;
5413 dbgs() <<
" intwiden=" << SelectedIntWidening <<
"\n";
5431 if (VecTy && VecTy->getElementType()->isFloatingPointTy() &&
5432 VecTy->getElementCount().getFixedValue() > 1) {
5433 LogSelection(
"direct-fp-vecty", VecTy, VecTy,
false);
5434 return {VecTy,
false, VecTy};
5439 auto [CommonUseTy, LargestIntTy] =
5442 TypeSize CommonUseSize =
DL.getTypeAllocSize(CommonUseTy);
5448 LogSelection(
"common-type-vecty", VecTy, VecTy,
false);
5449 return {VecTy,
false, VecTy};
5452 LogSelection(
"common-type", CommonUseTy,
nullptr, IntWiden);
5453 return {CommonUseTy, IntWiden,
nullptr};
5460 P.beginOffset(),
P.size())) {
5464 if (TypePartitionTy->isArrayTy() &&
5465 TypePartitionTy->getArrayElementType()->isIntegerTy() &&
5466 DL.isLegalInteger(
P.size() * 8))
5470 LogSelection(
"type-partition-int-widen", TypePartitionTy,
nullptr,
true);
5471 return {TypePartitionTy,
true,
nullptr};
5474 LogSelection(
"type-partition-vecty", VecTy, VecTy,
false);
5475 return {VecTy,
false, VecTy};
5480 DL.getTypeAllocSize(LargestIntTy).getFixedValue() >=
P.size() &&
5482 LogSelection(
"largest-int-int-widen", LargestIntTy,
nullptr,
true);
5483 return {LargestIntTy,
true,
nullptr};
5488 if (AggregateToVector) {
5491 LogSelection(
"struct-fallback-vecty", VTy,
nullptr,
false);
5492 return {VTy,
false,
nullptr};
5498 LogSelection(
"type-partition-fallback", TypePartitionTy,
nullptr,
false);
5499 return {TypePartitionTy,
false,
nullptr};
5504 DL.getTypeAllocSize(LargestIntTy).getFixedValue() >=
P.size()) {
5505 LogSelection(
"largest-int-fallback", LargestIntTy,
nullptr,
false);
5506 return {LargestIntTy,
false,
nullptr};
5510 if (
DL.isLegalInteger(
P.size() * 8)) {
5512 LogSelection(
"legal-int-fallback", IntTy,
nullptr,
false);
5513 return {IntTy,
false,
nullptr};
5518 LogSelection(
"byte-array-fallback", ArrayTy,
nullptr,
false);
5519 return {ArrayTy,
false,
nullptr};
5532std::pair<AllocaInst *, uint64_t>
5533SROA::rewritePartition(AllocaInst &AI, AllocaSlices &AS, Partition &
P) {
5534 const DataLayout &
DL = AI.getDataLayout();
5536 auto [PartitionTy, IsIntegerWideningViable, VecTy] =
5546 if (PartitionTy == AI.getAllocatedType() &&
P.beginOffset() == 0) {
5556 const bool IsUnconstrained = Alignment <=
DL.getABITypeAlign(PartitionTy);
5557 NewAI =
new AllocaInst(
5558 PartitionTy, AI.getAddressSpace(),
nullptr,
5559 IsUnconstrained ?
DL.getPrefTypeAlign(PartitionTy) : Alignment,
5560 AI.
getName() +
".sroa." + Twine(
P.begin() - AS.begin()),
5567 LLVM_DEBUG(
dbgs() <<
"Rewriting alloca partition " <<
"[" <<
P.beginOffset()
5568 <<
"," <<
P.endOffset() <<
") to: " << *NewAI <<
"\n");
5573 unsigned PPWOldSize = PostPromotionWorklist.size();
5574 unsigned NumUses = 0;
5575 SmallSetVector<PHINode *, 8> PHIUsers;
5576 SmallSetVector<SelectInst *, 8> SelectUsers;
5579 DL, AS, *
this, AI, *NewAI, PartitionTy,
P.beginOffset(),
P.endOffset(),
5580 IsIntegerWideningViable, VecTy, PHIUsers, SelectUsers);
5581 bool Promotable =
true;
5583 if (
auto DeletedValues =
Rewriter.rewriteTreeStructuredMerge(
P)) {
5584 NumUses += DeletedValues->
size() + 1;
5585 for (
Value *V : *DeletedValues)
5586 DeadInsts.push_back(V);
5588 for (Slice *S :
P.splitSliceTails()) {
5592 for (Slice &S :
P) {
5598 NumAllocaPartitionUses += NumUses;
5599 MaxUsesPerAllocaPartition.updateMax(NumUses);
5603 for (PHINode *
PHI : PHIUsers)
5607 SelectUsers.
clear();
5612 NewSelectsToRewrite;
5614 for (SelectInst *Sel : SelectUsers) {
5615 std::optional<RewriteableMemOps>
Ops =
5616 isSafeSelectToSpeculate(*Sel, PreserveCFG);
5620 SelectUsers.clear();
5621 NewSelectsToRewrite.
clear();
5628 for (Use *U : AS.getDeadUsesIfPromotable()) {
5630 Value::dropDroppableUse(*U);
5633 DeadInsts.push_back(OldInst);
5635 if (PHIUsers.empty() && SelectUsers.empty()) {
5637 PromotableAllocas.insert(NewAI);
5642 SpeculatablePHIs.insert_range(PHIUsers);
5643 SelectsToRewrite.reserve(SelectsToRewrite.size() +
5644 NewSelectsToRewrite.
size());
5646 std::make_move_iterator(NewSelectsToRewrite.
begin()),
5647 std::make_move_iterator(NewSelectsToRewrite.
end())))
5648 SelectsToRewrite.insert(std::move(KV));
5649 Worklist.insert(NewAI);
5653 while (PostPromotionWorklist.size() > PPWOldSize)
5654 PostPromotionWorklist.pop_back();
5659 return {
nullptr, 0};
5664 Worklist.insert(NewAI);
5667 return {NewAI,
DL.getTypeSizeInBits(PartitionTy).getFixedValue()};
5711 int64_t BitExtractOffset) {
5713 bool HasFragment =
false;
5714 bool HasBitExtract =
false;
5723 HasBitExtract =
true;
5724 int64_t ExtractOffsetInBits =
Op.getArg(0);
5725 int64_t ExtractSizeInBits =
Op.getArg(1);
5734 assert(BitExtractOffset <= 0);
5735 int64_t AdjustedOffset = ExtractOffsetInBits + BitExtractOffset;
5741 if (AdjustedOffset < 0)
5744 Ops.push_back(
Op.getOp());
5745 Ops.push_back(std::max<int64_t>(0, AdjustedOffset));
5746 Ops.push_back(ExtractSizeInBits);
5749 Op.appendToVector(
Ops);
5754 if (HasFragment && HasBitExtract)
5757 if (!HasBitExtract) {
5776 std::optional<DIExpression::FragmentInfo> NewFragment,
5777 int64_t BitExtractAdjustment) {
5787 BitExtractAdjustment);
5788 if (!NewFragmentExpr)
5794 BeforeInst->
getParent()->insertDbgRecordBefore(DVR,
5807 BeforeInst->
getParent()->insertDbgRecordBefore(DVR,
5813 if (!NewAddr->
hasMetadata(LLVMContext::MD_DIAssignID)) {
5821 LLVM_DEBUG(
dbgs() <<
"Created new DVRAssign: " << *NewAssign <<
"\n");
5827bool SROA::splitAlloca(AllocaInst &AI, AllocaSlices &AS) {
5828 if (AS.begin() == AS.end())
5831 unsigned NumPartitions = 0;
5833 const DataLayout &
DL = AI.getModule()->getDataLayout();
5836 Changed |= presplitLoadsAndStores(AI, AS);
5844 bool IsSorted =
true;
5846 uint64_t AllocaSize = AI.getAllocationSize(
DL)->getFixedValue();
5847 const uint64_t MaxBitVectorSize = 1024;
5848 if (AllocaSize <= MaxBitVectorSize) {
5851 SmallBitVector SplittableOffset(AllocaSize + 1,
true);
5853 for (
unsigned O = S.beginOffset() + 1;
5854 O < S.endOffset() && O < AllocaSize; O++)
5855 SplittableOffset.reset(O);
5857 for (Slice &S : AS) {
5858 if (!S.isSplittable())
5861 if ((S.beginOffset() > AllocaSize || SplittableOffset[S.beginOffset()]) &&
5862 (S.endOffset() > AllocaSize || SplittableOffset[S.endOffset()]))
5867 S.makeUnsplittable();
5874 for (Slice &S : AS) {
5875 if (!S.isSplittable())
5878 if (S.beginOffset() == 0 && S.endOffset() >= AllocaSize)
5883 S.makeUnsplittable();
5898 Fragment(AllocaInst *AI, uint64_t O, uint64_t S)
5904 for (
auto &
P : AS.partitions()) {
5905 auto [NewAI, ActiveBits] = rewritePartition(AI, AS, P);
5909 uint64_t SizeOfByte = 8;
5911 uint64_t Size = std::min(ActiveBits, P.size() * SizeOfByte);
5912 Fragments.push_back(
5913 Fragment(NewAI, P.beginOffset() * SizeOfByte, Size));
5919 NumAllocaPartitions += NumPartitions;
5920 MaxPartitionsPerAlloca.updateMax(NumPartitions);
5924 auto MigrateOne = [&](DbgVariableRecord *DbgVariable) {
5929 const Value *DbgPtr = DbgVariable->getAddress();
5931 DbgVariable->getFragmentOrEntireVariable();
5934 int64_t CurrentExprOffsetInBytes = 0;
5935 SmallVector<uint64_t> PostOffsetOps;
5937 ->extractLeadingOffset(CurrentExprOffsetInBytes, PostOffsetOps))
5941 int64_t ExtractOffsetInBits = 0;
5945 ExtractOffsetInBits =
Op.getArg(0);
5950 DIBuilder DIB(*AI.getModule(),
false);
5951 for (
auto Fragment : Fragments) {
5952 int64_t OffsetFromLocationInBits;
5953 std::optional<DIExpression::FragmentInfo> NewDbgFragment;
5958 DL, &AI, Fragment.Offset, Fragment.Size, DbgPtr,
5959 CurrentExprOffsetInBytes * 8, ExtractOffsetInBits, VarFrag,
5960 NewDbgFragment, OffsetFromLocationInBits))
5966 if (NewDbgFragment && !NewDbgFragment->SizeInBits)
5971 if (!NewDbgFragment)
5972 NewDbgFragment = DbgVariable->getFragment();
5976 int64_t OffestFromNewAllocaInBits =
5977 OffsetFromLocationInBits - ExtractOffsetInBits;
5980 int64_t BitExtractOffset =
5981 std::min<int64_t>(0, OffestFromNewAllocaInBits);
5986 OffestFromNewAllocaInBits =
5987 std::max(int64_t(0), OffestFromNewAllocaInBits);
5993 DIExpression *NewExpr = DIExpression::get(AI.getContext(), PostOffsetOps);
5994 if (OffestFromNewAllocaInBits > 0) {
5995 int64_t OffsetInBytes = (OffestFromNewAllocaInBits + 7) / 8;
6001 auto RemoveOne = [DbgVariable](
auto *OldDII) {
6002 auto SameVariableFragment = [](
const auto *
LHS,
const auto *
RHS) {
6003 return LHS->getVariable() ==
RHS->getVariable() &&
6004 LHS->getDebugLoc()->getInlinedAt() ==
6005 RHS->getDebugLoc()->getInlinedAt();
6007 if (SameVariableFragment(OldDII, DbgVariable))
6008 OldDII->eraseFromParent();
6013 NewDbgFragment, BitExtractOffset);
6027void SROA::clobberUse(Use &U) {
6037 DeadInsts.push_back(OldI);
6059bool SROA::propagateStoredValuesToLoads(AllocaInst &AI, AllocaSlices &AS) {
6064 LLVM_DEBUG(
dbgs() <<
"Attempting to propagate values on " << AI <<
"\n");
6065 bool AllSameAndValid =
true;
6066 Type *PartitionType =
nullptr;
6067 SmallVector<Instruction *> Insts;
6068 uint64_t BeginOffset = 0;
6069 uint64_t EndOffset = 0;
6071 auto Flush = [&]() {
6072 if (AllSameAndValid && !Insts.
empty()) {
6073 LLVM_DEBUG(
dbgs() <<
"Propagate values on slice [" << BeginOffset <<
", "
6074 << EndOffset <<
")\n");
6076 SSAUpdater
SSA(&NewPHIs);
6078 BasicLoadAndStorePromoter Promoter(Insts,
SSA, PartitionType);
6079 Promoter.run(Insts);
6081 AllSameAndValid =
true;
6082 PartitionType =
nullptr;
6086 for (Slice &S : AS) {
6090 dbgs() <<
"Ignoring slice: ";
6091 AS.print(
dbgs(), &S);
6095 if (S.beginOffset() >= EndOffset) {
6097 BeginOffset = S.beginOffset();
6098 EndOffset = S.endOffset();
6099 }
else if (S.beginOffset() != BeginOffset || S.endOffset() != EndOffset) {
6100 if (AllSameAndValid) {
6102 dbgs() <<
"Slice does not match range [" << BeginOffset <<
", "
6103 << EndOffset <<
")";
6104 AS.print(
dbgs(), &S);
6106 AllSameAndValid =
false;
6108 EndOffset = std::max(EndOffset, S.endOffset());
6115 if (!LI->
isSimple() || (PartitionType && UserTy != PartitionType))
6116 AllSameAndValid =
false;
6117 PartitionType = UserTy;
6120 Type *UserTy =
SI->getValueOperand()->getType();
6121 if (!
SI->isSimple() || (PartitionType && UserTy != PartitionType))
6122 AllSameAndValid =
false;
6123 PartitionType = UserTy;
6126 AllSameAndValid =
false;
6139std::pair<
bool ,
bool >
6140SROA::runOnAlloca(AllocaInst &AI) {
6142 bool CFGChanged =
false;
6145 ++NumAllocasAnalyzed;
6148 if (AI.use_empty()) {
6149 AI.eraseFromParent();
6153 const DataLayout &
DL = AI.getDataLayout();
6156 std::optional<TypeSize>
Size = AI.getAllocationSize(
DL);
6157 if (AI.isArrayAllocation() || !
Size ||
Size->isScalable() ||
Size->isZero())
6162 IRBuilderTy IRB(&AI);
6163 AggLoadStoreRewriter AggRewriter(
DL, IRB);
6164 Changed |= AggRewriter.rewrite(AI);
6167 AllocaSlices AS(
DL, AI);
6172 if (AS.isEscapedReadOnly()) {
6173 Changed |= propagateStoredValuesToLoads(AI, AS);
6178 for (Instruction *DeadUser : AS.getDeadUsers()) {
6180 for (Use &DeadOp : DeadUser->operands())
6187 DeadInsts.push_back(DeadUser);
6190 for (Use *DeadOp : AS.getDeadOperands()) {
6191 clobberUse(*DeadOp);
6196 if (AS.begin() == AS.end())
6199 Changed |= splitAlloca(AI, AS);
6202 while (!SpeculatablePHIs.empty())
6206 auto RemainingSelectsToRewrite = SelectsToRewrite.takeVector();
6207 while (!RemainingSelectsToRewrite.empty()) {
6208 const auto [
K,
V] = RemainingSelectsToRewrite.pop_back_val();
6225bool SROA::deleteDeadInstructions(
6226 SmallPtrSetImpl<AllocaInst *> &DeletedAllocas) {
6228 while (!DeadInsts.empty()) {
6238 DeletedAllocas.
insert(AI);
6240 OldDII->eraseFromParent();
6246 for (Use &Operand :
I->operands())
6251 DeadInsts.push_back(U);
6255 I->eraseFromParent();
6265bool SROA::promoteAllocas() {
6266 if (PromotableAllocas.empty())
6273 NumPromoted += PromotableAllocas.size();
6274 PromoteMemToReg(PromotableAllocas.getArrayRef(), DTU->getDomTree(), AC);
6277 PromotableAllocas.clear();
6281std::pair<
bool ,
bool > SROA::runSROA(Function &
F) {
6284 const DataLayout &
DL =
F.getDataLayout();
6289 std::optional<TypeSize>
Size = AI->getAllocationSize(
DL);
6291 PromotableAllocas.insert(AI);
6293 Worklist.insert(AI);
6298 bool CFGChanged =
false;
6301 SmallPtrSet<AllocaInst *, 4> DeletedAllocas;
6304 while (!Worklist.empty()) {
6305 auto [IterationChanged, IterationCFGChanged] =
6306 runOnAlloca(*Worklist.pop_back_val());
6308 CFGChanged |= IterationCFGChanged;
6310 Changed |= deleteDeadInstructions(DeletedAllocas);
6314 if (!DeletedAllocas.
empty()) {
6315 Worklist.set_subtract(DeletedAllocas);
6316 PostPromotionWorklist.set_subtract(DeletedAllocas);
6317 PromotableAllocas.set_subtract(DeletedAllocas);
6318 DeletedAllocas.
clear();
6324 Worklist = PostPromotionWorklist;
6325 PostPromotionWorklist.clear();
6326 }
while (!Worklist.empty());
6328 assert((!CFGChanged ||
Changed) &&
"Can not only modify the CFG.");
6329 assert((!CFGChanged || !PreserveCFG) &&
6330 "Should not have modified the CFG when told to preserve it.");
6333 for (
auto &BB :
F) {
6346 SROA(&
F.getContext(), &DTU, &AC, Options).runSROA(
F);
6359 OS, MapClassName2PassName);
6363 if (Options.AggregateToVector)
6364 OS <<
";aggregate-to-vector";
6385 if (skipFunction(
F))
6388 DominatorTree &DT = getAnalysis<DominatorTreeWrapperPass>().getDomTree();
6390 getAnalysis<AssumptionCacheTracker>().getAssumptionCache(
F);
6396 void getAnalysisUsage(AnalysisUsage &AU)
const override {
6403 StringRef getPassName()
const override {
return "SROA"; }
6408char SROALegacyPass::ID = 0;
6413 AggregateToVector));
6417 "Scalar Replacement Of Aggregates",
false,
false)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
#define LLVM_DUMP_METHOD
Mark debug helper function definitions like dump() that should not be stripped from debug builds.
This file contains the declarations for the subclasses of Constant, which represent the different fla...
DXIL Forward Handle Accesses
This file defines the DenseMap class.
static bool runOnFunction(Function &F, bool PostInlining)
This is the interface for a simple mod/ref and alias analysis over globals.
Module.h This file contains the declarations for the Module class.
This header defines various interfaces for pass management in LLVM.
This defines the Use class.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
print mir2vec MIR2Vec Vocabulary Printer Pass
This file implements a map that provides insertion order iteration.
static std::optional< AllocFnsTy > getAllocationSize(const CallBase *CB, const TargetLibraryInfo *TLI)
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t IntrinsicInst * II
PassBuilder PB(Machine, PassOpts->PTO, std::nullopt, &PIC)
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
This file defines the PointerIntPair class.
This file provides a collection of visitors which walk the (instruction) uses of a pointer.
const SmallVectorImpl< MachineOperand > & Cond
Remove Loads Into Fake Uses
bool isDead(const MachineInstr &MI, const MachineRegisterInfo &MRI)
Func getContext().diagnose(DiagnosticInfoUnsupported(Func
static void visit(BasicBlock &Start, std::function< bool(BasicBlock *)> op)
static void migrateDebugInfo(AllocaInst *OldAlloca, bool IsSplit, uint64_t OldAllocaOffsetInBits, uint64_t SliceSizeInBits, Instruction *OldInst, Instruction *Inst, Value *Dest, Value *Value, const DataLayout &DL)
Find linked dbg.assign and generate a new one with the correct FragmentInfo.
static VectorType * isVectorPromotionViable(Partition &P, const DataLayout &DL, unsigned VScale)
Test whether the given alloca partitioning and range of slices can be promoted to a vector.
static Align getAdjustedAlignment(Instruction *I, uint64_t Offset)
Compute the adjusted alignment for a load or store from an offset.
static VectorType * checkVectorTypesForPromotion(Partition &P, const DataLayout &DL, SmallVectorImpl< VectorType * > &CandidateTys, bool HaveCommonEltTy, Type *CommonEltTy, bool HaveVecPtrTy, bool HaveCommonVecPtrTy, VectorType *CommonVecPtrTy, unsigned VScale)
Test whether any vector type in CandidateTys is viable for promotion.
static std::pair< Type *, IntegerType * > findCommonType(AllocaSlices::const_iterator B, AllocaSlices::const_iterator E, uint64_t EndOffset)
Walk the range of a partitioning looking for a common type to cover this sequence of slices.
static Type * stripAggregateTypeWrapping(const DataLayout &DL, Type *Ty)
Strip aggregate type wrapping.
static FragCalcResult calculateFragment(DILocalVariable *Variable, uint64_t NewStorageSliceOffsetInBits, uint64_t NewStorageSliceSizeInBits, std::optional< DIExpression::FragmentInfo > StorageFragment, std::optional< DIExpression::FragmentInfo > CurrentFragment, DIExpression::FragmentInfo &Target)
static DIExpression * createOrReplaceFragment(const DIExpression *Expr, DIExpression::FragmentInfo Frag, int64_t BitExtractOffset)
Create or replace an existing fragment in a DIExpression with Frag.
static Value * insertInteger(const DataLayout &DL, IRBuilderTy &IRB, Value *Old, Value *V, uint64_t Offset, const Twine &Name)
static bool isVectorPromotionViableForSlice(Partition &P, const Slice &S, VectorType *Ty, uint64_t ElementSize, const DataLayout &DL, unsigned VScale)
Test whether the given slice use can be promoted to a vector.
static Value * getAdjustedPtr(IRBuilderTy &IRB, const DataLayout &DL, Value *Ptr, APInt Offset, Type *PointerTy, const Twine &NamePrefix)
Compute an adjusted pointer from Ptr by Offset bytes where the resulting pointer has PointerTy.
static bool isIntegerWideningViableForSlice(const Slice &S, uint64_t AllocBeginOffset, Type *AllocaTy, const DataLayout &DL, bool &WholeAllocaOp)
Test whether a slice of an alloca is valid for integer widening.
static Value * extractVector(IRBuilderTy &IRB, Value *V, unsigned BeginIndex, unsigned EndIndex, const Twine &Name)
static Value * foldPHINodeOrSelectInst(Instruction &I)
A helper that folds a PHI node or a select.
static bool rewriteSelectInstMemOps(SelectInst &SI, const RewriteableMemOps &Ops, IRBuilderTy &IRB, DomTreeUpdater *DTU)
static void rewriteMemOpOfSelect(SelectInst &SI, T &I, SelectHandSpeculativity Spec, DomTreeUpdater &DTU)
static Value * foldSelectInst(SelectInst &SI)
bool isKillAddress(const DbgVariableRecord *DVR)
static Value * insertVector(IRBuilderTy &IRB, Value *Old, Value *V, unsigned BeginIndex, const Twine &Name)
static bool isIntegerWideningViable(Partition &P, Type *AllocaTy, const DataLayout &DL)
Test whether the given alloca partition's integer operations can be widened to promotable ones.
static void speculatePHINodeLoads(IRBuilderTy &IRB, PHINode &PN)
static VectorType * createAndCheckVectorTypesForPromotion(SetVector< Type * > &OtherTys, ArrayRef< VectorType * > CandidateTysCopy, function_ref< void(Type *)> CheckCandidateType, Partition &P, const DataLayout &DL, SmallVectorImpl< VectorType * > &CandidateTys, bool &HaveCommonEltTy, Type *&CommonEltTy, bool &HaveVecPtrTy, bool &HaveCommonVecPtrTy, VectorType *&CommonVecPtrTy, unsigned VScale)
static DebugVariable getAggregateVariable(DbgVariableRecord *DVR)
static std::tuple< Type *, bool, VectorType * > selectPartitionType(Partition &P, const DataLayout &DL, AllocaInst &AI, LLVMContext &C, bool AggregateToVector)
Select a partition type for an alloca partition.
static bool isSafePHIToSpeculate(PHINode &PN)
PHI instructions that use an alloca and are subsequently loaded can be rewritten to load both input p...
static FixedVectorType * tryCanonicalizeStructToVector(StructType *STy, Partition &P, const DataLayout &DL)
Try to canonicalize a homogeneous struct partition to a vector type.
static Value * extractInteger(const DataLayout &DL, IRBuilderTy &IRB, Value *V, IntegerType *Ty, uint64_t Offset, const Twine &Name)
static void insertNewDbgInst(DIBuilder &DIB, DbgVariableRecord *Orig, AllocaInst *NewAddr, DIExpression *NewAddrExpr, Instruction *BeforeInst, std::optional< DIExpression::FragmentInfo > NewFragment, int64_t BitExtractAdjustment)
Insert a new DbgRecord.
static void speculateSelectInstLoads(SelectInst &SI, LoadInst &LI, IRBuilderTy &IRB)
static Value * mergeTwoVectors(Value *V0, Value *V1, const DataLayout &DL, Type *NewAIEltTy, IRBuilder<> &Builder)
This function takes two vector values and combines them into a single vector by concatenating their e...
const DIExpression * getAddressExpression(const DbgVariableRecord *DVR)
static Type * getTypePartition(const DataLayout &DL, Type *Ty, uint64_t Offset, uint64_t Size)
Try to find a partition of the aggregate type passed in for a given offset and size.
static bool canConvertValue(const DataLayout &DL, Type *OldTy, Type *NewTy, unsigned VScale=0)
Test whether we can convert a value from the old to the new type.
static SelectHandSpeculativity isSafeLoadOfSelectToSpeculate(LoadInst &LI, SelectInst &SI, bool PreserveCFG)
This file provides the interface for LLVM's Scalar Replacement of Aggregates pass.
This file implements a set that has insertion order iteration characteristics.
This file implements the SmallBitVector class.
This file defines the SmallPtrSet class.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
static SymbolRef::Type getType(const Symbol *Sym)
static unsigned getBitWidth(Type *Ty, const DataLayout &DL)
Returns the bitwidth of the given scalar or pointer type.
Virtual Register Rewriter
Builder for the alloca slices.
SliceBuilder(const DataLayout &DL, AllocaInst &AI, AllocaSlices &AS)
An iterator over partitions of the alloca's slices.
bool operator==(const partition_iterator &RHS) const
friend class AllocaSlices
partition_iterator & operator++()
Class for arbitrary precision integers.
an instruction to allocate memory on the stack
LLVM_ABI bool isStaticAlloca() const
Return true if this alloca is in the entry block of the function and is a constant size.
Align getAlign() const
Return the alignment of the memory that is being allocated by the instruction.
PointerType * getType() const
Overload to return most specific pointer type.
Type * getAllocatedType() const
Return the type that is being allocated by the instruction.
LLVM_ABI std::optional< TypeSize > getAllocationSize(const DataLayout &DL) const
Get allocation size in bytes.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
static LLVM_ABI ArrayType * get(Type *ElementType, uint64_t NumElements)
This static method is the primary way to construct an ArrayType.
A function analysis which provides an AssumptionCache.
An immutable pass that tracks lazily created AssumptionCache objects.
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
iterator begin()
Instruction iterator methods.
InstListType::iterator iterator
Instruction iterators...
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
Represents analyses that only rely on functions' control flow.
LLVM_ABI CaptureInfo getCaptureInfo(unsigned OpNo) const
Return which pointer components this operand may capture.
bool onlyReadsMemory(unsigned OpNo) const
bool isDataOperand(const Use *U) const
This is the shared class of boolean and integer constants.
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static DIAssignID * getDistinct(LLVMContext &Context)
LLVM_ABI DbgRecord * insertDbgAssign(Instruction *LinkedInstr, Value *Val, DILocalVariable *SrcVar, DIExpression *ValExpr, Value *Addr, DIExpression *AddrExpr, const DILocation *DL)
Insert a new dbg_assign record.
iterator_range< expr_op_iterator > expr_ops() const
DbgVariableFragmentInfo FragmentInfo
LLVM_ABI bool startsWithDeref() const
Return whether the first element a DW_OP_deref.
static LLVM_ABI bool calculateFragmentIntersect(const DataLayout &DL, const Value *SliceStart, uint64_t SliceOffsetInBits, uint64_t SliceSizeInBits, const Value *DbgPtr, int64_t DbgPtrOffsetInBits, int64_t DbgExtractOffsetInBits, DIExpression::FragmentInfo VarFrag, std::optional< DIExpression::FragmentInfo > &Result, int64_t &OffsetFromLocationInBits)
Computes a fragment, bit-extract operation if needed, and new constant offset to describe a part of a...
static LLVM_ABI std::optional< DIExpression * > createFragmentExpression(const DIExpression *Expr, unsigned OffsetInBits, unsigned SizeInBits)
Create a DIExpression to describe one part of an aggregate variable that is fragmented across multipl...
static LLVM_ABI DIExpression * prepend(const DIExpression *Expr, uint8_t Flags, int64_t Offset=0)
Prepend DIExpr with a deref and offset operation and optionally turn it into a stack value or/and an ...
A parsed version of the target data layout string in and methods for querying it.
LLVM_ABI void moveBefore(DbgRecord *MoveBefore)
DebugLoc getDebugLoc() const
void setDebugLoc(DebugLoc Loc)
Record of a variable value-assignment, aka a non instruction representation of the dbg....
LLVM_ABI void setKillAddress()
Kill the address component.
LLVM_ABI bool isKillLocation() const
LocationType getType() const
LLVM_ABI bool isKillAddress() const
Check whether this kills the address component.
LLVM_ABI void replaceVariableLocationOp(Value *OldValue, Value *NewValue, bool AllowEmpty=false)
Value * getValue(unsigned OpIdx=0) const
static LLVM_ABI DbgVariableRecord * createLinkedDVRAssign(Instruction *LinkedInstr, Value *Val, DILocalVariable *Variable, DIExpression *Expression, Value *Address, DIExpression *AddressExpression, const DILocation *DI)
LLVM_ABI void setAssignId(DIAssignID *New)
DIExpression * getExpression() const
static LLVM_ABI DbgVariableRecord * createDVRDeclare(Value *Address, DILocalVariable *DV, DIExpression *Expr, const DILocation *DI)
static LLVM_ABI DbgVariableRecord * createDbgVariableRecord(Value *Location, DILocalVariable *DV, DIExpression *Expr, const DILocation *DI)
DILocalVariable * getVariable() const
LLVM_ABI void setKillLocation()
bool isDbgDeclare() const
void setAddress(Value *V)
DIExpression * getAddressExpression() const
LLVM_ABI DILocation * getInlinedAt() const
Identifies a unique instance of a variable.
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
iterator find(const_arg_type_t< KeyT > Val)
size_type count(const_arg_type_t< KeyT > Val) const
Return 1 if the specified key is in the map, 0 otherwise.
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Analysis pass which computes a DominatorTree.
Legacy analysis pass which computes a DominatorTree.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
Class to represent fixed width SIMD vectors.
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
FunctionPass class - This class is used to implement most global optimizations.
unsigned getVScaleValue() const
Return the value for vscale based on the vscale_range attribute or 0 when unknown.
const BasicBlock & getEntryBlock() const
LLVM_ABI bool accumulateConstantOffset(const DataLayout &DL, APInt &Offset, function_ref< bool(Value &, APInt &)> ExternalAnalysis=nullptr) const
Accumulate the constant address offset of this GEP if possible.
Value * getPointerOperand()
iterator_range< op_iterator > indices()
Type * getSourceElementType() const
LLVM_ABI GEPNoWrapFlags getNoWrapFlags() const
Get the nowrap flags for the GEP instruction.
This provides the default implementation of the IRBuilder 'InsertHelper' method that is called whenev...
virtual void InsertHelper(Instruction *I, const Twine &Name, BasicBlock::iterator InsertPt) const
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
Base class for instruction visitors.
LLVM_ABI unsigned getNumSuccessors() const LLVM_READONLY
Return the number of successors that this instruction has.
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void setAAMetadata(const AAMDNodes &N)
Sets the AA metadata on this instruction from the AAMDNodes structure.
bool hasMetadata() const
Return true if this instruction has any metadata attached to it.
LLVM_ABI bool isAtomic() const LLVM_READONLY
Return true if this instruction has an AtomicOrdering of unordered or higher.
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
Instruction * user_back()
Specialize the methods defined in Value, as we know that an instruction can only be used by other ins...
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this Instruction.
LLVM_ABI bool mayHaveSideEffects() const LLVM_READONLY
Return true if the instruction may have side effects.
LLVM_ABI bool comesBefore(const Instruction *Other) const
Given an instruction Other in the same basic block as this instruction, return true if this instructi...
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI AAMDNodes getAAMetadata() const
Returns the AA metadata for this instruction.
void setDebugLoc(DebugLoc Loc)
Set the debug location information for this instruction.
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
Class to represent integer types.
@ MAX_INT_BITS
Maximum number of bits that can be specified.
unsigned getBitWidth() const
Get the number of bits in this IntegerType.
A wrapper class for inspecting calls to intrinsic functions.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
unsigned getPointerAddressSpace() const
Returns the address space of the pointer operand.
void setAlignment(Align Align)
Value * getPointerOperand()
bool isVolatile() const
Return true if this is a load from a volatile memory location.
void setAtomic(AtomicOrdering Ordering, SyncScope::ID SSID=SyncScope::System)
Sets the ordering constraint and the synchronization scope ID of this load instruction.
AtomicOrdering getOrdering() const
Returns the ordering constraint of this load instruction.
Type * getPointerOperandType() const
static unsigned getPointerOperandIndex()
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this load instruction.
Align getAlign() const
Return the alignment of the access that is being performed.
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
LLVMContext & getContext() const
LLVM_ABI StringRef getName() const
Return the name of the corresponding LLVM basic block, or an empty string.
This is the common base class for memset/memcpy/memmove.
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
op_range incoming_values()
BasicBlock * getIncomingBlock(unsigned i) const
Return incoming basic block number i.
Value * getIncomingValue(unsigned i) const
Return incoming value number x.
int getBasicBlockIndex(const BasicBlock *BB) const
Return the first index of the specified basic block in the value list for this PHI.
unsigned getNumIncomingValues() const
Return the number of incoming edges.
static PHINode * Create(Type *Ty, unsigned NumReservedValues, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Constructors - NumReservedValues is a hint for the number of incoming edges that this phi node will h...
static LLVM_ABI PassRegistry * getPassRegistry()
getPassRegistry - Access the global registry object, which is automatically initialized at applicatio...
PointerIntPair - This class implements a pair of a pointer and small integer.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
A set of analyses that are preserved following a run of a transformation pass.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
PreservedAnalyses & preserveSet()
Mark an analysis set as preserved.
PreservedAnalyses & preserve()
Mark an analysis as preserved.
PtrUseVisitor(const DataLayout &DL)
LLVM_ABI SROAPass(SROAOptions Options)
If PreserveCFG is set, then the pass is not allowed to modify CFG in any way, even if it would update...
LLVM_ABI PreservedAnalyses run(Function &F, FunctionAnalysisManager &AM)
Run the pass over the function.
LLVM_ABI void printPipeline(raw_ostream &OS, function_ref< StringRef(StringRef)> MapClassName2PassName)
Helper class for SSA formation on a set of values defined in multiple blocks.
This class represents the LLVM 'select' instruction.
A vector that has set insertion semantics.
size_type size() const
Determine the number of elements in the SetVector.
void clear()
Completely clear the SetVector.
bool insert(const value_type &X)
Insert a new element into the SetVector.
bool erase(PtrType Ptr)
Remove pointer from the set.
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
iterator erase(const_iterator CI)
typename SuperClass::const_iterator const_iterator
typename SuperClass::iterator iterator
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
void setAlignment(Align Align)
Value * getValueOperand()
static unsigned getPointerOperandIndex()
Value * getPointerOperand()
void setAtomic(AtomicOrdering Ordering, SyncScope::ID SSID=SyncScope::System)
Sets the ordering constraint and the synchronization scope ID of this store instruction.
Represent a constant reference to a string, i.e.
static constexpr size_t npos
constexpr StringRef substr(size_t Start, size_t N=npos) const
Return a reference to the substring from [Start, Start + N).
size_t rfind(char C, size_t From=npos) const
Search for the last character C in the string.
size_t find(char C, size_t From=0) const
Search for the first character C in the string.
LLVM_ABI size_t find_first_not_of(char C, size_t From=0) const
Find the first character in the string that is not C or npos if not found.
Used to lazily calculate structure layout information for a target machine, based on the DataLayout s...
TypeSize getSizeInBytes() const
LLVM_ABI unsigned getElementContainingOffset(uint64_t FixedOffset) const
Given a valid byte offset into the structure, returns the structure index that contains it.
TypeSize getElementOffset(unsigned Idx) const
TypeSize getSizeInBits() const
Class to represent struct types.
static LLVM_ABI StructType * get(LLVMContext &Context, ArrayRef< Type * > Elements, bool isPacked=false)
This static method is the primary way to create a literal StructType.
element_iterator element_end() const
ArrayRef< Type * > elements() const
element_iterator element_begin() const
unsigned getNumElements() const
Random access to the elements.
Type * getElementType(unsigned N) const
Type::subtype_iterator element_iterator
Target - Wrapper for Target specific information.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
static constexpr TypeSize getFixed(ScalarTy ExactSize)
The instances of the Type class are immutable: once they are created, they are never changed.
LLVM_ABI unsigned getIntegerBitWidth() const
bool isPointerTy() const
True if this is an instance of PointerType.
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
bool isSingleValueType() const
Return true if the type is a valid type for a register in codegen.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
bool isStructTy() const
True if this is an instance of StructType.
bool isTargetExtTy() const
Return true if this is a target extension type.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isPtrOrPtrVectorTy() const
Return true if this is a pointer type or a vector of pointer types.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
const Use & getOperandUse(unsigned i) const
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
user_iterator user_begin()
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
LLVMContext & getContext() const
All values hold a context through their type.
LLVM_ABI const Value * stripInBoundsOffsets(function_ref< void(const Value *)> Func=[](const Value *) {}) const
Strip off pointer casts and inbounds GEPs.
iterator_range< user_iterator > users()
LLVM_ABI void dropDroppableUsesIn(User &Usr)
Remove every use of this value in User that can safely be removed.
LLVM_ABI const Value * stripAndAccumulateConstantOffsets(const DataLayout &DL, APInt &Offset, bool AllowNonInbounds, bool AllowInvariantGroup=false, function_ref< bool(Value &Value, APInt &Offset)> ExternalAnalysis=nullptr, bool LookThroughIntToPtr=false) const
Accumulate the constant offset this value has compared to a base pointer.
iterator_range< use_iterator > uses()
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
static VectorType * getWithSizeAndScalar(VectorType *SizeTy, Type *EltTy)
This static method attempts to construct a VectorType with the same size-in-bits as SizeTy but with a...
static LLVM_ABI bool isValidElementType(Type *ElemTy)
Return true if the specified type is valid as a element type.
constexpr ScalarTy getFixedValue() const
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
An efficient, type-erasing, non-owning reference to a callable.
const ParentTy * getParent() const
self_iterator getIterator()
NodeTy * getNextNode()
Get the next node, or nullptr for the list tail.
CRTP base class which implements the entire standard iterator facade in terms of a minimal subset of ...
A range adaptor for a pair of iterators.
This class implements an extremely fast bulk output stream that can only output to a stream.
This provides a very simple, boring adaptor for a begin and end iterator into a range type.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char IsVolatile[]
Key for Kernel::Arg::Metadata::mIsVolatile.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
@ BasicBlock
Various leaf nodes.
SmallVector< DbgVariableRecord * > getDVRAssignmentMarkers(const Instruction *Inst)
Return a range of dbg_assign records for which Inst performs the assignment they encode.
LLVM_ABI void deleteAssignmentMarkers(const Instruction *Inst)
Delete the llvm.dbg.assign intrinsics linked to Inst.
initializer< Ty > init(const Ty &Val)
@ DW_OP_LLVM_extract_bits_zext
Only used in LLVM metadata.
@ DW_OP_LLVM_fragment
Only used in LLVM metadata.
@ DW_OP_LLVM_extract_bits_sext
Only used in LLVM metadata.
@ User
could "use" a pointer
NodeAddr< PhiNode * > Phi
NodeAddr< UseNode * > Use
friend class Instruction
Iterator for Instructions in a `BasicBlock.
LLVM_ABI iterator begin() const
unsigned getNumElements(Type *Ty)
This is an optimization pass for GlobalISel generic memory operations.
static cl::opt< bool > SROASkipMem2Reg("sroa-skip-mem2reg", cl::init(false), cl::Hidden)
Disable running mem2reg during SROA in order to test or debug SROA.
void dump(const SparseBitVector< ElementSize > &LHS, raw_ostream &out)
bool operator<(int64_t V1, const APSInt &V2)
void stable_sort(R &&Range)
LLVM_ABI bool RemoveRedundantDbgInstrs(BasicBlock *BB)
Try to remove redundant dbg.value instructions from given basic block.
LLVM_ABI cl::opt< bool > ProfcheckDisableMetadataFixes
UnaryFunction for_each(R &&Range, UnaryFunction F)
Provide wrappers to std::for_each which take ranges instead of having to pass begin/end explicitly.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Printable print(const GCNRegPressure &RP, const GCNSubtarget *ST=nullptr, unsigned DynamicVGPRBlockSize=0)
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
LLVM_ABI void PromoteMemToReg(ArrayRef< AllocaInst * > Allocas, DominatorTree &DT, AssumptionCache *AC=nullptr)
Promote the specified list of alloca instructions into scalar registers, inserting PHI nodes as appro...
LLVM_ABI bool isAssumeLikeIntrinsic(const Instruction *I)
Return true if it is an intrinsic that cannot be speculated but also cannot trap.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
auto successors(const MachineBasicBlock *BB)
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
bool operator!=(uint64_t V1, const APInt &V2)
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
LLVM_ABI std::optional< RegOrConstant > getVectorSplat(const MachineInstr &MI, const MachineRegisterInfo &MRI)
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
auto unique(Range &&R, Predicate P)
bool operator==(const AddressRangeValuePair &LHS, const AddressRangeValuePair &RHS)
LLVM_ABI bool isAllocaPromotable(const AllocaInst *AI)
Return true if this alloca is legal for promotion.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
auto dyn_cast_or_null(const Y &Val)
void erase(Container &C, ValueType V)
Wrapper function to remove a value from a container:
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI bool isInstructionTriviallyDead(Instruction *I, const TargetLibraryInfo *TLI=nullptr)
Return true if the result produced by the instruction is not used, and the instruction will return.
bool capturesFullProvenance(CaptureComponents CC)
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
void sort(IteratorTy Start, IteratorTy End)
LLVM_ABI void SplitBlockAndInsertIfThenElse(Value *Cond, BasicBlock::iterator SplitBefore, Instruction **ThenTerm, Instruction **ElseTerm, MDNode *BranchWeights=nullptr, DomTreeUpdater *DTU=nullptr, LoopInfo *LI=nullptr)
SplitBlockAndInsertIfThenElse is similar to SplitBlockAndInsertIfThen, but also creates the ElseBlock...
LLVM_ABI bool isSafeToLoadUnconditionally(Value *V, Align Alignment, const APInt &Size, const DataLayout &DL, Instruction *ScanFrom, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr, const TargetLibraryInfo *TLI=nullptr)
Return true if we know that executing a load from this value cannot trap.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
LLVM_ABI void initializeSROALegacyPassPass(PassRegistry &)
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
LLVM_ABI TinyPtrVector< DbgVariableRecord * > findDVRValues(Value *V)
As above, for DVRValues.
LLVM_ABI void llvm_unreachable_internal(const char *msg=nullptr, const char *file=nullptr, unsigned line=0)
This function calls abort(), and prints the optional message to stderr.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
constexpr int PoisonMaskElem
iterator_range(Container &&) -> iterator_range< llvm::detail::IterOfRange< Container > >
IRBuilder(LLVMContext &, FolderTy, InserterTy, MDNode *, ArrayRef< OperandBundleDef >) -> IRBuilder< FolderTy, InserterTy >
LLVM_ABI bool isAssignmentTrackingEnabled(const Module &M)
Return true if assignment tracking is enabled for module M.
DWARFExpression::Operation Op
LLVM_ABI FunctionPass * createSROAPass(bool PreserveCFG=true, bool AggregateToVector=false)
ArrayRef(const T &OneElt) -> ArrayRef< T >
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
constexpr auto seq(T Begin, T End)
Iterate over an integral type from Begin up to - but not including - End.
void erase_if(Container &C, UnaryPredicate P)
Provide a container algorithm similar to C++ Library Fundamentals v2's erase_if which is equivalent t...
LLVM_ABI TinyPtrVector< DbgVariableRecord * > findDVRDeclares(Value *V)
Finds dbg.declare records declaring local variables as living in the memory that 'V' points to.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Align commonAlignment(Align A, uint64_t Offset)
Returns the alignment that satisfies both alignments.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
LLVM_ABI Instruction * SplitBlockAndInsertIfThen(Value *Cond, BasicBlock::iterator SplitBefore, bool Unreachable, MDNode *BranchWeights=nullptr, DomTreeUpdater *DTU=nullptr, LoopInfo *LI=nullptr, BasicBlock *ThenBlock=nullptr)
Split the containing block at the specified instruction - everything before SplitBefore stays in the ...
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
LLVM_ABI llvm::SmallVector< int, 16 > createSequentialMask(unsigned Start, unsigned NumInts, unsigned NumUndefs)
Create a sequential shuffle mask.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
A collection of metadata nodes that might be associated with a memory access used by the alias-analys...
AAMDNodes shift(size_t Offset) const
Create a new AAMDNode that describes this AAMDNode after applying a constant offset to the start of t...
LLVM_ABI AAMDNodes adjustForAccess(unsigned AccessSize)
Create a new AAMDNode for accessing AccessSize bytes of this AAMDNode.
This struct is a compact representation of a valid (non-zero power of two) alignment.
Describes an element of a Bitfield.
static Bitfield::Type get(StorageType Packed)
Unpacks the field from the Packed value.
static void set(StorageType &Packed, typename Bitfield::Type Value)
Sets the typed value in the provided Packed value.
A CRTP mix-in to automatically provide informational APIs needed for passes.