243#include "llvm/IR/IntrinsicsAMDGPU.h"
263#define DEBUG_TYPE "amdgpu-lower-buffer-fat-pointers"
288 Type *remapType(
Type *SrcTy)
override;
289 void clear() { Map.clear(); }
295class BufferFatPtrToIntTypeMap :
public BufferFatPtrTypeLoweringBase {
296 using BufferFatPtrTypeLoweringBase::BufferFatPtrTypeLoweringBase;
306class BufferFatPtrToStructTypeMap :
public BufferFatPtrTypeLoweringBase {
307 using BufferFatPtrTypeLoweringBase::BufferFatPtrTypeLoweringBase;
316Type *BufferFatPtrTypeLoweringBase::remapTypeImpl(
Type *Ty) {
322 return *
Entry = remapScalar(PT);
328 return *
Entry = remapVector(VT);
336 bool IsUniqued = !TyAsStruct || TyAsStruct->
isLiteral();
345 Type *NewElem = remapTypeImpl(OldElem);
346 ElementTypes[
I] = NewElem;
347 Changed |= (OldElem != NewElem);
355 return *
Entry = ArrayType::get(ElementTypes[0], ArrTy->getNumElements());
357 return *
Entry = FunctionType::get(ElementTypes[0],
367 SmallString<16>
Name(STy->getName());
375Type *BufferFatPtrTypeLoweringBase::remapType(
Type *SrcTy) {
376 return remapTypeImpl(SrcTy);
379Type *BufferFatPtrToStructTypeMap::remapScalar(PointerType *PT) {
380 LLVMContext &Ctx = PT->getContext();
385Type *BufferFatPtrToStructTypeMap::remapVector(VectorType *VT) {
386 ElementCount
EC = VT->getElementCount();
387 LLVMContext &Ctx = VT->getContext();
406 if (!ST->isLiteral() || ST->getNumElements() != 2)
412 return MaybeRsrc && MaybeOff &&
421 return isBufferFatPtrOrVector(U.get()->getType());
434class StoreFatPtrsAsIntsAndExpandMemcpyVisitor
435 :
public InstVisitor<StoreFatPtrsAsIntsAndExpandMemcpyVisitor, bool> {
436 BufferFatPtrToIntTypeMap *TypeMap;
441 const TargetTransformInfo *
TTI;
453 StoreFatPtrsAsIntsAndExpandMemcpyVisitor(BufferFatPtrToIntTypeMap *TypeMap,
454 const DataLayout &
DL,
456 : TypeMap(TypeMap), IRB(Ctx, InstSimplifyFolder(
DL)) {}
458 ScalarEvolution *SE);
460 bool visitInstruction(Instruction &
I) {
return false; }
461 bool visitAllocaInst(AllocaInst &
I);
462 bool visitLoadInst(LoadInst &LI);
463 bool visitStoreInst(StoreInst &SI);
464 bool visitGetElementPtrInst(GetElementPtrInst &
I);
466 bool visitMemCpyInst(MemCpyInst &MCI);
467 bool visitMemMoveInst(MemMoveInst &MMI);
468 bool visitMemSetInst(MemSetInst &MSI);
469 bool visitMemSetPatternInst(MemSetPatternInst &MSPI);
473Value *StoreFatPtrsAsIntsAndExpandMemcpyVisitor::fatPtrsToInts(
478 return IRB.CreatePtrToInt(V, To, Name +
".int");
484 Type *FromPart = AT->getArrayElementType();
486 for (
uint64_t I = 0,
E = AT->getArrayNumElements();
I <
E; ++
I) {
489 fatPtrsToInts(
Field, FromPart, ToPart, Name +
"." + Twine(
I));
490 Ret = IRB.CreateInsertValue(Ret, NewField,
I);
493 for (
auto [Idx, FromPart, ToPart] :
495 Value *
Field = IRB.CreateExtractValue(V, Idx);
497 fatPtrsToInts(
Field, FromPart, ToPart, Name +
"." + Twine(Idx));
498 Ret = IRB.CreateInsertValue(Ret, NewField, Idx);
504Value *StoreFatPtrsAsIntsAndExpandMemcpyVisitor::intsToFatPtrs(
509 Value *Cast = IRB.CreateIntToPtr(V, To, Name +
".ptr");
519 for (
uint64_t I = 0,
E = AT->getArrayNumElements();
I <
E; ++
I) {
522 intsToFatPtrs(
Field, FromPart, ToPart, Name +
"." + Twine(
I));
523 Ret = IRB.CreateInsertValue(Ret, NewField,
I);
526 for (
auto [Idx, FromPart, ToPart] :
528 Value *
Field = IRB.CreateExtractValue(V, Idx);
530 intsToFatPtrs(
Field, FromPart, ToPart, Name +
"." + Twine(Idx));
531 Ret = IRB.CreateInsertValue(Ret, NewField, Idx);
537bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::processFunction(
538 Function &
F,
const TargetTransformInfo *
TTI, ScalarEvolution *SE) {
559bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitAllocaInst(AllocaInst &
I) {
560 Type *Ty =
I.getAllocatedType();
561 Type *NewTy = TypeMap->remapType(Ty);
564 I.setAllocatedType(NewTy);
568bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitGetElementPtrInst(
569 GetElementPtrInst &
I) {
570 Type *Ty =
I.getSourceElementType();
571 Type *NewTy = TypeMap->remapType(Ty);
576 I.setSourceElementType(NewTy);
577 I.setResultElementType(TypeMap->remapType(
I.getResultElementType()));
581bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitLoadInst(LoadInst &LI) {
583 Type *IntTy = TypeMap->remapType(Ty);
587 IRB.SetInsertPoint(&LI);
589 NLI->mutateType(IntTy);
590 NLI = IRB.Insert(NLI);
593 Value *CastBack = intsToFatPtrs(NLI, IntTy, Ty, NLI->getName());
599bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitStoreInst(StoreInst &SI) {
601 Type *Ty =
V->getType();
602 Type *IntTy = TypeMap->remapType(Ty);
606 IRB.SetInsertPoint(&SI);
607 Value *IntV = fatPtrsToInts(V, Ty, IntTy,
V->getName());
611 SI.setOperand(0, IntV);
615bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitMemCpyInst(
627bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitMemMoveInst(
633 "memmove() on buffer descriptors is not implemented because pointer "
634 "comparison on buffer descriptors isn't implemented\n");
637bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitMemSetInst(
646bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitMemSetPatternInst(
647 MemSetPatternInst &MSPI) {
676class LegalizeBufferContentTypesVisitor
677 :
public InstVisitor<LegalizeBufferContentTypesVisitor, bool> {
678 friend class InstVisitor<LegalizeBufferContentTypesVisitor, bool>;
682 const DataLayout &
DL;
684 ScalarEvolution *SE =
nullptr;
694 const TargetMachine *TM;
695 const GCNSubtarget *ST =
nullptr;
699 Type *scalarArrayTypeAsVector(
Type *MaybeArrayType);
700 Value *arrayToVector(
Value *V,
Type *TargetType,
const Twine &Name);
701 Value *vectorToArray(
Value *V,
Type *OrigType,
const Twine &Name);
705 struct OobProperties {
707 bool NoWrapFromMax =
false;
709 bool NoPartialOOB =
false;
711 OobProperties() =
delete;
713 OobProperties(
bool NoWrapFromMax,
bool NoPartialOOB)
714 : NoWrapFromMax(NoWrapFromMax), NoPartialOOB(NoPartialOOB) {}
744 uint64_t maxIntrinsicWidth(
Type *Ty, Align
A, OobProperties OobProps);
751 Value *makeLegalNonAggregate(
Value *V,
Type *TargetType,
const Twine &Name);
752 Value *makeIllegalNonAggregate(
Value *V,
Type *OrigType,
const Twine &Name);
766 SmallVectorImpl<VecSlice> &Slices);
768 Value *extractSlice(
Value *Vec, VecSlice S,
const Twine &Name);
769 Value *insertSlice(
Value *Whole,
Value *Part, VecSlice S,
const Twine &Name);
779 Type *intrinsicTypeFor(
Type *LegalType);
781 bool visitLoadImpl(LoadInst &OrigLI,
Type *PartType,
782 SmallVectorImpl<uint32_t> &AggIdxs,
uint64_t AggByteOffset,
783 Value *&Result,
const Twine &Name);
785 std::pair<bool, bool> visitStoreImpl(StoreInst &OrigSI,
Type *PartType,
786 SmallVectorImpl<uint32_t> &AggIdxs,
790 bool visitInstruction(Instruction &
I) {
return false; }
791 bool visitLoadInst(LoadInst &LI);
792 bool visitStoreInst(StoreInst &SI);
795 bool visitIntrinsicInst(IntrinsicInst &
II);
796 bool visitAddrSpaceCastInst(AddrSpaceCastInst &ASCI);
799 LegalizeBufferContentTypesVisitor(
const DataLayout &
DL, LLVMContext &Ctx,
800 const TargetMachine *TM)
801 : IRB(Ctx, InstSimplifyFolder(
DL)),
DL(
DL), TM(TM) {}
806Type *LegalizeBufferContentTypesVisitor::scalarArrayTypeAsVector(
Type *
T) {
810 Type *ET = AT->getElementType();
813 "should have recursed");
814 if (!
DL.typeSizeEqualsStoreSize(AT))
816 "loading padded arrays from buffer fat pinters should have recursed");
820Value *LegalizeBufferContentTypesVisitor::arrayToVector(
Value *V,
825 unsigned EC = VT->getNumElements();
826 for (
auto I : iota_range<unsigned>(0, EC,
false)) {
827 Value *Elem = IRB.CreateExtractValue(V,
I, Name +
".elem." + Twine(
I));
828 VectorRes = IRB.CreateInsertElement(VectorRes, Elem,
I,
829 Name +
".as.vec." + Twine(
I));
834Value *LegalizeBufferContentTypesVisitor::vectorToArray(
Value *V,
839 unsigned EC = AT->getNumElements();
840 for (
auto I : iota_range<unsigned>(0, EC,
false)) {
841 Value *Elem = IRB.CreateExtractElement(V,
I, Name +
".elem." + Twine(
I));
842 ArrayRes = IRB.CreateInsertValue(ArrayRes, Elem,
I,
843 Name +
".as.array." + Twine(
I));
848LegalizeBufferContentTypesVisitor::OobProperties
849LegalizeBufferContentTypesVisitor::analyzeOobProperties(
Value *Ptr,
Type *Ty,
851 OobProperties
Result(
false,
false);
854 return OobProperties(
true,
true);
860 const SCEV *PtrOp = SE->
getSCEV(Ptr);
866 Value *PtrBaseVal = PtrBase->getValue();
873 auto NumRecordsIfKnown = ZeroBasePointerToNumRecords.
find(PtrBaseVal);
874 if (NumRecordsIfKnown == ZeroBasePointerToNumRecords.
end())
877 unsigned TypeSize =
DL.getTypeStoreSize(Ty).getKnownMinValue();
882 Result.NoWrapFromMax =
true;
886 if (!NumRecordsIfKnown->second)
888 const SCEV *NumRecords = SE->
getSCEV(NumRecordsIfKnown->second);
892 Result.NoPartialOOB =
true;
894 const SCEV *BoundsDiff;
895 if (ST->has45BitNumRecordsBufferResource()) {
896 const SCEV *PtrDiffExt =
900 const SCEV *NumRecordsI32 =
907 Result.NoPartialOOB =
true;
912LegalizeBufferContentTypesVisitor::maxIntrinsicWidth(
Type *
T, Align
A,
913 OobProperties OobProps) {
919 TypeSize ElemBits =
DL.getTypeSizeInBits(VT->getElementType());
926 if (!OobProps.NoWrapFromMax)
945 if (!OobProps.NoPartialOOB)
950 return Result.value() * 8;
953Type *LegalizeBufferContentTypesVisitor::legalNonAggregateForMemOp(
955 TypeSize
Size =
DL.getTypeStoreSizeInBits(
T);
957 if (!
DL.typeSizeEqualsStoreSize(
T))
958 T = IRB.getIntNTy(
Size.getFixedValue());
959 Type *ElemTy =
T->getScalarType();
965 unsigned ElemSize =
DL.getTypeSizeInBits(ElemTy).getFixedValue();
966 if (
isPowerOf2_32(ElemSize) && ElemSize >= 16 && ElemSize <= MaxWidth) {
972 Type *BestVectorElemType =
nullptr;
973 if (
Size.isKnownMultipleOf(32) && MaxWidth >= 32)
974 BestVectorElemType = IRB.getInt32Ty();
975 else if (
Size.isKnownMultipleOf(16) && MaxWidth >= 16)
976 BestVectorElemType = IRB.getInt16Ty();
978 BestVectorElemType = IRB.getInt8Ty();
979 unsigned NumCastElems =
981 if (NumCastElems == 1)
982 return BestVectorElemType;
986Value *LegalizeBufferContentTypesVisitor::makeLegalNonAggregate(
987 Value *V,
Type *TargetType,
const Twine &Name) {
988 Type *SourceType =
V->getType();
989 TypeSize SourceSize =
DL.getTypeSizeInBits(SourceType);
990 TypeSize TargetSize =
DL.getTypeSizeInBits(TargetType);
991 if (SourceSize != TargetSize) {
994 Value *AsScalar = IRB.CreateBitCast(V, ShortScalarTy, Name +
".as.scalar");
995 Value *Zext = IRB.CreateZExt(AsScalar, ByteScalarTy, Name +
".zext");
997 SourceType = ByteScalarTy;
999 return IRB.CreateBitCast(V, TargetType, Name +
".legal");
1002Value *LegalizeBufferContentTypesVisitor::makeIllegalNonAggregate(
1003 Value *V,
Type *OrigType,
const Twine &Name) {
1004 Type *LegalType =
V->getType();
1005 TypeSize LegalSize =
DL.getTypeSizeInBits(LegalType);
1006 TypeSize OrigSize =
DL.getTypeSizeInBits(OrigType);
1007 if (LegalSize != OrigSize) {
1010 Value *AsScalar = IRB.CreateBitCast(V, ByteScalarTy, Name +
".bytes.cast");
1011 Value *Trunc = IRB.CreateTrunc(AsScalar, ShortScalarTy, Name +
".trunc");
1012 return IRB.CreateBitCast(Trunc, OrigType, Name +
".orig");
1014 return IRB.CreateBitCast(V, OrigType, Name +
".real.ty");
1017Type *LegalizeBufferContentTypesVisitor::intrinsicTypeFor(
Type *LegalType) {
1021 Type *ET = VT->getElementType();
1024 if (VT->getNumElements() == 1)
1026 if (
DL.getTypeSizeInBits(LegalType) == 96 &&
DL.getTypeSizeInBits(ET) < 32)
1029 switch (VT->getNumElements()) {
1033 return IRB.getInt8Ty();
1035 return IRB.getInt16Ty();
1037 return IRB.getInt32Ty();
1047void LegalizeBufferContentTypesVisitor::getVecSlices(
1048 Type *
T,
uint64_t MaxWidth, SmallVectorImpl<VecSlice> &Slices) {
1055 DL.getTypeSizeInBits(VT->getElementType()).getFixedValue();
1057 uint64_t ElemsPer4Words = 128 / ElemBitWidth;
1058 uint64_t ElemsPer2Words = ElemsPer4Words / 2;
1059 uint64_t ElemsPerWord = ElemsPer2Words / 2;
1060 uint64_t ElemsPerShort = ElemsPerWord / 2;
1061 uint64_t ElemsPerByte = ElemsPerShort / 2;
1065 uint64_t ElemsPer3Words = ElemsPerWord * 3;
1067 uint64_t TotalElems = VT->getNumElements();
1069 auto TrySlice = [&](
unsigned MaybeLen,
unsigned Width) {
1070 if (MaybeLen > 0 && Width <= MaxWidth && Index + MaybeLen <= TotalElems) {
1071 VecSlice Slice{
Index, MaybeLen};
1078 while (Index < TotalElems) {
1079 TrySlice(ElemsPer4Words, 128) || TrySlice(ElemsPer3Words, 96) ||
1080 TrySlice(ElemsPer2Words, 64) || TrySlice(ElemsPerWord, 32) ||
1081 TrySlice(ElemsPerShort, 16) || TrySlice(ElemsPerByte, 8);
1085Value *LegalizeBufferContentTypesVisitor::extractSlice(
Value *Vec, VecSlice S,
1086 const Twine &Name) {
1090 if (S.Length == VecVT->getNumElements() && S.Index == 0)
1093 return IRB.CreateExtractElement(Vec, S.Index,
1094 Name +
".slice." + Twine(S.Index));
1096 llvm::iota_range<int>(S.Index, S.Index + S.Length,
false));
1097 return IRB.CreateShuffleVector(Vec, Mask, Name +
".slice." + Twine(S.Index));
1100Value *LegalizeBufferContentTypesVisitor::insertSlice(
Value *Whole,
Value *Part,
1102 const Twine &Name) {
1106 if (S.Length == WholeVT->getNumElements() && S.Index == 0)
1108 if (S.Length == 1) {
1109 return IRB.CreateInsertElement(Whole, Part, S.Index,
1110 Name +
".slice." + Twine(S.Index));
1115 SmallVector<int> ExtPartMask(NumElems, -1);
1120 Value *ExtPart = IRB.CreateShuffleVector(Part, ExtPartMask,
1121 Name +
".ext." + Twine(S.Index));
1123 SmallVector<int>
Mask =
1128 return IRB.CreateShuffleVector(Whole, ExtPart, Mask,
1129 Name +
".parts." + Twine(S.Index));
1132bool LegalizeBufferContentTypesVisitor::visitLoadImpl(
1133 LoadInst &OrigLI,
Type *PartType, SmallVectorImpl<uint32_t> &AggIdxs,
1136 const StructLayout *Layout =
DL.getStructLayout(ST);
1138 for (
auto [
I, ElemTy,
Offset] :
1141 Changed |= visitLoadImpl(OrigLI, ElemTy, AggIdxs,
1142 AggByteOff +
Offset.getFixedValue(), Result,
1143 Name +
"." + Twine(
I));
1149 Type *ElemTy = AT->getElementType();
1152 TypeSize ElemAllocSize =
DL.getTypeAllocSize(ElemTy);
1154 for (
auto I : llvm::iota_range<uint32_t>(0, AT->getNumElements(),
1157 Changed |= visitLoadImpl(OrigLI, ElemTy, AggIdxs,
1159 Result, Name + Twine(
I));
1169 Type *ArrayAsVecType = scalarArrayTypeAsVector(PartType);
1170 OobProperties OobProps =
1172 uint64_t MaxWidth = maxIntrinsicWidth(ArrayAsVecType, PartAlign, OobProps);
1173 Type *LegalType = legalNonAggregateForMemOp(ArrayAsVecType, MaxWidth);
1176 getVecSlices(LegalType, MaxWidth, Slices);
1177 bool HasSlices = Slices.
size() > 1;
1178 bool IsAggPart = !AggIdxs.
empty();
1180 if (!HasSlices && !IsAggPart) {
1181 Type *LoadableType = intrinsicTypeFor(LegalType);
1182 if (LoadableType == PartType)
1185 IRB.SetInsertPoint(&OrigLI);
1187 NLI->mutateType(LoadableType);
1188 NLI = IRB.Insert(NLI);
1189 NLI->setName(Name +
".loadable");
1191 LoadsRes = IRB.CreateBitCast(NLI, LegalType, Name +
".from.loadable");
1193 IRB.SetInsertPoint(&OrigLI);
1201 unsigned ElemBytes =
DL.getTypeStoreSize(ElemType);
1203 if (IsAggPart && Slices.
empty())
1205 for (VecSlice S : Slices) {
1208 int64_t ByteOffset = AggByteOff + S.Index * ElemBytes;
1210 Value *NewPtr = IRB.CreateGEP(
1212 OrigPtr->
getName() +
".off.ptr." + Twine(ByteOffset),
1215 Type *LoadableType = intrinsicTypeFor(SliceType);
1216 LoadInst *NewLI = IRB.CreateAlignedLoad(
1218 Name +
".off." + Twine(ByteOffset));
1224 Value *
Loaded = IRB.CreateBitCast(NewLI, SliceType,
1225 NewLI->
getName() +
".from.loadable");
1226 LoadsRes = insertSlice(LoadsRes, Loaded, S, Name);
1229 if (LegalType != ArrayAsVecType)
1230 LoadsRes = makeIllegalNonAggregate(LoadsRes, ArrayAsVecType, Name);
1231 if (ArrayAsVecType != PartType)
1232 LoadsRes = vectorToArray(LoadsRes, PartType, Name);
1235 Result = IRB.CreateInsertValue(Result, LoadsRes, AggIdxs, Name);
1241bool LegalizeBufferContentTypesVisitor::visitLoadInst(LoadInst &LI) {
1245 SmallVector<uint32_t> AggIdxs;
1248 bool Changed = visitLoadImpl(LI, OrigType, AggIdxs, 0, Result, LI.
getName());
1257std::pair<bool, bool> LegalizeBufferContentTypesVisitor::visitStoreImpl(
1258 StoreInst &OrigSI,
Type *PartType, SmallVectorImpl<uint32_t> &AggIdxs,
1259 uint64_t AggByteOff,
const Twine &Name) {
1261 const StructLayout *Layout =
DL.getStructLayout(ST);
1263 for (
auto [
I, ElemTy,
Offset] :
1266 Changed |= std::get<0>(visitStoreImpl(OrigSI, ElemTy, AggIdxs,
1267 AggByteOff +
Offset.getFixedValue(),
1268 Name +
"." + Twine(
I)));
1271 return std::make_pair(
Changed,
false);
1274 Type *ElemTy = AT->getElementType();
1277 TypeSize ElemAllocSize =
DL.getTypeAllocSize(ElemTy);
1279 for (
auto I : llvm::iota_range<uint32_t>(0, AT->getNumElements(),
1282 Changed |= std::get<0>(visitStoreImpl(
1283 OrigSI, ElemTy, AggIdxs,
1287 return std::make_pair(
Changed,
false);
1292 Value *NewData = OrigData;
1294 bool IsAggPart = !AggIdxs.
empty();
1296 NewData = IRB.CreateExtractValue(NewData, AggIdxs, Name);
1298 Type *ArrayAsVecType = scalarArrayTypeAsVector(PartType);
1299 if (ArrayAsVecType != PartType) {
1300 NewData = arrayToVector(NewData, ArrayAsVecType, Name);
1304 OobProperties OobProps =
1306 uint64_t MaxWidth = maxIntrinsicWidth(ArrayAsVecType, PartAlign, OobProps);
1307 Type *LegalType = legalNonAggregateForMemOp(ArrayAsVecType, MaxWidth);
1308 if (LegalType != ArrayAsVecType) {
1309 NewData = makeLegalNonAggregate(NewData, LegalType, Name);
1313 getVecSlices(LegalType, MaxWidth, Slices);
1314 bool NeedToSplit = Slices.
size() > 1 || IsAggPart;
1316 Type *StorableType = intrinsicTypeFor(LegalType);
1317 if (StorableType == PartType)
1318 return std::make_pair(
false,
false);
1319 NewData = IRB.CreateBitCast(NewData, StorableType, Name +
".storable");
1321 return std::make_pair(
true,
true);
1326 if (IsAggPart && Slices.
empty())
1328 unsigned ElemBytes =
DL.getTypeStoreSize(ElemType);
1330 for (VecSlice S : Slices) {
1333 int64_t ByteOffset = AggByteOff + S.Index * ElemBytes;
1334 Value *NewPtr = IRB.CreateGEP(
1335 IRB.getInt8Ty(), OrigPtr, IRB.getInt32(ByteOffset),
1336 OrigPtr->
getName() +
".part." + Twine(S.Index),
1339 Value *DataSlice = extractSlice(NewData, S, Name);
1340 Type *StorableType = intrinsicTypeFor(SliceType);
1341 DataSlice = IRB.CreateBitCast(DataSlice, StorableType,
1342 DataSlice->
getName() +
".storable");
1346 NewSI->setOperand(0, DataSlice);
1347 NewSI->setOperand(1, NewPtr);
1350 return std::make_pair(
true,
false);
1353bool LegalizeBufferContentTypesVisitor::visitStoreInst(StoreInst &SI) {
1356 IRB.SetInsertPoint(&SI);
1357 SmallVector<uint32_t> AggIdxs;
1358 Value *OrigData =
SI.getValueOperand();
1359 auto [
Changed, ModifiedInPlace] =
1360 visitStoreImpl(SI, OrigData->
getType(), AggIdxs, 0, OrigData->
getName());
1361 if (
Changed && !ModifiedInPlace)
1362 SI.eraseFromParent();
1366bool LegalizeBufferContentTypesVisitor::visitAddrSpaceCastInst(
1367 AddrSpaceCastInst &AI) {
1372 auto Record = ZeroBasePointerToNumRecords.
find(Src);
1373 if (Record != ZeroBasePointerToNumRecords.
end())
1374 ZeroBasePointerToNumRecords.
insert({&AI,
Record->second});
1376 ZeroBasePointerToNumRecords.
insert({&AI,
nullptr});
1380bool LegalizeBufferContentTypesVisitor::visitIntrinsicInst(IntrinsicInst &
II) {
1381 if (
II.getIntrinsicID() != Intrinsic::amdgcn_make_buffer_rsrc)
1383 ZeroBasePointerToNumRecords.
insert({&
II,
II.getOperand(2)});
1387bool LegalizeBufferContentTypesVisitor::processFunction(
Function &
F,
1388 ScalarEvolution *SE) {
1395 ZeroBasePointerToNumRecords.
clear();
1402static std::pair<Constant *, Constant *>
1405 return std::make_pair(
C->getAggregateElement(0u),
C->getAggregateElement(1u));
1410class FatPtrConstMaterializer final :
public ValueMaterializer {
1411 BufferFatPtrToStructTypeMap *TypeMap;
1417 ValueMapper InternalMapper;
1419 Constant *materializeBufferFatPtrConst(Constant *
C);
1423 FatPtrConstMaterializer(BufferFatPtrToStructTypeMap *TypeMap,
1426 InternalMapper(UnderlyingMap,
RF_None, TypeMap, this) {}
1427 ~FatPtrConstMaterializer() =
default;
1433Constant *FatPtrConstMaterializer::materializeBufferFatPtrConst(Constant *
C) {
1434 Type *SrcTy =
C->getType();
1436 if (
C->isNullValue())
1437 return ConstantAggregateZero::getNullValue(NewTy);
1450 if (Constant *S =
VC->getSplatValue()) {
1455 auto EC =
VC->getType()->getElementCount();
1461 for (
Value *
Op :
VC->operand_values()) {
1476 "fat pointer) values are not supported");
1480 "constant exprs containing ptr addrspace(7) (buffer "
1481 "fat pointer) values should have been expanded earlier");
1486Value *FatPtrConstMaterializer::materialize(
Value *V) {
1494 return materializeBufferFatPtrConst(
C);
1502class SplitPtrStructs :
public InstVisitor<SplitPtrStructs, PtrParts> {
1545 void processConditionals();
1595void SplitPtrStructs::copyMetadata(
Value *Dest,
Value *Src) {
1599 if (!DestI || !SrcI)
1602 DestI->copyMetadata(*SrcI);
1607 "of something that wasn't rewritten");
1608 auto *RsrcEntry = &RsrcParts[
V];
1609 auto *OffEntry = &OffParts[
V];
1610 if (*RsrcEntry && *OffEntry)
1611 return {*RsrcEntry, *OffEntry};
1615 return {*RsrcEntry = Rsrc, *OffEntry =
Off};
1618 IRBuilder<InstSimplifyFolder>::InsertPointGuard Guard(IRB);
1623 return {*RsrcEntry = Rsrc, *OffEntry =
Off};
1626 IRB.SetInsertPoint(*
I->getInsertionPointAfterDef());
1627 IRB.SetCurrentDebugLocation(
I->getDebugLoc());
1629 IRB.SetInsertPointPastAllocas(
A->getParent());
1630 IRB.SetCurrentDebugLocation(
DebugLoc());
1632 Value *Rsrc = IRB.CreateExtractValue(V, 0,
V->getName() +
".rsrc");
1633 Value *
Off = IRB.CreateExtractValue(V, 1,
V->getName() +
".off");
1634 return {*RsrcEntry = Rsrc, *OffEntry =
Off};
1647 V =
GEP->getPointerOperand();
1649 V = ASC->getPointerOperand();
1653void SplitPtrStructs::getPossibleRsrcRoots(Instruction *
I,
1654 SmallPtrSetImpl<Value *> &Roots,
1655 SmallPtrSetImpl<Value *> &Seen) {
1659 for (
Value *In :
PHI->incoming_values()) {
1666 if (!Seen.
insert(SI).second)
1681void SplitPtrStructs::processConditionals() {
1682 SmallDenseMap<Value *, Value *> FoundRsrcs;
1683 SmallPtrSet<Value *, 4> Roots;
1684 SmallPtrSet<Value *, 4> Seen;
1685 for (Instruction *
I : Conditionals) {
1687 Value *Rsrc = RsrcParts[
I];
1689 assert(Rsrc && Off &&
"must have visited conditionals by now");
1691 std::optional<Value *> MaybeRsrc;
1692 auto MaybeFoundRsrc = FoundRsrcs.
find(
I);
1693 if (MaybeFoundRsrc != FoundRsrcs.
end()) {
1694 MaybeRsrc = MaybeFoundRsrc->second;
1696 IRBuilder<InstSimplifyFolder>::InsertPointGuard Guard(IRB);
1699 getPossibleRsrcRoots(
I, Roots, Seen);
1702 for (
Value *V : Roots)
1704 for (
Value *V : Seen)
1716 if (Diff.size() == 1) {
1717 Value *RootVal = *Diff.begin();
1721 MaybeRsrc = std::get<0>(getPtrParts(RootVal));
1723 MaybeRsrc = RootVal;
1731 IRB.SetInsertPoint(*
PHI->getInsertionPointAfterDef());
1732 IRB.SetCurrentDebugLocation(
PHI->getDebugLoc());
1734 NewRsrc = *MaybeRsrc;
1737 auto *RsrcPHI = IRB.CreatePHI(RsrcTy,
PHI->getNumIncomingValues());
1738 RsrcPHI->takeName(Rsrc);
1739 for (
auto [V, BB] :
llvm::zip(
PHI->incoming_values(),
PHI->blocks())) {
1740 Value *VRsrc = std::get<0>(getPtrParts(V));
1741 RsrcPHI->addIncoming(VRsrc, BB);
1743 copyMetadata(RsrcPHI,
PHI);
1748 auto *NewOff = IRB.CreatePHI(OffTy,
PHI->getNumIncomingValues());
1749 NewOff->takeName(Off);
1750 for (
auto [V, BB] :
llvm::zip(
PHI->incoming_values(),
PHI->blocks())) {
1751 assert(OffParts.
count(V) &&
"An offset part had to be created by now");
1752 Value *VOff = std::get<1>(getPtrParts(V));
1753 NewOff->addIncoming(VOff, BB);
1755 copyMetadata(NewOff,
PHI);
1765 RsrcInst->replaceAllUsesWith(NewRsrc);
1769 OffInst->replaceAllUsesWith(NewOff);
1774 for (
Value *V : Seen)
1775 FoundRsrcs[
V] = NewRsrc;
1780 if (RsrcInst != *MaybeRsrc) {
1782 RsrcInst->replaceAllUsesWith(*MaybeRsrc);
1785 for (
Value *V : Seen)
1786 FoundRsrcs[
V] = *MaybeRsrc;
1794void SplitPtrStructs::killAndReplaceSplitInstructions(
1795 SmallVectorImpl<Instruction *> &Origs) {
1796 for (Instruction *
I : ConditionalTemps)
1797 I->eraseFromParent();
1799 for (Instruction *
I : Origs) {
1805 for (DbgVariableRecord *Dbg : Dbgs) {
1806 auto &
DL =
I->getDataLayout();
1808 "We should've RAUW'd away loads, stores, etc. at this point");
1809 DbgVariableRecord *OffDbg =
Dbg->clone();
1810 auto [Rsrc,
Off] = getPtrParts(
I);
1812 int64_t RsrcSz =
DL.getTypeSizeInBits(Rsrc->
getType());
1813 int64_t OffSz =
DL.getTypeSizeInBits(
Off->getType());
1815 std::optional<DIExpression *> RsrcExpr =
1818 std::optional<DIExpression *> OffExpr =
1829 Dbg->setExpression(*RsrcExpr);
1830 Dbg->replaceVariableLocationOp(
I, Rsrc);
1837 I->replaceUsesWithIf(
Poison, [&](
const Use &U) ->
bool {
1843 if (
I->use_empty()) {
1844 I->eraseFromParent();
1847 IRB.SetInsertPoint(*
I->getInsertionPointAfterDef());
1848 IRB.SetCurrentDebugLocation(
I->getDebugLoc());
1849 auto [Rsrc,
Off] = getPtrParts(
I);
1851 Struct = IRB.CreateInsertValue(Struct, Rsrc, 0);
1852 Struct = IRB.CreateInsertValue(Struct, Off, 1);
1853 copyMetadata(Struct,
I);
1855 I->replaceAllUsesWith(Struct);
1856 I->eraseFromParent();
1860void SplitPtrStructs::setAlign(CallInst *Intr, Align
A,
unsigned RsrcArgIdx) {
1862 Intr->
addParamAttr(RsrcArgIdx, Attribute::getWithAlignment(Ctx,
A));
1868 case AtomicOrdering::Release:
1869 case AtomicOrdering::AcquireRelease:
1870 case AtomicOrdering::SequentiallyConsistent:
1871 IRB.CreateFence(AtomicOrdering::Release, SSID);
1881 case AtomicOrdering::Acquire:
1882 case AtomicOrdering::AcquireRelease:
1883 case AtomicOrdering::SequentiallyConsistent:
1884 IRB.CreateFence(AtomicOrdering::Acquire, SSID);
1891Value *SplitPtrStructs::handleMemoryInst(Instruction *
I,
Value *Arg,
Value *Ptr,
1892 Type *Ty, Align Alignment,
1895 IRB.SetInsertPoint(
I);
1897 auto [Rsrc,
Off] = getPtrParts(Ptr);
1900 Args.push_back(Arg);
1901 Args.push_back(Rsrc);
1902 Args.push_back(Off);
1903 insertPreMemOpFence(Order, SSID);
1907 Args.push_back(IRB.getInt32(0));
1912 Args.push_back(IRB.getInt32(Aux));
1916 IID = Order == AtomicOrdering::NotAtomic
1917 ? Intrinsic::amdgcn_raw_ptr_buffer_load
1918 : Intrinsic::amdgcn_raw_ptr_atomic_buffer_load;
1920 IID = Intrinsic::amdgcn_raw_ptr_buffer_store;
1922 switch (RMW->getOperation()) {
1924 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_swap;
1927 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_add;
1930 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_sub;
1933 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_and;
1936 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_or;
1939 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_xor;
1942 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_smax;
1945 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_smin;
1948 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_umax;
1951 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_umin;
1954 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_fadd;
1957 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_fmax;
1960 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_fmin;
1963 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_cond_sub_u32;
1966 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_sub_clamp_u32;
1970 "atomic floating point subtraction not supported for "
1971 "buffer resources and should've been expanded away");
1976 "atomic floating point fmaximum not supported for "
1977 "buffer resources and should've been expanded away");
1982 "atomic floating point fminimum not supported for "
1983 "buffer resources and should've been expanded away");
1988 "atomic floating point fmaximumnum not supported for "
1989 "buffer resources and should've been expanded away");
1994 "atomic floating point fminimumnum not supported for "
1995 "buffer resources and should've been expanded away");
2000 "atomic nand not supported for buffer resources and "
2001 "should've been expanded away");
2006 "wrapping increment/decrement not supported for "
2007 "buffer resources and should've been expanded away");
2014 CallInst *
Call = IRB.CreateIntrinsicWithoutFolding(IID, Ty, Args);
2015 copyMetadata(
Call,
I);
2016 setAlign(
Call, Alignment, Arg ? 1 : 0);
2019 insertPostMemOpFence(Order, SSID);
2023 I->replaceAllUsesWith(
Call);
2027PtrParts SplitPtrStructs::visitInstruction(Instruction &
I) {
2028 return {
nullptr,
nullptr};
2031PtrParts SplitPtrStructs::visitLoadInst(LoadInst &LI) {
2033 return {
nullptr,
nullptr};
2037 return {
nullptr,
nullptr};
2040PtrParts SplitPtrStructs::visitStoreInst(StoreInst &SI) {
2042 return {
nullptr,
nullptr};
2043 Value *Arg =
SI.getValueOperand();
2044 handleMemoryInst(&SI, Arg,
SI.getPointerOperand(), Arg->
getType(),
2045 SI.getAlign(),
SI.getOrdering(),
SI.isVolatile(),
2046 SI.getSyncScopeID());
2047 return {
nullptr,
nullptr};
2050PtrParts SplitPtrStructs::visitAtomicRMWInst(AtomicRMWInst &AI) {
2052 return {
nullptr,
nullptr};
2057 return {
nullptr,
nullptr};
2062PtrParts SplitPtrStructs::visitAtomicCmpXchgInst(AtomicCmpXchgInst &AI) {
2065 return {
nullptr,
nullptr};
2066 IRB.SetInsertPoint(&AI);
2071 bool IsNonTemporal = AI.
getMetadata(LLVMContext::MD_nontemporal);
2073 auto [Rsrc,
Off] = getPtrParts(Ptr);
2074 insertPreMemOpFence(Order, SSID);
2081 CallInst *
Call = IRB.CreateIntrinsicWithoutFolding(
2082 Intrinsic::amdgcn_raw_ptr_buffer_atomic_cmpswap, Ty,
2084 IRB.getInt32(0), IRB.getInt32(Aux)});
2085 copyMetadata(
Call, &AI);
2088 insertPostMemOpFence(Order, SSID);
2091 Res = IRB.CreateInsertValue(Res,
Call, 0);
2093 Res = IRB.CreateInsertValue(Res, Succeeded, 1);
2096 return {
nullptr,
nullptr};
2099PtrParts SplitPtrStructs::visitGetElementPtrInst(GetElementPtrInst &
GEP) {
2100 using namespace llvm::PatternMatch;
2101 Value *Ptr =
GEP.getPointerOperand();
2103 return {
nullptr,
nullptr};
2104 IRB.SetInsertPoint(&
GEP);
2106 auto [Rsrc,
Off] = getPtrParts(Ptr);
2107 const DataLayout &
DL =
GEP.getDataLayout();
2108 bool IsNUW =
GEP.hasNoUnsignedWrap();
2109 bool IsNUSW =
GEP.hasNoUnsignedSignedWrap();
2120 GEP.mutateType(FatPtrTy);
2122 GEP.mutateType(ResTy);
2124 if (BroadcastsPtr) {
2125 Rsrc = IRB.CreateVectorSplat(ResRsrcVecTy->getElementCount(), Rsrc,
2127 Off = IRB.CreateVectorSplat(ResRsrcVecTy->getElementCount(), Off,
2135 bool HasNonNegativeOff =
false;
2137 HasNonNegativeOff = !CI->isNegative();
2143 NewOff = IRB.CreateAdd(Off, OffAccum,
"",
2144 IsNUW || (IsNUSW && HasNonNegativeOff),
2147 copyMetadata(NewOff, &
GEP);
2150 return {Rsrc, NewOff};
2153PtrParts SplitPtrStructs::visitPtrToIntInst(PtrToIntInst &PI) {
2156 return {
nullptr,
nullptr};
2157 IRB.SetInsertPoint(&PI);
2162 auto [Rsrc,
Off] = getPtrParts(Ptr);
2168 Res = IRB.CreateIntCast(Off, ResTy,
false,
2171 Value *RsrcInt = IRB.CreatePtrToInt(Rsrc, ResTy, PI.
getName() +
".rsrc");
2172 Value *Shl = IRB.CreateShl(
2175 "", Width >= FatPtrWidth, Width > FatPtrWidth);
2176 Value *OffCast = IRB.CreateIntCast(Off, ResTy,
false,
2178 Res = IRB.CreateOr(Shl, OffCast);
2181 copyMetadata(Res, &PI);
2185 return {
nullptr,
nullptr};
2188PtrParts SplitPtrStructs::visitPtrToAddrInst(PtrToAddrInst &PA) {
2191 return {
nullptr,
nullptr};
2192 IRB.SetInsertPoint(&PA);
2194 auto [Rsrc,
Off] = getPtrParts(Ptr);
2195 Value *Res = IRB.CreateIntCast(Off, PA.
getType(),
false);
2196 copyMetadata(Res, &PA);
2200 return {
nullptr,
nullptr};
2203PtrParts SplitPtrStructs::visitIntToPtrInst(IntToPtrInst &IP) {
2205 return {
nullptr,
nullptr};
2206 IRB.SetInsertPoint(&IP);
2215 Type *RsrcTy = RetTy->getElementType(0);
2216 Type *OffTy = RetTy->getElementType(1);
2225 RsrcInt = IRB.CreateIntCast(RsrcPart, RsrcIntTy,
false);
2227 Value *Rsrc = IRB.CreateIntToPtr(RsrcInt, RsrcTy, IP.
getName() +
".rsrc");
2229 IRB.CreateIntCast(
Int, OffTy,
false, IP.
getName() +
".off");
2231 copyMetadata(Rsrc, &IP);
2236PtrParts SplitPtrStructs::visitAddrSpaceCastInst(AddrSpaceCastInst &
I) {
2240 return {
nullptr,
nullptr};
2241 IRB.SetInsertPoint(&
I);
2244 if (
In->getType() ==
I.getType()) {
2245 auto [Rsrc,
Off] = getPtrParts(In);
2251 Type *RsrcTy = ResTy->getElementType(0);
2252 Type *OffTy = ResTy->getElementType(1);
2258 if (InConst && InConst->isNullValue()) {
2261 return {NullRsrc, ZeroOff};
2267 return {PoisonRsrc, PoisonOff};
2273 return {UndefRsrc, UndefOff};
2278 "only buffer resources (addrspace 8) and null/poison pointers can be "
2279 "cast to buffer fat pointers (addrspace 7)");
2281 return {
In, ZeroOff};
2284PtrParts SplitPtrStructs::visitICmpInst(ICmpInst &Cmp) {
2287 return {
nullptr,
nullptr};
2289 IRB.SetInsertPoint(&Cmp);
2290 ICmpInst::Predicate Pred =
Cmp.getPredicate();
2292 assert((Pred == ICmpInst::ICMP_EQ || Pred == ICmpInst::ICMP_NE) &&
2293 "Pointer comparison is only equal or unequal");
2294 auto [LhsRsrc, LhsOff] = getPtrParts(Lhs);
2295 auto [RhsRsrc, RhsOff] = getPtrParts(Rhs);
2296 Value *Res = IRB.CreateICmp(Pred, LhsOff, RhsOff);
2297 copyMetadata(Res, &Cmp);
2300 Cmp.replaceAllUsesWith(Res);
2301 return {
nullptr,
nullptr};
2304PtrParts SplitPtrStructs::visitFreezeInst(FreezeInst &
I) {
2306 return {
nullptr,
nullptr};
2307 IRB.SetInsertPoint(&
I);
2308 auto [Rsrc,
Off] = getPtrParts(
I.getOperand(0));
2310 Value *RsrcRes = IRB.CreateFreeze(Rsrc,
I.getName() +
".rsrc");
2311 copyMetadata(RsrcRes, &
I);
2312 Value *OffRes = IRB.CreateFreeze(Off,
I.getName() +
".off");
2313 copyMetadata(OffRes, &
I);
2315 return {RsrcRes, OffRes};
2318PtrParts SplitPtrStructs::visitExtractElementInst(ExtractElementInst &
I) {
2320 return {
nullptr,
nullptr};
2321 IRB.SetInsertPoint(&
I);
2322 Value *Vec =
I.getVectorOperand();
2323 Value *Idx =
I.getIndexOperand();
2324 auto [Rsrc,
Off] = getPtrParts(Vec);
2326 Value *RsrcRes = IRB.CreateExtractElement(Rsrc, Idx,
I.getName() +
".rsrc");
2327 copyMetadata(RsrcRes, &
I);
2328 Value *OffRes = IRB.CreateExtractElement(Off, Idx,
I.getName() +
".off");
2329 copyMetadata(OffRes, &
I);
2331 return {RsrcRes, OffRes};
2334PtrParts SplitPtrStructs::visitInsertElementInst(InsertElementInst &
I) {
2338 return {
nullptr,
nullptr};
2339 IRB.SetInsertPoint(&
I);
2340 Value *Vec =
I.getOperand(0);
2341 Value *Elem =
I.getOperand(1);
2342 Value *Idx =
I.getOperand(2);
2343 auto [VecRsrc, VecOff] = getPtrParts(Vec);
2344 auto [ElemRsrc, ElemOff] = getPtrParts(Elem);
2347 IRB.CreateInsertElement(VecRsrc, ElemRsrc, Idx,
I.getName() +
".rsrc");
2348 copyMetadata(RsrcRes, &
I);
2350 IRB.CreateInsertElement(VecOff, ElemOff, Idx,
I.getName() +
".off");
2351 copyMetadata(OffRes, &
I);
2353 return {RsrcRes, OffRes};
2356PtrParts SplitPtrStructs::visitShuffleVectorInst(ShuffleVectorInst &
I) {
2359 return {
nullptr,
nullptr};
2360 IRB.SetInsertPoint(&
I);
2363 Value *V2 =
I.getOperand(1);
2364 ArrayRef<int>
Mask =
I.getShuffleMask();
2365 auto [V1Rsrc, V1Off] = getPtrParts(
V1);
2366 auto [V2Rsrc, V2Off] = getPtrParts(V2);
2369 IRB.CreateShuffleVector(V1Rsrc, V2Rsrc, Mask,
I.getName() +
".rsrc");
2370 copyMetadata(RsrcRes, &
I);
2372 IRB.CreateShuffleVector(V1Off, V2Off, Mask,
I.getName() +
".off");
2373 copyMetadata(OffRes, &
I);
2375 return {RsrcRes, OffRes};
2378PtrParts SplitPtrStructs::visitPHINode(PHINode &
PHI) {
2380 return {
nullptr,
nullptr};
2381 IRB.SetInsertPoint(*
PHI.getInsertionPointAfterDef());
2387 Value *TmpRsrc = IRB.CreateExtractValue(&
PHI, 0,
PHI.getName() +
".rsrc");
2388 Value *TmpOff = IRB.CreateExtractValue(&
PHI, 1,
PHI.getName() +
".off");
2389 Conditionals.push_back(&
PHI);
2391 return {TmpRsrc, TmpOff};
2394PtrParts SplitPtrStructs::visitSelectInst(SelectInst &SI) {
2396 return {
nullptr,
nullptr};
2397 IRB.SetInsertPoint(&SI);
2400 Value *True =
SI.getTrueValue();
2401 Value *False =
SI.getFalseValue();
2402 auto [TrueRsrc, TrueOff] = getPtrParts(True);
2403 auto [FalseRsrc, FalseOff] = getPtrParts(False);
2406 IRB.CreateSelect(
Cond, TrueRsrc, FalseRsrc,
SI.getName() +
".rsrc", &SI);
2407 copyMetadata(RsrcRes, &SI);
2408 Conditionals.push_back(&SI);
2410 IRB.CreateSelect(
Cond, TrueOff, FalseOff,
SI.getName() +
".off", &SI);
2411 copyMetadata(OffRes, &SI);
2413 return {RsrcRes, OffRes};
2424 case Intrinsic::amdgcn_make_buffer_rsrc:
2425 case Intrinsic::ptrmask:
2426 case Intrinsic::invariant_start:
2427 case Intrinsic::invariant_end:
2428 case Intrinsic::launder_invariant_group:
2429 case Intrinsic::strip_invariant_group:
2430 case Intrinsic::memcpy:
2431 case Intrinsic::memcpy_inline:
2432 case Intrinsic::memmove:
2433 case Intrinsic::memset:
2434 case Intrinsic::memset_inline:
2435 case Intrinsic::experimental_memset_pattern:
2436 case Intrinsic::amdgcn_load_to_lds:
2437 case Intrinsic::amdgcn_load_async_to_lds:
2442PtrParts SplitPtrStructs::visitIntrinsicInst(IntrinsicInst &
I) {
2447 case Intrinsic::amdgcn_make_buffer_rsrc: {
2449 return {
nullptr,
nullptr};
2451 Value *Stride =
I.getArgOperand(1);
2452 Value *NumRecords =
I.getArgOperand(2);
2455 Type *RsrcType = SplitType->getElementType(0);
2456 Type *OffType = SplitType->getElementType(1);
2457 IRB.SetInsertPoint(&
I);
2458 Value *Rsrc = IRB.CreateIntrinsic(
2459 IID, {RsrcType,
Base->getType(), NumRecords->
getType()},
2461 copyMetadata(Rsrc, &
I);
2465 return {Rsrc,
Zero};
2467 case Intrinsic::ptrmask: {
2468 Value *Ptr =
I.getArgOperand(0);
2470 return {
nullptr,
nullptr};
2472 IRB.SetInsertPoint(&
I);
2473 auto [Rsrc,
Off] = getPtrParts(Ptr);
2474 if (
Mask->getType() !=
Off->getType())
2476 "pointer (data layout not set up correctly?)");
2477 Value *OffRes = IRB.CreateAnd(Off, Mask,
I.getName() +
".off");
2478 copyMetadata(OffRes, &
I);
2480 return {Rsrc, OffRes};
2484 case Intrinsic::invariant_start: {
2485 Value *Ptr =
I.getArgOperand(1);
2487 return {
nullptr,
nullptr};
2488 IRB.SetInsertPoint(&
I);
2489 auto [Rsrc,
Off] = getPtrParts(Ptr);
2491 auto *NewRsrc = IRB.CreateIntrinsic(IID, {NewTy}, {
I.getOperand(0), Rsrc});
2492 copyMetadata(NewRsrc, &
I);
2495 I.replaceAllUsesWith(NewRsrc);
2496 return {
nullptr,
nullptr};
2498 case Intrinsic::invariant_end: {
2499 Value *RealPtr =
I.getArgOperand(2);
2501 return {
nullptr,
nullptr};
2502 IRB.SetInsertPoint(&
I);
2503 Value *RealRsrc = getPtrParts(RealPtr).first;
2504 Value *InvPtr =
I.getArgOperand(0);
2506 Value *NewRsrc = IRB.CreateIntrinsic(IID, {RealRsrc->
getType()},
2507 {InvPtr,
Size, RealRsrc});
2508 copyMetadata(NewRsrc, &
I);
2511 I.replaceAllUsesWith(NewRsrc);
2512 return {
nullptr,
nullptr};
2514 case Intrinsic::launder_invariant_group:
2515 case Intrinsic::strip_invariant_group: {
2516 Value *Ptr =
I.getArgOperand(0);
2518 return {
nullptr,
nullptr};
2519 IRB.SetInsertPoint(&
I);
2520 auto [Rsrc,
Off] = getPtrParts(Ptr);
2521 Value *NewRsrc = IRB.CreateIntrinsic(IID, {Rsrc->
getType()}, {Rsrc});
2522 copyMetadata(NewRsrc, &
I);
2525 return {NewRsrc,
Off};
2527 case Intrinsic::amdgcn_load_to_lds:
2528 case Intrinsic::amdgcn_load_async_to_lds: {
2529 Value *Ptr =
I.getArgOperand(0);
2531 return {
nullptr,
nullptr};
2532 IRB.SetInsertPoint(&
I);
2533 auto [Rsrc,
Off] = getPtrParts(Ptr);
2534 Value *LDSPtr =
I.getArgOperand(1);
2535 Value *LoadSize =
I.getArgOperand(2);
2536 Value *ImmOff =
I.getArgOperand(3);
2537 Value *Aux =
I.getArgOperand(4);
2538 Value *SOffset = IRB.getInt32(0);
2540 IID == Intrinsic::amdgcn_load_to_lds
2541 ? Intrinsic::amdgcn_raw_ptr_buffer_load_lds
2542 : Intrinsic::amdgcn_raw_ptr_buffer_load_async_lds;
2543 Instruction *NewLoad = IRB.CreateIntrinsicWithoutFolding(
2544 NewIntr, {}, {Rsrc, LDSPtr, LoadSize,
Off, SOffset, ImmOff, Aux});
2545 copyMetadata(NewLoad, &
I);
2547 I.replaceAllUsesWith(NewLoad);
2548 return {
nullptr,
nullptr};
2551 return {
nullptr,
nullptr};
2554void SplitPtrStructs::processFunction(
Function &
F) {
2556 SmallVector<Instruction *, 0> Originals(
2558 LLVM_DEBUG(
dbgs() <<
"Splitting pointer structs in function: " <<
F.getName()
2560 for (Instruction *
I : Originals) {
2568 assert(((Rsrc && Off) || (!Rsrc && !Off)) &&
2569 "Can't have a resource but no offset");
2571 RsrcParts[
I] = Rsrc;
2575 processConditionals();
2576 killAndReplaceSplitInstructions(Originals);
2582 Conditionals.clear();
2583 ConditionalTemps.clear();
2587class AMDGPULowerBufferFatPointers :
public ModulePass {
2591 AMDGPULowerBufferFatPointers() : ModulePass(
ID) {}
2594 bool runOnModule(
Module &M)
override;
2596 void getAnalysisUsage(AnalysisUsage &AU)
const override;
2604 BufferFatPtrToStructTypeMap *TypeMap) {
2605 bool HasFatPointers =
false;
2608 HasFatPointers |= (
I.getType() != TypeMap->remapType(
I.getType()));
2610 for (
const Value *V :
I.operand_values())
2611 HasFatPointers |= (V->getType() != TypeMap->remapType(V->getType()));
2613 return HasFatPointers;
2617 BufferFatPtrToStructTypeMap *TypeMap) {
2618 Type *Ty =
F.getFunctionType();
2619 return Ty != TypeMap->remapType(Ty);
2635 while (!OldF->
empty()) {
2649 CloneMap[&NewArg] = &OldArg;
2650 NewArg.takeName(&OldArg);
2651 Type *OldArgTy = OldArg.getType(), *NewArgTy = NewArg.getType();
2653 NewArg.mutateType(OldArgTy);
2654 OldArg.replaceAllUsesWith(&NewArg);
2655 NewArg.mutateType(NewArgTy);
2659 if (OldArgTy != NewArgTy && !IsIntrinsic)
2662 AttributeFuncs::typeIncompatible(NewArgTy, ArgAttr));
2669 AttributeFuncs::typeIncompatible(NewF->
getReturnType(), RetAttrs));
2671 NewF->
getContext(), OldAttrs.getFnAttrs(), RetAttrs, ArgAttrs));
2679 CloneMap[&BB] = &BB;
2685bool AMDGPULowerBufferFatPointers::run(
Module &M,
const TargetMachine &TM,
2688 const DataLayout &
DL =
M.getDataLayout();
2694 LLVMContext &Ctx =
M.getContext();
2696 BufferFatPtrToStructTypeMap StructTM(
DL);
2697 BufferFatPtrToIntTypeMap IntTM(
DL);
2701 Ctx.
emitError(
"global variables with a buffer fat pointer address "
2702 "space (7) are not supported");
2704 GV.eraseFromParent();
2709 Type *VT = GV.getValueType();
2710 if (VT != StructTM.remapType(VT)) {
2712 Ctx.
emitError(
"global variables that contain buffer fat pointers "
2713 "(address space 7 pointers) are unsupported. Use "
2714 "buffer resource pointers (address space 8) instead");
2716 GV.eraseFromParent();
2732 SmallPtrSet<Constant *, 8> Visited;
2733 SetVector<Constant *> BufferFatPtrConsts;
2734 while (!Worklist.
empty()) {
2736 if (!Visited.
insert(
C).second)
2752 StoreFatPtrsAsIntsAndExpandMemcpyVisitor MemOpsRewrite(&IntTM,
DL,
2754 LegalizeBufferContentTypesVisitor BufferContentsTypeRewrite(
2755 DL,
M.getContext(), &TM);
2759 const TargetTransformInfo *
TTI = GetTTI(
F);
2760 ScalarEvolution *SE = GetSE(
F);
2761 Changed |= MemOpsRewrite.processFunction(
F,
TTI, SE);
2762 if (InterfaceChange || BodyChanges) {
2763 NeedsRemap.
push_back(std::make_pair(&
F, InterfaceChange));
2764 Changed |= BufferContentsTypeRewrite.processFunction(
F, SE);
2767 if (NeedsRemap.
empty())
2774 FatPtrConstMaterializer Materializer(&StructTM, CloneMap);
2776 ValueMapper LowerInFuncs(CloneMap,
RF_None, &StructTM, &Materializer);
2777 for (
auto [
F, InterfaceChange] : NeedsRemap) {
2779 if (InterfaceChange)
2785 LowerInFuncs.remapFunction(*NewF);
2790 if (InterfaceChange) {
2791 F->replaceAllUsesWith(NewF);
2792 F->eraseFromParent();
2800 SplitPtrStructs Splitter(
DL,
M.getContext(), &TM);
2802 Splitter.processFunction(*
F);
2807 F->eraseFromParent();
2811 F->replaceAllUsesWith(*NewF);
2817bool AMDGPULowerBufferFatPointers::runOnModule(
Module &M) {
2818 TargetPassConfig &TPC = getAnalysis<TargetPassConfig>();
2819 const TargetMachine &TM = TPC.
getTM<TargetMachine>();
2820 auto GetTTI = [&](
Function &
F) ->
const TargetTransformInfo * {
2821 if (
F.isDeclaration())
2823 return &getAnalysis<TargetTransformInfoWrapperPass>().getTTI(
F);
2825 auto GetSE = [&](
Function &
F) -> ScalarEvolution * {
2826 if (
F.isDeclaration())
2828 return &getAnalysis<ScalarEvolutionWrapperPass>(
F).getSE();
2830 return run(M, TM, GetTTI, GetSE);
2833char AMDGPULowerBufferFatPointers::ID = 0;
2837void AMDGPULowerBufferFatPointers::getAnalysisUsage(
AnalysisUsage &AU)
const {
2843#define PASS_DESC "Lower buffer fat pointer operations to buffer resources"
2854 return new AMDGPULowerBufferFatPointers();
2861 if (
F.isDeclaration())
2866 if (
F.isDeclaration())
2870 return AMDGPULowerBufferFatPointers().run(M, TM, GetTTI, GetSE)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU address space definition.
function_ref< const TargetTransformInfo *(Function &)> GetTTIFn
static Function * moveFunctionAdaptingType(Function *OldF, FunctionType *NewTy, ValueToValueMapTy &CloneMap)
Move the body of OldF into a new function, returning it.
static void makeCloneInPraceMap(Function *F, ValueToValueMapTy &CloneMap)
static bool isBufferFatPtrOrVector(Type *Ty)
static bool isSplitFatPtr(Type *Ty)
std::pair< Value *, Value * > PtrParts
static bool hasFatPointerInterface(const Function &F, BufferFatPtrToStructTypeMap *TypeMap)
static bool isRemovablePointerIntrinsic(Intrinsic::ID IID)
Returns true if this intrinsic needs to be removed when it is applied to ptr addrspace(7) values.
static bool containsBufferFatPointers(const Function &F, BufferFatPtrToStructTypeMap *TypeMap)
Returns true if there are values that have a buffer fat pointer in them, which means we'll need to pe...
static Value * rsrcPartRoot(Value *V)
Returns the instruction that defines the resource part of the value V.
static constexpr unsigned BufferOffsetWidth
function_ref< ScalarEvolution *(Function &)> GetSEFn
static bool isBufferFatPtrConst(Constant *C)
static std::pair< Constant *, Constant * > splitLoweredFatBufferConst(Constant *C)
Return the ptr addrspace(8) and i32 (resource and offset parts) in a lowered buffer fat pointer const...
The AMDGPU TargetMachine interface definition for hw codegen targets.
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Expand Atomic instructions
Atomic ordering constants.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
AMD GCN specific subclass of TargetSubtarget.
This header defines various interfaces for pass management in LLVM.
Machine Check Debug Module
static bool processFunction(Function &F, NVPTXTargetMachine &TM)
uint64_t IntrinsicInst * II
OptimizedStructLayoutField Field
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
const SmallVectorImpl< MachineOperand > & Cond
static void visit(BasicBlock &Start, std::function< bool(BasicBlock *)> op)
This file defines generic set operations that may be used on set's of different types,...
This file defines the SmallVector class.
static SymbolRef::Type getType(const Symbol *Sym)
Target-Independent Code Generator Pass Configuration Options pass.
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
bool ule(const APInt &RHS) const
Unsigned less or equal comparison.
bool sge(const APInt &RHS) const
Signed greater or equal comparison.
This class represents a conversion between pointers from one address space to another.
Value * getPointerOperand()
Gets the pointer operand.
unsigned getSrcAddressSpace() const
Returns the address space of the pointer operand.
unsigned getDestAddressSpace() const
Returns the address space of the result.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
This class represents an incoming formal argument to a Function.
An instruction that atomically checks whether a specified value is in a memory location,...
Value * getNewValOperand()
AtomicOrdering getMergedOrdering() const
Returns a single ordering which is at least as strong as both the success and failure orderings for t...
bool isVolatile() const
Return true if this is a cmpxchg from a volatile memory location.
Value * getCompareOperand()
Value * getPointerOperand()
Align getAlign() const
Return the alignment of the memory that is being allocated by the instruction.
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this cmpxchg instruction.
an instruction that atomically reads a memory location, combines it with another value,...
Align getAlign() const
Return the alignment of the memory that is being allocated by the instruction.
bool isVolatile() const
Return true if this is a RMW on a volatile memory location.
@ USubCond
Subtract only if no unsigned overflow.
@ FMinimum
*p = minimum(old, v) minimum matches the behavior of llvm.minimum.
@ Min
*p = old <signed v ? old : v
@ USubSat
*p = usub.sat(old, v) usub.sat matches the behavior of llvm.usub.sat.
@ FMaximum
*p = maximum(old, v) maximum matches the behavior of llvm.maximum.
@ UIncWrap
Increment one up to a maximum value.
@ Max
*p = old >signed v ? old : v
@ UMin
*p = old <unsigned v ? old : v
@ FMin
*p = minnum(old, v) minnum matches the behavior of llvm.minnum.
@ UMax
*p = old >unsigned v ? old : v
@ FMaximumNum
*p = maximumnum(old, v) maximumnum matches the behavior of llvm.maximumnum.
@ FMax
*p = maxnum(old, v) maxnum matches the behavior of llvm.maxnum.
@ UDecWrap
Decrement one until a minimum value or zero.
@ FMinimumNum
*p = minimumnum(old, v) minimumnum matches the behavior of llvm.minimumnum.
Value * getPointerOperand()
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this rmw instruction.
AtomicOrdering getOrdering() const
Returns the ordering constraint of this rmw instruction.
This class holds the attributes for a particular argument, parameter, function, or return value.
LLVM_ABI AttributeSet removeAttributes(LLVMContext &C, const AttributeMask &AttrsToRemove) const
Remove the specified attributes from this set.
LLVM Basic Block Representation.
LLVM_ABI void removeFromParent()
Unlink 'this' from the containing function, but do not delete it.
LLVM_ABI void insertInto(Function *Parent, BasicBlock *InsertBefore=nullptr)
Insert unlinked basic block into a function.
void addParamAttr(unsigned ArgNo, Attribute::AttrKind Kind)
Adds the attribute to the indicated argument.
This class represents a function call, abstracting a target machine's calling convention.
static LLVM_ABI Constant * get(StructType *T, ArrayRef< Constant * > V)
static LLVM_ABI Constant * getSplat(ElementCount EC, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
static LLVM_ABI Constant * get(ArrayRef< Constant * > V)
This is an important base class in LLVM.
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
static LLVM_ABI std::optional< DIExpression * > createFragmentExpression(const DIExpression *Expr, unsigned OffsetInBits, unsigned SizeInBits)
Create a DIExpression to describe one part of an aggregate variable that is fragmented across multipl...
A parsed version of the target data layout string in and methods for querying it.
LLVM_ABI void insertBefore(DbgRecord *InsertBefore)
LLVM_ABI void eraseFromParent()
LLVM_ABI void replaceVariableLocationOp(Value *OldValue, Value *NewValue, bool AllowEmpty=false)
void setExpression(DIExpression *NewExpr)
iterator find(const_arg_type_t< KeyT > Val)
Implements a dense probed hash-table based set.
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
This class represents a freeze function that returns random concrete value if an operand is either a ...
static Function * Create(FunctionType *Ty, LinkageTypes Linkage, unsigned AddrSpace, const Twine &N="", Module *M=nullptr)
const BasicBlock & front() const
iterator_range< arg_iterator > args()
AttributeList getAttributes() const
Return the attribute list for this Function.
bool isIntrinsic() const
isIntrinsic - Returns true if the function's name starts with "llvm.".
void setAttributes(AttributeList Attrs)
Set the attribute list for this Function.
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
void updateAfterNameChange()
Update internal caches that depend on the function name (such as the intrinsic ID and libcall cache).
Type * getReturnType() const
Returns the type of the ret val.
void copyAttributesFrom(const Function *Src)
copyAttributesFrom - copy all additional attributes (those not needed to create a Function) from the ...
bool hasRelaxedBufferOOBMode() const
bool hasUnalignedBufferAccessEnabled() const
static GEPNoWrapFlags noUnsignedWrap()
static GEPNoWrapFlags none()
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
LLVM_ABI void copyMetadata(const GlobalObject *Src, unsigned Offset)
Copy metadata from Src, adjusting offsets by Offset.
LinkageTypes getLinkage() const
void setDLLStorageClass(DLLStorageClassTypes C)
unsigned getAddressSpace() const
Module * getParent()
Get the module that this global value is contained inside of...
DLLStorageClassTypes getDLLStorageClass() const
This instruction compares its operands according to the predicate given to the constructor.
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
This instruction inserts a single (scalar) element into a VectorType value.
InstSimplifyFolder - Use InstructionSimplify to fold operations to existing values.
Base class for instruction visitors.
LLVM_ABI Instruction * clone() const
Create a copy of 'this' instruction that is identical in all ways except the following:
LLVM_ABI void setAAMetadata(const AAMDNodes &N)
Sets the AA metadata on this instruction from the AAMDNodes structure.
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this Instruction.
LLVM_ABI AAMDNodes getAAMetadata() const
Returns the AA metadata for this instruction.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
This class represents a cast from an integer to a pointer.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
A wrapper class for inspecting calls to intrinsic functions.
This is an important class for using LLVM in a threaded context.
LLVM_ABI void emitError(const Instruction *I, const Twine &ErrorStr)
emitError - Emit an error message to the currently installed error handler with optional location inf...
An instruction for reading from memory.
unsigned getPointerAddressSpace() const
Returns the address space of the pointer operand.
Value * getPointerOperand()
bool isVolatile() const
Return true if this is a load from a volatile memory location.
void setAtomic(AtomicOrdering Ordering, SyncScope::ID SSID=SyncScope::System)
Sets the ordering constraint and the synchronization scope ID of this load instruction.
AtomicOrdering getOrdering() const
Returns the ordering constraint of this load instruction.
Type * getPointerOperandType() const
void setVolatile(bool V)
Specify whether this is a volatile load or not.
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this load instruction.
Align getAlign() const
Return the alignment of the access that is being performed.
unsigned getDestAddressSpace() const
unsigned getSourceAddressSpace() const
ModulePass class - This class is used to implement unstructured interprocedural optimizations and ana...
A Module instance is used to store all the information related to an LLVM module.
const FunctionListType & getFunctionList() const
Get the Module's list of functions (constant).
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
A set of analyses that are preserved following a run of a transformation pass.
static PreservedAnalyses none()
Convenience factory function for the empty preserved set.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
This class represents a cast from a pointer to an address (non-capturing ptrtoint).
Value * getPointerOperand()
Gets the pointer operand.
This class represents a cast from a pointer to an integer.
Value * getPointerOperand()
Gets the pointer operand.
LLVM_ABI bool isAllOnesValue() const
Return true if the expression is a constant all-ones value.
Type * getType() const
Return the LLVM type of this SCEV expression.
Analysis pass that exposes the ScalarEvolution for a function.
The main scalar evolution driver.
LLVM_ABI bool isKnownNonNegative(const SCEV *S)
Test if the given expression is known to be non-negative.
LLVM_ABI bool isKnownNonPositive(const SCEV *S)
Test if the given expression is known to be non-positive.
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
LLVM_ABI const SCEV * getSCEV(Value *V)
Return a SCEV expression for the full generality of the specified expression.
LLVM_ABI const SCEV * getMinusSCEV(SCEVUse LHS, SCEVUse RHS, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap, unsigned Depth=0)
Return LHS-RHS.
LLVM_ABI const SCEV * getTruncateOrNoop(const SCEV *V, Type *Ty)
Return a SCEV corresponding to a conversion of the input value to the specified type.
LLVM_ABI bool isSCEVable(Type *Ty) const
Test if values of the given type are analyzable within the SCEV framework.
APInt getSignedRangeMin(const SCEV *S)
Determine the min of the signed range for a particular SCEV.
LLVM_ABI const SCEV * getNoopOrZeroExtend(const SCEV *V, Type *Ty)
Return a SCEV corresponding to a conversion of the input value to the specified type.
LLVM_ABI const SCEV * getPointerBase(const SCEV *V)
Transitively follow the chain of pointer-type operands until reaching a SCEV that does not have a sin...
APInt getUnsignedRangeMax(const SCEV *S)
Determine the max of the unsigned range for a particular SCEV.
LLVM_ABI const SCEV * getAddExpr(SmallVectorImpl< SCEVUse > &Ops, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap, unsigned Depth=0)
Get a canonical add expression, or something simpler if possible.
This class represents the LLVM 'select' instruction.
ArrayRef< value_type > getArrayRef() const
bool insert(const value_type &X)
Insert a new element into the SetVector.
This instruction constructs a fixed permutation of two input vectors.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Value * getValueOperand()
Value * getPointerOperand()
MutableArrayRef< TypeSize > getMemberOffsets()
static LLVM_ABI StructType * get(LLVMContext &Context, ArrayRef< Type * > Elements, bool isPacked=false)
This static method is the primary way to create a literal StructType.
static LLVM_ABI StructType * create(LLVMContext &Context, StringRef Name)
This creates an identified struct.
bool isLiteral() const
Return true if this type is uniqued by structural equivalence, false if it is a struct definition.
Type * getElementType(unsigned N) const
Analysis pass providing the TargetTransformInfo.
Primary interface to the complete machine description for the target machine.
const STC & getSubtarget(const Function &F) const
This method returns a pointer to the specified type of TargetSubtargetInfo.
Target-Independent Code Generator Pass Configuration Options.
TMC & getTM() const
Get the right type of TargetMachine for this target.
The instances of the Type class are immutable: once they are created, they are never changed.
LLVM_ABI unsigned getIntegerBitWidth() const
bool isVectorTy() const
True if this is an instance of VectorType.
Type * getArrayElementType() const
ArrayRef< Type * > subtypes() const
bool isSingleValueType() const
Return true if the type is a valid type for a register in codegen.
unsigned getNumContainedTypes() const
Return the number of types in the derived type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isIntegerTy() const
True if this is an instance of IntegerType.
Type * getContainedType(unsigned i) const
This method is used to implement the type iterator (defined at the end of the file).
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
void setOperand(unsigned i, Value *Val)
Value * getOperand(unsigned i) const
This is a class that can be implemented by clients to remap types when cloning constants and instruct...
size_type count(const KeyT &Val) const
Return 1 if the specified key is in the map, 0 otherwise.
iterator find(const KeyT &Val)
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
LLVM_ABI Constant * mapConstant(const Constant &C)
LLVM_ABI Value * mapValue(const Value &V)
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
LLVMContext & getContext() const
All values hold a context through their type.
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
std::pair< iterator, bool > insert(const ValueT &V)
bool contains(const_arg_type_t< ValueT > V) const
Check if the set contains the given element.
constexpr bool isKnownMultipleOf(ScalarTy RHS) const
This function tells the caller whether the element count is known at compile time to be a multiple of...
constexpr ScalarTy getFixedValue() const
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
An efficient, type-erasing, non-owning reference to a callable.
self_iterator getIterator()
iterator insertAfter(iterator where, pointer New)
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ BUFFER_FAT_POINTER
Address space for 160-bit buffer fat pointers.
@ BUFFER_RESOURCE
Address space for 128-bit buffer resources.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
LLVM_ABI std::optional< Function * > remangleIntrinsicFunction(Function *F)
bool match(Val *V, const Pattern &P)
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
SmallVector< DbgVariableRecord * > getDVRAssignmentMarkers(const Instruction *Inst)
Return a range of dbg_assign records for which Inst performs the assignment they encode.
DXILDebugInfoMap run(Module &M)
friend class Instruction
Iterator for Instructions in a `BasicBlock.
This is an optimization pass for GlobalISel generic memory operations.
detail::zippy< detail::zip_shortest, T, U, Args... > zip(T &&t, U &&u, Args &&...args)
zip iterator for two or more iteratable types.
LLVM_ABI void findDbgValues(Value *V, SmallVectorImpl< DbgVariableRecord * > &DbgVariableRecords)
Finds the dbg.values describing a value.
ModulePass * createAMDGPULowerBufferFatPointersPass()
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
LLVM_ABI void copyMetadataForLoad(LoadInst &Dest, const LoadInst &Source)
Copy the metadata from the source instruction to the destination (the replacement for the source inst...
bool set_is_subset(const S1Ty &S1, const S2Ty &S2)
set_is_subset(A, B) - Return true iff A in B
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
InnerAnalysisManagerProxy< FunctionAnalysisManager, Module > FunctionAnalysisManagerModuleProxy
Provide the FunctionAnalysisManager to Module proxy.
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
auto dyn_cast_or_null(const Y &Val)
LLVM_ABI bool convertUsersOfConstantsToInstructions(ArrayRef< Constant * > Consts, Function *RestrictToFunc=nullptr, bool RemoveDeadConstants=true, bool IncludeSelf=false)
Replace constant expressions users of the given constants with instructions.
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI Value * emitGEPOffset(IRBuilderBase *Builder, const DataLayout &DL, User *GEP, bool NoAssumptions=false)
Given a getelementptr instruction/constantexpr, emit the code necessary to compute the offset from th...
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
char & AMDGPULowerBufferFatPointersID
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
MutableArrayRef(T &OneElt) -> MutableArrayRef< T >
AtomicOrdering
Atomic ordering for LLVM's memory model.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
IRBuilder(LLVMContext &, FolderTy, InserterTy, MDNode *, ArrayRef< OperandBundleDef >) -> IRBuilder< FolderTy, InserterTy >
DWARFExpression::Operation Op
S1Ty set_difference(const S1Ty &S1, const S2Ty &S2)
set_difference(A, B) - Return A - B
ArrayRef(const T &OneElt) -> ArrayRef< T >
ValueMap< const Value *, WeakTrackingVH > ValueToValueMapTy
LLVM_ABI void expandMemSetAsLoop(MemSetInst *MemSet, const TargetTransformInfo *TTI=nullptr)
Expand MemSet as a loop.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
LLVM_ABI void expandMemSetPatternAsLoop(MemSetPatternInst *MemSet, const TargetTransformInfo *TTI=nullptr)
Expand MemSetPattern as a loop.
iterator_range< pointer_iterator< WrappedIteratorT > > make_pointer_range(RangeT &&Range)
Align commonAlignment(Align A, uint64_t Offset)
Returns the alignment that satisfies both alignments.
LLVM_ABI void expandMemCpyAsLoop(MemCpyInst *MemCpy, const TargetTransformInfo &TTI, ScalarEvolution *SE=nullptr)
Expand MemCpy as a loop. MemCpy is not deleted.
AnalysisManager< Module > ModuleAnalysisManager
Convenience typedef for the Module analysis manager.
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
LLVM_ABI AAMDNodes adjustForAccess(unsigned AccessSize)
Create a new AAMDNode for accessing AccessSize bytes of this AAMDNode.
PreservedAnalyses run(Module &M, ModuleAnalysisManager &AM)
This struct is a compact representation of a valid (non-zero power of two) alignment.