28#include "llvm/IR/IntrinsicsAMDGPU.h"
36#define DEBUG_TYPE "AMDGPUtti"
40struct AMDGPUImageDMaskIntrinsic {
44#define GET_AMDGPUImageDMaskIntrinsicTable_IMPL
45#include "AMDGPUGenSearchableTables.inc"
56 "nans handled separately");
74 bool AllowI16SExt =
false) {
75 Type *VTy = V.getType();
84 APFloat FloatValue(ConstFloat->getValueAPF());
85 bool LosesInfo =
true;
94 APInt IntValue(ConstInt->getValue());
103 Value *CastCandidate;
110 if (!IsExt && !IsFloat && AllowI16SExt)
123 Type *VTy = V.getType();
132 return Builder.CreateExtractElement(VecCast->
getOperand(0), Idx);
156 Func(Args, OverloadTys);
172 bool RemoveOldIntr = &OldIntr != &InstToReplace;
181static std::optional<Instruction *>
186 if (
const auto *LZMappingInfo =
188 if (
auto *ConstantLod =
190 if (ConstantLod->isZero() || ConstantLod->isNegative()) {
195 II,
II, NewImageDimIntr->
Intr, IC, [&](
auto &Args,
auto &ArgTys) {
196 Args.erase(Args.begin() + ImageDimIntr->LodIndex);
203 if (
const auto *MIPMappingInfo =
205 if (
auto *ConstantMip =
207 if (ConstantMip->isZero()) {
212 II,
II, NewImageDimIntr->
Intr, IC, [&](
auto &Args,
auto &ArgTys) {
213 Args.erase(Args.begin() + ImageDimIntr->MipIndex);
220 if (
const auto *BiasMappingInfo =
222 if (
auto *ConstantBias =
224 if (ConstantBias->isZero()) {
229 II,
II, NewImageDimIntr->
Intr, IC, [&](
auto &Args,
auto &ArgTys) {
230 Args.erase(Args.begin() + ImageDimIntr->BiasIndex);
231 ArgTys.erase(ArgTys.begin() + ImageDimIntr->BiasTyArg);
238 if (
const auto *OffsetMappingInfo =
240 if (
auto *ConstantOffset =
242 if (ConstantOffset->isZero()) {
245 OffsetMappingInfo->NoOffset, ImageDimIntr->
Dim);
247 II,
II, NewImageDimIntr->
Intr, IC, [&](
auto &Args,
auto &ArgTys) {
248 Args.erase(Args.begin() + ImageDimIntr->OffsetIndex);
255 if (ST->hasD16Images()) {
265 if (
II.hasOneUse()) {
268 if (
User->getOpcode() == Instruction::FPTrunc &&
272 [&](
auto &Args,
auto &ArgTys) {
275 ArgTys[0] = User->getType();
284 bool AllHalfExtracts =
true;
286 for (
User *U :
II.users()) {
288 if (!Ext || !Ext->hasOneUse()) {
289 AllHalfExtracts =
false;
294 if (!Tr || !Tr->getType()->isHalfTy()) {
295 AllHalfExtracts =
false;
302 if (!ExtractTruncPairs.
empty() && AllHalfExtracts) {
313 OverloadTys[0] = HalfVecTy;
316 M, ImageDimIntr->
Intr, OverloadTys);
318 II.mutateType(HalfVecTy);
319 II.setCalledFunction(HalfDecl);
322 for (
auto &[Ext, Tr] : ExtractTruncPairs) {
323 Value *Idx = Ext->getIndexOperand();
325 Builder.SetInsertPoint(Tr);
327 Value *HalfExtract = Builder.CreateExtractElement(&
II, Idx);
330 Tr->replaceAllUsesWith(HalfExtract);
333 for (
auto &[Ext, Tr] : ExtractTruncPairs) {
344 if (!ST->hasA16() && !ST->hasG16())
351 bool FloatCoord =
false;
353 bool OnlyDerivatives =
false;
358 bool AllowI16SExt = !HasSampler;
361 OperandIndex < ImageDimIntr->VAddrEnd; OperandIndex++) {
362 Value *Coord =
II.getOperand(OperandIndex);
365 if (OperandIndex < ImageDimIntr->CoordStart ||
370 OnlyDerivatives =
true;
379 if (!OnlyDerivatives && !ST->hasA16())
380 OnlyDerivatives =
true;
383 if (!OnlyDerivatives && ImageDimIntr->
NumBiasArgs != 0) {
386 "Only image instructions with a sampler can have a bias");
388 OnlyDerivatives =
true;
391 if (OnlyDerivatives && (!ST->hasG16() || ImageDimIntr->
GradientStart ==
399 II,
II,
II.getIntrinsicID(), IC, [&](
auto &Args,
auto &ArgTys) {
400 ArgTys[ImageDimIntr->GradientTyArg] = CoordType;
401 if (!OnlyDerivatives) {
402 ArgTys[ImageDimIntr->CoordTyArg] = CoordType;
405 if (ImageDimIntr->NumBiasArgs != 0)
406 ArgTys[ImageDimIntr->BiasTyArg] = Type::getHalfTy(II.getContext());
412 OperandIndex < EndIndex; OperandIndex++) {
414 convertTo16Bit(*II.getOperand(OperandIndex), IC.Builder);
419 Value *Bias = II.getOperand(ImageDimIntr->BiasIndex);
420 Args[ImageDimIntr->BiasIndex] = convertTo16Bit(*Bias, IC.Builder);
454 if (
I.hasNoSignedZeros() &&
464 Value *Src =
nullptr;
467 if (Src->getType()->isHalfTy())
484 unsigned VWidth = VTy->getNumElements();
487 for (
int i = VWidth - 1; i > 0; --i) {
509 unsigned VWidth = VTy->getNumElements();
515 SVI->getShuffleMask(ShuffleMask);
517 for (
int I = VWidth - 1;
I > 0; --
I) {
518 if (ShuffleMask.empty()) {
569 unsigned LaneArgIdx)
const {
570 unsigned MaskBits = ST->getWavefrontSizeLog2();
577 if (!
Known.isConstant())
584 Value *LaneArg =
II.getArgOperand(LaneArgIdx);
586 ConstantInt::get(LaneArg->
getType(),
Known.getConstant() & DemandedMask);
587 if (MaskedConst != LaneArg) {
588 II.getOperandUse(LaneArgIdx).set(MaskedConst);
600 CallInst *NewCall =
B.CreateCall(&NewCallee,
Ops, OpBundles);
616 if (ST.isWave32() &&
match(V, W32Pred))
618 if (ST.isWave64() &&
match(V, W64Pred))
627 const auto IID =
II.getIntrinsicID();
628 assert(IID == Intrinsic::amdgcn_readlane ||
629 IID == Intrinsic::amdgcn_readfirstlane ||
630 IID == Intrinsic::amdgcn_permlane64);
640 const bool IsReadLane = (IID == Intrinsic::amdgcn_readlane);
644 Value *LaneID =
nullptr;
646 LaneID =
II.getOperand(1);
660 const auto DoIt = [&](
unsigned OpIdx,
664 Ops.push_back(LaneID);
680 return DoIt(0,
II.getCalledFunction());
684 Type *SrcTy = Src->getType();
690 return DoIt(0, Remangled);
698 return DoIt(1,
II.getCalledFunction());
700 return DoIt(0,
II.getCalledFunction());
711 unsigned Depth = 0) {
721 return CI->getZExtValue();
730 std::optional<unsigned>
LHS =
734 std::optional<unsigned>
RHS =
743 return CI ? std::optional<unsigned>(CI->getZExtValue()) : std::nullopt;
751 unsigned WaveSize = ST.getWavefrontSize();
753 for (
unsigned Lane :
seq(WaveSize)) {
755 if (!Val || *Val >= WaveSize)
764template <
unsigned Period>
766 static_assert(
isPowerOf2_32(Period),
"Period must be a power of two");
767 for (
unsigned I = Period,
E = Ids.
size();
I <
E; ++
I)
768 if (Ids[
I] != Ids[
I % Period] + (
I & ~(Period - 1)))
776 for (
unsigned I = 0;
I <
N; ++
I)
792 return Ids[3] << 6 | Ids[2] << 4 | Ids[1] << 2 | Ids[0];
799 for (
unsigned J = 0; J <
N; ++J)
800 if (Ids[J] != (
N - 1) - J)
812 for (
unsigned J = 1; J < 16; ++J)
813 if (Ids[J] != (Ids[0] + J) % 16)
831 unsigned Mask = Ids[0];
834 for (
unsigned J = 0; J < 16; ++J)
835 if (Ids[J] != (Mask ^ J))
845 unsigned Selector = 0;
846 for (
unsigned J = 0; J < 8; ++J)
847 Selector |= Ids[J] << (J * 3);
856 for (
unsigned J = 0; J < 16; ++J)
857 Sel |=
static_cast<uint64_t>(Ids[J] & 0xF) << (J * 4);
864 if (Ids.
size() != 64)
866 for (
unsigned J = 0; J < 64; ++J)
867 if (Ids[J] != (J ^ 32))
878 for (
unsigned J = 0; J < 16; ++J) {
879 if (Ids[J] < 16 || Ids[J] >= 32)
881 if (Ids[J + 16] != Ids[J] - 16)
892static std::optional<unsigned>
901 unsigned AndMask = 0, OrMask = 0, XorMask = 0;
902 for (
unsigned B = 0;
B < 5; ++
B) {
903 unsigned Bit0 = (Ids[0] >>
B) & 1;
904 unsigned Bit1 = (Ids[1u <<
B] >>
B) & 1;
907 XorMask |= Bit0 <<
B;
915 for (
unsigned I :
seq(32u)) {
916 unsigned Expected = ((
I & AndMask) | OrMask) ^ XorMask;
931static std::optional<unsigned>
942 for (
unsigned I = 0;
I < 32; ++
I)
943 if (Ids[
I] != (
I +
N) % 32)
955 return B.CreateIntrinsic(Intrinsic::amdgcn_update_dpp, {Ty},
957 B.getInt32(0xF),
B.getInt32(0xF),
B.getTrue()});
962 return B.CreateIntrinsic(Intrinsic::amdgcn_mov_dpp8, {Val->
getType()},
963 {Val,
B.getInt32(Selector)});
970 return B.CreateIntrinsic(Intrinsic::amdgcn_permlane16, {Ty},
972 B.getInt32(
Hi),
B.getFalse(),
B.getFalse()});
980 return B.CreateIntrinsic(Intrinsic::amdgcn_permlanex16, {Ty},
982 B.getInt32(
Hi),
B.getFalse(),
B.getFalse()});
990 assert(
DL.getTypeSizeInBits(OrigTy) == 32 &&
991 "ds_swizzle only supports 32-bit operands");
995 Src =
B.CreatePtrToInt(Src, I32Ty);
996 else if (OrigTy != I32Ty)
997 Src =
B.CreateBitCast(Src, I32Ty);
998 Value *Result =
B.CreateIntrinsic(Intrinsic::amdgcn_ds_swizzle, {},
1001 return B.CreateIntToPtr(Result, OrigTy);
1002 if (OrigTy != I32Ty)
1003 return B.CreateBitCast(Result, OrigTy);
1009 return B.CreateIntrinsic(Intrinsic::amdgcn_permlane64, {Val->
getType()},
1020 [](
const auto &
E) {
return E.value() ==
E.index(); }))
1044 if (ST.hasDPPRowShare()) {
1049 if (ST.hasDPP() && ST.hasGFX10Insts()) {
1059 if (ST.hasPermlane16Insts()) {
1079 if (ST.hasDsSwizzleRotateMode()) {
1092static std::optional<Instruction *>
1096 if (
DL.getTypeSizeInBits(
II.getType()) != 32)
1097 return std::nullopt;
1099 if (!ST.isWaveSizeKnown())
1100 return std::nullopt;
1102 unsigned WaveSize = ST.getWavefrontSize();
1103 bool IsBpermute =
II.getIntrinsicID() == Intrinsic::amdgcn_ds_bpermute;
1104 Value *Src =
II.getArgOperand(IsBpermute ? 1 : 0);
1105 Value *Index =
II.getArgOperand(IsBpermute ? 0 : 1);
1110 for (
unsigned Lane :
seq(WaveSize)) {
1112 if (!Val || (*Val & 3) || (*Val >> 2) >= WaveSize)
1113 return std::nullopt;
1114 Ids[Lane] = *Val >> 2;
1118 return std::nullopt;
1123 return std::nullopt;
1127std::optional<Instruction *>
1131 case Intrinsic::amdgcn_implicitarg_ptr: {
1132 if (
II.getFunction()->hasFnAttribute(
"amdgpu-no-implicitarg-ptr"))
1134 uint64_t ImplicitArgBytes = ST->getImplicitArgNumBytes(*
II.getFunction());
1137 II.getAttributes().getRetDereferenceableOrNullBytes();
1138 if (CurrentOrNullBytes != 0) {
1141 uint64_t NewBytes = std::max(CurrentOrNullBytes, ImplicitArgBytes);
1144 II.removeRetAttr(Attribute::DereferenceableOrNull);
1148 uint64_t CurrentBytes =
II.getAttributes().getRetDereferenceableBytes();
1149 uint64_t NewBytes = std::max(CurrentBytes, ImplicitArgBytes);
1150 if (NewBytes != CurrentBytes) {
1156 return std::nullopt;
1158 case Intrinsic::amdgcn_rcp: {
1159 Value *Src =
II.getArgOperand(0);
1170 if (
II.isStrictFP())
1188 auto IID = SrcCI->getIntrinsicID();
1193 if (IID == Intrinsic::amdgcn_sqrt || IID == Intrinsic::sqrt) {
1203 SrcCI->getModule(), Intrinsic::amdgcn_rsq, {SrcCI->getType()});
1206 II.setFastMathFlags(InnerFMF);
1208 II.setCalledFunction(NewDecl);
1214 case Intrinsic::amdgcn_sqrt:
1215 case Intrinsic::amdgcn_rsq:
1216 case Intrinsic::amdgcn_tanh: {
1217 Value *Src =
II.getArgOperand(0);
1229 if (IID == Intrinsic::amdgcn_sqrt && Src->getType()->isHalfTy()) {
1231 II.getModule(), Intrinsic::sqrt, {II.getType()});
1232 II.setCalledFunction(NewDecl);
1238 case Intrinsic::amdgcn_log:
1239 case Intrinsic::amdgcn_exp2: {
1240 const bool IsLog = IID == Intrinsic::amdgcn_log;
1241 const bool IsExp = IID == Intrinsic::amdgcn_exp2;
1242 Value *Src =
II.getArgOperand(0);
1252 if (
C->isInfinity()) {
1255 if (!
C->isNegative())
1259 if (IsExp &&
C->isNegative())
1263 if (
II.isStrictFP())
1267 Constant *Quieted = ConstantFP::get(Ty,
C->getValue().makeQuiet());
1272 if (
C->isZero() || (
C->getValue().isDenormal() && Ty->isFloatTy())) {
1274 : ConstantFP::get(Ty, 1.0);
1278 if (IsLog &&
C->isNegative())
1286 case Intrinsic::amdgcn_frexp_mant:
1287 case Intrinsic::amdgcn_frexp_exp: {
1288 Value *Src =
II.getArgOperand(0);
1294 if (IID == Intrinsic::amdgcn_frexp_mant) {
1296 II, ConstantFP::get(
II.getContext(), Significand));
1316 case Intrinsic::amdgcn_class: {
1317 Value *Src0 =
II.getArgOperand(0);
1318 Value *Src1 =
II.getArgOperand(1);
1322 II.getModule(), Intrinsic::is_fpclass, Src0->
getType()));
1325 II.setArgOperand(1, ConstantInt::get(Src1->
getType(),
1346 case Intrinsic::amdgcn_cvt_pkrtz: {
1347 auto foldFPTruncToF16RTZ = [](
Value *Arg) ->
Value * {
1360 return ConstantFP::get(HalfTy, Val);
1363 Value *Src =
nullptr;
1365 if (Src->getType()->isHalfTy())
1372 if (
Value *Src0 = foldFPTruncToF16RTZ(
II.getArgOperand(0))) {
1373 if (
Value *Src1 = foldFPTruncToF16RTZ(
II.getArgOperand(1))) {
1383 case Intrinsic::amdgcn_cvt_pknorm_i16:
1384 case Intrinsic::amdgcn_cvt_pknorm_u16:
1385 case Intrinsic::amdgcn_cvt_pk_i16:
1386 case Intrinsic::amdgcn_cvt_pk_u16: {
1387 Value *Src0 =
II.getArgOperand(0);
1388 Value *Src1 =
II.getArgOperand(1);
1400 case Intrinsic::amdgcn_cvt_off_f32_i4: {
1401 Value* Arg =
II.getArgOperand(0);
1415 constexpr size_t ResValsSize = 16;
1416 static constexpr float ResVals[ResValsSize] = {
1417 0.0, 0.0625, 0.125, 0.1875, 0.25, 0.3125, 0.375, 0.4375,
1418 -0.5, -0.4375, -0.375, -0.3125, -0.25, -0.1875, -0.125, -0.0625};
1420 ConstantFP::get(Ty, ResVals[CArg->
getZExtValue() & (ResValsSize - 1)]);
1423 case Intrinsic::amdgcn_ubfe:
1424 case Intrinsic::amdgcn_sbfe: {
1426 Value *Src =
II.getArgOperand(0);
1433 unsigned IntSize = Ty->getIntegerBitWidth();
1438 if ((Width & (IntSize - 1)) == 0) {
1443 if (Width >= IntSize) {
1445 II, 2, ConstantInt::get(CWidth->
getType(), Width & (IntSize - 1)));
1456 ConstantInt::get(COffset->
getType(),
Offset & (IntSize - 1)));
1460 bool Signed = IID == Intrinsic::amdgcn_sbfe;
1462 if (!CWidth || !COffset)
1472 if (
Offset + Width < IntSize) {
1476 RightShift->takeName(&
II);
1483 RightShift->takeName(&
II);
1486 case Intrinsic::amdgcn_exp:
1487 case Intrinsic::amdgcn_exp_row:
1488 case Intrinsic::amdgcn_exp_compr: {
1494 bool IsCompr = IID == Intrinsic::amdgcn_exp_compr;
1496 for (
int I = 0;
I < (IsCompr ? 2 : 4); ++
I) {
1497 if ((!IsCompr && (EnBits & (1 <<
I)) == 0) ||
1498 (IsCompr && ((EnBits & (0x3 << (2 *
I))) == 0))) {
1499 Value *Src =
II.getArgOperand(
I + 2);
1513 case Intrinsic::amdgcn_fmed3: {
1514 Value *Src0 =
II.getArgOperand(0);
1515 Value *Src1 =
II.getArgOperand(1);
1516 Value *Src2 =
II.getArgOperand(2);
1518 for (
Value *Src : {Src0, Src1, Src2}) {
1523 if (
II.isStrictFP())
1560 const APFloat *ConstSrc0 =
nullptr;
1561 const APFloat *ConstSrc1 =
nullptr;
1562 const APFloat *ConstSrc2 =
nullptr;
1567 const bool IsPosInfinity = ConstSrc0 && ConstSrc0->
isPosInfinity();
1587 const bool IsPosInfinity = ConstSrc1 && ConstSrc1->
isPosInfinity();
1610 auto *Quieted = ConstantFP::get(
II.getType(), ConstSrc2->
makeQuiet());
1630 CI->copyFastMathFlags(&
II);
1656 II.setArgOperand(0, Src0);
1657 II.setArgOperand(1, Src1);
1658 II.setArgOperand(2, Src2);
1668 ConstantFP::get(
II.getType(), Result));
1673 if (!ST->hasMed3_16())
1682 IID, {
X->getType()}, {
X,
Y, Z}, &
II,
II.getName());
1690 case Intrinsic::amdgcn_icmp:
1691 case Intrinsic::amdgcn_fcmp: {
1695 bool IsInteger = IID == Intrinsic::amdgcn_icmp;
1702 Value *Src0 =
II.getArgOperand(0);
1703 Value *Src1 =
II.getArgOperand(1);
1730 II.setArgOperand(0, Src1);
1731 II.setArgOperand(1, Src0);
1733 2, ConstantInt::get(CC->
getType(),
static_cast<int>(SwapPred)));
1780 ? Intrinsic::amdgcn_fcmp
1781 : Intrinsic::amdgcn_icmp;
1786 unsigned Width = CmpType->getBitWidth();
1787 unsigned NewWidth = Width;
1795 else if (Width <= 32)
1797 else if (Width <= 64)
1802 if (Width != NewWidth) {
1812 }
else if (!Ty->isFloatTy() && !Ty->isDoubleTy() && !Ty->isHalfTy())
1815 Value *Args[] = {SrcLHS, SrcRHS,
1816 ConstantInt::get(CC->
getType(), SrcPred)};
1818 NewIID, {
II.getType(), SrcLHS->
getType()}, Args);
1825 case Intrinsic::amdgcn_mbcnt_hi:
1830 case Intrinsic::amdgcn_mbcnt_lo: {
1843 if (std::optional<ConstantRange> ExistingRange =
II.getRange()) {
1844 ComputedRange = ComputedRange.
intersectWith(*ExistingRange);
1845 if (ComputedRange == *ExistingRange)
1849 II.addRangeRetAttr(ComputedRange);
1852 case Intrinsic::amdgcn_ballot: {
1853 Value *Arg =
II.getArgOperand(0);
1858 if (Src->isZero()) {
1863 if (ST->isWave32() &&
II.getType()->getIntegerBitWidth() == 64) {
1870 {IC.Builder.getInt32Ty()},
1871 {II.getArgOperand(0)}),
1878 case Intrinsic::amdgcn_wavefrontsize: {
1879 if (ST->isWaveSizeKnown())
1881 II, ConstantInt::get(
II.getType(), ST->getWavefrontSize()));
1884 case Intrinsic::amdgcn_wqm_vote: {
1891 case Intrinsic::amdgcn_kill: {
1893 if (!
C || !
C->getZExtValue())
1899 case Intrinsic::amdgcn_s_sendmsg:
1900 case Intrinsic::amdgcn_s_sendmsghalt: {
1906 Value *M0Val =
II.getArgOperand(1);
1912 decodeMsg(MsgImm->getZExtValue(), MsgId, OpId, StreamId, *ST);
1914 if (!msgDoesNotUseM0(MsgId, *ST))
1918 II.dropUBImplyingAttrsAndMetadata();
1922 case Intrinsic::amdgcn_update_dpp: {
1923 Value *Old =
II.getArgOperand(0);
1928 if (BC->isNullValue() || RM->getZExtValue() != 0xF ||
1935 case Intrinsic::amdgcn_permlane16:
1936 case Intrinsic::amdgcn_permlane16_var:
1937 case Intrinsic::amdgcn_permlanex16:
1938 case Intrinsic::amdgcn_permlanex16_var: {
1940 Value *VDstIn =
II.getArgOperand(0);
1945 unsigned int FiIdx = (IID == Intrinsic::amdgcn_permlane16 ||
1946 IID == Intrinsic::amdgcn_permlanex16)
1953 unsigned int BcIdx = FiIdx + 1;
1962 case Intrinsic::amdgcn_wave_shuffle:
1964 case Intrinsic::amdgcn_permlane64:
1965 case Intrinsic::amdgcn_readfirstlane:
1966 case Intrinsic::amdgcn_readlane:
1967 case Intrinsic::amdgcn_ds_bpermute: {
1969 unsigned SrcIdx = IID == Intrinsic::amdgcn_ds_bpermute ? 1 : 0;
1970 const Use &Src =
II.getArgOperandUse(SrcIdx);
1974 if (IID == Intrinsic::amdgcn_readlane &&
1981 if (IID == Intrinsic::amdgcn_ds_bpermute) {
1982 const Use &Lane =
II.getArgOperandUse(0);
1986 II.getModule(), Intrinsic::amdgcn_readlane,
II.getType());
1987 II.setCalledFunction(NewDecl);
1988 II.setOperand(0, Src);
1989 II.setOperand(1, NewLane);
1994 if (IID == Intrinsic::amdgcn_ds_bpermute)
2000 return std::nullopt;
2002 case Intrinsic::amdgcn_writelane: {
2006 return std::nullopt;
2008 case Intrinsic::amdgcn_trig_preop: {
2011 if (!
II.getType()->isDoubleTy())
2014 Value *Src =
II.getArgOperand(0);
2015 Value *Segment =
II.getArgOperand(1);
2024 if (StrippedSign != Src)
2027 if (
II.isStrictFP())
2049 unsigned Shift = SegmentVal * 53;
2054 static const uint32_t TwoByPi[] = {
2055 0xa2f9836e, 0x4e441529, 0xfc2757d1, 0xf534ddc0, 0xdb629599, 0x3c439041,
2056 0xfe5163ab, 0xdebbc561, 0xb7246e3a, 0x424dd2e0, 0x06492eea, 0x09d1921c,
2057 0xfe1deb1c, 0xb129a73e, 0xe88235f5, 0x2ebb4484, 0xe99c7026, 0xb45f7e41,
2058 0x3991d639, 0x835339f4, 0x9c845f8b, 0xbdf9283b, 0x1ff897ff, 0xde05980f,
2059 0xef2f118b, 0x5a0a6d1f, 0x6d367ecf, 0x27cb09b7, 0x4f463f66, 0x9e5fea2d,
2060 0x7527bac7, 0xebe5f17b, 0x3d0739f7, 0x8a5292ea, 0x6bfb5fb1, 0x1f8d5d08,
2064 unsigned Idx = Shift >> 5;
2065 if (Idx + 2 >= std::size(TwoByPi)) {
2070 unsigned BShift = Shift & 0x1f;
2074 Thi = (Thi << BShift) | (Tlo >> (64 - BShift));
2078 int Scale = -53 - Shift;
2085 case Intrinsic::amdgcn_fmul_legacy: {
2086 Value *Op0 =
II.getArgOperand(0);
2087 Value *Op1 =
II.getArgOperand(1);
2089 for (
Value *Src : {Op0, Op1}) {
2110 case Intrinsic::amdgcn_fma_legacy: {
2111 Value *Op0 =
II.getArgOperand(0);
2112 Value *Op1 =
II.getArgOperand(1);
2113 Value *Op2 =
II.getArgOperand(2);
2115 for (
Value *Src : {Op0, Op1, Op2}) {
2137 II.getModule(), Intrinsic::fma,
II.getType()));
2142 case Intrinsic::amdgcn_is_shared:
2143 case Intrinsic::amdgcn_is_private: {
2144 Value *Src =
II.getArgOperand(0);
2154 case Intrinsic::amdgcn_make_buffer_rsrc: {
2155 Value *Src =
II.getArgOperand(0);
2158 return std::nullopt;
2160 case Intrinsic::amdgcn_raw_buffer_store_format:
2161 case Intrinsic::amdgcn_struct_buffer_store_format:
2162 case Intrinsic::amdgcn_raw_tbuffer_store:
2163 case Intrinsic::amdgcn_struct_tbuffer_store:
2164 case Intrinsic::amdgcn_image_store_1d:
2165 case Intrinsic::amdgcn_image_store_1darray:
2166 case Intrinsic::amdgcn_image_store_2d:
2167 case Intrinsic::amdgcn_image_store_2darray:
2168 case Intrinsic::amdgcn_image_store_2darraymsaa:
2169 case Intrinsic::amdgcn_image_store_2dmsaa:
2170 case Intrinsic::amdgcn_image_store_3d:
2171 case Intrinsic::amdgcn_image_store_cube:
2172 case Intrinsic::amdgcn_image_store_mip_1d:
2173 case Intrinsic::amdgcn_image_store_mip_1darray:
2174 case Intrinsic::amdgcn_image_store_mip_2d:
2175 case Intrinsic::amdgcn_image_store_mip_2darray:
2176 case Intrinsic::amdgcn_image_store_mip_3d:
2177 case Intrinsic::amdgcn_image_store_mip_cube: {
2182 if (ST->hasDefaultComponentBroadcast())
2184 else if (ST->hasDefaultComponentZero())
2189 int DMaskIdx = getAMDGPUImageDMaskIntrinsic(
II.getIntrinsicID()) ? 1 : -1;
2197 case Intrinsic::amdgcn_prng_b32: {
2198 auto *Src =
II.getArgOperand(0);
2202 return std::nullopt;
2204 case Intrinsic::amdgcn_mfma_scale_f32_16x16x128_f8f6f4:
2205 case Intrinsic::amdgcn_mfma_scale_f32_32x32x64_f8f6f4: {
2206 Value *Src0 =
II.getArgOperand(0);
2207 Value *Src1 =
II.getArgOperand(1);
2213 auto getFormatNumRegs = [](
unsigned FormatVal) {
2214 switch (FormatVal) {
2228 bool MadeChange =
false;
2229 unsigned Src0NumElts = getFormatNumRegs(CBSZ);
2230 unsigned Src1NumElts = getFormatNumRegs(BLGP);
2234 if (Src0Ty->getNumElements() > Src0NumElts) {
2241 if (Src1Ty->getNumElements() > Src1NumElts) {
2249 return std::nullopt;
2260 case Intrinsic::amdgcn_wmma_f32_16x16x128_f8f6f4:
2261 case Intrinsic::amdgcn_wmma_scale_f32_16x16x128_f8f6f4:
2262 case Intrinsic::amdgcn_wmma_scale16_f32_16x16x128_f8f6f4: {
2263 Value *Src0 =
II.getArgOperand(1);
2264 Value *Src1 =
II.getArgOperand(3);
2270 bool MadeChange =
false;
2276 if (Src0Ty->getNumElements() > Src0NumElts) {
2283 if (Src1Ty->getNumElements() > Src1NumElts) {
2291 return std::nullopt;
2308 return std::nullopt;
2321 int DMaskIdx,
bool IsLoad) {
2324 :
II.getOperand(0)->getType());
2325 unsigned VWidth = IIVTy->getNumElements();
2328 Type *EltTy = IIVTy->getElementType();
2340 const unsigned UnusedComponentsAtFront = DemandedElts.
countr_zero();
2345 DemandedElts = (1 << ActiveBits) - 1;
2347 if (UnusedComponentsAtFront > 0) {
2348 static const unsigned InvalidOffsetIdx = 0xf;
2351 switch (
II.getIntrinsicID()) {
2352 case Intrinsic::amdgcn_raw_buffer_load:
2353 case Intrinsic::amdgcn_raw_ptr_buffer_load:
2356 case Intrinsic::amdgcn_s_buffer_load:
2357 case Intrinsic::amdgcn_ptr_s_buffer_load:
2361 if (ActiveBits == 4 && UnusedComponentsAtFront == 1)
2362 OffsetIdx = InvalidOffsetIdx;
2366 case Intrinsic::amdgcn_struct_buffer_load:
2367 case Intrinsic::amdgcn_struct_ptr_buffer_load:
2372 OffsetIdx = InvalidOffsetIdx;
2376 if (OffsetIdx != InvalidOffsetIdx) {
2378 DemandedElts &= ~((1 << UnusedComponentsAtFront) - 1);
2379 auto *
Offset = Args[OffsetIdx];
2380 unsigned SingleComponentSizeInBits =
2382 unsigned OffsetAdd =
2383 UnusedComponentsAtFront * SingleComponentSizeInBits / 8;
2384 auto *OffsetAddVal = ConstantInt::get(
Offset->getType(), OffsetAdd);
2404 unsigned NewDMaskVal = 0;
2405 unsigned OrigLdStIdx = 0;
2406 for (
unsigned SrcIdx = 0; SrcIdx < 4; ++SrcIdx) {
2407 const unsigned Bit = 1 << SrcIdx;
2408 if (!!(DMaskVal & Bit)) {
2409 if (!!DemandedElts[OrigLdStIdx])
2415 if (DMaskVal != NewDMaskVal)
2416 Args[DMaskIdx] = ConstantInt::get(DMask->
getType(), NewDMaskVal);
2419 unsigned NewNumElts = DemandedElts.
popcount();
2423 if (NewNumElts >= VWidth && DemandedElts.
isMask()) {
2425 II.setArgOperand(DMaskIdx, Args[DMaskIdx]);
2437 OverloadTys[0] = NewTy;
2441 for (
unsigned OrigStoreIdx = 0; OrigStoreIdx < VWidth; ++OrigStoreIdx)
2442 if (DemandedElts[OrigStoreIdx])
2445 if (NewNumElts == 1)
2452 II.getIntrinsicID(), OverloadTys, Args);
2455 AttributeList OldAttrList =
II.getAttributes();
2459 if (NewNumElts == 1) {
2465 unsigned NewLoadIdx = 0;
2466 for (
unsigned OrigLoadIdx = 0; OrigLoadIdx < VWidth; ++OrigLoadIdx) {
2467 if (!!DemandedElts[OrigLoadIdx])
2483 APInt &UndefElts)
const {
2488 const unsigned FirstElt = DemandedElts.
countr_zero();
2490 const unsigned MaskLen = LastElt - FirstElt + 1;
2492 unsigned OldNumElts = VT->getNumElements();
2493 if (MaskLen == OldNumElts && MaskLen != 1)
2496 Type *EltTy = VT->getElementType();
2504 Value *Src =
II.getArgOperand(0);
2509 II.getOperandBundlesAsDefs(OpBundles);
2526 for (
unsigned I = 0;
I != MaskLen; ++
I) {
2527 if (DemandedElts[FirstElt +
I])
2528 ExtractMask[
I] = FirstElt +
I;
2537 for (
unsigned I = 0;
I != MaskLen; ++
I) {
2538 if (DemandedElts[FirstElt +
I])
2539 InsertMask[FirstElt +
I] =
I;
2551 SimplifyAndSetOp)
const {
2552 switch (
II.getIntrinsicID()) {
2553 case Intrinsic::amdgcn_readfirstlane:
2554 SimplifyAndSetOp(&
II, 0, DemandedElts, UndefElts);
2556 case Intrinsic::amdgcn_raw_buffer_load:
2557 case Intrinsic::amdgcn_raw_ptr_buffer_load:
2558 case Intrinsic::amdgcn_raw_buffer_load_format:
2559 case Intrinsic::amdgcn_raw_ptr_buffer_load_format:
2560 case Intrinsic::amdgcn_raw_tbuffer_load:
2561 case Intrinsic::amdgcn_raw_ptr_tbuffer_load:
2562 case Intrinsic::amdgcn_s_buffer_load:
2563 case Intrinsic::amdgcn_ptr_s_buffer_load:
2564 case Intrinsic::amdgcn_struct_buffer_load:
2565 case Intrinsic::amdgcn_struct_ptr_buffer_load:
2566 case Intrinsic::amdgcn_struct_buffer_load_format:
2567 case Intrinsic::amdgcn_struct_ptr_buffer_load_format:
2568 case Intrinsic::amdgcn_struct_tbuffer_load:
2569 case Intrinsic::amdgcn_struct_ptr_tbuffer_load:
2572 if (getAMDGPUImageDMaskIntrinsic(
II.getIntrinsicID())) {
2578 return std::nullopt;
for(const MachineOperand &MO :llvm::drop_begin(OldMI.operands(), Desc.getNumOperands()))
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static Value * createPermlane16(IRBuilderBase &B, Value *Val, uint32_t Lo, uint32_t Hi)
Emit v_permlane16 with the precomputed lane-select halves.
static std::optional< unsigned > matchRowSharePattern(ArrayRef< uint8_t > Ids)
Match a row-share pattern: all 16 lanes of each row read the same source lane.
static bool matchMirrorPattern(ArrayRef< uint8_t > Ids)
Match an N-lane reversal (mirror) pattern.
static bool canSafelyConvertTo16Bit(Value &V, bool IsFloat, bool AllowI16SExt=false)
static bool tryBuildShuffleMap(Value *Index, const GCNSubtarget &ST, SmallVectorImpl< uint8_t > &Ids, const DataLayout &DL)
Build the per-lane shuffle map by evaluating Index for every lane in the wave.
static std::optional< unsigned > matchQuadPermPattern(ArrayRef< uint8_t > Ids)
Match a 4-lane (quad) permutation, encoded as the v_mov_b32_dpp QUAD_PERM control word: bits[1:0]=Ids...
static std::optional< unsigned > matchDsSwizzleRotatePattern(ArrayRef< uint8_t > Ids)
Match a GFX9+ DS_SWIZZLE rotate-mode permutation: a cyclic left-rotation of all 32 lanes within each ...
static std::optional< unsigned > matchHalfRowPermPattern(ArrayRef< uint8_t > Ids)
Match an 8-lane arbitrary permutation, encoded as the v_mov_b32_dpp8 24-bit selector (three bits per ...
static std::optional< unsigned > matchRowXMaskPattern(ArrayRef< uint8_t > Ids)
Match an XOR mask pattern within each 16-lane row: Ids[J] == Mask ^ J, with Mask in [1,...
static constexpr auto matchHalfRowMirrorPattern
static Value * createPermlaneX16(IRBuilderBase &B, Value *Val, uint32_t Lo, uint32_t Hi)
Emit v_permlanex16 with the precomputed lane-select halves.
static bool isRowPattern(ArrayRef< uint8_t > Ids)
Match an N-lane row pattern: each lane in [0, N) reads from a source lane in the same N-lane row,...
static bool canContractSqrtToRsq(const FPMathOperator *SqrtOp)
Return true if it's legal to contract llvm.amdgcn.rcp(llvm.sqrt)
static bool isTriviallyUniform(const Use &U)
Return true if we can easily prove that use U is uniform.
static CallInst * rewriteCall(IRBuilderBase &B, CallInst &Old, Function &NewCallee, ArrayRef< Value * > Ops)
static Value * convertTo16Bit(Value &V, InstCombiner::BuilderTy &Builder)
static constexpr auto isFullRowPattern
static constexpr auto isQuadPattern
static APInt trimTrailingZerosInVector(InstCombiner &IC, Value *UseV, Instruction *I)
static uint64_t computePermlane16Masks(ArrayRef< uint8_t > Ids)
Pack a 16-lane permutation into a single 64-bit value: four bits per output lane, lane J in bits [J*4...
static bool matchHalfWaveSwapPattern(ArrayRef< uint8_t > Ids)
Match a half-wave swap: lane J reads from lane J ^ 32.
static bool hasPeriodicLayout(ArrayRef< uint8_t > Ids)
Lanes are partitioned into groups of Period; each group is a translated copy of the first: Ids[I] = I...
static std::optional< Instruction * > tryOptimizeShufflePattern(InstCombiner &IC, IntrinsicInst &II, const GCNSubtarget &ST)
Try to fold a wave_shuffle/ds_bpermute whose lane index is a constant function of the lane ID into a ...
static constexpr auto isHalfRowPattern
static APInt defaultComponentBroadcast(Value *V)
static std::optional< unsigned > matchDsSwizzleBitmaskPattern(ArrayRef< uint8_t > Ids)
Match a DS_SWIZZLE bitmask-mode permutation: dst_lane = ((src_lane & AND) | OR) ^ XOR with each mask ...
static Value * createDsSwizzle(IRBuilderBase &B, Value *Val, unsigned Offset, const DataLayout &DL)
Emit ds_swizzle with the given immediate, bitcasting/converting between pointer/float types and i32 a...
static std::optional< Instruction * > modifyIntrinsicCall(IntrinsicInst &OldIntr, Instruction &InstToReplace, unsigned NewIntr, InstCombiner &IC, std::function< void(SmallVectorImpl< Value * > &, SmallVectorImpl< Type * > &)> Func)
Applies Func(OldIntr.Args, OldIntr.ArgTys), creates intrinsic call with modified arguments (based on ...
static Value * matchShuffleToHWIntrinsic(IRBuilderBase &B, Value *Src, ArrayRef< uint8_t > Ids, const GCNSubtarget &ST, const DataLayout &DL)
Given a shuffle map, try to emit the best hardware intrinsic.
static std::optional< unsigned > matchRowRotatePattern(ArrayRef< uint8_t > Ids)
Match a 16-lane cyclic rotation; returns the rotation amount in [1, 15].
static bool isCrossRowPattern(ArrayRef< uint8_t > Ids)
Match a cross-row permutation suitable for v_permlanex16: every lane in the low 16-lane half reads fr...
static bool isThreadID(const GCNSubtarget &ST, Value *V)
static Value * createUpdateDpp(IRBuilderBase &B, Value *Val, unsigned Ctrl)
Emit v_mov_b32_dpp with the given control word, row/bank masks 0xF, and bound_ctrl=1 so out-of-bounds...
static APFloat fmed3AMDGCN(const APFloat &Src0, const APFloat &Src1, const APFloat &Src2)
static Value * simplifyAMDGCNMemoryIntrinsicDemanded(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, int DMaskIdx=-1, bool IsLoad=true)
Implement SimplifyDemandedVectorElts for amdgcn buffer and image intrinsics.
static std::optional< Instruction * > simplifyAMDGCNImageIntrinsic(const GCNSubtarget *ST, const AMDGPU::ImageDimIntrinsicInfo *ImageDimIntr, IntrinsicInst &II, InstCombiner &IC)
static Value * createMovDpp8(IRBuilderBase &B, Value *Val, unsigned Selector)
Emit v_mov_b32_dpp8 with the given 24-bit lane selector.
static Value * matchFPExtFromF16(Value *Arg)
Match an fpext from half to float, or a constant we can convert.
static constexpr auto matchFullRowMirrorPattern
static std::optional< unsigned > evalLaneExpr(Value *V, unsigned Lane, const GCNSubtarget &ST, const DataLayout &DL, unsigned Depth=0)
Evaluate V as a function of the lane ID and return its value on Lane, or std::nullopt if V is not a c...
static Value * createPermlane64(IRBuilderBase &B, Value *Val)
Emit v_permlane64 (swap of the two 32-lane halves of a wave64).
Contains the definition of a TargetInstrInfo class that is common to all AMD GPUs.
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
Utilities for dealing with flags related to floating point properties and mode controls.
AMD GCN specific subclass of TargetSubtarget.
This file provides the interface for the instcombine pass implementation.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
MachineInstr unsigned OpIdx
uint64_t IntrinsicInst * II
Provides some synthesis utilities to produce sequences of values.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static const fltSemantics & IEEEsingle()
static constexpr roundingMode rmTowardZero
static constexpr roundingMode rmNearestTiesToEven
static const fltSemantics & IEEEhalf()
static APFloat getQNaN(const fltSemantics &Sem, bool Negative=false, const APInt *payload=nullptr)
Factory for QNaN values.
LLVM_ABI opStatus convert(const fltSemantics &ToSemantics, roundingMode RM, bool *losesInfo)
bool bitwiseIsEqual(const APFloat &RHS) const
bool isPosInfinity() const
APFloat makeQuiet() const
Assuming this is an IEEE-754 NaN value, quiet its signaling bit.
APInt bitcastToAPInt() const
static APFloat getZero(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Zero.
Class for arbitrary precision integers.
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
void clearBit(unsigned BitPosition)
Set a given bit to 0.
uint64_t getZExtValue() const
Get zero extended value.
unsigned popcount() const
Count the number of bits set.
LLVM_ABI uint64_t extractBitsAsZExtValue(unsigned numBits, unsigned bitPosition) const
unsigned getActiveBits() const
Compute the number of active bits in the value.
LLVM_ABI APInt trunc(unsigned width) const
Truncate to new width.
unsigned countr_zero() const
Count the number of trailing zero bits.
bool isMask(unsigned numBits) const
Represent a constant reference to an array (0 or more elements consecutively in memory),...
ArrayRef< T > take_front(size_t N=1) const
Return a copy of *this with only the first N elements.
size_t size() const
Get the array size.
static LLVM_ABI Attribute getWithDereferenceableBytes(LLVMContext &Context, uint64_t Bytes)
LLVM_ABI const Module * getModule() const
Return the module owning the function this basic block belongs to, or nullptr if the function does no...
bool isTypeLegal(Type *Ty) const override
LLVM_ABI void getOperandBundlesAsDefs(SmallVectorImpl< OperandBundleDef > &Defs) const
Return the list of operand bundles attached to this instruction as a vector of OperandBundleDefs.
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
void setAttributes(AttributeList A)
Set the attributes for this call.
iterator_range< User::op_iterator > args()
Iteration adapter for range-for loops.
AttributeList getAttributes() const
Return the attributes for this call.
This class represents a function call, abstracting a target machine's calling convention.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
bool isFPPredicate() const
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
ConstantFP - Floating Point Values [float, double].
const APFloat & getValueAPF() const
static LLVM_ABI ConstantFP * getZero(Type *Ty, bool Negative=false)
static LLVM_ABI ConstantFP * getNaN(Type *Ty, bool Negative=false, uint64_t Payload=0)
static LLVM_ABI ConstantFP * getInfinity(Type *Ty, bool Negative=false)
This is the shared class of boolean and integer constants.
static ConstantInt * getSigned(IntegerType *Ty, int64_t V, bool ImplicitTrunc=false)
Return a ConstantInt with the specified value for the specified type.
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
const APInt & getValue() const
Return the constant as an APInt value reference.
This class represents a range of values.
LLVM_ABI ConstantRange add(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an addition of a value in this ran...
LLVM_ABI bool isFullSet() const
Return true if this set contains all of the elements possible for this data-type.
LLVM_ABI ConstantRange intersectWith(const ConstantRange &CR, PreferredRangeType Type=Smallest) const
Return the range that results from the intersection of this range with another range.
This is an important base class in LLVM.
bool isNullValue() const
Return true if this is the value that would be returned by getNullValue.
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
TypeSize getTypeSizeInBits(Type *Ty) const
Size examples:
LLVM_ABI bool dominates(const BasicBlock *BB, const Use &U) const
Return true if the (end of the) basic block BB dominates the use U.
Tagged union holding either a T or a Error.
This class represents an extension of floating point types.
Utility class for floating point operations which can have information about relaxed accuracy require...
FastMathFlags getFastMathFlags() const
Convenience function for getting all the fast-math flags.
bool hasApproxFunc() const
Test if this operation allows approximations of math library functions or intrinsics.
LLVM_ABI float getFPAccuracy() const
Get the maximum error permitted by this operation in ULPs.
Convenience struct for specifying and reasoning about fast-math flags.
bool allowContract() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
bool simplifyDemandedLaneMaskArg(InstCombiner &IC, IntrinsicInst &II, unsigned LaneAgIdx) const
Simplify a lane index operand (e.g.
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
Instruction * hoistLaneIntrinsicThroughOperand(InstCombiner &IC, IntrinsicInst &II) const
std::optional< Value * > simplifyDemandedVectorEltsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp) const override
KnownIEEEMode fpenvIEEEMode(const Instruction &I) const
Return KnownIEEEMode::On if we know if the use context can assume "amdgpu-ieee"="true" and KnownIEEEM...
Value * simplifyAMDGCNLaneIntrinsicDemanded(InstCombiner &IC, IntrinsicInst &II, const APInt &DemandedElts, APInt &UndefElts) const
bool canSimplifyLegacyMulToMul(const Instruction &I, const Value *Op0, const Value *Op1, InstCombiner &IC) const
Common base class shared among various IRBuilders.
LLVM_ABI CallInst * CreateIntrinsicWithoutFolding(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={})
Create a call to intrinsic ID with Args, mangled using OverloadTypes.
Value * CreateInsertElement(Type *VecTy, Value *NewElt, Value *Idx, const Twine &Name="")
Value * CreateExtractElement(Value *Vec, Value *Idx, const Twine &Name="")
IntegerType * getIntNTy(unsigned N)
Fetch the type representing an N-bit integer.
Value * CreateZExtOrTrunc(Value *V, Type *DestTy, const Twine &Name="")
Create a ZExt or Trunc from the integer value V to DestTy.
ConstantInt * getTrue()
Get the constant value for i1 true.
Value * CreateSExt(Value *V, Type *DestTy, const Twine &Name="")
Value * CreateLShr(Value *LHS, Value *RHS, const Twine &Name="", bool isExact=false)
Value * CreateExtractVector(Type *DstType, Value *SrcVec, Value *Idx, const Twine &Name="")
Create a call to the vector.extract intrinsic.
BasicBlock * GetInsertBlock() const
Value * CreateICmpNE(Value *LHS, Value *RHS, const Twine &Name="")
ConstantInt * getInt64(uint64_t C)
Get a constant 64-bit value.
Value * CreateMaxNum(Value *LHS, Value *RHS, FMFSource FMFSource={}, const Twine &Name="")
Create call to the maxnum intrinsic.
Value * CreateShl(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Value * CreateZExt(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNeg=false)
Value * CreateShuffleVector(Value *V1, Value *V2, Value *Mask, const Twine &Name="")
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
Value * CreateMaximumNum(Value *LHS, Value *RHS, const Twine &Name="")
Create call to the maximum intrinsic.
Value * CreateMinNum(Value *LHS, Value *RHS, FMFSource FMFSource={}, const Twine &Name="")
Create call to the minnum intrinsic.
Value * CreateAdd(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
CallInst * CreateCall(FunctionType *FTy, Value *Callee, ArrayRef< Value * > Args={}, const Twine &Name="", MDNode *FPMathTag=nullptr)
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
Value * CreateFAddFMF(Value *L, Value *R, FMFSource FMFSource, const Twine &Name="", MDNode *FPMD=nullptr)
Value * CreateMinimumNum(Value *LHS, Value *RHS, const Twine &Name="")
Create call to the minimumnum intrinsic.
Value * CreateAShr(Value *LHS, Value *RHS, const Twine &Name="", bool isExact=false)
Value * CreateFMulFMF(Value *L, Value *R, FMFSource FMFSource, const Twine &Name="", MDNode *FPMD=nullptr)
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
The core instruction combiner logic.
const DataLayout & getDataLayout() const
virtual Instruction * eraseInstFromFunction(Instruction &I)=0
Combiner aware instruction erasure.
DominatorTree & getDominatorTree() const
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
virtual bool SimplifyDemandedBits(Instruction *I, unsigned OpNo, const APInt &DemandedMask, KnownBits &Known, const SimplifyQuery &Q, unsigned Depth=0)=0
IRBuilder< TargetFolder, IRBuilderInstCombineInserter > BuilderTy
An IRBuilder that automatically inserts new instructions into the worklist.
static Value * stripSignOnlyFPOps(Value *Val)
Ignore all operations which only change the sign of a value, returning the underlying magnitude value...
Instruction * replaceOperand(Instruction &I, unsigned OpNum, Value *V)
Replace operand of instruction and add old operand to the worklist.
const SimplifyQuery & getSimplifyQuery() const
LLVM_ABI Instruction * clone() const
Create a copy of 'this' instruction that is identical in all ways except the following:
LLVM_ABI void copyFastMathFlags(FastMathFlags FMF)
Convenience function for transferring all fast-math flag values to this instruction,...
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
Class to represent integer types.
A wrapper class for inspecting calls to intrinsic functions.
A Module instance is used to store all the information related to an LLVM module.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
The instances of the Type class are immutable: once they are created, they are never changed.
bool isPointerTy() const
True if this is an instance of PointerType.
bool isFloatTy() const
Return true if this is 'float', a 32-bit IEEE fp type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
static LLVM_ABI IntegerType * getInt16Ty(LLVMContext &C)
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI Type * getHalfTy(LLVMContext &C)
bool isVoidTy() const
Return true if this is 'void'.
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
const Use & getOperandUse(unsigned i) const
void setOperand(unsigned i, Value *Val)
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
LLVM_ABI bool hasOneUser() const
Return true if there is exactly one user of this value.
LLVMContext & getContext() const
All values hold a context through their type.
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
const ParentTy * getParent() const
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_READONLY const MIMGOffsetMappingInfo * getMIMGOffsetMappingInfo(unsigned Offset)
uint8_t wmmaScaleF8F6F4FormatToNumRegs(unsigned Fmt)
const ImageDimIntrinsicInfo * getImageDimIntrinsicByBaseOpcode(unsigned BaseOpcode, unsigned Dim)
LLVM_READONLY const MIMGMIPMappingInfo * getMIMGMIPMappingInfo(unsigned MIP)
bool isArgPassedInSGPR(const Argument *A)
bool isIntrinsicAlwaysUniform(unsigned IntrID)
LLVM_READONLY const MIMGBiasMappingInfo * getMIMGBiasMappingInfo(unsigned Bias)
std::optional< APFloat > evaluateRcp(const APFloat &Val)
Evaluate the constant-folded result of v_rcp for Val, accounting for the hardware's denormal flushing...
LLVM_READONLY const MIMGLZMappingInfo * getMIMGLZMappingInfo(unsigned L)
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
const ImageDimIntrinsicInfo * getImageDimIntrinsicInfo(unsigned Intr)
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
LLVM_ABI bool isSignatureValid(Intrinsic::ID ID, FunctionType *FT, SmallVectorImpl< Type * > &OverloadTys, raw_ostream &OS=nulls())
Returns true if FT is a valid function type for intrinsic ID.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
cst_pred_ty< is_all_ones > m_AllOnes()
Match an integer or vector with all bits set.
auto m_Cmp()
Matches any compare instruction and ignore it.
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
cstfp_pred_ty< is_any_zero_fp > m_AnyZeroFP()
Match a floating-point negative zero or positive zero.
ap_match< APFloat > m_APFloat(const APFloat *&Res)
Match a ConstantFP or splatted ConstantVector, binding the specified pointer to the contained APFloat...
TwoOps_match< Val_t, Idx_t, Instruction::ExtractElement > m_ExtractElt(const Val_t &Val, const Idx_t &Idx)
Matches ExtractElementInst.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
auto m_Value()
Match an arbitrary value and ignore it.
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
auto m_ConstantFP()
Match an arbitrary ConstantFP and ignore it.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI KnownFPClass computeKnownFPClass(const Value *V, const APInt &DemandedElts, FPClassTest InterestedClasses, const SimplifyQuery &SQ, unsigned Depth=0)
Determine which floating-point classes are valid for V, and return them in KnownFPClass bit sets.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
constexpr bool isMask_32(uint32_t Value)
Return true if the argument is a non-empty sequence of ones starting at the least significant bit wit...
LLVM_ABI Constant * ConstantFoldCompareInstOperands(unsigned Predicate, Constant *LHS, Constant *RHS, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, const Instruction *I=nullptr)
Attempt to constant fold a compare instruction (icmp/fcmp) with the specified operands.
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
APFloat frexp(const APFloat &X, int &Exp, APFloat::roundingMode RM)
Equivalent of C standard library function.
auto dyn_cast_or_null(const Y &Val)
LLVM_READONLY APFloat maxnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 maxNum semantics.
constexpr unsigned MaxAnalysisRecursionDepth
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
APFloat scalbn(APFloat X, int Exp, APFloat::roundingMode RM)
Returns: X * 2^Exp for integral exponents.
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
constexpr int PoisonMaskElem
LLVM_ABI Value * findScalarElement(Value *V, unsigned EltNo)
Given a vector and an element number, see if the scalar value is already around as a register,...
@ NearestTiesToEven
roundTiesToEven.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
constexpr auto seq(T Begin, T End)
Iterate over an integral type from Begin up to - but not including - End.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
LLVM_ABI Constant * ConstantFoldInstOperands(const Instruction *I, ArrayRef< Constant * > Ops, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, bool AllowNonDeterministic=true)
ConstantFoldInstOperands - Attempt to constant fold an instruction with the specified operands.
constexpr uint64_t Make_64(uint32_t High, uint32_t Low)
Make a 64-bit integer from a high / low pair of 32-bit integers.
LLVM_ABI ConstantRange computeConstantRange(const Value *V, bool ForSigned, const SimplifyQuery &SQ, unsigned Depth=0)
Determine the possible constant range of an integer or vector of integer value.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Represent subnormal handling kind for floating point instruction inputs and outputs.
bool isKnownNeverInfOrNaN() const
Return true if it's known this can never be an infinity or nan.
LLVM_ABI bool isKnownNeverLogicalZero(DenormalMode Mode) const
Return true if it's known this can never be interpreted as a zero.
SimplifyQuery getWithInstruction(const Instruction *I) const
LLVM_ABI bool isUndefValue(Value *V) const
If CanUseUndef is true, returns whether V is undef.