33#include "llvm/IR/IntrinsicsAMDGPU.h"
41#define DEBUG_TYPE "si-instr-info"
43#define GET_INSTRINFO_CTOR_DTOR
44#include "AMDGPUGenInstrInfo.inc"
47#define GET_ImageDimIntrinsicTable_IMPL
48#define GET_RsrcIntrinsics_IMPL
49#define GET_GFX1250BlockingCyclesTable_DECL
50#define GET_GFX1250BlockingCyclesTable_IMPL
57#include "AMDGPUGenSearchableTables.inc"
65 cl::desc(
"Restrict range of branch instructions (DEBUG)"));
68 "amdgpu-fix-16-bit-physreg-copies",
69 cl::desc(
"Fix copies between 32 and 16 bit registers by extending to 32 bit"),
85 unsigned N =
Node->getNumOperands();
86 while (
N &&
Node->getOperand(
N - 1).getValueType() == MVT::Glue)
98 int Op0Idx = AMDGPU::getNamedOperandIdx(Opc0,
OpName);
99 int Op1Idx = AMDGPU::getNamedOperandIdx(Opc1,
OpName);
101 if (Op0Idx == -1 && Op1Idx == -1)
105 if ((Op0Idx == -1 && Op1Idx != -1) ||
106 (Op1Idx == -1 && Op0Idx != -1))
127 return !
MI.memoperands_empty() &&
129 return MMO->isLoad() && MMO->isInvariant();
138static std::tuple<unsigned, unsigned, unsigned>
146 unsigned LoReloc, HiReloc;
176 return {BaseFlags, LoReloc, HiReloc};
194 if (!
MI.hasImplicitDef() &&
195 MI.getNumImplicitOperands() ==
MI.getDesc().implicit_uses().size() &&
196 !
MI.mayRaiseFPException())
205 if (!
MI.getNumOperands() || !
MI.getOperand(0).isReg())
220 if (
MI.isNotDuplicable() ||
MI.mayStore() ||
MI.mayRaiseFPException() ||
221 MI.hasUnmodeledSideEffects())
226 if (
MI.isInlineAsm())
230 if (
MI.mayLoad() && !
MI.isDereferenceableInvariantLoad())
245 if (Reg.isPhysical()) {
261 if (MO.isDef() && Reg != DefReg)
271 case AMDGPU::V_SUBREV_U16_e32:
272 case AMDGPU::V_SUBREV_U16_e64:
274 case AMDGPU::V_SUBREV_U32_e32:
275 case AMDGPU::V_SUBREV_U32_e64:
277 case AMDGPU::V_SUBREV_CO_U32_e32:
278 case AMDGPU::V_SUBREV_CO_U32_e64:
280 case AMDGPU::V_SUBBREV_U32_e32:
281 case AMDGPU::V_SUBBREV_U32_e64:
284 case AMDGPU::V_ASHRREV_I16_e32:
285 case AMDGPU::V_ASHRREV_I16_e64:
286 case AMDGPU::V_ASHRREV_I32_e32:
287 case AMDGPU::V_ASHRREV_I32_e64:
288 case AMDGPU::V_ASHRREV_I64_e64:
289 case AMDGPU::V_LSHLREV_B16_e32:
290 case AMDGPU::V_LSHLREV_B16_e64:
291 case AMDGPU::V_LSHLREV_B32_e32:
292 case AMDGPU::V_LSHLREV_B32_e64:
293 case AMDGPU::V_LSHLREV_B64_e64:
294 case AMDGPU::V_LSHRREV_B16_e32:
295 case AMDGPU::V_LSHRREV_B16_e64:
296 case AMDGPU::V_LSHRREV_B32_e32:
297 case AMDGPU::V_LSHRREV_B32_e64:
298 case AMDGPU::V_LSHRREV_B64_e64:
299 return !ST.hasGFX11Insts();
306bool SIInstrInfo::resultDependsOnExec(
const MachineInstr &
MI)
const {
310 if (
MI.isConvergent())
338 if (
MI.getOpcode() == AMDGPU::SI_IF_BREAK)
343 for (
auto Op :
MI.uses()) {
344 if (
Op.isReg() &&
Op.getReg().isVirtual() &&
358 while (FromCycle && !(ToCycle && CI->
contains(FromCycle, ToCycle))) {
378 int64_t &Offset1)
const {
386 if (!
get(Opc0).mayLoad() || !
get(Opc1).mayLoad())
390 if (!
get(Opc0).getNumDefs() || !
get(Opc1).getNumDefs())
406 int Offset0Idx = AMDGPU::getNamedOperandIdx(Opc0, AMDGPU::OpName::offset);
407 int Offset1Idx = AMDGPU::getNamedOperandIdx(Opc1, AMDGPU::OpName::offset);
408 if (Offset0Idx == -1 || Offset1Idx == -1)
415 Offset0Idx -=
get(Opc0).NumDefs;
416 Offset1Idx -=
get(Opc1).NumDefs;
446 if (!Load0Offset || !Load1Offset)
463 int OffIdx0 = AMDGPU::getNamedOperandIdx(Opc0, AMDGPU::OpName::offset);
464 int OffIdx1 = AMDGPU::getNamedOperandIdx(Opc1, AMDGPU::OpName::offset);
466 if (OffIdx0 == -1 || OffIdx1 == -1)
472 OffIdx0 -=
get(Opc0).NumDefs;
473 OffIdx1 -=
get(Opc1).NumDefs;
492 case AMDGPU::DS_READ2ST64_B32:
493 case AMDGPU::DS_READ2ST64_B64:
494 case AMDGPU::DS_WRITE2ST64_B32:
495 case AMDGPU::DS_WRITE2ST64_B64:
510 OffsetIsScalable =
false;
527 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst);
529 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::data0);
530 if (
Opc == AMDGPU::DS_ATOMIC_ASYNC_BARRIER_ARRIVE_B64)
543 unsigned Offset0 = Offset0Op->
getImm() & 0xff;
544 unsigned Offset1 = Offset1Op->
getImm() & 0xff;
545 if (Offset0 + 1 != Offset1)
556 int Data0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::data0);
564 Offset = EltSize * Offset0;
566 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst);
567 if (DataOpIdx == -1) {
568 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::data0);
570 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::data1);
586 if (BaseOp && !BaseOp->
isFI())
594 if (SOffset->
isReg())
600 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst);
602 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdata);
611 isMIMG(LdSt) ? AMDGPU::OpName::srsrc : AMDGPU::OpName::rsrc;
612 int SRsrcIdx = AMDGPU::getNamedOperandIdx(
Opc, RsrcOpName);
614 int VAddr0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr0);
615 if (VAddr0Idx >= 0) {
617 for (
int I = VAddr0Idx;
I < SRsrcIdx; ++
I)
624 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdata);
639 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::sdst);
656 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst);
658 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdata);
675 if (BaseOps1.
front()->isIdenticalTo(*BaseOps2.
front()))
683 if (MO1->getAddrSpace() != MO2->getAddrSpace())
686 const auto *Base1 = MO1->getValue();
687 const auto *Base2 = MO2->getValue();
688 if (!Base1 || !Base2)
696 return Base1 == Base2;
700 int64_t Offset1,
bool OffsetIsScalable1,
702 int64_t Offset2,
bool OffsetIsScalable2,
703 unsigned ClusterSize,
704 unsigned NumBytes)
const {
717 }
else if (!BaseOps1.
empty() || !BaseOps2.
empty()) {
736 const unsigned LoadSize = NumBytes / ClusterSize;
737 const unsigned NumDWords = ((LoadSize + 3) / 4) * ClusterSize;
738 return NumDWords <= MaxMemoryClusterDWords;
752 int64_t Offset0, int64_t Offset1,
753 unsigned NumLoads)
const {
754 assert(Offset1 > Offset0 &&
755 "Second offset should be larger than first offset!");
760 return (NumLoads <= 16 && (Offset1 - Offset0) < 64);
767 const char *
Msg =
"illegal VGPR to SGPR copy") {
786 assert((
TII.getSubtarget().hasMAIInsts() &&
787 !
TII.getSubtarget().hasGFX90AInsts()) &&
788 "Expected GFX908 subtarget.");
791 AMDGPU::AGPR_32RegClass.
contains(SrcReg)) &&
792 "Source register of the copy should be either an SGPR or an AGPR.");
795 "Destination register of the copy should be an AGPR.");
804 for (
auto Def =
MI,
E =
MBB.begin(); Def !=
E; ) {
807 if (!Def->modifiesRegister(SrcReg, &RI))
810 if (Def->getOpcode() != AMDGPU::V_ACCVGPR_WRITE_B32_e64 ||
811 Def->getOperand(0).getReg() != SrcReg)
818 bool SafeToPropagate =
true;
821 for (
auto I = Def;
I !=
MI && SafeToPropagate; ++
I)
822 if (
I->modifiesRegister(DefOp.
getReg(), &RI))
823 SafeToPropagate =
false;
825 if (!SafeToPropagate)
828 for (
auto I = Def;
I !=
MI; ++
I)
829 I->clearRegisterKills(DefOp.
getReg(), &RI);
837 if (ImpUseSuperReg) {
838 Builder.addReg(ImpUseSuperReg,
846 RS.enterBasicBlockEnd(
MBB);
847 RS.backward(std::next(
MI));
856 unsigned RegNo = (DestReg - AMDGPU::AGPR0) % 3;
859 assert(
MBB.getParent()->getRegInfo().isReserved(Tmp) &&
860 "VGPR used for an intermediate copy should have been reserved.");
865 Register Tmp2 = RS.scavengeRegisterBackwards(AMDGPU::VGPR_32RegClass,
MI,
875 unsigned TmpCopyOp = AMDGPU::V_MOV_B32_e32;
876 if (AMDGPU::AGPR_32RegClass.
contains(SrcReg)) {
877 TmpCopyOp = AMDGPU::V_ACCVGPR_READ_B32_e64;
884 if (ImpUseSuperReg) {
885 UseBuilder.
addReg(ImpUseSuperReg,
902 for (
unsigned Idx = 0; Idx < BaseIndices.
size(); ++Idx) {
903 int16_t SubIdx = BaseIndices[Idx];
904 Register DestSubReg = RI.getSubReg(DestReg, SubIdx);
905 Register SrcSubReg = RI.getSubReg(SrcReg, SubIdx);
906 assert(DestSubReg && SrcSubReg &&
"Failed to find subregs!");
907 unsigned Opcode = AMDGPU::S_MOV_B32;
910 bool AlignedDest = ((DestSubReg - AMDGPU::SGPR0) % 2) == 0;
911 bool AlignedSrc = ((SrcSubReg - AMDGPU::SGPR0) % 2) == 0;
912 if (AlignedDest && AlignedSrc && (Idx + 1 < BaseIndices.
size())) {
916 DestSubReg = RI.getSubReg(DestReg, SubIdx);
917 SrcSubReg = RI.getSubReg(SrcReg, SubIdx);
918 assert(DestSubReg && SrcSubReg &&
"Failed to find subregs!");
919 Opcode = AMDGPU::S_MOV_B64;
934 assert(FirstMI && LastMI);
939 LastMI->addRegisterKilled(SrcReg, &RI);
945 Register SrcReg,
bool KillSrc,
bool RenamableDest,
946 bool RenamableSrc)
const {
948 unsigned Size = RI.getRegSizeInBits(*RC);
950 unsigned SrcSize = RI.getRegSizeInBits(*SrcRC);
956 if (((
Size == 16) != (SrcSize == 16))) {
958 assert(ST.useRealTrue16Insts());
960 MCRegister SubReg = RI.getSubReg(RegToFix, AMDGPU::lo16);
963 if (DestReg == SrcReg) {
969 RC = RI.getPhysRegBaseClass(DestReg);
970 Size = RI.getRegSizeInBits(*RC);
971 SrcRC = RI.getPhysRegBaseClass(SrcReg);
972 SrcSize = RI.getRegSizeInBits(*SrcRC);
976 if (RC == &AMDGPU::VGPR_32RegClass) {
978 AMDGPU::SReg_32RegClass.
contains(SrcReg) ||
979 AMDGPU::AGPR_32RegClass.
contains(SrcReg));
980 unsigned Opc = AMDGPU::AGPR_32RegClass.contains(SrcReg) ?
981 AMDGPU::V_ACCVGPR_READ_B32_e64 : AMDGPU::V_MOV_B32_e32;
987 if (RC == &AMDGPU::SReg_32_XM0RegClass ||
988 RC == &AMDGPU::SReg_32RegClass) {
989 if (SrcReg == AMDGPU::SCC) {
996 if (!AMDGPU::SReg_32RegClass.
contains(SrcReg)) {
997 if (DestReg == AMDGPU::VCC_LO) {
1015 if (RC == &AMDGPU::SReg_64RegClass) {
1016 if (SrcReg == AMDGPU::SCC) {
1023 if (!AMDGPU::SReg_64_EncodableRegClass.
contains(SrcReg)) {
1024 if (DestReg == AMDGPU::VCC) {
1042 if (DestReg == AMDGPU::SCC) {
1045 if (AMDGPU::SReg_64RegClass.
contains(SrcReg)) {
1049 assert(ST.hasScalarCompareEq64());
1063 if (RC == &AMDGPU::AGPR_32RegClass) {
1064 if (AMDGPU::VGPR_32RegClass.
contains(SrcReg) ||
1065 (ST.hasGFX90AInsts() && AMDGPU::SReg_32RegClass.contains(SrcReg))) {
1071 if (AMDGPU::AGPR_32RegClass.
contains(SrcReg) && ST.hasGFX90AInsts()) {
1080 const bool Overlap = RI.regsOverlap(SrcReg, DestReg);
1087 AMDGPU::SReg_LO16RegClass.
contains(SrcReg) ||
1088 AMDGPU::AGPR_LO16RegClass.
contains(SrcReg));
1090 bool IsSGPRDst = AMDGPU::SReg_LO16RegClass.contains(DestReg);
1091 bool IsSGPRSrc = AMDGPU::SReg_LO16RegClass.contains(SrcReg);
1092 bool IsAGPRDst = AMDGPU::AGPR_LO16RegClass.contains(DestReg);
1093 bool IsAGPRSrc = AMDGPU::AGPR_LO16RegClass.contains(SrcReg);
1096 MCRegister NewDestReg = RI.get32BitRegister(DestReg);
1097 MCRegister NewSrcReg = RI.get32BitRegister(SrcReg);
1110 if (IsAGPRDst || IsAGPRSrc) {
1111 if (!DstLow || !SrcLow) {
1113 "Cannot use hi16 subreg with an AGPR!");
1120 if (ST.useRealTrue16Insts()) {
1126 if (AMDGPU::VGPR_16_Lo128RegClass.
contains(DestReg) &&
1127 (IsSGPRSrc || AMDGPU::VGPR_16_Lo128RegClass.
contains(SrcReg))) {
1139 if (IsSGPRSrc && !ST.hasSDWAScalar()) {
1140 if (!DstLow || !SrcLow) {
1142 "Cannot use hi16 subreg on VI!");
1168 unsigned SrcOp = 1) {
1172 return DstOpRC && SrcOpRC && DstOpRC->
contains(Dst) &&
1176 if (RC == RI.getVGPR64Class() && (SrcRC == RC || RI.isSGPRClass(SrcRC))) {
1177 if (ST.hasVMovB64Inst() &&
1178 CanCopyWith(AMDGPU::V_MOV_B64_e32, DestReg, SrcReg)) {
1183 if (ST.hasPkMovB32() &&
1184 CanCopyWith(AMDGPU::V_PK_MOV_B32, DestReg, SrcReg, 2)) {
1200 const bool Forward = RI.getHWRegIndex(DestReg) <= RI.getHWRegIndex(SrcReg);
1201 if (RI.isSGPRClass(RC)) {
1202 if (!RI.isSGPRClass(SrcRC)) {
1206 const bool CanKillSuperReg = KillSrc && !RI.regsOverlap(SrcReg, DestReg);
1212 unsigned Opcode = AMDGPU::V_MOV_B32_e32;
1213 unsigned WideOpcode = AMDGPU::INSTRUCTION_LIST_END;
1214 if (RI.isAGPRClass(RC)) {
1215 if (ST.hasGFX90AInsts() && RI.isAGPRClass(SrcRC))
1216 Opcode = AMDGPU::V_ACCVGPR_MOV_B32;
1217 else if (RI.hasVGPRs(SrcRC) ||
1218 (ST.hasGFX90AInsts() && RI.isSGPRClass(SrcRC)))
1219 Opcode = AMDGPU::V_ACCVGPR_WRITE_B32_e64;
1221 Opcode = AMDGPU::INSTRUCTION_LIST_END;
1222 }
else if (RI.hasVGPRs(RC) && RI.isAGPRClass(SrcRC)) {
1223 Opcode = AMDGPU::V_ACCVGPR_READ_B32_e64;
1224 }
else if (RI.isVGPRClass(RC)) {
1225 if (ST.hasVMovB64Inst())
1226 WideOpcode = AMDGPU::V_MOV_B64_e32;
1227 else if (ST.hasPkMovB32())
1228 WideOpcode = AMDGPU::V_PK_MOV_B32;
1232 if (WideOpcode != AMDGPU::INSTRUCTION_LIST_END) {
1234 unsigned SrcOp = WideOpcode == AMDGPU::V_PK_MOV_B32 ? 2 : 1;
1241 const bool Overlap = RI.regsOverlap(SrcReg, DestReg);
1242 const bool CanKillSuperReg = KillSrc && !Overlap;
1249 std::unique_ptr<RegScavenger> RS;
1250 if (Opcode == AMDGPU::INSTRUCTION_LIST_END)
1251 RS = std::make_unique<RegScavenger>();
1255 for (
unsigned Idx{}; Idx < SubIndices.
size();) {
1256 unsigned NumRegs = 1;
1257 unsigned ThisOpcode = Opcode;
1259 Forward ? SubIndices[Idx] : SubIndices[SubIndices.
size() - Idx - 1];
1261 if (WideDstRC && WideSrcRC && Idx + 1 < SubIndices.
size()) {
1262 unsigned Channel = RI.getChannelFromSubReg(SubIdx);
1266 unsigned WideSubIdx = RI.getSubRegFromChannel(Channel, 2);
1267 Register WideDst = RI.getSubReg(DestReg, WideSubIdx);
1268 Register WideSrc = RI.getSubReg(SrcReg, WideSubIdx);
1270 if (WideDst && WideSrc && WideDstRC->
contains(WideDst) &&
1271 WideSrcRC->contains(WideSrc)) {
1272 SubIdx = WideSubIdx;
1274 ThisOpcode = WideOpcode;
1278 Register DestSubReg = RI.getSubReg(DestReg, SubIdx);
1279 Register SrcSubReg = RI.getSubReg(SrcReg, SubIdx);
1280 assert(DestSubReg && SrcSubReg &&
"Failed to find subregs!");
1283 bool UseKill = CanKillSuperReg && Idx == SubIndices.
size();
1285 if (ThisOpcode == AMDGPU::INSTRUCTION_LIST_END) {
1288 *RS, Overlap, ImpUseSuper);
1289 }
else if (ThisOpcode == AMDGPU::V_PK_MOV_B32) {
1330 int64_t &ImmVal)
const {
1331 switch (
MI.getOpcode()) {
1332 case AMDGPU::V_MOV_B32_e32:
1333 case AMDGPU::S_MOV_B32:
1334 case AMDGPU::S_MOVK_I32:
1335 case AMDGPU::S_MOV_B64:
1336 case AMDGPU::V_MOV_B64_e32:
1337 case AMDGPU::V_ACCVGPR_WRITE_B32_e64:
1338 case AMDGPU::AV_MOV_B32_IMM_PSEUDO:
1339 case AMDGPU::AV_MOV_B64_IMM_PSEUDO:
1340 case AMDGPU::S_MOV_B64_IMM_PSEUDO:
1341 case AMDGPU::V_MOV_B64_PSEUDO:
1342 case AMDGPU::V_MOV_B16_t16_e32: {
1346 return MI.getOperand(0).getReg() == Reg;
1351 case AMDGPU::V_MOV_B16_t16_e64: {
1353 if (Src0.
isImm() && !
MI.getOperand(1).getImm()) {
1355 return MI.getOperand(0).getReg() == Reg;
1360 case AMDGPU::S_BREV_B32:
1361 case AMDGPU::V_BFREV_B32_e32:
1362 case AMDGPU::V_BFREV_B32_e64: {
1366 return MI.getOperand(0).getReg() == Reg;
1371 case AMDGPU::S_NOT_B32:
1372 case AMDGPU::V_NOT_B32_e32:
1373 case AMDGPU::V_NOT_B32_e64: {
1376 ImmVal =
static_cast<int64_t
>(~static_cast<int32_t>(Src0.
getImm()));
1377 return MI.getOperand(0).getReg() == Reg;
1387std::optional<int64_t>
1397 if (!
Op.isReg() || !
Op.getReg().isVirtual())
1398 return std::nullopt;
1400 if (Def && Def->isMoveImmediate()) {
1402 if (ImmSrc.
isImm()) {
1409 return std::nullopt;
1412std::optional<int64_t>
1421 if (RI.isAGPRClass(DstRC))
1422 return AMDGPU::COPY;
1423 if (RI.getRegSizeInBits(*DstRC) == 16) {
1426 return RI.isSGPRClass(DstRC) ? AMDGPU::COPY : AMDGPU::V_MOV_B16_t16_e64;
1428 if (RI.getRegSizeInBits(*DstRC) == 32)
1429 return RI.isSGPRClass(DstRC) ? AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32;
1430 if (RI.getRegSizeInBits(*DstRC) == 64 && RI.isSGPRClass(DstRC))
1431 return AMDGPU::S_MOV_B64;
1432 if (RI.getRegSizeInBits(*DstRC) == 64 && !RI.isSGPRClass(DstRC))
1433 return AMDGPU::V_MOV_B64_PSEUDO;
1434 return AMDGPU::COPY;
1439 bool IsIndirectSrc)
const {
1440 if (IsIndirectSrc) {
1442 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V1);
1444 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V2);
1446 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V3);
1448 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V4);
1450 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V5);
1452 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V6);
1454 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V7);
1456 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V8);
1458 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V9);
1460 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V10);
1462 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V11);
1464 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V12);
1466 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V16);
1467 if (VecSize <= 1024)
1468 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V32);
1474 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V1);
1476 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V2);
1478 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V3);
1480 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V4);
1482 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V5);
1484 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V6);
1486 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V7);
1488 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V8);
1490 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V9);
1492 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V10);
1494 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V11);
1496 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V12);
1498 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V16);
1499 if (VecSize <= 1024)
1500 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V32);
1507 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V1;
1509 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V2;
1511 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V3;
1513 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V4;
1515 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V5;
1517 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V6;
1519 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V7;
1521 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V8;
1523 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V9;
1525 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V10;
1527 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V11;
1529 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V12;
1531 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V16;
1532 if (VecSize <= 1024)
1533 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V32;
1540 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V1;
1542 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V2;
1544 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V3;
1546 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V4;
1548 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V5;
1550 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V6;
1552 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V7;
1554 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V8;
1556 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V9;
1558 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V10;
1560 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V11;
1562 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V12;
1564 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V16;
1565 if (VecSize <= 1024)
1566 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V32;
1573 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V1;
1575 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V2;
1577 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V4;
1579 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V8;
1580 if (VecSize <= 1024)
1581 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V16;
1588 bool IsSGPR)
const {
1600 assert(EltSize == 32 &&
"invalid reg indexing elt size");
1607 return NeedsCFI ? AMDGPU::SI_SPILL_S32_CFI_SAVE : AMDGPU::SI_SPILL_S32_SAVE;
1609 return NeedsCFI ? AMDGPU::SI_SPILL_S64_CFI_SAVE : AMDGPU::SI_SPILL_S64_SAVE;
1611 return NeedsCFI ? AMDGPU::SI_SPILL_S96_CFI_SAVE : AMDGPU::SI_SPILL_S96_SAVE;
1613 return NeedsCFI ? AMDGPU::SI_SPILL_S128_CFI_SAVE
1614 : AMDGPU::SI_SPILL_S128_SAVE;
1616 return NeedsCFI ? AMDGPU::SI_SPILL_S160_CFI_SAVE
1617 : AMDGPU::SI_SPILL_S160_SAVE;
1619 return NeedsCFI ? AMDGPU::SI_SPILL_S192_CFI_SAVE
1620 : AMDGPU::SI_SPILL_S192_SAVE;
1622 return NeedsCFI ? AMDGPU::SI_SPILL_S224_CFI_SAVE
1623 : AMDGPU::SI_SPILL_S224_SAVE;
1625 return AMDGPU::SI_SPILL_S256_SAVE;
1627 return AMDGPU::SI_SPILL_S288_SAVE;
1629 return AMDGPU::SI_SPILL_S320_SAVE;
1631 return AMDGPU::SI_SPILL_S352_SAVE;
1633 return AMDGPU::SI_SPILL_S384_SAVE;
1635 return NeedsCFI ? AMDGPU::SI_SPILL_S512_CFI_SAVE
1636 : AMDGPU::SI_SPILL_S512_SAVE;
1638 return NeedsCFI ? AMDGPU::SI_SPILL_S1024_CFI_SAVE
1639 : AMDGPU::SI_SPILL_S1024_SAVE;
1648 return AMDGPU::SI_SPILL_V16_SAVE;
1650 return NeedsCFI ? AMDGPU::SI_SPILL_V32_CFI_SAVE : AMDGPU::SI_SPILL_V32_SAVE;
1652 return NeedsCFI ? AMDGPU::SI_SPILL_V64_CFI_SAVE : AMDGPU::SI_SPILL_V64_SAVE;
1654 return NeedsCFI ? AMDGPU::SI_SPILL_V96_CFI_SAVE : AMDGPU::SI_SPILL_V96_SAVE;
1656 return NeedsCFI ? AMDGPU::SI_SPILL_V128_CFI_SAVE
1657 : AMDGPU::SI_SPILL_V128_SAVE;
1659 return NeedsCFI ? AMDGPU::SI_SPILL_V160_CFI_SAVE
1660 : AMDGPU::SI_SPILL_V160_SAVE;
1662 return NeedsCFI ? AMDGPU::SI_SPILL_V192_CFI_SAVE
1663 : AMDGPU::SI_SPILL_V192_SAVE;
1665 return NeedsCFI ? AMDGPU::SI_SPILL_V224_CFI_SAVE
1666 : AMDGPU::SI_SPILL_V224_SAVE;
1668 return NeedsCFI ? AMDGPU::SI_SPILL_V256_CFI_SAVE
1669 : AMDGPU::SI_SPILL_V256_SAVE;
1671 return NeedsCFI ? AMDGPU::SI_SPILL_V288_CFI_SAVE
1672 : AMDGPU::SI_SPILL_V288_SAVE;
1674 return NeedsCFI ? AMDGPU::SI_SPILL_V320_CFI_SAVE
1675 : AMDGPU::SI_SPILL_V320_SAVE;
1677 return NeedsCFI ? AMDGPU::SI_SPILL_V352_CFI_SAVE
1678 : AMDGPU::SI_SPILL_V352_SAVE;
1680 return NeedsCFI ? AMDGPU::SI_SPILL_V384_CFI_SAVE
1681 : AMDGPU::SI_SPILL_V384_SAVE;
1683 return NeedsCFI ? AMDGPU::SI_SPILL_V512_CFI_SAVE
1684 : AMDGPU::SI_SPILL_V512_SAVE;
1686 return NeedsCFI ? AMDGPU::SI_SPILL_V1024_CFI_SAVE
1687 : AMDGPU::SI_SPILL_V1024_SAVE;
1696 return NeedsCFI ? AMDGPU::SI_SPILL_AV32_CFI_SAVE
1697 : AMDGPU::SI_SPILL_AV32_SAVE;
1699 return NeedsCFI ? AMDGPU::SI_SPILL_AV64_CFI_SAVE
1700 : AMDGPU::SI_SPILL_AV64_SAVE;
1702 return NeedsCFI ? AMDGPU::SI_SPILL_AV96_CFI_SAVE
1703 : AMDGPU::SI_SPILL_AV96_SAVE;
1705 return NeedsCFI ? AMDGPU::SI_SPILL_AV128_CFI_SAVE
1706 : AMDGPU::SI_SPILL_AV128_SAVE;
1708 return NeedsCFI ? AMDGPU::SI_SPILL_AV160_CFI_SAVE
1709 : AMDGPU::SI_SPILL_AV160_SAVE;
1711 return NeedsCFI ? AMDGPU::SI_SPILL_AV192_CFI_SAVE
1712 : AMDGPU::SI_SPILL_AV192_SAVE;
1714 return NeedsCFI ? AMDGPU::SI_SPILL_AV224_CFI_SAVE
1715 : AMDGPU::SI_SPILL_AV224_SAVE;
1717 return NeedsCFI ? AMDGPU::SI_SPILL_AV256_CFI_SAVE
1718 : AMDGPU::SI_SPILL_AV256_SAVE;
1720 return AMDGPU::SI_SPILL_AV288_SAVE;
1722 return AMDGPU::SI_SPILL_AV320_SAVE;
1724 return AMDGPU::SI_SPILL_AV352_SAVE;
1726 return AMDGPU::SI_SPILL_AV384_SAVE;
1728 return NeedsCFI ? AMDGPU::SI_SPILL_AV512_CFI_SAVE
1729 : AMDGPU::SI_SPILL_AV512_SAVE;
1731 return NeedsCFI ? AMDGPU::SI_SPILL_AV1024_CFI_SAVE
1732 : AMDGPU::SI_SPILL_AV1024_SAVE;
1739 bool IsVectorSuperClass) {
1744 if (IsVectorSuperClass)
1745 return AMDGPU::SI_SPILL_WWM_AV32_SAVE;
1747 return AMDGPU::SI_SPILL_WWM_V32_SAVE;
1753 bool IsVectorSuperClass = RI.isVectorSuperClass(RC);
1760 if (ST.hasMAIInsts())
1766void SIInstrInfo::storeRegToStackSlotImpl(
1779 FrameInfo.getObjectAlign(FrameIndex));
1780 unsigned SpillSize = RI.getSpillSize(*RC);
1786 assert(SrcReg != AMDGPU::M0 &&
"m0 should not be spilled");
1787 assert(SrcReg != AMDGPU::EXEC_LO && SrcReg != AMDGPU::EXEC_HI &&
1788 SrcReg != AMDGPU::EXEC &&
"exec should not be spilled");
1797 if (SrcReg.
isVirtual() && SpillSize == 4) {
1811 SpillSize, *MFI, NeedsCFI);
1826 storeRegToStackSlotImpl(
MBB,
MI, SrcReg, isKill, FrameIndex, RC, VReg, Flags,
1835 storeRegToStackSlotImpl(
MBB,
MI, SrcReg, isKill, FrameIndex, RC,
Register(),
1842 return AMDGPU::SI_SPILL_S32_RESTORE;
1844 return AMDGPU::SI_SPILL_S64_RESTORE;
1846 return AMDGPU::SI_SPILL_S96_RESTORE;
1848 return AMDGPU::SI_SPILL_S128_RESTORE;
1850 return AMDGPU::SI_SPILL_S160_RESTORE;
1852 return AMDGPU::SI_SPILL_S192_RESTORE;
1854 return AMDGPU::SI_SPILL_S224_RESTORE;
1856 return AMDGPU::SI_SPILL_S256_RESTORE;
1858 return AMDGPU::SI_SPILL_S288_RESTORE;
1860 return AMDGPU::SI_SPILL_S320_RESTORE;
1862 return AMDGPU::SI_SPILL_S352_RESTORE;
1864 return AMDGPU::SI_SPILL_S384_RESTORE;
1866 return AMDGPU::SI_SPILL_S512_RESTORE;
1868 return AMDGPU::SI_SPILL_S1024_RESTORE;
1877 return AMDGPU::SI_SPILL_V16_RESTORE;
1879 return AMDGPU::SI_SPILL_V32_RESTORE;
1881 return AMDGPU::SI_SPILL_V64_RESTORE;
1883 return AMDGPU::SI_SPILL_V96_RESTORE;
1885 return AMDGPU::SI_SPILL_V128_RESTORE;
1887 return AMDGPU::SI_SPILL_V160_RESTORE;
1889 return AMDGPU::SI_SPILL_V192_RESTORE;
1891 return AMDGPU::SI_SPILL_V224_RESTORE;
1893 return AMDGPU::SI_SPILL_V256_RESTORE;
1895 return AMDGPU::SI_SPILL_V288_RESTORE;
1897 return AMDGPU::SI_SPILL_V320_RESTORE;
1899 return AMDGPU::SI_SPILL_V352_RESTORE;
1901 return AMDGPU::SI_SPILL_V384_RESTORE;
1903 return AMDGPU::SI_SPILL_V512_RESTORE;
1905 return AMDGPU::SI_SPILL_V1024_RESTORE;
1914 return AMDGPU::SI_SPILL_AV32_RESTORE;
1916 return AMDGPU::SI_SPILL_AV64_RESTORE;
1918 return AMDGPU::SI_SPILL_AV96_RESTORE;
1920 return AMDGPU::SI_SPILL_AV128_RESTORE;
1922 return AMDGPU::SI_SPILL_AV160_RESTORE;
1924 return AMDGPU::SI_SPILL_AV192_RESTORE;
1926 return AMDGPU::SI_SPILL_AV224_RESTORE;
1928 return AMDGPU::SI_SPILL_AV256_RESTORE;
1930 return AMDGPU::SI_SPILL_AV288_RESTORE;
1932 return AMDGPU::SI_SPILL_AV320_RESTORE;
1934 return AMDGPU::SI_SPILL_AV352_RESTORE;
1936 return AMDGPU::SI_SPILL_AV384_RESTORE;
1938 return AMDGPU::SI_SPILL_AV512_RESTORE;
1940 return AMDGPU::SI_SPILL_AV1024_RESTORE;
1947 bool IsVectorSuperClass) {
1952 if (IsVectorSuperClass)
1953 return AMDGPU::SI_SPILL_WWM_AV32_RESTORE;
1955 return AMDGPU::SI_SPILL_WWM_V32_RESTORE;
1961 bool IsVectorSuperClass = RI.isVectorSuperClass(RC);
1968 if (ST.hasMAIInsts())
1971 assert(!RI.isAGPRClass(RC));
1985 unsigned SpillSize = RI.getSpillSize(*RC);
1992 FrameInfo.getObjectAlign(FrameIndex));
1994 if (RI.isSGPRClass(RC)) {
1997 assert(DestReg != AMDGPU::M0 &&
"m0 should not be reloaded into");
1998 assert(DestReg != AMDGPU::EXEC_LO && DestReg != AMDGPU::EXEC_HI &&
1999 DestReg != AMDGPU::EXEC &&
"exec should not be spilled");
2004 if (DestReg.
isVirtual() && SpillSize == 4) {
2033 unsigned Quantity)
const {
2035 unsigned MaxSNopCount = 1u << ST.getSNopBits();
2036 while (Quantity > 0) {
2037 unsigned Arg = std::min(Quantity, MaxSNopCount);
2048 constexpr unsigned DoorbellIDMask = 0x3ff;
2049 constexpr unsigned ECQueueWaveAbort = 0x400;
2054 if (!
MBB.succ_empty() || std::next(
MI.getIterator()) !=
MBB.end()) {
2055 MBB.splitAt(
MI,
false);
2059 MBB.addSuccessor(TrapBB);
2069 BuildMI(*TrapBB, TrapBB->
end(),
DL,
get(AMDGPU::S_MOV_B32), AMDGPU::TTMP2)
2073 BuildMI(*TrapBB, TrapBB->
end(),
DL,
get(AMDGPU::S_AND_B32), DoorbellRegMasked)
2079 BuildMI(*TrapBB, TrapBB->
end(),
DL,
get(AMDGPU::S_OR_B32), SetWaveAbortBit)
2080 .
addUse(DoorbellRegMasked)
2081 .
addImm(ECQueueWaveAbort)
2083 BuildMI(*TrapBB, TrapBB->
end(),
DL,
get(AMDGPU::S_MOV_B32), AMDGPU::M0)
2084 .
addUse(SetWaveAbortBit);
2087 BuildMI(*TrapBB, TrapBB->
end(),
DL,
get(AMDGPU::S_MOV_B32), AMDGPU::M0)
2098 return MBB.getNextNode();
2102 switch (
MI.getOpcode()) {
2104 if (
MI.isMetaInstruction())
2109 return MI.getOperand(0).getImm() + 1;
2120 switch (
MI.getOpcode()) {
2122 case AMDGPU::S_MOV_B64_term:
2125 MI.setDesc(
get(AMDGPU::S_MOV_B64));
2128 case AMDGPU::S_MOV_B32_term:
2131 MI.setDesc(
get(AMDGPU::S_MOV_B32));
2134 case AMDGPU::S_XOR_B64_term:
2137 MI.setDesc(
get(AMDGPU::S_XOR_B64));
2140 case AMDGPU::S_XOR_B32_term:
2143 MI.setDesc(
get(AMDGPU::S_XOR_B32));
2145 case AMDGPU::S_OR_B64_term:
2148 MI.setDesc(
get(AMDGPU::S_OR_B64));
2150 case AMDGPU::S_OR_B32_term:
2153 MI.setDesc(
get(AMDGPU::S_OR_B32));
2156 case AMDGPU::S_ANDN2_B64_term:
2159 MI.setDesc(
get(AMDGPU::S_ANDN2_B64));
2162 case AMDGPU::S_ANDN2_B32_term:
2165 MI.setDesc(
get(AMDGPU::S_ANDN2_B32));
2168 case AMDGPU::S_AND_B64_term:
2171 MI.setDesc(
get(AMDGPU::S_AND_B64));
2174 case AMDGPU::S_AND_B32_term:
2177 MI.setDesc(
get(AMDGPU::S_AND_B32));
2180 case AMDGPU::S_AND_SAVEEXEC_B64_term:
2183 MI.setDesc(
get(AMDGPU::S_AND_SAVEEXEC_B64));
2186 case AMDGPU::S_AND_SAVEEXEC_B32_term:
2189 MI.setDesc(
get(AMDGPU::S_AND_SAVEEXEC_B32));
2192 case AMDGPU::V_CMPX_EQ_U32_nosdst_e32_term:
2193 MI.setDesc(
get(AMDGPU::V_CMPX_EQ_U32_nosdst_e32));
2195 case AMDGPU::V_CMPX_EQ_U64_nosdst_e32_term:
2196 MI.setDesc(
get(AMDGPU::V_CMPX_EQ_U64_nosdst_e32));
2199 case AMDGPU::SI_SPILL_S32_TO_VGPR:
2200 MI.setDesc(
get(AMDGPU::V_WRITELANE_B32));
2203 case AMDGPU::SI_RESTORE_S32_FROM_VGPR:
2204 MI.setDesc(
get(AMDGPU::V_READLANE_B32));
2206 case AMDGPU::AV_MOV_B32_IMM_PSEUDO: {
2210 get(IsAGPR ? AMDGPU::V_ACCVGPR_WRITE_B32_e64 : AMDGPU::V_MOV_B32_e32));
2213 case AMDGPU::AV_MOV_B64_IMM_PSEUDO: {
2216 int64_t
Imm =
MI.getOperand(1).getImm();
2218 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2219 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2224 MI.eraseFromParent();
2230 case AMDGPU::V_MOV_B64_PSEUDO: {
2232 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2233 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2241 if (ST.hasVMovB64Inst() && Mov64RC->
contains(Dst)) {
2242 MI.setDesc(Mov64Desc);
2246 (
SrcOp.isGlobal() && ST.has64BitLiterals()))
2249 if (
SrcOp.isGlobal()) {
2254 unsigned BaseFlags, LoReloc, HiReloc;
2255 std::tie(BaseFlags, LoReloc, HiReloc) =
2262 }
else if (
SrcOp.isImm()) {
2264 APInt Lo(32,
Imm.getLoBits(32).getZExtValue());
2265 APInt Hi(32,
Imm.getHiBits(32).getZExtValue());
2289 if (ST.hasPkMovB32() &&
2308 MI.eraseFromParent();
2311 case AMDGPU::V_MOV_B64_DPP_PSEUDO: {
2315 case AMDGPU::S_MOV_B64_IMM_PSEUDO: {
2319 if (ST.has64BitLiterals()) {
2320 MI.setDesc(
get(AMDGPU::S_MOV_B64));
2324 if (
SrcOp.isGlobal()) {
2326 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2327 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2330 unsigned BaseFlags, LoReloc, HiReloc;
2331 std::tie(BaseFlags, LoReloc, HiReloc) =
2338 MI.eraseFromParent();
2345 MI.setDesc(
get(AMDGPU::S_MOV_B64));
2350 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2351 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2353 APInt Lo(32,
Imm.getLoBits(32).getZExtValue());
2354 APInt Hi(32,
Imm.getHiBits(32).getZExtValue());
2359 MI.eraseFromParent();
2362 case AMDGPU::V_SET_INACTIVE_B32: {
2366 .
add(
MI.getOperand(3))
2367 .
add(
MI.getOperand(4))
2368 .
add(
MI.getOperand(1))
2369 .
add(
MI.getOperand(2))
2370 .
add(
MI.getOperand(5));
2371 MI.eraseFromParent();
2374 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V1:
2375 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V2:
2376 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V3:
2377 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V4:
2378 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V5:
2379 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V6:
2380 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V7:
2381 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V8:
2382 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V9:
2383 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V10:
2384 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V11:
2385 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V12:
2386 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V16:
2387 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V32:
2388 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V1:
2389 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V2:
2390 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V3:
2391 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V4:
2392 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V5:
2393 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V6:
2394 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V7:
2395 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V8:
2396 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V9:
2397 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V10:
2398 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V11:
2399 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V12:
2400 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V16:
2401 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V32:
2402 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V1:
2403 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V2:
2404 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V4:
2405 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V8:
2406 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V16: {
2410 if (RI.hasVGPRs(EltRC)) {
2411 Opc = AMDGPU::V_MOVRELD_B32_e32;
2413 Opc = RI.getRegSizeInBits(*EltRC) == 64 ? AMDGPU::S_MOVRELD_B64
2414 : AMDGPU::S_MOVRELD_B32;
2419 bool IsUndef =
MI.getOperand(1).isUndef();
2420 unsigned SubReg =
MI.getOperand(3).getImm();
2421 assert(VecReg ==
MI.getOperand(1).getReg());
2426 .
add(
MI.getOperand(2))
2430 const int ImpDefIdx =
2432 const int ImpUseIdx = ImpDefIdx + 1;
2434 MI.eraseFromParent();
2437 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V1:
2438 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V2:
2439 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V3:
2440 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V4:
2441 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V5:
2442 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V6:
2443 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V7:
2444 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V8:
2445 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V9:
2446 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V10:
2447 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V11:
2448 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V12:
2449 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V16:
2450 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V32: {
2451 assert(ST.useVGPRIndexMode());
2453 bool IsUndef =
MI.getOperand(1).isUndef();
2462 const MCInstrDesc &OpDesc =
get(AMDGPU::V_MOV_B32_indirect_write);
2466 .
add(
MI.getOperand(2))
2470 const int ImpDefIdx =
2472 const int ImpUseIdx = ImpDefIdx + 1;
2479 MI.eraseFromParent();
2482 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V1:
2483 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V2:
2484 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V3:
2485 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V4:
2486 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V5:
2487 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V6:
2488 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V7:
2489 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V8:
2490 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V9:
2491 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V10:
2492 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V11:
2493 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V12:
2494 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V16:
2495 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V32: {
2496 assert(ST.useVGPRIndexMode());
2499 bool IsUndef =
MI.getOperand(1).isUndef();
2503 .
add(
MI.getOperand(2))
2516 MI.eraseFromParent();
2519 case AMDGPU::SI_PC_ADD_REL_OFFSET: {
2522 Register RegLo = RI.getSubReg(Reg, AMDGPU::sub0);
2523 Register RegHi = RI.getSubReg(Reg, AMDGPU::sub1);
2542 if (ST.hasGetPCZeroExtension()) {
2546 BuildMI(MF,
DL,
get(AMDGPU::S_SEXT_I32_I16), RegHi).addReg(RegHi));
2553 BuildMI(MF,
DL,
get(AMDGPU::S_ADD_U32), RegLo).addReg(RegLo).add(OpLo));
2563 MI.eraseFromParent();
2566 case AMDGPU::SI_PC_ADD_REL_OFFSET64: {
2576 Op.setOffset(
Op.getOffset() + 4);
2578 BuildMI(MF,
DL,
get(AMDGPU::S_ADD_U64), Reg).addReg(Reg).add(
Op));
2582 MI.eraseFromParent();
2585 case AMDGPU::ENTER_STRICT_WWM: {
2591 case AMDGPU::ENTER_STRICT_WQM: {
2598 MI.eraseFromParent();
2601 case AMDGPU::EXIT_STRICT_WWM:
2602 case AMDGPU::EXIT_STRICT_WQM: {
2608 case AMDGPU::SI_RETURN: {
2622 MI.eraseFromParent();
2626 case AMDGPU::S_MUL_U64_U32_PSEUDO:
2627 case AMDGPU::S_MUL_I64_I32_PSEUDO:
2628 MI.setDesc(
get(AMDGPU::S_MUL_U64));
2631 case AMDGPU::S_GETPC_B64_pseudo:
2632 MI.setDesc(
get(AMDGPU::S_GETPC_B64));
2633 if (ST.hasGetPCZeroExtension()) {
2635 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2644 case AMDGPU::V_MAX_BF16_PSEUDO_e64: {
2645 assert(ST.hasBF16PackedInsts());
2646 MI.setDesc(
get(AMDGPU::V_PK_MAX_NUM_BF16));
2657 case AMDGPU::GET_STACK_BASE:
2660 if (ST.getFrameLowering()->mayReserveScratchForCWSR(*
MBB.getParent())) {
2667 Register DestReg =
MI.getOperand(0).getReg();
2677 MI.getOperand(
MI.getNumExplicitOperands()).setIsDead(
false);
2678 MI.getOperand(
MI.getNumExplicitOperands()).setIsUse();
2679 MI.setDesc(
get(AMDGPU::S_CMOVK_I32));
2682 MI.setDesc(
get(AMDGPU::S_MOV_B32));
2685 MI.getNumExplicitOperands());
2703 case AMDGPU::S_MOV_B64:
2704 case AMDGPU::S_MOV_B64_IMM_PSEUDO: {
2713 if (UsedLanes.
all())
2718 unsigned LoSubReg = RI.composeSubRegIndices(OrigSubReg, AMDGPU::sub0);
2719 unsigned HiSubReg = RI.composeSubRegIndices(OrigSubReg, AMDGPU::sub1);
2721 bool NeedLo = (UsedLanes & RI.getSubRegIndexLaneMask(LoSubReg)).any();
2722 bool NeedHi = (UsedLanes & RI.getSubRegIndexLaneMask(HiSubReg)).any();
2724 if (NeedLo && NeedHi)
2728 int32_t Imm32 = NeedLo ?
Lo_32(Imm64) :
Hi_32(Imm64);
2730 unsigned UseSubReg = NeedLo ? LoSubReg : HiSubReg;
2739 case AMDGPU::S_LOAD_DWORDX16_IMM:
2740 case AMDGPU::S_LOAD_DWORDX8_IMM: {
2753 for (
auto &CandMO :
I->operands()) {
2754 if (!CandMO.isReg() || CandMO.getReg() != RegToFind || CandMO.isDef())
2762 if (!UseMO || UseMO->
getSubReg() == AMDGPU::NoSubRegister)
2766 unsigned SubregSize = RI.getSubRegIdxSize(UseMO->
getSubReg());
2772 unsigned NewOpcode = -1;
2773 if (SubregSize == 256)
2774 NewOpcode = AMDGPU::S_LOAD_DWORDX8_IMM;
2775 else if (SubregSize == 128)
2776 NewOpcode = AMDGPU::S_LOAD_DWORDX4_IMM;
2786 UseMO->
setSubReg(AMDGPU::NoSubRegister);
2791 MI->getOperand(0).setReg(DestReg);
2792 MI->getOperand(0).setSubReg(AMDGPU::NoSubRegister);
2796 OffsetMO->
setImm(FinalOffset);
2802 MI->setMemRefs(*MF, NewMMOs);
2815std::pair<MachineInstr*, MachineInstr*>
2817 assert (
MI.getOpcode() == AMDGPU::V_MOV_B64_DPP_PSEUDO);
2819 if (ST.hasVMovB64Inst() && ST.hasFeature(AMDGPU::FeatureDPALU_DPP) &&
2822 MI.setDesc(
get(AMDGPU::V_MOV_B64_dpp));
2823 return std::pair(&
MI,
nullptr);
2834 for (
auto Sub : { AMDGPU::sub0, AMDGPU::sub1 }) {
2836 if (Dst.isPhysical()) {
2837 MovDPP.addDef(RI.getSubReg(Dst,
Sub));
2844 for (
unsigned I = 1;
I <= 2; ++
I) {
2847 if (
SrcOp.isImm()) {
2849 Imm.ashrInPlace(Part * 32);
2850 MovDPP.addImm(
Imm.getLoBits(32).getZExtValue());
2854 if (Src.isPhysical())
2855 MovDPP.addReg(RI.getSubReg(Src,
Sub));
2862 MovDPP.addImm(MO.getImm());
2864 Split[Part] = MovDPP;
2868 if (Dst.isVirtual())
2875 MI.eraseFromParent();
2876 return std::pair(Split[0], Split[1]);
2879std::optional<DestSourcePair>
2881 if (
MI.getOpcode() == AMDGPU::WWM_COPY)
2884 return std::nullopt;
2888 AMDGPU::OpName Src0OpName,
2890 AMDGPU::OpName Src1OpName)
const {
2897 "All commutable instructions have both src0 and src1 modifiers");
2899 int Src0ModsVal = Src0Mods->
getImm();
2900 int Src1ModsVal = Src1Mods->
getImm();
2902 Src1Mods->
setImm(Src0ModsVal);
2903 Src0Mods->
setImm(Src1ModsVal);
2912 bool IsKill = RegOp.
isKill();
2914 bool IsUndef = RegOp.
isUndef();
2915 bool IsDebug = RegOp.
isDebug();
2917 if (NonRegOp.
isImm())
2919 else if (NonRegOp.
isFI())
2940 int64_t NonRegVal = NonRegOp1.
getImm();
2943 NonRegOp2.
setImm(NonRegVal);
2950 unsigned OpIdx1)
const {
2955 unsigned Opc =
MI.getOpcode();
2956 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
2966 if ((
int)OpIdx0 == Src0Idx && !MO0.
isReg() &&
2969 if ((
int)OpIdx1 == Src0Idx && !MO1.
isReg() &&
2974 if ((
int)OpIdx1 != Src0Idx && MO0.
isReg()) {
2980 if ((
int)OpIdx0 != Src0Idx && MO1.
isReg()) {
3002 unsigned Src1Idx)
const {
3003 assert(!NewMI &&
"this should never be used");
3008 unsigned Opc =
MI.getOpcode();
3010 if (CommutedOpcode == -1)
3013 if (Src0Idx > Src1Idx)
3016 assert(AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0) ==
3017 static_cast<int>(Src0Idx) &&
3018 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1) ==
3019 static_cast<int>(Src1Idx) &&
3020 "inconsistency with findCommutedOpIndices");
3045 Src1, AMDGPU::OpName::src1_modifiers);
3048 AMDGPU::OpName::src1_sel);
3060 unsigned &SrcOpIdx0,
3061 unsigned &SrcOpIdx1)
const {
3069 unsigned &SrcOpIdx0,
3070 unsigned &SrcOpIdx1)
const {
3071 if (!
Desc.isCommutable())
3074 unsigned Opc =
Desc.getOpcode();
3075 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
3079 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
3083 return fixCommutedOpIndices(SrcOpIdx0, SrcOpIdx1, Src0Idx, Src1Idx);
3087 int64_t BrOffset)
const {
3104 return MI.getOperand(0).getMBB();
3109 if (
MI.getOpcode() == AMDGPU::SI_IF ||
MI.getOpcode() == AMDGPU::SI_ELSE ||
3110 MI.getOpcode() == AMDGPU::SI_LOOP ||
3111 MI.getOpcode() == AMDGPU::SI_WATERFALL_LOOP)
3123 "new block should be inserted for expanding unconditional branch");
3126 "restore block should be inserted for restoring clobbered registers");
3134 if (ST.useAddPC64Inst()) {
3136 MCCtx.createTempSymbol(
"offset",
true);
3140 MCCtx.createTempSymbol(
"post_addpc",
true);
3141 AddPC->setPostInstrSymbol(*MF, PostAddPCLabel);
3145 Offset->setVariableValue(OffsetExpr);
3149 assert(RS &&
"RegScavenger required for long branching");
3157 const bool FlushSGPRWrites = (ST.isWave64() && ST.hasVALUMaskWriteHazard()) ||
3158 ST.hasVALUReadSGPRHazard();
3159 auto ApplyHazardWorkarounds = [
this, &
MBB, &
I, &
DL, FlushSGPRWrites]() {
3160 if (FlushSGPRWrites)
3168 ApplyHazardWorkarounds();
3171 MCCtx.createTempSymbol(
"post_getpc",
true);
3175 MCCtx.createTempSymbol(
"offset_lo",
true);
3177 MCCtx.createTempSymbol(
"offset_hi",
true);
3180 .
addReg(PCReg, {}, AMDGPU::sub0)
3184 .
addReg(PCReg, {}, AMDGPU::sub1)
3186 ApplyHazardWorkarounds();
3227 if (LongBranchReservedReg) {
3228 RS->enterBasicBlock(
MBB);
3229 Scav = LongBranchReservedReg;
3231 RS->enterBasicBlockEnd(
MBB);
3232 Scav = RS->scavengeRegisterBackwards(
3237 RS->setRegUsed(Scav);
3245 TRI->spillEmergencySGPR(GetPC, RestoreBB, AMDGPU::SGPR0_SGPR1, RS);
3262unsigned SIInstrInfo::getBranchOpcode(SIInstrInfo::BranchPredicate
Cond) {
3264 case SIInstrInfo::SCC_TRUE:
3265 return AMDGPU::S_CBRANCH_SCC1;
3266 case SIInstrInfo::SCC_FALSE:
3267 return AMDGPU::S_CBRANCH_SCC0;
3268 case SIInstrInfo::VCCNZ:
3269 return AMDGPU::S_CBRANCH_VCCNZ;
3270 case SIInstrInfo::VCCZ:
3271 return AMDGPU::S_CBRANCH_VCCZ;
3272 case SIInstrInfo::EXECNZ:
3273 return AMDGPU::S_CBRANCH_EXECNZ;
3274 case SIInstrInfo::EXECZ:
3275 return AMDGPU::S_CBRANCH_EXECZ;
3281SIInstrInfo::BranchPredicate SIInstrInfo::getBranchPredicate(
unsigned Opcode) {
3283 case AMDGPU::S_CBRANCH_SCC0:
3285 case AMDGPU::S_CBRANCH_SCC1:
3287 case AMDGPU::S_CBRANCH_VCCNZ:
3289 case AMDGPU::S_CBRANCH_VCCZ:
3291 case AMDGPU::S_CBRANCH_EXECNZ:
3293 case AMDGPU::S_CBRANCH_EXECZ:
3305 bool AllowModify)
const {
3306 if (
I->getOpcode() == AMDGPU::S_BRANCH) {
3308 TBB =
I->getOperand(0).getMBB();
3312 BranchPredicate Pred = getBranchPredicate(
I->getOpcode());
3313 if (Pred == INVALID_BR)
3318 Cond.push_back(
I->getOperand(1));
3322 if (
I ==
MBB.end()) {
3328 if (
I->getOpcode() == AMDGPU::S_BRANCH) {
3330 FBB =
I->getOperand(0).getMBB();
3340 bool AllowModify)
const {
3348 while (
I != E && !
I->isBranch() && !
I->isReturn()) {
3349 switch (
I->getOpcode()) {
3350 case AMDGPU::S_MOV_B64_term:
3351 case AMDGPU::S_XOR_B64_term:
3352 case AMDGPU::S_OR_B64_term:
3353 case AMDGPU::S_ANDN2_B64_term:
3354 case AMDGPU::S_AND_B64_term:
3355 case AMDGPU::S_AND_SAVEEXEC_B64_term:
3356 case AMDGPU::S_MOV_B32_term:
3357 case AMDGPU::S_XOR_B32_term:
3358 case AMDGPU::S_OR_B32_term:
3359 case AMDGPU::S_ANDN2_B32_term:
3360 case AMDGPU::S_AND_B32_term:
3361 case AMDGPU::S_AND_SAVEEXEC_B32_term:
3362 case AMDGPU::V_CMPX_EQ_U32_nosdst_e32_term:
3363 case AMDGPU::V_CMPX_EQ_U64_nosdst_e32_term:
3366 case AMDGPU::SI_ELSE:
3367 case AMDGPU::SI_KILL_I1_TERMINATOR:
3368 case AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR:
3385 int *BytesRemoved)
const {
3387 unsigned RemovedSize = 0;
3390 if (
MI.isBranch() ||
MI.isReturn()) {
3392 MI.eraseFromParent();
3398 *BytesRemoved = RemovedSize;
3415 int *BytesAdded)
const {
3416 if (!FBB &&
Cond.empty()) {
3420 *BytesAdded = ST.hasOffset3fBug() ? 8 : 4;
3427 = getBranchOpcode(
static_cast<BranchPredicate
>(
Cond[0].
getImm()));
3439 *BytesAdded = ST.hasOffset3fBug() ? 8 : 4;
3457 *BytesAdded = ST.hasOffset3fBug() ? 16 : 8;
3464 if (
Cond.size() != 2) {
3468 if (
Cond[0].isImm()) {
3489 bool shouldIgnoreForPipelining(
const MachineInstr *
MI)
const override {
3493 std::optional<bool> createTripCountGreaterCondition(
3494 int TC, MachineBasicBlock &
MBB,
3495 SmallVectorImpl<MachineOperand> &CondParam)
override {
3496 CondParam = this->
Cond;
3500 void adjustTripCount(
int TripCountAdjust)
override {}
3502 void setPreheader(MachineBasicBlock *NewPreheader)
override {}
3506std::unique_ptr<TargetInstrInfo::PipelinerLoopInfo>
3515 if (
TBB == LoopBB && FBB == LoopBB)
3522 assert((
TBB == LoopBB || FBB == LoopBB) &&
3523 "The Loop must be a single-basic-block loop");
3526 BranchPredicate Pred =
static_cast<BranchPredicate
>(
Cond[0].getImm());
3527 if (Pred != SCC_TRUE && Pred != SCC_FALSE)
3532 if (
MI.isCall() ||
MI.isInlineAsm())
3547 if (CmpI == Instructions.end() || CmpI->isPHI())
3551 return std::make_unique<AMDGPUPipelinerLoopInfo>(
CmpInst,
Cond);
3557 Register FalseReg,
int &CondCycles,
3558 int &TrueCycles,
int &FalseCycles)
const {
3568 CondCycles = TrueCycles = FalseCycles = NumInsts;
3571 return RI.hasVGPRs(RC) && NumInsts <= 6;
3585 if (NumInsts % 2 == 0)
3588 CondCycles = TrueCycles = FalseCycles = NumInsts;
3589 return RI.isSGPRClass(RC);
3600 BranchPredicate Pred =
static_cast<BranchPredicate
>(
Cond[0].getImm());
3601 if (Pred == VCCZ || Pred == SCC_FALSE) {
3602 Pred =
static_cast<BranchPredicate
>(-Pred);
3608 unsigned DstSize = RI.getRegSizeInBits(*DstRC);
3610 if (DstSize == 32) {
3612 if (Pred == SCC_TRUE) {
3627 if (DstSize == 64 && Pred == SCC_TRUE) {
3637 static const int16_t Sub0_15[] = {
3638 AMDGPU::sub0, AMDGPU::sub1, AMDGPU::sub2, AMDGPU::sub3,
3639 AMDGPU::sub4, AMDGPU::sub5, AMDGPU::sub6, AMDGPU::sub7,
3640 AMDGPU::sub8, AMDGPU::sub9, AMDGPU::sub10, AMDGPU::sub11,
3641 AMDGPU::sub12, AMDGPU::sub13, AMDGPU::sub14, AMDGPU::sub15,
3644 static const int16_t Sub0_15_64[] = {
3645 AMDGPU::sub0_sub1, AMDGPU::sub2_sub3,
3646 AMDGPU::sub4_sub5, AMDGPU::sub6_sub7,
3647 AMDGPU::sub8_sub9, AMDGPU::sub10_sub11,
3648 AMDGPU::sub12_sub13, AMDGPU::sub14_sub15,
3651 unsigned SelOp = AMDGPU::V_CNDMASK_B32_e32;
3653 const int16_t *SubIndices = Sub0_15;
3654 int NElts = DstSize / 32;
3658 if (Pred == SCC_TRUE) {
3660 SelOp = AMDGPU::S_CSELECT_B32;
3661 EltRC = &AMDGPU::SGPR_32RegClass;
3663 SelOp = AMDGPU::S_CSELECT_B64;
3664 EltRC = &AMDGPU::SGPR_64RegClass;
3665 SubIndices = Sub0_15_64;
3671 MBB,
I,
DL,
get(AMDGPU::REG_SEQUENCE), DstReg);
3676 for (
int Idx = 0; Idx != NElts; ++Idx) {
3680 unsigned SubIdx = SubIndices[Idx];
3683 if (SelOp == AMDGPU::V_CNDMASK_B32_e32) {
3685 .
addReg(FalseReg, {}, SubIdx)
3686 .addReg(TrueReg, {}, SubIdx);
3689 .
addReg(TrueReg, {}, SubIdx)
3690 .addReg(FalseReg, {}, SubIdx);
3703 if (
MI.isBranch() ||
MI.isCall() ||
MI.isReturn() ||
MI.isIndirectBranch())
3706 switch (
MI.getOpcode()) {
3707 case AMDGPU::S_ENDPGM:
3708 case AMDGPU::S_ENDPGM_SAVED:
3709 case AMDGPU::S_TRAP:
3710 case AMDGPU::S_GETREG_B32:
3711 case AMDGPU::S_SETREG_B32:
3712 case AMDGPU::S_SETREG_B32_mode:
3713 case AMDGPU::S_SETREG_IMM32_B32:
3714 case AMDGPU::S_SETREG_IMM32_B32_mode:
3715 case AMDGPU::S_SENDMSG:
3716 case AMDGPU::S_SENDMSGHALT:
3717 case AMDGPU::S_SENDMSG_RTN_B32:
3718 case AMDGPU::S_SENDMSG_RTN_B64:
3719 case AMDGPU::S_BARRIER_WAIT:
3720 case AMDGPU::S_BARRIER_SIGNAL_M0:
3721 case AMDGPU::S_BARRIER_SIGNAL_IMM:
3722 case AMDGPU::S_BARRIER_SIGNAL_ISFIRST_M0:
3723 case AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM:
3731 switch (
MI.getOpcode()) {
3732 case AMDGPU::V_MOV_B16_t16_e32:
3733 case AMDGPU::V_MOV_B16_t16_e64:
3734 case AMDGPU::V_MOV_B32_e32:
3735 case AMDGPU::V_MOV_B32_e64:
3736 case AMDGPU::V_MOV_B64_PSEUDO:
3737 case AMDGPU::V_MOV_B64_e32:
3738 case AMDGPU::V_MOV_B64_e64:
3739 case AMDGPU::S_MOV_B32:
3740 case AMDGPU::S_MOV_B64:
3741 case AMDGPU::S_MOV_B64_IMM_PSEUDO:
3743 case AMDGPU::WWM_COPY:
3744 case AMDGPU::V_ACCVGPR_WRITE_B32_e64:
3745 case AMDGPU::V_ACCVGPR_READ_B32_e64:
3746 case AMDGPU::V_ACCVGPR_MOV_B32:
3747 case AMDGPU::AV_MOV_B32_IMM_PSEUDO:
3748 case AMDGPU::AV_MOV_B64_IMM_PSEUDO:
3756 switch (
MI.getOpcode()) {
3757 case AMDGPU::V_MOV_B16_t16_e32:
3758 case AMDGPU::V_MOV_B16_t16_e64:
3760 case AMDGPU::V_MOV_B32_e32:
3761 case AMDGPU::V_MOV_B32_e64:
3762 case AMDGPU::V_MOV_B64_PSEUDO:
3763 case AMDGPU::V_MOV_B64_e32:
3764 case AMDGPU::V_MOV_B64_e64:
3765 case AMDGPU::S_MOV_B32:
3766 case AMDGPU::S_MOV_B64:
3767 case AMDGPU::S_MOV_B64_IMM_PSEUDO:
3769 case AMDGPU::WWM_COPY:
3770 case AMDGPU::V_ACCVGPR_WRITE_B32_e64:
3771 case AMDGPU::V_ACCVGPR_READ_B32_e64:
3772 case AMDGPU::V_ACCVGPR_MOV_B32:
3773 case AMDGPU::AV_MOV_B32_IMM_PSEUDO:
3774 case AMDGPU::AV_MOV_B64_IMM_PSEUDO:
3782 AMDGPU::OpName::src0_modifiers, AMDGPU::OpName::src1_modifiers,
3783 AMDGPU::OpName::src2_modifiers, AMDGPU::OpName::clamp,
3784 AMDGPU::OpName::omod, AMDGPU::OpName::op_sel};
3787 unsigned Opc =
MI.getOpcode();
3789 int Idx = AMDGPU::getNamedOperandIdx(
Opc, Name);
3791 MI.removeOperand(Idx);
3797 MI.setDesc(NewDesc);
3803 unsigned NumOps =
Desc.getNumOperands() +
Desc.implicit_uses().size() +
3804 Desc.implicit_defs().size();
3806 for (
unsigned I =
MI.getNumOperands() - 1;
I >=
NumOps; --
I)
3807 MI.removeOperand(
I);
3811 unsigned SubRegIndex) {
3812 switch (SubRegIndex) {
3813 case AMDGPU::NoSubRegister:
3823 case AMDGPU::sub1_lo16:
3825 case AMDGPU::sub1_hi16:
3828 return std::nullopt;
3836 case AMDGPU::V_MAC_F16_e32:
3837 case AMDGPU::V_MAC_F16_e64:
3838 case AMDGPU::V_MAD_F16_e64:
3839 return AMDGPU::V_MADAK_F16;
3840 case AMDGPU::V_MAC_F32_e32:
3841 case AMDGPU::V_MAC_F32_e64:
3842 case AMDGPU::V_MAD_F32_e64:
3843 return AMDGPU::V_MADAK_F32;
3844 case AMDGPU::V_FMAC_F32_e32:
3845 case AMDGPU::V_FMAC_F32_e64:
3846 case AMDGPU::V_FMA_F32_e64:
3847 return AMDGPU::V_FMAAK_F32;
3848 case AMDGPU::V_FMAC_F16_e32:
3849 case AMDGPU::V_FMAC_F16_e64:
3850 case AMDGPU::V_FMAC_F16_t16_e64:
3851 case AMDGPU::V_FMAC_F16_fake16_e64:
3852 case AMDGPU::V_FMAC_F16_t16_e32:
3853 case AMDGPU::V_FMAC_F16_fake16_e32:
3854 case AMDGPU::V_FMA_F16_e64:
3855 return ST.hasTrue16BitInsts() ? ST.useRealTrue16Insts()
3856 ? AMDGPU::V_FMAAK_F16_t16
3857 : AMDGPU::V_FMAAK_F16_fake16
3858 : AMDGPU::V_FMAAK_F16;
3859 case AMDGPU::V_FMAC_F64_e32:
3860 case AMDGPU::V_FMAC_F64_e64:
3861 case AMDGPU::V_FMA_F64_e64:
3862 return AMDGPU::V_FMAAK_F64;
3870 case AMDGPU::V_MAC_F16_e32:
3871 case AMDGPU::V_MAC_F16_e64:
3872 case AMDGPU::V_MAD_F16_e64:
3873 return AMDGPU::V_MADMK_F16;
3874 case AMDGPU::V_MAC_F32_e32:
3875 case AMDGPU::V_MAC_F32_e64:
3876 case AMDGPU::V_MAD_F32_e64:
3877 return AMDGPU::V_MADMK_F32;
3878 case AMDGPU::V_FMAC_F32_e32:
3879 case AMDGPU::V_FMAC_F32_e64:
3880 case AMDGPU::V_FMA_F32_e64:
3881 return AMDGPU::V_FMAMK_F32;
3882 case AMDGPU::V_FMAC_F16_e32:
3883 case AMDGPU::V_FMAC_F16_e64:
3884 case AMDGPU::V_FMAC_F16_t16_e64:
3885 case AMDGPU::V_FMAC_F16_fake16_e64:
3886 case AMDGPU::V_FMAC_F16_t16_e32:
3887 case AMDGPU::V_FMAC_F16_fake16_e32:
3888 case AMDGPU::V_FMA_F16_e64:
3889 return ST.hasTrue16BitInsts() ? ST.useRealTrue16Insts()
3890 ? AMDGPU::V_FMAMK_F16_t16
3891 : AMDGPU::V_FMAMK_F16_fake16
3892 : AMDGPU::V_FMAMK_F16;
3893 case AMDGPU::V_FMAC_F64_e32:
3894 case AMDGPU::V_FMAC_F64_e64:
3895 case AMDGPU::V_FMA_F64_e64:
3896 return AMDGPU::V_FMAMK_F64;
3910 assert(!
DefMI.getOperand(0).getSubReg() &&
"Expected SSA form");
3913 if (
Opc == AMDGPU::COPY) {
3914 assert(!
UseMI.getOperand(0).getSubReg() &&
"Expected SSA form");
3921 if (HasMultipleUses) {
3924 unsigned ImmDefSize = RI.getRegSizeInBits(*MRI->
getRegClass(Reg));
3927 if (UseSubReg != AMDGPU::NoSubRegister && ImmDefSize == 64)
3935 if (ImmDefSize == 32 &&
3940 bool Is16Bit = UseSubReg != AMDGPU::NoSubRegister &&
3941 RI.getSubRegIdxSize(UseSubReg) == 16;
3944 if (RI.hasVGPRs(DstRC))
3947 if (DstReg.
isVirtual() && UseSubReg != AMDGPU::lo16)
3953 unsigned NewOpc = AMDGPU::INSTRUCTION_LIST_END;
3960 for (
unsigned MovOp :
3961 {AMDGPU::S_MOV_B32, AMDGPU::V_MOV_B32_e32, AMDGPU::S_MOV_B64,
3962 AMDGPU::V_MOV_B64_PSEUDO, AMDGPU::V_ACCVGPR_WRITE_B32_e64}) {
3970 MovDstRC = RI.getMatchingSuperRegClass(MovDstRC, DstRC, AMDGPU::lo16);
3974 if (MovDstPhysReg) {
3978 RI.getMatchingSuperReg(MovDstPhysReg, AMDGPU::lo16, MovDstRC);
3985 if (MovDstPhysReg) {
3986 if (!MovDstRC->
contains(MovDstPhysReg))
4002 if (!RI.opCanUseLiteralConstant(OpInfo.OperandType) &&
4010 if (NewOpc == AMDGPU::INSTRUCTION_LIST_END)
4014 UseMI.getOperand(0).setSubReg(AMDGPU::NoSubRegister);
4016 UseMI.getOperand(0).setReg(MovDstPhysReg);
4021 UseMI.setDesc(NewMCID);
4022 UseMI.getOperand(1).ChangeToImmediate(*SubRegImm);
4023 UseMI.addImplicitDefUseOperands(*MF);
4027 if (HasMultipleUses)
4030 if (
Opc == AMDGPU::V_MAD_F32_e64 ||
Opc == AMDGPU::V_MAC_F32_e64 ||
4031 Opc == AMDGPU::V_MAD_F16_e64 ||
Opc == AMDGPU::V_MAC_F16_e64 ||
4032 Opc == AMDGPU::V_FMA_F32_e64 ||
Opc == AMDGPU::V_FMAC_F32_e64 ||
4033 Opc == AMDGPU::V_FMA_F16_e64 ||
Opc == AMDGPU::V_FMAC_F16_e64 ||
4034 Opc == AMDGPU::V_FMAC_F16_t16_e64 ||
4035 Opc == AMDGPU::V_FMAC_F16_fake16_e64 ||
Opc == AMDGPU::V_FMA_F64_e64 ||
4036 Opc == AMDGPU::V_FMAC_F64_e64) {
4045 int Src0Idx = getNamedOperandIdx(
UseMI.getOpcode(), AMDGPU::OpName::src0);
4056 auto CopyRegOperandToNarrowerRC =
4059 if (!
MI.getOperand(OpNo).isReg())
4063 if (RI.getCommonSubClass(RC, NewRC) != NewRC)
4066 BuildMI(*
MI.getParent(),
MI.getIterator(),
MI.getDebugLoc(),
4067 get(AMDGPU::COPY), Tmp)
4069 MI.getOperand(OpNo).setReg(Tmp);
4070 MI.getOperand(OpNo).setIsKill();
4077 Src1->
isReg() && Src1->
getReg() == Reg ? Src0 : Src1;
4078 if (!RegSrc->
isReg())
4081 ST.getConstantBusLimit(
Opc) < 2)
4096 if (Def && Def->isMoveImmediate() &&
4111 unsigned SrcSubReg = RegSrc->
getSubReg();
4116 if (
Opc == AMDGPU::V_MAC_F32_e64 ||
Opc == AMDGPU::V_MAC_F16_e64 ||
4117 Opc == AMDGPU::V_FMAC_F32_e64 ||
Opc == AMDGPU::V_FMAC_F16_t16_e64 ||
4118 Opc == AMDGPU::V_FMAC_F16_fake16_e64 ||
4119 Opc == AMDGPU::V_FMAC_F16_e64 ||
Opc == AMDGPU::V_FMAC_F64_e64)
4120 UseMI.untieRegOperand(
4121 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2));
4128 if (NewOpc == AMDGPU::V_FMAMK_F16_t16 ||
4129 NewOpc == AMDGPU::V_FMAMK_F16_fake16) {
4133 UseMI.getDebugLoc(),
get(AMDGPU::COPY),
4134 UseMI.getOperand(0).getReg())
4136 UseMI.getOperand(0).setReg(Tmp);
4137 CopyRegOperandToNarrowerRC(
UseMI, 1, NewRC);
4138 CopyRegOperandToNarrowerRC(
UseMI, 3, NewRC);
4143 DefMI.eraseFromParent();
4150 if (ST.getConstantBusLimit(
Opc) < 2) {
4153 bool Src0Inlined =
false;
4154 if (Src0->
isReg()) {
4159 if (Def && Def->isMoveImmediate() &&
4164 }
else if (ST.getConstantBusLimit(
Opc) <= 1 &&
4165 RI.isSGPRReg(*MRI, Src0->
getReg())) {
4171 if (Src1->
isReg() && !Src0Inlined) {
4174 if (Def && Def->isMoveImmediate() &&
4178 else if (RI.isSGPRReg(*MRI, Src1->
getReg()))
4191 if (
Opc == AMDGPU::V_MAC_F32_e64 ||
Opc == AMDGPU::V_MAC_F16_e64 ||
4192 Opc == AMDGPU::V_FMAC_F32_e64 ||
Opc == AMDGPU::V_FMAC_F16_t16_e64 ||
4193 Opc == AMDGPU::V_FMAC_F16_fake16_e64 ||
4194 Opc == AMDGPU::V_FMAC_F16_e64 ||
Opc == AMDGPU::V_FMAC_F64_e64)
4195 UseMI.untieRegOperand(
4196 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2));
4198 const std::optional<int64_t> SubRegImm =
4208 if (NewOpc == AMDGPU::V_FMAAK_F16_t16 ||
4209 NewOpc == AMDGPU::V_FMAAK_F16_fake16) {
4213 UseMI.getDebugLoc(),
get(AMDGPU::COPY),
4214 UseMI.getOperand(0).getReg())
4216 UseMI.getOperand(0).setReg(Tmp);
4217 CopyRegOperandToNarrowerRC(
UseMI, 1, NewRC);
4218 CopyRegOperandToNarrowerRC(
UseMI, 2, NewRC);
4227 AMDGPU::getNamedOperandIdx(
UseMI.getOpcode(), AMDGPU::OpName::src0);
4233 DefMI.eraseFromParent();
4245 if (BaseOps1.
size() != BaseOps2.
size())
4247 for (
size_t I = 0,
E = BaseOps1.
size();
I <
E; ++
I) {
4248 if (!BaseOps1[
I]->isIdenticalTo(*BaseOps2[
I]))
4256 int LowOffset = OffsetA < OffsetB ? OffsetA : OffsetB;
4257 int HighOffset = OffsetA < OffsetB ? OffsetB : OffsetA;
4258 LocationSize LowWidth = (LowOffset == OffsetA) ? WidthA : WidthB;
4260 LowOffset + (int)LowWidth.
getValue() <= HighOffset;
4263bool SIInstrInfo::checkInstOffsetsDoNotOverlap(
const MachineInstr &MIa,
4266 int64_t Offset0, Offset1;
4269 bool Offset0IsScalable, Offset1IsScalable;
4283 LocationSize Width0 = MIa.
memoperands().front()->getSize();
4284 LocationSize Width1 = MIb.
memoperands().front()->getSize();
4291 "MIa must load from or modify a memory location");
4293 "MIb must load from or modify a memory location");
4315 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4322 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4332 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4346 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4357 case AMDGPU::V_MAC_F16_e32:
4358 case AMDGPU::V_MAC_F16_e64:
4359 return AMDGPU::V_MAD_F16_e64;
4360 case AMDGPU::V_MAC_F32_e32:
4361 case AMDGPU::V_MAC_F32_e64:
4362 return AMDGPU::V_MAD_F32_e64;
4363 case AMDGPU::V_MAC_LEGACY_F32_e32:
4364 case AMDGPU::V_MAC_LEGACY_F32_e64:
4365 return AMDGPU::V_MAD_LEGACY_F32_e64;
4366 case AMDGPU::V_FMAC_LEGACY_F32_e32:
4367 case AMDGPU::V_FMAC_LEGACY_F32_e64:
4368 return AMDGPU::V_FMA_LEGACY_F32_e64;
4369 case AMDGPU::V_FMAC_F16_e32:
4370 case AMDGPU::V_FMAC_F16_e64:
4371 case AMDGPU::V_FMAC_F16_t16_e64:
4372 case AMDGPU::V_FMAC_F16_fake16_e64:
4373 return ST.hasTrue16BitInsts() ? ST.useRealTrue16Insts()
4374 ? AMDGPU::V_FMA_F16_gfx9_t16_e64
4375 : AMDGPU::V_FMA_F16_gfx9_fake16_e64
4376 : AMDGPU::V_FMA_F16_gfx9_e64;
4377 case AMDGPU::V_FMAC_F32_e32:
4378 case AMDGPU::V_FMAC_F32_e64:
4379 return AMDGPU::V_FMA_F32_e64;
4380 case AMDGPU::V_FMAC_F64_e32:
4381 case AMDGPU::V_FMAC_F64_e64:
4382 return AMDGPU::V_FMA_F64_e64;
4401 if (
MI.isBundle()) {
4404 if (
MI.getBundleSize() != 1)
4406 CandidateMI =
MI.getNextNode();
4410 MachineInstr *NewMI = convertToThreeAddressImpl(*CandidateMI, U);
4414 if (
MI.isBundle()) {
4419 MI.untieRegOperand(MO.getOperandNo());
4426 if (Def.isEarlyClobber() && Def.isReg() &&
4431 auto UpdateDefIndex = [&](
LiveRange &LR) {
4432 auto *S = LR.find(OldIndex);
4433 if (S != LR.end() && S->start == OldIndex) {
4434 assert(S->valno && S->valno->def == OldIndex);
4435 S->start = NewIndex;
4436 S->valno->def = NewIndex;
4440 for (
auto &SR : LI.subranges())
4446 if (U.RemoveMIUse) {
4449 Register DefReg = U.RemoveMIUse->getOperand(0).getReg();
4453 U.RemoveMIUse->setDesc(
get(AMDGPU::IMPLICIT_DEF));
4454 U.RemoveMIUse->getOperand(0).setIsDead(
true);
4455 for (
unsigned I = U.RemoveMIUse->getNumOperands() - 1;
I != 0; --
I)
4456 U.RemoveMIUse->removeOperand(
I);
4459 if (
MI.isBundle()) {
4463 if (MO.isReg() && MO.getReg() == DefReg) {
4464 assert(MO.getSubReg() == 0 &&
4465 "tied sub-registers in bundles currently not supported");
4466 MI.removeOperand(MO.getOperandNo());
4483 if (MIOp.isReg() && MIOp.getReg() == DefReg) {
4484 MIOp.setIsUndef(
true);
4485 MIOp.setReg(DummyReg);
4489 if (
MI.isBundle()) {
4493 if (MIOp.isReg() && MIOp.getReg() == DefReg) {
4494 MIOp.setIsUndef(
true);
4495 MIOp.setReg(DummyReg);
4508 return MI.isBundle() ? &
MI : NewMI;
4513 ThreeAddressUpdates &U)
const {
4515 unsigned Opc =
MI.getOpcode();
4519 if (NewMFMAOpc != -1) {
4522 for (
unsigned I = 0, E =
MI.getNumExplicitOperands();
I != E; ++
I)
4523 MIB.
add(
MI.getOperand(
I));
4531 for (
unsigned I = 0,
E =
MI.getNumExplicitOperands();
I !=
E; ++
I)
4536 assert(
Opc != AMDGPU::V_FMAC_F16_t16_e32 &&
4537 Opc != AMDGPU::V_FMAC_F16_fake16_e32 &&
4538 "V_FMAC_F16_t16/fake16_e32 is not supported and not expected to be "
4542 bool IsF64 =
Opc == AMDGPU::V_FMAC_F64_e32 ||
Opc == AMDGPU::V_FMAC_F64_e64;
4543 bool IsLegacy =
Opc == AMDGPU::V_MAC_LEGACY_F32_e32 ||
4544 Opc == AMDGPU::V_MAC_LEGACY_F32_e64 ||
4545 Opc == AMDGPU::V_FMAC_LEGACY_F32_e32 ||
4546 Opc == AMDGPU::V_FMAC_LEGACY_F32_e64;
4547 bool Src0Literal =
false;
4552 case AMDGPU::V_MAC_F16_e64:
4553 case AMDGPU::V_FMAC_F16_e64:
4554 case AMDGPU::V_FMAC_F16_t16_e64:
4555 case AMDGPU::V_FMAC_F16_fake16_e64:
4556 case AMDGPU::V_MAC_F32_e64:
4557 case AMDGPU::V_MAC_LEGACY_F32_e64:
4558 case AMDGPU::V_FMAC_F32_e64:
4559 case AMDGPU::V_FMAC_LEGACY_F32_e64:
4560 case AMDGPU::V_FMAC_F64_e64:
4562 case AMDGPU::V_MAC_F16_e32:
4563 case AMDGPU::V_FMAC_F16_e32:
4564 case AMDGPU::V_MAC_F32_e32:
4565 case AMDGPU::V_MAC_LEGACY_F32_e32:
4566 case AMDGPU::V_FMAC_F32_e32:
4567 case AMDGPU::V_FMAC_LEGACY_F32_e32:
4568 case AMDGPU::V_FMAC_F64_e32: {
4569 int Src0Idx = AMDGPU::getNamedOperandIdx(
MI.getOpcode(),
4570 AMDGPU::OpName::src0);
4571 const MachineOperand *Src0 = &
MI.getOperand(Src0Idx);
4582 MachineInstrBuilder MIB;
4585 const MachineOperand *Src0Mods =
4588 const MachineOperand *Src1Mods =
4591 const MachineOperand *Src2Mods =
4597 if (!Src0Mods && !Src1Mods && !Src2Mods && !Clamp && !Omod && !IsLegacy &&
4598 (!IsF64 || ST.hasFmaakFmamkF64Insts()) &&
4600 (ST.getConstantBusLimit(
Opc) > 1 || !Src0->
isReg() ||
4602 MachineInstr *
DefMI =
nullptr;
4604 std::optional<int64_t> ImmOpt;
4639 MI, AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::src0),
4655 if (Src0Literal && !ST.hasVOP3Literal())
4683 switch (
MI.getOpcode()) {
4684 case AMDGPU::S_SET_GPR_IDX_ON:
4685 case AMDGPU::S_SET_GPR_IDX_MODE:
4686 case AMDGPU::S_SET_GPR_IDX_OFF:
4704 if (
MI.isTerminator() ||
MI.isPosition())
4708 if (
MI.getOpcode() == TargetOpcode::INLINEASM_BR)
4711 if (
MI.getOpcode() == AMDGPU::SCHED_BARRIER &&
MI.getOperand(0).getImm() == 0)
4717 return MI.modifiesRegister(AMDGPU::EXEC, &RI) ||
4718 MI.getOpcode() == AMDGPU::S_SETREG_IMM32_B32 ||
4719 MI.getOpcode() == AMDGPU::S_SETREG_B32 ||
4720 MI.getOpcode() == AMDGPU::S_SETPRIO ||
4721 MI.getOpcode() == AMDGPU::S_SETPRIO_INC_WG ||
4726 return Opcode == AMDGPU::DS_ORDERED_COUNT ||
4727 Opcode == AMDGPU::DS_ADD_GS_REG_RTN ||
4728 Opcode == AMDGPU::DS_SUB_GS_REG_RTN ||
isGWS(Opcode);
4742 if (
MI.getMF()->getFunction().hasFnAttribute(
"amdgpu-no-flat-scratch-init"))
4747 if (
MI.memoperands_empty())
4752 unsigned AS = Memop->getAddrSpace();
4753 if (AS == AMDGPUAS::FLAT_ADDRESS) {
4754 const MDNode *MD = Memop->getAAInfo().NoAliasAddrSpace;
4755 return !MD || !AMDGPU::hasValueInRangeLikeMetadata(
4756 *MD, AMDGPUAS::PRIVATE_ADDRESS);
4771 if (
MI.memoperands_empty())
4780 unsigned AS = Memop->getAddrSpace();
4790 bool TgSplit)
const {
4803 if (
MI.memoperands_empty())
4808 unsigned AS = Memop->getAddrSpace();
4824 unsigned Opcode =
MI.getOpcode();
4839 if (Opcode == AMDGPU::S_SENDMSG || Opcode == AMDGPU::S_SENDMSGHALT ||
4840 isEXP(Opcode) || Opcode == AMDGPU::DS_ORDERED_COUNT ||
4841 Opcode == AMDGPU::S_TRAP || Opcode == AMDGPU::S_WAIT_EVENT ||
4842 Opcode == AMDGPU::S_SETHALT)
4845 if (
MI.isCall() ||
MI.isInlineAsm())
4851 if (ST.hasVPermPk16Hazard() &&
isVPermPk16(Opcode))
4867 if (Opcode == AMDGPU::V_READFIRSTLANE_B32 ||
4868 Opcode == AMDGPU::V_READLANE_B32 || Opcode == AMDGPU::V_WRITELANE_B32 ||
4869 Opcode == AMDGPU::SI_RESTORE_S32_FROM_VGPR ||
4870 Opcode == AMDGPU::SI_SPILL_S32_TO_VGPR)
4878 if (
MI.isMetaInstruction())
4882 if (
MI.isCopyLike()) {
4883 if (!RI.isSGPRReg(MRI,
MI.getOperand(0).getReg()))
4887 return MI.readsRegister(AMDGPU::EXEC, &RI);
4898 return !
isSALU(
MI) ||
MI.readsRegister(AMDGPU::EXEC, &RI);
4902 switch (
Imm.getBitWidth()) {
4908 ST.hasInv2PiInlineImm());
4911 ST.hasInv2PiInlineImm());
4913 return ST.has16BitInsts() &&
4915 ST.hasInv2PiInlineImm());
4922 APInt IntImm =
Imm.bitcastToAPInt();
4924 bool HasInv2Pi = ST.hasInv2PiInlineImm();
4932 return ST.has16BitInsts() &&
4935 return ST.has16BitInsts() &&
4945 switch (OperandType) {
4955 int32_t Trunc =
static_cast<int32_t
>(
Imm);
4999 int16_t Trunc =
static_cast<int16_t
>(
Imm);
5000 return ST.has16BitInsts() &&
5009 int16_t Trunc =
static_cast<int16_t
>(
Imm);
5010 return ST.has16BitInsts() &&
5062 if (!RI.opCanUseLiteralConstant(OpInfo.OperandType))
5068 return ST.hasVOP3Literal();
5072 int64_t ImmVal)
const {
5074 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
5075 if (Src1Idx != -1 &&
isDPP(
Opc) && !ST.hasDPPSrc1SGPR() &&
5076 OpNo ==
static_cast<unsigned>(Src1Idx))
5081 if (
isMAI(InstDesc) && ST.hasMFMAInlineLiteralBug() &&
5082 OpNo == (
unsigned)AMDGPU::getNamedOperandIdx(InstDesc.
getOpcode(),
5083 AMDGPU::OpName::src2))
5086 if (ST.hasBF16InlineConstFromUpperFP32() &&
isVOP1(
Opc)) {
5093 return RI.opCanUseInlineConstant(OpInfo.OperandType);
5105 "unexpected imm-like operand kind");
5118 if (Opcode == AMDGPU::V_MUL_LEGACY_F32_e64 && ST.hasGFX90AInsts())
5138 return Op32 != -1 &&
TII.isVOPC(Op32);
5143 unsigned Depth)
const {
5144 assert(MRI.
isSSA() &&
"isMaskedByExec requires SSA form");
5153 constexpr unsigned MaxDepth = 6;
5154 if (
Depth >= MaxDepth || !Reg.isVirtual())
5160 if (!Def || Def->getParent() !=
MBB)
5168 auto Recurse = [&](
unsigned OpIdx) {
5174 unsigned Opc = Def->getOpcode();
5175 if (
Opc == AMDGPU::COPY && Recurse(1))
5177 if (
Opc == LMC.
AndOpc && (Recurse(1) || Recurse(2)))
5197 AMDGPU::OpName
OpName)
const {
5199 return Mods && Mods->
getImm();
5212 switch (
MI.getOpcode()) {
5213 default:
return false;
5215 case AMDGPU::V_ADDC_U32_e64:
5216 case AMDGPU::V_SUBB_U32_e64:
5217 case AMDGPU::V_SUBBREV_U32_e64: {
5220 if (!Src1->
isReg() || !RI.isVGPR(MRI, Src1->
getReg()))
5225 case AMDGPU::V_MAC_F16_e64:
5226 case AMDGPU::V_MAC_F32_e64:
5227 case AMDGPU::V_MAC_LEGACY_F32_e64:
5228 case AMDGPU::V_FMAC_F16_e64:
5229 case AMDGPU::V_FMAC_F16_t16_e64:
5230 case AMDGPU::V_FMAC_F16_fake16_e64:
5231 case AMDGPU::V_FMAC_F32_e64:
5232 case AMDGPU::V_FMAC_F64_e64:
5233 case AMDGPU::V_FMAC_LEGACY_F32_e64:
5234 if (!Src2->
isReg() || !RI.isVGPR(MRI, Src2->
getReg()) ||
5239 case AMDGPU::V_CNDMASK_B32_e64:
5245 if (Src1 && (!Src1->
isReg() || !RI.isVGPR(MRI, Src1->
getReg()) ||
5258 if (Src0 && Src0->
isImm()) {
5261 get(Op32), AMDGPU::getNamedOperandIdx(Op32, AMDGPU::OpName::src0),
5283 (
Use.getReg() == AMDGPU::VCC ||
Use.getReg() == AMDGPU::VCC_LO)) {
5292 unsigned Op32)
const {
5306 Inst32.
add(
MI.getOperand(
I));
5310 int Idx =
MI.getNumExplicitDefs();
5312 int OpTy =
MI.getDesc().operands()[Idx++].OperandType;
5317 if (AMDGPU::getNamedOperandIdx(Op32, AMDGPU::OpName::src2) == -1) {
5337 if (OldSDst && OldSDst->
isDead()) {
5340 NewVCC->setIsDead();
5349 if (Reg == AMDGPU::SGPR_NULL || Reg == AMDGPU::SGPR_NULL64)
5357 return Reg == AMDGPU::VCC || Reg == AMDGPU::VCC_LO || Reg == AMDGPU::M0;
5360 return AMDGPU::SReg_32RegClass.contains(Reg) ||
5361 AMDGPU::SReg_64RegClass.contains(Reg);
5389 switch (MO.getReg()) {
5391 case AMDGPU::VCC_LO:
5392 case AMDGPU::VCC_HI:
5394 case AMDGPU::FLAT_SCR:
5407 switch (
MI.getOpcode()) {
5408 case AMDGPU::V_READLANE_B32:
5409 case AMDGPU::SI_RESTORE_S32_FROM_VGPR:
5410 case AMDGPU::V_WRITELANE_B32:
5411 case AMDGPU::SI_SPILL_S32_TO_VGPR:
5418 if (
MI.isPreISelOpcode() ||
5419 SIInstrInfo::isGenericOpcode(
MI.getOpcode()) ||
5437 return SubReg.
getSubReg() != AMDGPU::NoSubRegister &&
5448 if (RI.isVectorRegister(MRI, SrcReg) && RI.isSGPRReg(MRI, DstReg)) {
5449 ErrInfo =
"illegal copy from vector register to SGPR";
5467 if (!MRI.
isSSA() &&
MI.isCopy())
5468 return verifyCopy(
MI, MRI, ErrInfo);
5470 if (SIInstrInfo::isGenericOpcode(Opcode))
5473 int Src0Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0);
5474 int Src1Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src1);
5475 int Src2Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src2);
5477 if (Src0Idx == -1) {
5479 Src0Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0X);
5480 Src1Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vsrc1X);
5481 Src2Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0Y);
5482 Src3Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vsrc1Y);
5487 if (!
Desc.isVariadic() &&
5488 Desc.getNumOperands() !=
MI.getNumExplicitOperands()) {
5489 ErrInfo =
"Instruction has wrong number of operands.";
5493 if (
MI.isInlineAsm()) {
5506 if (!Reg.isVirtual() && !RC->
contains(Reg)) {
5507 ErrInfo =
"inlineasm operand has incorrect register class.";
5515 if (
isImage(
MI) &&
MI.memoperands_empty() &&
MI.mayLoadOrStore()) {
5516 ErrInfo =
"missing memory operand from image instruction.";
5521 for (
int i = 0, e =
Desc.getNumOperands(); i != e; ++i) {
5524 ErrInfo =
"FPImm Machine Operands are not supported. ISel should bitcast "
5525 "all fp values to integers.";
5531 switch (OpInfo.OperandType) {
5533 if (
MI.getOperand(i).isImm() ||
MI.getOperand(i).isGlobal()) {
5534 ErrInfo =
"Illegal immediate value for operand.";
5566 ErrInfo =
"Illegal immediate value for operand.";
5575 if (ST.has64BitLiterals() &&
Desc.getSize() != 4 && MO.
isImm() &&
5578 OpInfo.OperandType ==
5580 ErrInfo =
"illegal 64-bit immediate value for operand.";
5587 ErrInfo =
"Expected inline constant for operand.";
5601 if (!
MI.getOperand(i).isImm() && !
MI.getOperand(i).isFI()) {
5602 ErrInfo =
"Expected immediate, but got non-immediate";
5611 if (OpInfo.isGenericType())
5619 if (!ST.hasSDWA()) {
5620 ErrInfo =
"SDWA is not supported on this target";
5624 for (
auto Op : {AMDGPU::OpName::src0_sel, AMDGPU::OpName::src1_sel,
5625 AMDGPU::OpName::dst_sel}) {
5631 ErrInfo =
"Invalid SDWA selection";
5636 int DstIdx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vdst);
5638 for (
int OpIdx : {DstIdx, Src0Idx, Src1Idx, Src2Idx}) {
5643 if (!ST.hasSDWAScalar()) {
5645 if (!MO.
isReg() || !RI.hasVGPRs(RI.getRegClassForReg(MRI, MO.
getReg()))) {
5646 ErrInfo =
"Only VGPRs allowed as operands in SDWA instructions on VI";
5653 "Only reg allowed as operands in SDWA instructions on GFX9+";
5659 if (!ST.hasSDWAOmod()) {
5662 if (OMod !=
nullptr &&
5664 ErrInfo =
"OMod not allowed in SDWA instructions on VI";
5669 if (Opcode == AMDGPU::V_CVT_F32_FP8_sdwa ||
5670 Opcode == AMDGPU::V_CVT_F32_BF8_sdwa ||
5671 Opcode == AMDGPU::V_CVT_PK_F32_FP8_sdwa ||
5672 Opcode == AMDGPU::V_CVT_PK_F32_BF8_sdwa) {
5675 unsigned Mods = Src0ModsMO->
getImm();
5678 ErrInfo =
"sext, abs and neg are not allowed on this instruction";
5684 if (
isVOPC(BasicOpcode)) {
5685 if (!ST.hasSDWASdst() && DstIdx != -1) {
5688 if (!Dst.isReg() || Dst.getReg() != AMDGPU::VCC) {
5689 ErrInfo =
"Only VCC allowed as dst in SDWA instructions on VI";
5692 }
else if (!ST.hasSDWAOutModsVOPC()) {
5695 if (Clamp && (!Clamp->
isImm() || Clamp->
getImm() != 0)) {
5696 ErrInfo =
"Clamp not allowed in VOPC SDWA instructions on VI";
5702 if (OMod && (!OMod->
isImm() || OMod->
getImm() != 0)) {
5703 ErrInfo =
"OMod not allowed in VOPC SDWA instructions on VI";
5710 if (DstUnused && DstUnused->isImm() &&
5713 if (!Dst.isReg() || !Dst.isTied()) {
5714 ErrInfo =
"Dst register should have tied register";
5719 MI.getOperand(
MI.findTiedOperandIdx(DstIdx));
5722 "Dst register should be tied to implicit use of preserved register";
5726 ErrInfo =
"Dst register should use same physical register as preserved";
5732 if (
isDPP(
MI) && !ST.hasDPPSrc1SGPR() && Src1Idx != -1) {
5734 if (Src1MO.
isReg() && RI.isSGPRReg(MRI, Src1MO.
getReg())) {
5735 ErrInfo =
"DPP src1 cannot be SGPR on this subtarget";
5738 if (Src1MO.
isImm()) {
5739 ErrInfo =
"DPP src1 cannot be an immediate on this subtarget";
5745 if (
isImage(Opcode) && !
MI.mayStore()) {
5750 uint64_t DMaskImm = DMask->
getImm();
5757 if (D16 && D16->getImm() && !ST.hasUnpackedD16VMem())
5765 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vdata);
5769 uint32_t DstSize = RI.getRegSizeInBits(*DstRC) / 32;
5770 if (RegCount > DstSize) {
5771 ErrInfo =
"Image instruction returns too many registers for dst "
5781 Desc.getOpcode() != AMDGPU::V_WRITELANE_B32) {
5782 unsigned ConstantBusCount = 0;
5783 bool UsesLiteral =
false;
5786 int ImmIdx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::imm);
5790 LiteralVal = &
MI.getOperand(ImmIdx);
5799 for (
int OpIdx : {Src0Idx, Src1Idx, Src2Idx, Src3Idx}) {
5810 }
else if (!MO.
isFI()) {
5817 ErrInfo =
"VOP2/VOP3 instruction uses more than one literal";
5827 if (
llvm::all_of(SGPRsUsed, [
this, SGPRUsed](
unsigned SGPR) {
5828 return !RI.regsOverlap(SGPRUsed, SGPR);
5837 if (ConstantBusCount > ST.getConstantBusLimit(Opcode) &&
5838 Opcode != AMDGPU::V_WRITELANE_B32) {
5839 ErrInfo =
"VOP* instruction violates constant bus restriction";
5843 if (
isVOP3(
MI) && UsesLiteral && !ST.hasVOP3Literal()) {
5844 ErrInfo =
"VOP3 instruction uses literal";
5851 if (
Desc.getOpcode() == AMDGPU::V_WRITELANE_B32) {
5852 unsigned SGPRCount = 0;
5855 for (
int OpIdx : {Src0Idx, Src1Idx}) {
5863 if (MO.
getReg() != SGPRUsed)
5868 if (SGPRCount > ST.getConstantBusLimit(Opcode)) {
5869 ErrInfo =
"WRITELANE instruction violates constant bus restriction";
5876 if (
Desc.getOpcode() == AMDGPU::V_DIV_SCALE_F32_e64 ||
5877 Desc.getOpcode() == AMDGPU::V_DIV_SCALE_F64_e64) {
5884 ErrInfo =
"v_div_scale_{f32|f64} require src0 = src1 or src2";
5894 ErrInfo =
"ABS not allowed in VOP3B instructions";
5907 ErrInfo =
"SOP2/SOPC instruction requires too many immediate constants";
5914 if (
Desc.isBranch()) {
5916 ErrInfo =
"invalid branch target for SOPK instruction";
5920 uint64_t
Imm =
Op->getImm();
5923 ErrInfo =
"invalid immediate for SOPK instruction";
5928 ErrInfo =
"invalid immediate for SOPK instruction";
5935 if (
Desc.getOpcode() == AMDGPU::V_MOVRELS_B32_e32 ||
5936 Desc.getOpcode() == AMDGPU::V_MOVRELS_B32_e64 ||
5937 Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e32 ||
5938 Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e64) {
5939 const bool IsDst =
Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e32 ||
5940 Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e64;
5942 const unsigned StaticNumOps =
5943 Desc.getNumOperands() +
Desc.implicit_uses().size();
5944 const unsigned NumImplicitOps = IsDst ? 2 : 1;
5950 if (
MI.getNumOperands() < StaticNumOps + NumImplicitOps) {
5951 ErrInfo =
"missing implicit register operands";
5957 if (!Dst->isUse()) {
5958 ErrInfo =
"v_movreld_b32 vdst should be a use operand";
5963 if (!
MI.isRegTiedToUseOperand(StaticNumOps, &UseOpIdx) ||
5964 UseOpIdx != StaticNumOps + 1) {
5965 ErrInfo =
"movrel implicit operands should be tied";
5972 =
MI.getOperand(StaticNumOps + NumImplicitOps - 1);
5974 !
isSubRegOf(RI, ImpUse, IsDst ? *Dst : Src0)) {
5975 ErrInfo =
"src0 should be subreg of implicit vector use";
5983 if (!
MI.hasRegisterImplicitUseOperand(AMDGPU::EXEC)) {
5984 ErrInfo =
"VALU instruction does not implicitly read exec mask";
5990 if (
MI.mayStore() &&
5995 if (Soff && Soff->
getReg() != AMDGPU::M0) {
5996 ErrInfo =
"scalar stores must use m0 as offset register";
6002 if (
isFLAT(
MI) && !ST.hasFlatInstOffsets()) {
6004 if (
Offset->getImm() != 0) {
6005 ErrInfo =
"subtarget does not support offsets in flat instructions";
6010 if (
isDS(
MI) && !ST.hasGDS()) {
6012 if (GDSOp && GDSOp->
getImm() != 0) {
6013 ErrInfo =
"GDS is not supported on this subtarget";
6021 int VAddr0Idx = AMDGPU::getNamedOperandIdx(Opcode,
6022 AMDGPU::OpName::vaddr0);
6023 AMDGPU::OpName RSrcOpName =
6024 isMIMG(
MI) ? AMDGPU::OpName::srsrc : AMDGPU::OpName::rsrc;
6025 int RsrcIdx = AMDGPU::getNamedOperandIdx(Opcode, RSrcOpName);
6033 ErrInfo =
"dim is out of range";
6038 if (ST.hasR128A16()) {
6040 IsA16 = R128A16->
getImm() != 0;
6041 }
else if (ST.hasA16()) {
6043 IsA16 = A16->
getImm() != 0;
6046 bool IsNSA = RsrcIdx - VAddr0Idx > 1;
6048 unsigned AddrWords =
6051 unsigned VAddrWords;
6053 VAddrWords = RsrcIdx - VAddr0Idx;
6054 if (ST.hasPartialNSAEncoding() &&
6056 unsigned LastVAddrIdx = RsrcIdx - 1;
6057 VAddrWords +=
getOpSize(
MI, LastVAddrIdx) / 4 - 1;
6065 if (VAddrWords != AddrWords) {
6067 <<
" but got " << VAddrWords <<
"\n");
6068 ErrInfo =
"bad vaddr size";
6078 unsigned DC = DppCt->
getImm();
6079 if (DC == DppCtrl::DPP_UNUSED1 || DC == DppCtrl::DPP_UNUSED2 ||
6080 DC == DppCtrl::DPP_UNUSED3 || DC > DppCtrl::DPP_LAST ||
6081 (DC >= DppCtrl::DPP_UNUSED4_FIRST && DC <= DppCtrl::DPP_UNUSED4_LAST) ||
6082 (DC >= DppCtrl::DPP_UNUSED5_FIRST && DC <= DppCtrl::DPP_UNUSED5_LAST) ||
6083 (DC >= DppCtrl::DPP_UNUSED6_FIRST && DC <= DppCtrl::DPP_UNUSED6_LAST) ||
6084 (DC >= DppCtrl::DPP_UNUSED7_FIRST && DC <= DppCtrl::DPP_UNUSED7_LAST) ||
6085 (DC >= DppCtrl::DPP_UNUSED8_FIRST && DC <= DppCtrl::DPP_UNUSED8_LAST)) {
6086 ErrInfo =
"Invalid dpp_ctrl value";
6089 if (DC >= DppCtrl::WAVE_SHL1 && DC <= DppCtrl::WAVE_ROR1 &&
6090 !ST.hasDPPWavefrontShifts()) {
6091 ErrInfo =
"Invalid dpp_ctrl value: "
6092 "wavefront shifts are not supported on GFX10+";
6095 if (DC >= DppCtrl::BCAST15 && DC <= DppCtrl::BCAST31 &&
6096 !ST.hasDPPBroadcasts()) {
6097 ErrInfo =
"Invalid dpp_ctrl value: "
6098 "broadcasts are not supported on GFX10+";
6101 if (DC >= DppCtrl::ROW_SHARE_FIRST && DC <= DppCtrl::ROW_XMASK_LAST &&
6103 if (DC >= DppCtrl::ROW_NEWBCAST_FIRST &&
6104 DC <= DppCtrl::ROW_NEWBCAST_LAST &&
6105 !ST.hasGFX90AInsts()) {
6106 ErrInfo =
"Invalid dpp_ctrl value: "
6107 "row_newbroadcast/row_share is not supported before "
6111 if (DC > DppCtrl::ROW_NEWBCAST_LAST || !ST.hasGFX90AInsts()) {
6112 ErrInfo =
"Invalid dpp_ctrl value: "
6113 "row_share and row_xmask are not supported before GFX10";
6118 if (Opcode != AMDGPU::V_MOV_B64_DPP_PSEUDO &&
6120 ST.hasFeature(AMDGPU::FeatureDPALU_DPP) &&
6122 ErrInfo =
"Invalid dpp_ctrl value: "
6123 "DP ALU dpp only support row_newbcast";
6130 AMDGPU::OpName DataName =
6131 isDS(Opcode) ? AMDGPU::OpName::data0 : AMDGPU::OpName::vdata;
6137 if (!ST.hasGFX90AInsts()) {
6138 if ((Dst && RI.isAGPR(MRI, Dst->getReg())) ||
6139 (
Data && RI.isAGPR(MRI,
Data->getReg())) ||
6140 (Data2 && RI.isAGPR(MRI, Data2->
getReg()))) {
6141 ErrInfo =
"Invalid register class: "
6142 "agpr loads and stores not supported on this GPU";
6148 if (ST.needsAlignedVGPRs()) {
6149 const auto isAlignedReg = [&
MI, &MRI,
this](AMDGPU::OpName
OpName) ->
bool {
6154 if (Reg.isPhysical())
6155 return !(RI.getHWRegIndex(Reg) & 1);
6157 return RI.getRegSizeInBits(RC) > 32 && RI.isProperlyAlignedRC(RC) &&
6158 !(RI.getChannelFromSubReg(
Op->getSubReg()) & 1);
6162 if (!isAlignedReg(AMDGPU::OpName::vaddr)) {
6163 ErrInfo =
"Subtarget requires even aligned vector registers "
6164 "for vaddr operand of image instructions";
6170 if (Opcode == AMDGPU::V_ACCVGPR_WRITE_B32_e64 && !ST.hasGFX90AInsts()) {
6172 if (Src->isReg() && RI.isSGPRReg(MRI, Src->getReg())) {
6173 ErrInfo =
"Invalid register class: "
6174 "v_accvgpr_write with an SGPR is not supported on this GPU";
6179 if (
Desc.getOpcode() == AMDGPU::G_AMDGPU_WAVE_ADDRESS) {
6182 ErrInfo =
"pseudo expects only physical SGPRs";
6189 if (!ST.hasScaleOffset()) {
6190 ErrInfo =
"Subtarget does not support offset scaling";
6194 ErrInfo =
"Instruction does not support offset scaling";
6202 for (
unsigned I = 0;
I < 3; ++
I) {
6208 if (ST.hasFlatScratchHiInB64InstHazard() &&
isSALU(
MI) &&
6209 MI.readsRegister(AMDGPU::SRC_FLAT_SCRATCH_BASE_HI,
nullptr)) {
6211 if ((Dst && RI.getRegClassForReg(MRI, Dst->getReg()) ==
6212 &AMDGPU::SReg_64RegClass) ||
6213 Opcode == AMDGPU::S_BITCMP0_B64 || Opcode == AMDGPU::S_BITCMP1_B64) {
6214 ErrInfo =
"Instruction cannot read flat_scratch_base_hi";
6223 if (
MI.getOpcode() == AMDGPU::S_MOV_B32) {
6225 return MI.getOperand(1).isReg() || RI.isAGPR(MRI,
MI.getOperand(0).getReg())
6227 : AMDGPU::V_MOV_B32_e32;
6237 default:
return AMDGPU::INSTRUCTION_LIST_END;
6238 case AMDGPU::REG_SEQUENCE:
return AMDGPU::REG_SEQUENCE;
6239 case AMDGPU::COPY:
return AMDGPU::COPY;
6240 case AMDGPU::PHI:
return AMDGPU::PHI;
6241 case AMDGPU::INSERT_SUBREG:
return AMDGPU::INSERT_SUBREG;
6242 case AMDGPU::WQM:
return AMDGPU::WQM;
6243 case AMDGPU::SOFT_WQM:
return AMDGPU::SOFT_WQM;
6244 case AMDGPU::STRICT_WWM:
return AMDGPU::STRICT_WWM;
6245 case AMDGPU::STRICT_WQM:
return AMDGPU::STRICT_WQM;
6246 case AMDGPU::S_ADD_I32:
6247 return ST.hasAddNoCarryInsts() ? AMDGPU::V_ADD_U32_e64 : AMDGPU::V_ADD_CO_U32_e32;
6248 case AMDGPU::S_ADDC_U32:
6249 return AMDGPU::V_ADDC_U32_e32;
6250 case AMDGPU::S_SUB_I32:
6251 return ST.hasAddNoCarryInsts() ? AMDGPU::V_SUB_U32_e64 : AMDGPU::V_SUB_CO_U32_e32;
6254 case AMDGPU::S_ADD_U32:
6255 return AMDGPU::V_ADD_CO_U32_e32;
6256 case AMDGPU::S_SUB_U32:
6257 return AMDGPU::V_SUB_CO_U32_e32;
6258 case AMDGPU::S_ADD_U64_PSEUDO:
6259 return AMDGPU::V_ADD_U64_PSEUDO;
6260 case AMDGPU::S_SUB_U64_PSEUDO:
6261 return AMDGPU::V_SUB_U64_PSEUDO;
6262 case AMDGPU::S_SUBB_U32:
return AMDGPU::V_SUBB_U32_e32;
6263 case AMDGPU::S_MUL_I32:
return AMDGPU::V_MUL_LO_U32_e64;
6264 case AMDGPU::S_MUL_HI_U32:
return AMDGPU::V_MUL_HI_U32_e64;
6265 case AMDGPU::S_MUL_HI_I32:
return AMDGPU::V_MUL_HI_I32_e64;
6266 case AMDGPU::S_AND_B32:
return AMDGPU::V_AND_B32_e64;
6267 case AMDGPU::S_OR_B32:
return AMDGPU::V_OR_B32_e64;
6268 case AMDGPU::S_XOR_B32:
return AMDGPU::V_XOR_B32_e64;
6269 case AMDGPU::S_XNOR_B32:
6270 return ST.hasDLInsts() ? AMDGPU::V_XNOR_B32_e64 : AMDGPU::INSTRUCTION_LIST_END;
6271 case AMDGPU::S_MIN_I32:
return AMDGPU::V_MIN_I32_e64;
6272 case AMDGPU::S_MIN_U32:
return AMDGPU::V_MIN_U32_e64;
6273 case AMDGPU::S_MAX_I32:
return AMDGPU::V_MAX_I32_e64;
6274 case AMDGPU::S_MAX_U32:
return AMDGPU::V_MAX_U32_e64;
6275 case AMDGPU::S_ASHR_I32:
return AMDGPU::V_ASHR_I32_e32;
6276 case AMDGPU::S_ASHR_I64:
return AMDGPU::V_ASHR_I64_e64;
6277 case AMDGPU::S_LSHL_B32:
return AMDGPU::V_LSHL_B32_e32;
6278 case AMDGPU::S_LSHL_B64:
return AMDGPU::V_LSHL_B64_e64;
6279 case AMDGPU::S_LSHR_B32:
return AMDGPU::V_LSHR_B32_e32;
6280 case AMDGPU::S_LSHR_B64:
return AMDGPU::V_LSHR_B64_e64;
6281 case AMDGPU::S_SEXT_I32_I8:
return AMDGPU::V_BFE_I32_e64;
6282 case AMDGPU::S_SEXT_I32_I16:
return AMDGPU::V_BFE_I32_e64;
6283 case AMDGPU::S_BFE_U32:
return AMDGPU::V_BFE_U32_e64;
6284 case AMDGPU::S_BFE_I32:
return AMDGPU::V_BFE_I32_e64;
6285 case AMDGPU::S_BFM_B32:
return AMDGPU::V_BFM_B32_e64;
6286 case AMDGPU::S_BREV_B32:
return AMDGPU::V_BFREV_B32_e32;
6287 case AMDGPU::S_NOT_B32:
return AMDGPU::V_NOT_B32_e32;
6288 case AMDGPU::S_NOT_B64:
return AMDGPU::V_NOT_B32_e32;
6289 case AMDGPU::S_CMP_EQ_I32:
return AMDGPU::V_CMP_EQ_I32_e64;
6290 case AMDGPU::S_CMP_LG_I32:
return AMDGPU::V_CMP_NE_I32_e64;
6291 case AMDGPU::S_CMP_GT_I32:
return AMDGPU::V_CMP_GT_I32_e64;
6292 case AMDGPU::S_CMP_GE_I32:
return AMDGPU::V_CMP_GE_I32_e64;
6293 case AMDGPU::S_CMP_LT_I32:
return AMDGPU::V_CMP_LT_I32_e64;
6294 case AMDGPU::S_CMP_LE_I32:
return AMDGPU::V_CMP_LE_I32_e64;
6295 case AMDGPU::S_CMP_EQ_U32:
return AMDGPU::V_CMP_EQ_U32_e64;
6296 case AMDGPU::S_CMP_LG_U32:
return AMDGPU::V_CMP_NE_U32_e64;
6297 case AMDGPU::S_CMP_GT_U32:
return AMDGPU::V_CMP_GT_U32_e64;
6298 case AMDGPU::S_CMP_GE_U32:
return AMDGPU::V_CMP_GE_U32_e64;
6299 case AMDGPU::S_CMP_LT_U32:
return AMDGPU::V_CMP_LT_U32_e64;
6300 case AMDGPU::S_CMP_LE_U32:
return AMDGPU::V_CMP_LE_U32_e64;
6301 case AMDGPU::S_CMP_EQ_U64:
return AMDGPU::V_CMP_EQ_U64_e64;
6302 case AMDGPU::S_CMP_LG_U64:
return AMDGPU::V_CMP_NE_U64_e64;
6303 case AMDGPU::S_BCNT1_I32_B32:
return AMDGPU::V_BCNT_U32_B32_e64;
6304 case AMDGPU::S_FF1_I32_B32:
return AMDGPU::V_FFBL_B32_e32;
6305 case AMDGPU::S_FLBIT_I32_B32:
return AMDGPU::V_FFBH_U32_e32;
6306 case AMDGPU::S_FLBIT_I32:
return AMDGPU::V_FFBH_I32_e64;
6307 case AMDGPU::S_CBRANCH_SCC0:
return AMDGPU::S_CBRANCH_VCCZ;
6308 case AMDGPU::S_CBRANCH_SCC1:
return AMDGPU::S_CBRANCH_VCCNZ;
6309 case AMDGPU::S_CVT_F32_I32:
return AMDGPU::V_CVT_F32_I32_e64;
6310 case AMDGPU::S_CVT_F32_U32:
return AMDGPU::V_CVT_F32_U32_e64;
6311 case AMDGPU::S_CVT_I32_F32:
return AMDGPU::V_CVT_I32_F32_e64;
6312 case AMDGPU::S_CVT_U32_F32:
return AMDGPU::V_CVT_U32_F32_e64;
6313 case AMDGPU::S_CVT_F32_F16:
6314 case AMDGPU::S_CVT_HI_F32_F16:
6315 return ST.useRealTrue16Insts() ? AMDGPU::V_CVT_F32_F16_t16_e64
6316 : AMDGPU::V_CVT_F32_F16_fake16_e64;
6317 case AMDGPU::S_CVT_F16_F32:
6318 return ST.useRealTrue16Insts() ? AMDGPU::V_CVT_F16_F32_t16_e64
6319 : AMDGPU::V_CVT_F16_F32_fake16_e64;
6320 case AMDGPU::S_CEIL_F32:
return AMDGPU::V_CEIL_F32_e64;
6321 case AMDGPU::S_FLOOR_F32:
return AMDGPU::V_FLOOR_F32_e64;
6322 case AMDGPU::S_TRUNC_F32:
return AMDGPU::V_TRUNC_F32_e64;
6323 case AMDGPU::S_RNDNE_F32:
return AMDGPU::V_RNDNE_F32_e64;
6324 case AMDGPU::S_CEIL_F16:
6325 return ST.useRealTrue16Insts() ? AMDGPU::V_CEIL_F16_t16_e64
6326 : AMDGPU::V_CEIL_F16_fake16_e64;
6327 case AMDGPU::S_FLOOR_F16:
6328 return ST.useRealTrue16Insts() ? AMDGPU::V_FLOOR_F16_t16_e64
6329 : AMDGPU::V_FLOOR_F16_fake16_e64;
6330 case AMDGPU::S_TRUNC_F16:
6331 return ST.useRealTrue16Insts() ? AMDGPU::V_TRUNC_F16_t16_e64
6332 : AMDGPU::V_TRUNC_F16_fake16_e64;
6333 case AMDGPU::S_RNDNE_F16:
6334 return ST.useRealTrue16Insts() ? AMDGPU::V_RNDNE_F16_t16_e64
6335 : AMDGPU::V_RNDNE_F16_fake16_e64;
6336 case AMDGPU::S_ADD_F32:
return AMDGPU::V_ADD_F32_e64;
6337 case AMDGPU::S_SUB_F32:
return AMDGPU::V_SUB_F32_e64;
6338 case AMDGPU::S_MIN_F32:
return AMDGPU::V_MIN_F32_e64;
6339 case AMDGPU::S_MAX_F32:
return AMDGPU::V_MAX_F32_e64;
6340 case AMDGPU::S_MINIMUM_F32:
return AMDGPU::V_MINIMUM_F32_e64;
6341 case AMDGPU::S_MAXIMUM_F32:
return AMDGPU::V_MAXIMUM_F32_e64;
6342 case AMDGPU::S_MUL_F32:
return AMDGPU::V_MUL_F32_e64;
6343 case AMDGPU::S_ADD_F16:
6344 return ST.useRealTrue16Insts() ? AMDGPU::V_ADD_F16_t16_e64
6345 : AMDGPU::V_ADD_F16_fake16_e64;
6346 case AMDGPU::S_SUB_F16:
6347 return ST.useRealTrue16Insts() ? AMDGPU::V_SUB_F16_t16_e64
6348 : AMDGPU::V_SUB_F16_fake16_e64;
6349 case AMDGPU::S_MIN_F16:
6350 return ST.useRealTrue16Insts() ? AMDGPU::V_MIN_F16_t16_e64
6351 : AMDGPU::V_MIN_F16_fake16_e64;
6352 case AMDGPU::S_MAX_F16:
6353 return ST.useRealTrue16Insts() ? AMDGPU::V_MAX_F16_t16_e64
6354 : AMDGPU::V_MAX_F16_fake16_e64;
6355 case AMDGPU::S_MINIMUM_F16:
6356 return ST.useRealTrue16Insts() ? AMDGPU::V_MINIMUM_F16_t16_e64
6357 : AMDGPU::V_MINIMUM_F16_fake16_e64;
6358 case AMDGPU::S_MAXIMUM_F16:
6359 return ST.useRealTrue16Insts() ? AMDGPU::V_MAXIMUM_F16_t16_e64
6360 : AMDGPU::V_MAXIMUM_F16_fake16_e64;
6361 case AMDGPU::S_MUL_F16:
6362 return ST.useRealTrue16Insts() ? AMDGPU::V_MUL_F16_t16_e64
6363 : AMDGPU::V_MUL_F16_fake16_e64;
6364 case AMDGPU::S_CVT_PK_RTZ_F16_F32:
return AMDGPU::V_CVT_PKRTZ_F16_F32_e64;
6365 case AMDGPU::S_FMAC_F32:
return AMDGPU::V_FMAC_F32_e64;
6366 case AMDGPU::S_FMAC_F16:
6367 return ST.useRealTrue16Insts() ? AMDGPU::V_FMAC_F16_t16_e64
6368 : AMDGPU::V_FMAC_F16_fake16_e64;
6369 case AMDGPU::S_FMAMK_F32:
return AMDGPU::V_FMAMK_F32;
6370 case AMDGPU::S_FMAAK_F32:
return AMDGPU::V_FMAAK_F32;
6371 case AMDGPU::S_CMP_LT_F32:
return AMDGPU::V_CMP_LT_F32_e64;
6372 case AMDGPU::S_CMP_EQ_F32:
return AMDGPU::V_CMP_EQ_F32_e64;
6373 case AMDGPU::S_CMP_LE_F32:
return AMDGPU::V_CMP_LE_F32_e64;
6374 case AMDGPU::S_CMP_GT_F32:
return AMDGPU::V_CMP_GT_F32_e64;
6375 case AMDGPU::S_CMP_LG_F32:
return AMDGPU::V_CMP_LG_F32_e64;
6376 case AMDGPU::S_CMP_GE_F32:
return AMDGPU::V_CMP_GE_F32_e64;
6377 case AMDGPU::S_CMP_O_F32:
return AMDGPU::V_CMP_O_F32_e64;
6378 case AMDGPU::S_CMP_U_F32:
return AMDGPU::V_CMP_U_F32_e64;
6379 case AMDGPU::S_CMP_NGE_F32:
return AMDGPU::V_CMP_NGE_F32_e64;
6380 case AMDGPU::S_CMP_NLG_F32:
return AMDGPU::V_CMP_NLG_F32_e64;
6381 case AMDGPU::S_CMP_NGT_F32:
return AMDGPU::V_CMP_NGT_F32_e64;
6382 case AMDGPU::S_CMP_NLE_F32:
return AMDGPU::V_CMP_NLE_F32_e64;
6383 case AMDGPU::S_CMP_NEQ_F32:
return AMDGPU::V_CMP_NEQ_F32_e64;
6384 case AMDGPU::S_CMP_NLT_F32:
return AMDGPU::V_CMP_NLT_F32_e64;
6385 case AMDGPU::S_CMP_LT_F16:
6386 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_LT_F16_t16_e64
6387 : AMDGPU::V_CMP_LT_F16_fake16_e64;
6388 case AMDGPU::S_CMP_EQ_F16:
6389 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_EQ_F16_t16_e64
6390 : AMDGPU::V_CMP_EQ_F16_fake16_e64;
6391 case AMDGPU::S_CMP_LE_F16:
6392 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_LE_F16_t16_e64
6393 : AMDGPU::V_CMP_LE_F16_fake16_e64;
6394 case AMDGPU::S_CMP_GT_F16:
6395 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_GT_F16_t16_e64
6396 : AMDGPU::V_CMP_GT_F16_fake16_e64;
6397 case AMDGPU::S_CMP_LG_F16:
6398 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_LG_F16_t16_e64
6399 : AMDGPU::V_CMP_LG_F16_fake16_e64;
6400 case AMDGPU::S_CMP_GE_F16:
6401 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_GE_F16_t16_e64
6402 : AMDGPU::V_CMP_GE_F16_fake16_e64;
6403 case AMDGPU::S_CMP_O_F16:
6404 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_O_F16_t16_e64
6405 : AMDGPU::V_CMP_O_F16_fake16_e64;
6406 case AMDGPU::S_CMP_U_F16:
6407 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_U_F16_t16_e64
6408 : AMDGPU::V_CMP_U_F16_fake16_e64;
6409 case AMDGPU::S_CMP_NGE_F16:
6410 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NGE_F16_t16_e64
6411 : AMDGPU::V_CMP_NGE_F16_fake16_e64;
6412 case AMDGPU::S_CMP_NLG_F16:
6413 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NLG_F16_t16_e64
6414 : AMDGPU::V_CMP_NLG_F16_fake16_e64;
6415 case AMDGPU::S_CMP_NGT_F16:
6416 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NGT_F16_t16_e64
6417 : AMDGPU::V_CMP_NGT_F16_fake16_e64;
6418 case AMDGPU::S_CMP_NLE_F16:
6419 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NLE_F16_t16_e64
6420 : AMDGPU::V_CMP_NLE_F16_fake16_e64;
6421 case AMDGPU::S_CMP_NEQ_F16:
6422 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NEQ_F16_t16_e64
6423 : AMDGPU::V_CMP_NEQ_F16_fake16_e64;
6424 case AMDGPU::S_CMP_NLT_F16:
6425 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NLT_F16_t16_e64
6426 : AMDGPU::V_CMP_NLT_F16_fake16_e64;
6427 case AMDGPU::V_S_EXP_F32_e64:
return AMDGPU::V_EXP_F32_e64;
6428 case AMDGPU::V_S_EXP_F16_e64:
6429 return ST.useRealTrue16Insts() ? AMDGPU::V_EXP_F16_t16_e64
6430 : AMDGPU::V_EXP_F16_fake16_e64;
6431 case AMDGPU::V_S_LOG_F32_e64:
return AMDGPU::V_LOG_F32_e64;
6432 case AMDGPU::V_S_LOG_F16_e64:
6433 return ST.useRealTrue16Insts() ? AMDGPU::V_LOG_F16_t16_e64
6434 : AMDGPU::V_LOG_F16_fake16_e64;
6435 case AMDGPU::V_S_RCP_F32_e64:
return AMDGPU::V_RCP_F32_e64;
6436 case AMDGPU::V_S_RCP_F16_e64:
6437 return ST.useRealTrue16Insts() ? AMDGPU::V_RCP_F16_t16_e64
6438 : AMDGPU::V_RCP_F16_fake16_e64;
6439 case AMDGPU::V_S_RSQ_F32_e64:
return AMDGPU::V_RSQ_F32_e64;
6440 case AMDGPU::V_S_RSQ_F16_e64:
6441 return ST.useRealTrue16Insts() ? AMDGPU::V_RSQ_F16_t16_e64
6442 : AMDGPU::V_RSQ_F16_fake16_e64;
6443 case AMDGPU::V_S_SQRT_F32_e64:
return AMDGPU::V_SQRT_F32_e64;
6444 case AMDGPU::V_S_SQRT_F16_e64:
6445 return ST.useRealTrue16Insts() ? AMDGPU::V_SQRT_F16_t16_e64
6446 : AMDGPU::V_SQRT_F16_fake16_e64;
6449 "Unexpected scalar opcode without corresponding vector one!");
6498 "Not a whole wave func");
6501 if (
MI.getOpcode() == AMDGPU::SI_WHOLE_WAVE_FUNC_SETUP ||
6502 MI.getOpcode() == AMDGPU::G_AMDGPU_WHOLE_WAVE_FUNC_SETUP)
6509 unsigned OpNo)
const {
6511 if (
MI.isVariadic() || OpNo >=
Desc.getNumOperands() ||
6512 Desc.operands()[OpNo].RegClass == -1) {
6515 if (Reg.isVirtual()) {
6519 return RI.getPhysRegBaseClass(Reg);
6522 int16_t RegClass = getOpRegClassID(
Desc.operands()[OpNo]);
6523 return RegClass < 0 ? nullptr : RI.getRegClass(RegClass);
6528 constexpr AMDGPU::OpName OpNames[] = {
6529 AMDGPU::OpName::src0, AMDGPU::OpName::src1, AMDGPU::OpName::src2};
6532 int SrcIdx = AMDGPU::getNamedOperandIdx(
MI.getOpcode(), OpNames[
I]);
6533 if (
static_cast<unsigned>(SrcIdx) == OpIdx)
6545 unsigned RCID = getOpRegClassID(
get(
MI.getOpcode()).operands()[OpIdx]);
6547 unsigned Size = RI.getRegSizeInBits(*RC);
6548 unsigned Opcode = (
Size == 64) ? AMDGPU::V_MOV_B64_PSEUDO
6549 :
Size == 16 ? AMDGPU::V_MOV_B16_t16_e64
6550 : AMDGPU::V_MOV_B32_e32;
6552 Opcode = AMDGPU::COPY;
6553 else if (RI.isSGPRClass(RC))
6554 Opcode = (
Size == 64) ? AMDGPU::S_MOV_B64 : AMDGPU::S_MOV_B32;
6579 .
addImm(AMDGPU::sub0_sub1)
6581 .
addImm(AMDGPU::sub2_sub3);
6582 }
else if (Opcode == AMDGPU::V_MOV_B16_t16_e64) {
6599 return RI.getSubReg(SuperReg.
getReg(), SubIdx);
6605 unsigned NewSubIdx = RI.composeSubRegIndices(SuperReg.
getSubReg(), SubIdx);
6616 if (SubIdx == AMDGPU::sub0)
6618 if (SubIdx == AMDGPU::sub1)
6630void SIInstrInfo::swapOperands(
MachineInstr &Inst)
const {
6646 if (Reg.isPhysical())
6653 RI.getLargestLegalSuperClass(RC, MRI.
getMF());
6656 return RI.getMatchingSuperRegClass(SuperRC, DRC, MO.
getSubReg()) !=
nullptr;
6659 return RI.getCommonSubClass(DRC, RC) !=
nullptr;
6666 unsigned Opc =
MI.getOpcode();
6669 if (MO.
isReg() && RI.isSGPRReg(MRI, MO.
getReg()) &&
6679 bool IsAGPR = RI.isAGPR(MRI, MO.
getReg());
6680 if (IsAGPR && !ST.hasMAIInsts())
6686 const int VDstIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst);
6687 const int DataIdx = AMDGPU::getNamedOperandIdx(
6688 Opc,
isDS(
Opc) ? AMDGPU::OpName::data0 : AMDGPU::OpName::vdata);
6689 if ((
int)OpIdx == VDstIdx && DataIdx != -1 &&
6690 MI.getOperand(DataIdx).isReg() &&
6691 RI.isAGPR(MRI,
MI.getOperand(DataIdx).getReg()) != IsAGPR)
6693 if ((
int)OpIdx == DataIdx) {
6694 if (VDstIdx != -1 &&
6695 RI.isAGPR(MRI,
MI.getOperand(VDstIdx).getReg()) != IsAGPR)
6698 const int Data1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::data1);
6699 if (Data1Idx != -1 &&
MI.getOperand(Data1Idx).isReg() &&
6700 RI.isAGPR(MRI,
MI.getOperand(Data1Idx).getReg()) != IsAGPR)
6705 if (
Opc == AMDGPU::V_ACCVGPR_WRITE_B32_e64 && !ST.hasGFX90AInsts() &&
6706 (
int)OpIdx == AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0) &&
6707 RI.isSGPRReg(MRI, MO.
getReg()))
6710 if (ST.hasFlatScratchHiInB64InstHazard() &&
6717 if (
Opc == AMDGPU::S_BITCMP0_B64 ||
Opc == AMDGPU::S_BITCMP1_B64)
6720 if (!ST.hasDPPSrc1SGPR() &&
isDPP(
MI) && RI.isSGPRReg(MRI, MO.
getReg()) &&
6721 (
int)OpIdx == AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1))
6741 constexpr unsigned NumOps = 3;
6742 constexpr AMDGPU::OpName OpNames[
NumOps * 2] = {
6743 AMDGPU::OpName::src0, AMDGPU::OpName::src1,
6744 AMDGPU::OpName::src2, AMDGPU::OpName::src0_modifiers,
6745 AMDGPU::OpName::src1_modifiers, AMDGPU::OpName::src2_modifiers};
6750 int SrcIdx = AMDGPU::getNamedOperandIdx(
MI.getOpcode(), OpNames[SrcN]);
6753 MO = &
MI.getOperand(SrcIdx);
6756 if (!MO->
isReg() || !RI.isSGPRReg(MRI, MO->
getReg()))
6760 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), OpNames[
NumOps + SrcN]);
6764 unsigned Mods =
MI.getOperand(ModsIdx).getImm();
6768 return !OpSel && !OpSelHi;
6777 int64_t RegClass = getOpRegClassID(OpInfo);
6779 RegClass != -1 ? RI.getRegClass(RegClass) :
nullptr;
6781 MO = &
MI.getOperand(OpIdx);
6785 if (
isVALU(
MI,
false) && !IsInlineConst &&
6789 int ConstantBusLimit = ST.getConstantBusLimit(
MI.getOpcode());
6790 int LiteralLimit = !
isVOP3(
MI) || ST.hasVOP3Literal() ? 1 : 0;
6794 if (!LiteralLimit--)
6804 for (
unsigned i = 0, e =
MI.getNumOperands(); i != e; ++i) {
6812 if (--ConstantBusLimit <= 0)
6824 if (!LiteralLimit--)
6826 if (--ConstantBusLimit <= 0)
6832 for (
unsigned i = 0, e =
MI.getNumOperands(); i != e; ++i) {
6836 if (!
Op.isReg() && !
Op.isFI() && !
Op.isRegMask() &&
6838 !
Op.isIdenticalTo(*MO))
6860 bool Is64BitOp = Is64BitFPOp ||
6868 (!ST.has64BitLiterals() || InstDesc.
getSize() != 4))
6877 if (!Is64BitFPOp && (int32_t)
Imm < 0 &&
6895 bool IsGFX950Only = ST.hasGFX950Insts();
6896 bool IsGFX940Only = ST.hasGFX940Insts();
6898 if (!IsGFX950Only && !IsGFX940Only)
6916 unsigned Opcode =
MI.getOpcode();
6918 case AMDGPU::V_CVT_PK_BF8_F32_e64:
6919 case AMDGPU::V_CVT_PK_FP8_F32_e64:
6920 case AMDGPU::V_MQSAD_PK_U16_U8_e64:
6921 case AMDGPU::V_MQSAD_U32_U8_e64:
6922 case AMDGPU::V_PK_ADD_F16:
6923 case AMDGPU::V_PK_ADD_F32:
6924 case AMDGPU::V_PK_ADD_I16:
6925 case AMDGPU::V_PK_ADD_U16:
6926 case AMDGPU::V_PK_ASHRREV_I16:
6927 case AMDGPU::V_PK_FMA_F16:
6928 case AMDGPU::V_PK_FMA_F32:
6929 case AMDGPU::V_PK_FMAC_F16_e32:
6930 case AMDGPU::V_PK_FMAC_F16_e64:
6931 case AMDGPU::V_PK_LSHLREV_B16:
6932 case AMDGPU::V_PK_LSHRREV_B16:
6933 case AMDGPU::V_PK_MAD_I16:
6934 case AMDGPU::V_PK_MAD_U16:
6935 case AMDGPU::V_PK_MAX_F16:
6936 case AMDGPU::V_PK_MAX_I16:
6937 case AMDGPU::V_PK_MAX_U16:
6938 case AMDGPU::V_PK_MIN_F16:
6939 case AMDGPU::V_PK_MIN_I16:
6940 case AMDGPU::V_PK_MIN_U16:
6941 case AMDGPU::V_PK_MOV_B32:
6942 case AMDGPU::V_PK_MUL_F16:
6943 case AMDGPU::V_PK_MUL_F32:
6944 case AMDGPU::V_PK_MUL_LO_U16:
6945 case AMDGPU::V_PK_SUB_I16:
6946 case AMDGPU::V_PK_SUB_U16:
6947 case AMDGPU::V_QSAD_PK_U16_U8_e64:
6956 unsigned Opc =
MI.getOpcode();
6959 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
6962 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
6968 if (HasImplicitSGPR && ST.getConstantBusLimit(
Opc) <= 1 && Src0.
isReg() &&
6969 RI.isSGPRReg(MRI, Src0.
getReg()))
6975 if (
Opc == AMDGPU::V_WRITELANE_B32) {
6977 if (Src0.
isReg() && RI.isVGPR(MRI, Src0.
getReg())) {
6983 if (Src1.
isReg() && RI.isVGPR(MRI, Src1.
getReg())) {
6994 if (
Opc == AMDGPU::V_FMAC_F32_e32 ||
Opc == AMDGPU::V_FMAC_F16_e32) {
6995 int Src2Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2);
6996 if (!RI.isVGPR(MRI,
MI.getOperand(Src2Idx).getReg()))
7008 if (
Opc == AMDGPU::V_READLANE_B32 && Src1.
isReg() &&
7009 RI.isVGPR(MRI, Src1.
getReg())) {
7022 if (HasImplicitSGPR || !
MI.isCommutable()) {
7039 if (CommutedOpc == -1) {
7044 MI.setDesc(
get(CommutedOpc));
7048 bool Src0Kill = Src0.
isKill();
7052 else if (Src1.
isReg()) {
7067 unsigned Opc =
MI.getOpcode();
7070 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0),
7071 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1),
7072 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2)
7075 if (
Opc == AMDGPU::V_PERMLANE16_B32_e64 ||
7076 Opc == AMDGPU::V_PERMLANEX16_B32_e64 ||
7077 Opc == AMDGPU::V_PERMLANE_BCAST_B32_e64 ||
7078 Opc == AMDGPU::V_PERMLANE_UP_B32_e64 ||
7079 Opc == AMDGPU::V_PERMLANE_DOWN_B32_e64 ||
7080 Opc == AMDGPU::V_PERMLANE_XOR_B32_e64 ||
7081 Opc == AMDGPU::V_PERMLANE_IDX_GEN_B32_e64) {
7091 if (VOP3Idx[2] != -1) {
7103 int ConstantBusLimit = ST.getConstantBusLimit(
Opc);
7104 int LiteralLimit = ST.hasVOP3Literal() ? 1 : 0;
7106 Register SGPRReg = findUsedSGPR(
MI, VOP3Idx);
7108 SGPRsUsed.
insert(SGPRReg);
7112 for (
int Idx : VOP3Idx) {
7121 if (LiteralLimit > 0 && ConstantBusLimit > 0) {
7133 if (!RI.isSGPRClass(RI.getRegClassForReg(MRI, MO.
getReg())))
7140 if (ConstantBusLimit > 0) {
7152 if ((
Opc == AMDGPU::V_FMAC_F32_e64 ||
Opc == AMDGPU::V_FMAC_F16_e64) &&
7153 !RI.isVGPR(MRI,
MI.getOperand(VOP3Idx[2]).getReg()))
7159 for (
unsigned I = 0;
I < 3; ++
I) {
7172 SRC = RI.getCommonSubClass(SRC, DstRC);
7175 unsigned SubRegs = RI.getRegSizeInBits(*VRC) / 32;
7177 if (RI.hasAGPRs(VRC)) {
7178 VRC = RI.getEquivalentVGPRClass(VRC);
7181 get(TargetOpcode::COPY), NewSrcReg)
7188 get(AMDGPU::V_READFIRSTLANE_B32), DstReg)
7194 for (
unsigned i = 0; i < SubRegs; ++i) {
7197 get(AMDGPU::V_READFIRSTLANE_B32), SGPR)
7198 .
addReg(SrcReg, {}, RI.getSubRegFromChannel(i));
7204 get(AMDGPU::REG_SEQUENCE), DstReg);
7205 for (
unsigned i = 0; i < SubRegs; ++i) {
7207 MIB.
addImm(RI.getSubRegFromChannel(i));
7220 if (SBase && !RI.isSGPRClass(MRI.
getRegClass(SBase->getReg()))) {
7222 SBase->setReg(SGPR);
7225 if (SOff && !RI.isSGPRReg(MRI, SOff->
getReg())) {
7233 int OldSAddrIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::saddr);
7234 if (OldSAddrIdx < 0)
7247 if (RI.isSGPRReg(MRI, SAddr.
getReg()))
7250 int NewVAddrIdx = AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::vaddr);
7251 if (NewVAddrIdx < 0)
7254 int OldVAddrIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr);
7258 if (OldVAddrIdx >= 0) {
7272 if (OldVAddrIdx == NewVAddrIdx) {
7283 assert(OldSAddrIdx == NewVAddrIdx);
7285 if (OldVAddrIdx >= 0) {
7286 int NewVDstIn = AMDGPU::getNamedOperandIdx(NewOpc,
7287 AMDGPU::OpName::vdst_in);
7291 if (NewVDstIn != -1) {
7292 int OldVDstIn = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst_in);
7298 if (NewVDstIn != -1) {
7299 int NewVDst = AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::vdst);
7340 unsigned OpSubReg =
Op.getSubReg();
7343 RI.getRegClassForReg(MRI, OpReg), OpSubReg);
7350 auto Copy =
BuildMI(InsertMBB,
I,
DL,
get(AMDGPU::COPY), DstReg)
7351 .
addReg(OpReg, {}, OpSubReg);
7353 Op.setSubReg(AMDGPU::NoSubRegister);
7360 if (Def->isMoveImmediate() && DstRC != &AMDGPU::VReg_1RegClass)
7363 bool ImpDef = Def->isImplicitDef();
7364 while (!ImpDef && Def && Def->isCopy()) {
7365 if (Def->getOperand(1).getReg().isPhysical())
7368 ImpDef = Def && Def->isImplicitDef();
7370 if (!RI.isSGPRClass(DstRC) && !Copy->readsRegister(AMDGPU::EXEC, &RI) &&
7386 const auto *BoolXExecRC =
TRI->getWaveMaskRegClass();
7391 bool UseNewExecInstructions =
7400 if (UseNewExecInstructions) {
7435 for (
auto [Idx, ScalarOp] :
enumerate(ScalarOps)) {
7436 unsigned RegSize =
TRI->getRegSizeInBits(ScalarOp->getReg(), MRI);
7437 unsigned NumSubRegs =
RegSize / 32;
7438 Register VScalarOp = ScalarOp->getReg();
7441 TII.getRegClass(
TII.get(AMDGPU::V_READFIRSTLANE_B32), 1);
7443 if (NumSubRegs == 1) {
7446 TRI->getCommonSubClass(VScalarOpRC, RFLSrcRC);
7447 Common != VScalarOpRC) {
7454 BuildMI(LoopBB,
I,
DL,
TII.get(AMDGPU::V_READFIRSTLANE_B32), CurReg)
7457 if (UseNewExecInstructions) {
7459 TII.get(AMDGPU::V_CMPX_EQ_U32_nosdst_e32_term))
7462 if (
I == LoopBB.
end())
7467 BuildMI(LoopBB,
I,
DL,
TII.get(AMDGPU::V_CMP_EQ_U32_e64), NewCondReg)
7473 CondReg = NewCondReg;
7485 if (PhySGPRs.empty() || !PhySGPRs[Idx].isValid())
7486 ScalarOp->setReg(CurReg);
7489 BuildMI(*ScalarOp->getParent()->getParent(), ScalarOp->getParent(),
DL,
7490 TII.get(AMDGPU::COPY), PhySGPRs[Idx])
7492 ScalarOp->setReg(PhySGPRs[Idx]);
7494 ScalarOp->setIsKill();
7498 assert(NumSubRegs % 2 == 0 && NumSubRegs <= 32 &&
7499 "Unhandled register size");
7501 for (
unsigned Idx = 0; Idx < NumSubRegs; Idx += 2) {
7508 BuildMI(LoopBB,
I,
DL,
TII.get(AMDGPU::V_READFIRSTLANE_B32), CurRegLo)
7509 .
addReg(VScalarOp, VScalarOpUndef,
TRI->getSubRegFromChannel(Idx));
7512 BuildMI(LoopBB,
I,
DL,
TII.get(AMDGPU::V_READFIRSTLANE_B32), CurRegHi)
7513 .
addReg(VScalarOp, VScalarOpUndef,
7514 TRI->getSubRegFromChannel(Idx + 1));
7521 BuildMI(LoopBB,
I,
DL,
TII.get(AMDGPU::REG_SEQUENCE), CurReg)
7528 NumSubRegs <= 2 ? 0 :
TRI->getSubRegFromChannel(Idx, 2);
7530 if (UseNewExecInstructions) {
7532 TII.get(AMDGPU::V_CMPX_EQ_U64_nosdst_e32_term))
7534 .
addReg(VScalarOp, VScalarOpUndef, SubReg);
7535 if (
I == LoopBB.
end())
7539 BuildMI(LoopBB,
I,
DL,
TII.get(AMDGPU::V_CMP_EQ_U64_e64), NewCondReg)
7541 .
addReg(VScalarOp, VScalarOpUndef, SubReg);
7545 CondReg = NewCondReg;
7557 const auto *SScalarOpRC =
7563 BuildMI(LoopBB,
I,
DL,
TII.get(AMDGPU::REG_SEQUENCE), SScalarOp);
7564 unsigned Channel = 0;
7565 for (
Register Piece : ReadlanePieces) {
7566 Merge.addReg(Piece).addImm(
TRI->getSubRegFromChannel(Channel++));
7570 if (PhySGPRs.empty() || !PhySGPRs[Idx].isValid())
7571 ScalarOp->setReg(SScalarOp);
7573 BuildMI(*ScalarOp->getParent()->getParent(), ScalarOp->getParent(),
DL,
7574 TII.get(AMDGPU::COPY), PhySGPRs[Idx])
7576 ScalarOp->setReg(PhySGPRs[Idx]);
7578 ScalarOp->setIsKill();
7585 if (!UseNewExecInstructions) {
7598 if (UseNewExecInstructions) {
7631 assert((PhySGPRs.empty() || PhySGPRs.size() == ScalarOps.
size()) &&
7632 "Physical SGPRs must be empty or match the number of scalar operands");
7638 if (!Begin.isValid())
7640 if (!End.isValid()) {
7646 const auto *BoolXExecRC =
TRI->getWaveMaskRegClass();
7655 std::numeric_limits<unsigned>::max()) !=
7673 for (
auto I = Begin;
I != AfterMI;
I++) {
7674 for (
auto &MO :
I->all_uses())
7710 for (
auto &Succ : RemainderBB->
successors()) {
7735static std::tuple<unsigned, unsigned>
7743 TII.buildExtractSubReg(
MI, MRI, Rsrc, &AMDGPU::VReg_128RegClass,
7744 AMDGPU::sub0_sub1, &AMDGPU::VReg_64RegClass);
7751 uint64_t RsrcDataFormat =
TII.getDefaultRsrcDataFormat();
7768 .
addImm(AMDGPU::sub0_sub1)
7774 return std::tuple(RsrcPtr, NewSRsrc);
7785 if (ST.useRealTrue16Insts())
7815 if (
MI.getOpcode() == AMDGPU::PHI) {
7817 assert(!RI.isSGPRClass(VRC));
7820 for (
unsigned I = 1, E =
MI.getNumOperands();
I != E;
I += 2) {
7822 if (!
Op.isReg() || !
Op.getReg().isVirtual())
7838 if (
MI.getOpcode() == AMDGPU::REG_SEQUENCE) {
7841 if (RI.hasVGPRs(DstRC)) {
7845 for (
unsigned I = 1, E =
MI.getNumOperands();
I != E;
I += 2) {
7847 if (!
Op.isReg() || !
Op.getReg().isVirtual())
7865 if (
MI.getOpcode() == AMDGPU::INSERT_SUBREG) {
7870 if (DstRC != Src0RC) {
7879 if (
MI.getOpcode() == AMDGPU::SI_INIT_M0) {
7881 if (Src.isReg() && RI.hasVectorRegisters(MRI.
getRegClass(Src.getReg())))
7887 if (
MI.getOpcode() == AMDGPU::S_BITREPLICATE_B64_B32 ||
7888 MI.getOpcode() == AMDGPU::S_QUADMASK_B32 ||
7889 MI.getOpcode() == AMDGPU::S_QUADMASK_B64 ||
7890 MI.getOpcode() == AMDGPU::S_WQM_B32 ||
7891 MI.getOpcode() == AMDGPU::S_WQM_B64 ||
7892 MI.getOpcode() == AMDGPU::S_INVERSE_BALLOT_U32 ||
7893 MI.getOpcode() == AMDGPU::S_INVERSE_BALLOT_U64) {
7895 if (Src.isReg() && RI.hasVectorRegisters(MRI.
getRegClass(Src.getReg())))
7908 ? AMDGPU::OpName::rsrc
7909 : AMDGPU::OpName::srsrc;
7914 AMDGPU::OpName SampOpName =
7915 isMIMG(
MI) ? AMDGPU::OpName::ssamp : AMDGPU::OpName::samp;
7924 if (
MI.getOpcode() == AMDGPU::SI_CALL_ISEL) {
7932 if (
MI.getOpcode() == AMDGPU::S_SLEEP_VAR) {
7936 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::src0);
7946 if (
MI.getOpcode() == AMDGPU::TENSOR_LOAD_TO_LDS_d2 ||
7947 MI.getOpcode() == AMDGPU::TENSOR_LOAD_TO_LDS_d4 ||
7948 MI.getOpcode() == AMDGPU::TENSOR_STORE_FROM_LDS_d2 ||
7949 MI.getOpcode() == AMDGPU::TENSOR_STORE_FROM_LDS_d4) {
7951 if (Src.isReg() && RI.hasVectorRegisters(MRI.
getRegClass(Src.getReg())))
7958 bool isSoffsetLegal =
true;
7960 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::soffset);
7961 if (SoffsetIdx != -1) {
7965 isSoffsetLegal =
false;
7969 bool isRsrcLegal =
true;
7971 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::srsrc);
7972 if (RsrcIdx != -1) {
7974 if (Rsrc->
isReg() && !RI.isSGPRReg(MRI, Rsrc->
getReg()))
7975 isRsrcLegal =
false;
7979 if (isRsrcLegal && isSoffsetLegal)
8007 const auto *BoolXExecRC = RI.getWaveMaskRegClass();
8011 unsigned RsrcPtr, NewSRsrc;
8018 .
addReg(RsrcPtr, {}, AMDGPU::sub0)
8019 .addReg(VAddr->
getReg(), {}, AMDGPU::sub0)
8025 .
addReg(RsrcPtr, {}, AMDGPU::sub1)
8026 .addReg(VAddr->
getReg(), {}, AMDGPU::sub1)
8039 }
else if (!VAddr && ST.hasAddr64()) {
8043 "FIXME: Need to emit flat atomics here");
8045 unsigned RsrcPtr, NewSRsrc;
8071 MIB.
addImm(CPol->getImm());
8076 MIB.
addImm(TFE->getImm());
8096 MI.removeFromParent();
8101 .
addReg(RsrcPtr, {}, AMDGPU::sub0)
8102 .addImm(AMDGPU::sub0)
8103 .
addReg(RsrcPtr, {}, AMDGPU::sub1)
8104 .addImm(AMDGPU::sub1);
8107 if (!isSoffsetLegal) {
8118 if (!isSoffsetLegal) {
8127 if (InSet.insert(
MI).second)
8131 AMDGPU::getNamedOperandIdx(
MI->getOpcode(), AMDGPU::OpName::srsrc);
8132 if (RsrcIdx != -1) {
8133 DeferredList.insert(
MI);
8138 return DeferredList.contains(
MI);
8148 if (!ST.useRealTrue16Insts())
8151 unsigned Opcode =
MI.getOpcode();
8154 if (OpIdx >=
MI.getNumExplicitOperands() ||
8155 OpIdx >=
get(Opcode).getNumOperands() ||
8156 get(Opcode).operands()[OpIdx].RegClass == -1)
8160 if (!
Op.isReg() || !
Op.getReg().isVirtual() ||
Op.isDef())
8164 if (!RI.isVGPRClass(CurrRC))
8167 int16_t RCID = getOpRegClassID(
get(Opcode).operands()[OpIdx]);
8169 if (RI.getMatchingSuperRegClass(CurrRC, ExpectedRC, AMDGPU::lo16)) {
8171 if (
Op.getSubReg() == AMDGPU::NoSubRegister)
8172 Op.setSubReg(AMDGPU::lo16);
8177 RI.getSubRegisterClass(CurrRC,
Op.getSubReg());
8178 if (RI.getMatchingSuperRegClass(ExpectedRC, CurrSRC, AMDGPU::lo16)) {
8188 Op.setReg(NewDstReg);
8189 Op.setSubReg(AMDGPU::NoSubRegister);
8194 for (
unsigned OpIdx = 0; OpIdx <
MI.getNumExplicitOperands(); OpIdx++)
8202 assert(
MI->getOpcode() == AMDGPU::SI_CALL_ISEL &&
8203 "This only handle waterfall for SI_CALL_ISEL");
8210 while (Start->getOpcode() != AMDGPU::ADJCALLSTACKUP)
8213 while (End->getOpcode() != AMDGPU::ADJCALLSTACKDOWN)
8218 while (End !=
MBB.end() && End->isCopy() &&
8219 MI->definesRegister(End->getOperand(1).getReg(), &RI))
8229 while (!Worklist.
empty()) {
8235 moveToVALUImpl(Worklist, MDT, Inst, WaterFalls, V2SPhyCopiesToErase);
8241 moveToVALUImpl(Worklist, MDT, *Inst, WaterFalls, V2SPhyCopiesToErase);
8243 "Deferred MachineInstr are not supposed to re-populate worklist");
8246 for (
auto &Entry : WaterFalls) {
8247 if (Entry.first->getOpcode() == AMDGPU::SI_CALL_ISEL)
8249 Entry.second.SGPRs);
8252 for (std::pair<MachineInstr *, bool> Entry : V2SPhyCopiesToErase)
8254 Entry.first->eraseFromParent();
8262 if (SubRegIndices.
size() <= 1) {
8265 get(AMDGPU::V_READFIRSTLANE_B32), NewDst)
8272 for (int16_t Indice : SubRegIndices) {
8275 get(AMDGPU::V_READFIRSTLANE_B32), NewDst)
8282 get(AMDGPU::REG_SEQUENCE), DstReg);
8283 for (
unsigned i = 0; i < SubRegIndices.size(); ++i) {
8285 MIB.
addImm(RI.getSubRegFromChannel(i));
8295 if (DstReg == AMDGPU::M0) {
8308 if (
I->getOpcode() == AMDGPU::SI_CALL_ISEL) {
8310 for (
unsigned i = 0; i <
UseMI->getNumOperands(); ++i) {
8311 if (
UseMI->getOperand(i).isReg() &&
8312 UseMI->getOperand(i).getReg() == DstReg) {
8316 V2SCopyInfo.MOs.push_back(MO);
8317 V2SCopyInfo.SGPRs.push_back(DstReg);
8321 }
else if (
I->getOpcode() == AMDGPU::SI_RETURN_TO_EPILOG &&
8322 I->getOperand(0).isReg() &&
8323 I->getOperand(0).getReg() == DstReg) {
8326 }
else if (
I->readsRegister(DstReg, &RI)) {
8328 V2SPhyCopiesToErase[&Inst] =
false;
8330 if (
I->findRegisterDefOperand(DstReg, &RI))
8352 case AMDGPU::S_ADD_I32:
8353 case AMDGPU::S_SUB_I32: {
8357 std::tie(
Changed, CreatedBBTmp) = moveScalarAddSub(Worklist, Inst, MDT);
8365 case AMDGPU::S_MUL_U64:
8366 if (ST.useVMulU64Inst()) {
8367 NewOpcode = AMDGPU::V_MUL_U64_e64;
8371 splitScalarSMulU64(Worklist, Inst, MDT);
8375 case AMDGPU::S_MUL_U64_U32_PSEUDO:
8376 case AMDGPU::S_MUL_I64_I32_PSEUDO:
8379 splitScalarSMulPseudo(Worklist, Inst, MDT);
8383 case AMDGPU::S_AND_B64:
8384 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_AND_B32, MDT);
8388 case AMDGPU::S_OR_B64:
8389 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_OR_B32, MDT);
8393 case AMDGPU::S_XOR_B64:
8394 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_XOR_B32, MDT);
8398 case AMDGPU::S_NAND_B64:
8399 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_NAND_B32, MDT);
8403 case AMDGPU::S_NOR_B64:
8404 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_NOR_B32, MDT);
8408 case AMDGPU::S_XNOR_B64:
8409 if (ST.hasDLInsts())
8410 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_XNOR_B32, MDT);
8412 splitScalar64BitXnor(Worklist, Inst, MDT);
8416 case AMDGPU::S_ANDN2_B64:
8417 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_ANDN2_B32, MDT);
8421 case AMDGPU::S_ORN2_B64:
8422 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_ORN2_B32, MDT);
8426 case AMDGPU::S_BREV_B64:
8427 splitScalar64BitUnaryOp(Worklist, Inst, AMDGPU::S_BREV_B32,
true);
8431 case AMDGPU::S_NOT_B64:
8432 splitScalar64BitUnaryOp(Worklist, Inst, AMDGPU::S_NOT_B32);
8436 case AMDGPU::S_BCNT1_I32_B64:
8437 splitScalar64BitBCNT(Worklist, Inst);
8441 case AMDGPU::S_BFE_I64:
8442 splitScalar64BitBFE(Worklist, Inst);
8446 case AMDGPU::S_FLBIT_I32_B64:
8447 splitScalar64BitCountOp(Worklist, Inst, AMDGPU::V_FFBH_U32_e32);
8450 case AMDGPU::S_FF1_I32_B64:
8451 splitScalar64BitCountOp(Worklist, Inst, AMDGPU::V_FFBL_B32_e32);
8455 case AMDGPU::S_LSHL_B32:
8456 if (ST.hasOnlyRevVALUShifts()) {
8457 NewOpcode = AMDGPU::V_LSHLREV_B32_e64;
8461 case AMDGPU::S_ASHR_I32:
8462 if (ST.hasOnlyRevVALUShifts()) {
8463 NewOpcode = AMDGPU::V_ASHRREV_I32_e64;
8467 case AMDGPU::S_LSHR_B32:
8468 if (ST.hasOnlyRevVALUShifts()) {
8469 NewOpcode = AMDGPU::V_LSHRREV_B32_e64;
8473 case AMDGPU::S_LSHL_B64:
8474 if (ST.hasOnlyRevVALUShifts()) {
8476 ? AMDGPU::V_LSHLREV_B64_pseudo_e64
8477 : AMDGPU::V_LSHLREV_B64_e64;
8481 case AMDGPU::S_ASHR_I64:
8482 if (ST.hasOnlyRevVALUShifts()) {
8483 NewOpcode = AMDGPU::V_ASHRREV_I64_e64;
8487 case AMDGPU::S_LSHR_B64:
8488 if (ST.hasOnlyRevVALUShifts()) {
8489 NewOpcode = AMDGPU::V_LSHRREV_B64_e64;
8494 case AMDGPU::S_ABS_I32:
8495 lowerScalarAbs(Worklist, Inst);
8499 case AMDGPU::S_ABSDIFF_I32:
8500 lowerScalarAbsDiff(Worklist, Inst);
8504 case AMDGPU::S_CBRANCH_SCC0:
8505 case AMDGPU::S_CBRANCH_SCC1: {
8508 bool IsSCC = CondReg == AMDGPU::SCC;
8517 case AMDGPU::S_BFE_U64:
8518 case AMDGPU::S_BFM_B64:
8521 case AMDGPU::S_PACK_LL_B32_B16:
8522 case AMDGPU::S_PACK_LH_B32_B16:
8523 case AMDGPU::S_PACK_HL_B32_B16:
8524 case AMDGPU::S_PACK_HH_B32_B16:
8525 movePackToVALU(Worklist, MRI, Inst);
8529 case AMDGPU::S_XNOR_B32:
8530 lowerScalarXnor(Worklist, Inst);
8534 case AMDGPU::S_NAND_B32:
8535 splitScalarNotBinop(Worklist, Inst, AMDGPU::S_AND_B32);
8539 case AMDGPU::S_NOR_B32:
8540 splitScalarNotBinop(Worklist, Inst, AMDGPU::S_OR_B32);
8544 case AMDGPU::S_ANDN2_B32:
8545 splitScalarBinOpN2(Worklist, Inst, AMDGPU::S_AND_B32);
8549 case AMDGPU::S_ORN2_B32:
8550 splitScalarBinOpN2(Worklist, Inst, AMDGPU::S_OR_B32);
8558 case AMDGPU::S_ADD_CO_PSEUDO:
8559 case AMDGPU::S_SUB_CO_PSEUDO: {
8560 unsigned Opc = (Inst.
getOpcode() == AMDGPU::S_ADD_CO_PSEUDO)
8561 ? AMDGPU::V_ADDC_U32_e64
8562 : AMDGPU::V_SUBB_U32_e64;
8563 const auto *CarryRC = RI.getWaveMaskRegClass();
8585 addUsersToMoveToVALUWorklist(DestReg, MRI, Worklist);
8589 case AMDGPU::S_UADDO_PSEUDO:
8590 case AMDGPU::S_USUBO_PSEUDO: {
8596 unsigned Opc = (Inst.
getOpcode() == AMDGPU::S_UADDO_PSEUDO)
8597 ? AMDGPU::V_ADD_CO_U32_e64
8598 : AMDGPU::V_SUB_CO_U32_e64;
8610 addUsersToMoveToVALUWorklist(DestReg, MRI, Worklist);
8614 case AMDGPU::S_LSHL1_ADD_U32:
8615 case AMDGPU::S_LSHL2_ADD_U32:
8616 case AMDGPU::S_LSHL3_ADD_U32:
8617 case AMDGPU::S_LSHL4_ADD_U32: {
8621 unsigned ShiftAmt = (Opcode == AMDGPU::S_LSHL1_ADD_U32 ? 1
8622 : Opcode == AMDGPU::S_LSHL2_ADD_U32 ? 2
8623 : Opcode == AMDGPU::S_LSHL3_ADD_U32 ? 3
8637 addUsersToMoveToVALUWorklist(DestReg, MRI, Worklist);
8641 case AMDGPU::S_CSELECT_B32:
8642 case AMDGPU::S_CSELECT_B64:
8643 lowerSelect(Worklist, Inst, MDT);
8646 case AMDGPU::S_CMP_EQ_I32:
8647 case AMDGPU::S_CMP_LG_I32:
8648 case AMDGPU::S_CMP_GT_I32:
8649 case AMDGPU::S_CMP_GE_I32:
8650 case AMDGPU::S_CMP_LT_I32:
8651 case AMDGPU::S_CMP_LE_I32:
8652 case AMDGPU::S_CMP_EQ_U32:
8653 case AMDGPU::S_CMP_LG_U32:
8654 case AMDGPU::S_CMP_GT_U32:
8655 case AMDGPU::S_CMP_GE_U32:
8656 case AMDGPU::S_CMP_LT_U32:
8657 case AMDGPU::S_CMP_LE_U32:
8658 case AMDGPU::S_CMP_EQ_U64:
8659 case AMDGPU::S_CMP_LG_U64:
8660 case AMDGPU::S_CMP_LT_F32:
8661 case AMDGPU::S_CMP_EQ_F32:
8662 case AMDGPU::S_CMP_LE_F32:
8663 case AMDGPU::S_CMP_GT_F32:
8664 case AMDGPU::S_CMP_LG_F32:
8665 case AMDGPU::S_CMP_GE_F32:
8666 case AMDGPU::S_CMP_O_F32:
8667 case AMDGPU::S_CMP_U_F32:
8668 case AMDGPU::S_CMP_NGE_F32:
8669 case AMDGPU::S_CMP_NLG_F32:
8670 case AMDGPU::S_CMP_NGT_F32:
8671 case AMDGPU::S_CMP_NLE_F32:
8672 case AMDGPU::S_CMP_NEQ_F32:
8673 case AMDGPU::S_CMP_NLT_F32: {
8678 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::src0_modifiers) >=
8692 addSCCDefUsersToVALUWorklist(SCCOp, Inst, Worklist, CondReg);
8696 case AMDGPU::S_CMP_LT_F16:
8697 case AMDGPU::S_CMP_EQ_F16:
8698 case AMDGPU::S_CMP_LE_F16:
8699 case AMDGPU::S_CMP_GT_F16:
8700 case AMDGPU::S_CMP_LG_F16:
8701 case AMDGPU::S_CMP_GE_F16:
8702 case AMDGPU::S_CMP_O_F16:
8703 case AMDGPU::S_CMP_U_F16:
8704 case AMDGPU::S_CMP_NGE_F16:
8705 case AMDGPU::S_CMP_NLG_F16:
8706 case AMDGPU::S_CMP_NGT_F16:
8707 case AMDGPU::S_CMP_NLE_F16:
8708 case AMDGPU::S_CMP_NEQ_F16:
8709 case AMDGPU::S_CMP_NLT_F16: {
8731 addSCCDefUsersToVALUWorklist(SCCOp, Inst, Worklist, CondReg);
8735 case AMDGPU::S_CVT_HI_F32_F16: {
8738 if (ST.useRealTrue16Insts()) {
8743 .
addReg(TmpReg, {}, AMDGPU::hi16)
8759 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8763 case AMDGPU::S_MINIMUM_F32:
8764 case AMDGPU::S_MAXIMUM_F32: {
8776 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8780 case AMDGPU::S_MINIMUM_F16:
8781 case AMDGPU::S_MAXIMUM_F16: {
8783 ? &AMDGPU::VGPR_16RegClass
8784 : &AMDGPU::VGPR_32RegClass);
8795 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8799 case AMDGPU::V_S_EXP_F16_e64:
8800 case AMDGPU::V_S_LOG_F16_e64:
8801 case AMDGPU::V_S_RCP_F16_e64:
8802 case AMDGPU::V_S_RSQ_F16_e64:
8803 case AMDGPU::V_S_SQRT_F16_e64: {
8805 ? &AMDGPU::VGPR_16RegClass
8806 : &AMDGPU::VGPR_32RegClass);
8817 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8823 if (NewOpcode == AMDGPU::INSTRUCTION_LIST_END) {
8831 if (NewOpcode == Opcode) {
8838 V2SPhyCopiesToErase);
8846 RI.getCommonSubClass(NewDstRC, SrcRC)) {
8853 addUsersToMoveToVALUWorklist(DstReg, MRI, Worklist);
8859 RI.composeSubRegIndices(SrcSubReg, UseMO.getSubReg()));
8860 UseMO.setReg(NewDstReg);
8879 unsigned OpIdx =
UseMI.getOperandNo(&UseMO);
8892 if (ST.useRealTrue16Insts() && Inst.
isCopy() &&
8896 if (RI.getMatchingSuperRegClass(NewDstRC, SrcRegRC, AMDGPU::lo16)) {
8902 get(AMDGPU::REG_SEQUENCE), NewDstReg)
8909 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
8911 }
else if (RI.getMatchingSuperRegClass(SrcRegRC, NewDstRC,
8916 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
8924 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
8934 if (AMDGPU::getNamedOperandIdx(NewOpcode,
8935 AMDGPU::OpName::src0_modifiers) >= 0)
8939 NewInstr->addOperand(Src);
8942 if (Opcode == AMDGPU::S_SEXT_I32_I8 || Opcode == AMDGPU::S_SEXT_I32_I16) {
8945 unsigned Size = (Opcode == AMDGPU::S_SEXT_I32_I8) ? 8 : 16;
8947 NewInstr.addImm(
Size);
8948 }
else if (Opcode == AMDGPU::S_BCNT1_I32_B32) {
8952 }
else if (Opcode == AMDGPU::S_BFE_I32 || Opcode == AMDGPU::S_BFE_U32) {
8957 "Scalar BFE is only implemented for constant width and offset");
8965 if (AMDGPU::getNamedOperandIdx(NewOpcode,
8966 AMDGPU::OpName::src1_modifiers) >= 0)
8968 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::src1) >= 0)
8970 if (AMDGPU::getNamedOperandIdx(NewOpcode,
8971 AMDGPU::OpName::src2_modifiers) >= 0)
8973 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::src2) >= 0)
8975 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::clamp) >= 0)
8977 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::omod) >= 0)
8979 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::op_sel) >= 0)
8985 NewInstr->addOperand(
Op);
8991 bool DeadSCCDef =
false;
8993 if (
Op.getReg() == AMDGPU::SCC) {
8999 addSCCDefUsersToVALUWorklist(
Op, Inst, Worklist);
9003 addSCCDefsToVALUWorklist(NewInstr, Worklist);
9008 if (NewInstr->getOperand(0).isReg() && NewInstr->getOperand(0).isDef()) {
9009 Register DstReg = NewInstr->getOperand(0).getReg();
9023 NewInstr->findRegisterDefOperand(RI.getVCC(), &RI))
9024 VCCDef->setIsDead();
9030 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
9034std::pair<bool, MachineBasicBlock *>
9037 if (ST.hasAddNoCarryInsts()) {
9049 assert(
Opc == AMDGPU::S_ADD_I32 ||
Opc == AMDGPU::S_SUB_I32);
9051 unsigned NewOpc =
Opc == AMDGPU::S_ADD_I32 ?
9052 AMDGPU::V_ADD_U32_e64 : AMDGPU::V_SUB_U32_e64;
9063 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9064 return std::pair(
true, NewBB);
9067 return std::pair(
false,
nullptr);
9084 bool IsSCC = (CondReg == AMDGPU::SCC);
9092 for (MachineOperand &UseMO :
9094 MachineInstr &
UseMI = *UseMO.getParent();
9095 switch (
UseMI.getOpcode()) {
9096 case AMDGPU::V_CNDMASK_B16_fake16_e32:
9097 case AMDGPU::V_CNDMASK_B16_fake16_e64:
9098 case AMDGPU::V_CNDMASK_B16_t16_e32:
9099 case AMDGPU::V_CNDMASK_B16_t16_e64:
9100 case AMDGPU::V_CNDMASK_B32_e32:
9101 case AMDGPU::V_CNDMASK_B32_e64:
9102 case AMDGPU::V_CNDMASK_B64_PSEUDO:
9103 if (UseMO.isImplicit() ||
9105 UseMO.setReg(CondReg);
9119 bool CopyFound =
false;
9120 for (MachineInstr &CandI :
9123 if (CandI.findRegisterDefOperandIdx(AMDGPU::SCC, &RI,
false,
false) !=
9125 if (CandI.isCopy() && CandI.getOperand(0).getReg() == AMDGPU::SCC) {
9127 .
addReg(CandI.getOperand(1).getReg());
9139 ST.isWave64() ? AMDGPU::S_CSELECT_B64 : AMDGPU::S_CSELECT_B32;
9148 MachineInstr *NewInst;
9149 if (Inst.
getOpcode() == AMDGPU::S_CSELECT_B32) {
9150 NewInst =
BuildMI(
MBB, MII,
DL,
get(AMDGPU::V_CNDMASK_B32_e64), NewDestReg)
9165 addUsersToMoveToVALUWorklist(NewDestReg, MRI, Worklist);
9180 bool HasCarryOut = !ST.hasAddNoCarryInsts();
9182 HasCarryOut ? AMDGPU::V_SUB_CO_U32_e32 : AMDGPU::V_SUB_U32_e32;
9184 MachineInstrBuilder
Sub =
9194 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9211 bool HasCarryOut = !ST.hasAddNoCarryInsts();
9213 HasCarryOut ? AMDGPU::V_SUB_CO_U32_e32 : AMDGPU::V_SUB_U32_e32;
9215 MachineInstrBuilder Sub1 =
BuildMI(
MBB, MII,
DL,
get(SubOp), SubResultReg)
9219 MachineInstrBuilder Sub2 =
9232 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9246 if (ST.hasDLInsts()) {
9256 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9262 bool Src0IsSGPR = Src0.
isReg() &&
9264 bool Src1IsSGPR = Src1.
isReg() &&
9278 }
else if (Src1IsSGPR) {
9296 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9302 unsigned Opcode)
const {
9326 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9331 unsigned Opcode)
const {
9355 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9370 const MCInstrDesc &InstDesc =
get(Opcode);
9373 &AMDGPU::SGPR_32RegClass;
9376 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9379 AMDGPU::sub0, Src0SubRC);
9384 RI.getSubRegisterClass(NewDestRC, AMDGPU::sub0);
9387 MachineInstr &LoHalf = *
BuildMI(
MBB, MII,
DL, InstDesc, DestSub0).
add(SrcReg0Sub0);
9390 AMDGPU::sub1, Src0SubRC);
9393 MachineInstr &HiHalf = *
BuildMI(
MBB, MII,
DL, InstDesc, DestSub1).
add(SrcReg0Sub1);
9407 Worklist.
insert(&LoHalf);
9408 Worklist.
insert(&HiHalf);
9414 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9438 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9439 if (RI.isSGPRClass(Src0SubRC))
9440 Src0SubRC = RI.getEquivalentVGPRClass(Src0SubRC);
9442 RI.getSubRegisterClass(Src1RC, AMDGPU::sub0);
9443 if (RI.isSGPRClass(Src1SubRC))
9444 Src1SubRC = RI.getEquivalentVGPRClass(Src1SubRC);
9448 MachineOperand Op0L =
9450 MachineOperand Op1L =
9452 MachineOperand Op0H =
9454 MachineOperand Op1H =
9473 MachineInstr *Op1L_Op0H =
9479 MachineInstr *Op1H_Op0L =
9485 MachineInstr *Carry =
9490 MachineInstr *LoHalf =
9500 MachineInstr *HiHalf =
9523 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9547 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9548 if (RI.isSGPRClass(Src0SubRC))
9549 Src0SubRC = RI.getEquivalentVGPRClass(Src0SubRC);
9551 RI.getSubRegisterClass(Src1RC, AMDGPU::sub0);
9552 if (RI.isSGPRClass(Src1SubRC))
9553 Src1SubRC = RI.getEquivalentVGPRClass(Src1SubRC);
9557 MachineOperand Op0L =
9559 MachineOperand Op1L =
9563 unsigned NewOpc =
Opc == AMDGPU::S_MUL_U64_U32_PSEUDO
9564 ? AMDGPU::V_MUL_HI_U32_e64
9565 : AMDGPU::V_MUL_HI_I32_e64;
9566 MachineInstr *HiHalf =
9569 MachineInstr *LoHalf =
9588 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9604 const MCInstrDesc &InstDesc =
get(Opcode);
9607 &AMDGPU::SGPR_32RegClass;
9610 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9613 &AMDGPU::SGPR_32RegClass;
9616 RI.getSubRegisterClass(Src1RC, AMDGPU::sub0);
9619 AMDGPU::sub0, Src0SubRC);
9621 AMDGPU::sub0, Src1SubRC);
9623 AMDGPU::sub1, Src0SubRC);
9625 AMDGPU::sub1, Src1SubRC);
9630 RI.getSubRegisterClass(NewDestRC, AMDGPU::sub0);
9633 MachineInstr &LoHalf = *
BuildMI(
MBB, MII,
DL, InstDesc, DestSub0)
9638 MachineInstr &HiHalf = *
BuildMI(
MBB, MII,
DL, InstDesc, DestSub1)
9651 Worklist.
insert(&LoHalf);
9652 Worklist.
insert(&HiHalf);
9655 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9675 MachineOperand* Op0;
9676 MachineOperand* Op1;
9678 if (Src0.
isReg() && RI.isSGPRReg(MRI, Src0.
getReg())) {
9711 const MCInstrDesc &InstDesc =
get(AMDGPU::V_BCNT_U32_B32_e64);
9714 &AMDGPU::SGPR_32RegClass;
9720 RI.getSubRegisterClass(SrcRC, AMDGPU::sub0);
9723 AMDGPU::sub0, SrcSubRC);
9725 AMDGPU::sub1, SrcSubRC);
9735 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9754 Offset == 0 &&
"Not implemented");
9777 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9787 .
addReg(Src.getReg(), {}, AMDGPU::sub0);
9790 .
addReg(Src.getReg(), {}, AMDGPU::sub0)
9796 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9815 const MCInstrDesc &InstDesc =
get(Opcode);
9817 bool IsCtlz = Opcode == AMDGPU::V_FFBH_U32_e32;
9820 Src.isReg() ? MRI.
getRegClass(Src.getReg()) : &AMDGPU::SGPR_32RegClass;
9822 RI.getSubRegisterClass(SrcRC, AMDGPU::sub0);
9824 MachineOperand SrcRegSub0 =
9826 MachineOperand SrcRegSub1 =
9840 .
addReg(IsCtlz ? MidReg1 : MidReg2);
9844 .
addReg(IsCtlz ? MidReg2 : MidReg1);
9848 addUsersToMoveToVALUWorklist(MidReg4, MRI, Worklist);
9851void SIInstrInfo::addUsersToMoveToVALUWorklist(
9855 MachineInstr &
UseMI = *MO.getParent();
9859 switch (
UseMI.getOpcode()) {
9862 case AMDGPU::SOFT_WQM:
9863 case AMDGPU::STRICT_WWM:
9864 case AMDGPU::STRICT_WQM:
9865 case AMDGPU::REG_SEQUENCE:
9867 case AMDGPU::INSERT_SUBREG:
9870 OpNo = MO.getOperandNo();
9877 if (!RI.hasVectorRegisters(OpRC))
9894 if (ST.useRealTrue16Insts()) {
9896 if (!Src0.
isReg() || !RI.isVGPR(MRI, Src0.
getReg())) {
9899 get(Src0.
isImm() ? AMDGPU::V_MOV_B32_e32 : AMDGPU::COPY), SrcReg0)
9905 if (!Src1.
isReg() || !RI.isVGPR(MRI, Src1.
getReg())) {
9908 get(Src1.
isImm() ? AMDGPU::V_MOV_B32_e32 : AMDGPU::COPY), SrcReg1)
9917 auto NewMI =
BuildMI(*
MBB, Inst,
DL,
get(AMDGPU::REG_SEQUENCE), ResultReg);
9919 case AMDGPU::S_PACK_LL_B32_B16:
9921 .addReg(SrcReg0, {},
9922 isSrc0Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9923 .addImm(AMDGPU::lo16)
9924 .addReg(SrcReg1, {},
9925 isSrc1Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9926 .addImm(AMDGPU::hi16);
9928 case AMDGPU::S_PACK_LH_B32_B16:
9930 .addReg(SrcReg0, {},
9931 isSrc0Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9932 .addImm(AMDGPU::lo16)
9933 .addReg(SrcReg1, {}, AMDGPU::hi16)
9934 .addImm(AMDGPU::hi16);
9936 case AMDGPU::S_PACK_HL_B32_B16:
9937 NewMI.addReg(SrcReg0, {}, AMDGPU::hi16)
9938 .addImm(AMDGPU::lo16)
9939 .addReg(SrcReg1, {},
9940 isSrc1Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9941 .addImm(AMDGPU::hi16);
9943 case AMDGPU::S_PACK_HH_B32_B16:
9944 NewMI.addReg(SrcReg0, {}, AMDGPU::hi16)
9945 .addImm(AMDGPU::lo16)
9946 .addReg(SrcReg1, {}, AMDGPU::hi16)
9947 .addImm(AMDGPU::hi16);
9955 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9960 case AMDGPU::S_PACK_LL_B32_B16: {
9979 case AMDGPU::S_PACK_LH_B32_B16: {
9989 case AMDGPU::S_PACK_HL_B32_B16: {
10000 case AMDGPU::S_PACK_HH_B32_B16: {
10020 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
10029 assert(
Op.isReg() &&
Op.getReg() == AMDGPU::SCC &&
Op.isDef() &&
10030 !
Op.isDead() &&
Op.getParent() == &SCCDefInst);
10031 SmallVector<MachineInstr *, 4> CopyToDelete;
10034 for (MachineInstr &
MI :
10038 int SCCIdx =
MI.findRegisterUseOperandIdx(AMDGPU::SCC, &RI,
false);
10039 if (SCCIdx != -1) {
10042 Register DestReg =
MI.getOperand(0).getReg();
10049 MI.getOperand(SCCIdx).setReg(NewCond);
10055 if (
MI.findRegisterDefOperandIdx(AMDGPU::SCC, &RI,
false,
false) != -1)
10058 for (
auto &Copy : CopyToDelete)
10059 Copy->eraseFromParent();
10067void SIInstrInfo::addSCCDefsToVALUWorklist(
MachineInstr *SCCUseInst,
10073 for (MachineInstr &
MI :
10076 if (
MI.modifiesRegister(AMDGPU::VCC, &RI))
10078 if (
MI.definesRegister(AMDGPU::SCC, &RI)) {
10095 case AMDGPU::REG_SEQUENCE:
10096 case AMDGPU::INSERT_SUBREG:
10098 case AMDGPU::SOFT_WQM:
10099 case AMDGPU::STRICT_WWM:
10100 case AMDGPU::STRICT_WQM: {
10102 if (RI.isAGPRClass(SrcRC)) {
10103 if (RI.isAGPRClass(NewDstRC))
10108 case AMDGPU::REG_SEQUENCE:
10109 case AMDGPU::INSERT_SUBREG:
10110 NewDstRC = RI.getEquivalentAGPRClass(NewDstRC);
10113 NewDstRC = RI.getEquivalentVGPRClass(NewDstRC);
10119 if (!RI.isSGPRClass(NewDstRC) || NewDstRC == &AMDGPU::VReg_1RegClass)
10122 NewDstRC = RI.getEquivalentVGPRClass(NewDstRC);
10136 int OpIndices[3])
const {
10137 const MCInstrDesc &
Desc =
MI.getDesc();
10155 for (
unsigned i = 0; i < 3; ++i) {
10156 int Idx = OpIndices[i];
10160 const MachineOperand &MO =
MI.getOperand(Idx);
10167 RI.getRegClass(getOpRegClassID(
Desc.operands()[Idx]));
10168 bool IsRequiredSGPR = RI.isSGPRClass(OpRC);
10169 if (IsRequiredSGPR)
10175 if (RI.isSGPRClass(RegRC))
10176 UsedSGPRs[i] =
Reg;
10192 if (UsedSGPRs[0]) {
10193 if (UsedSGPRs[0] == UsedSGPRs[1] || UsedSGPRs[0] == UsedSGPRs[2])
10194 SGPRReg = UsedSGPRs[0];
10197 if (!SGPRReg && UsedSGPRs[1]) {
10198 if (UsedSGPRs[1] == UsedSGPRs[2])
10199 SGPRReg = UsedSGPRs[1];
10206 AMDGPU::OpName OperandName)
const {
10207 if (OperandName == AMDGPU::OpName::NUM_OPERAND_NAMES)
10210 int Idx = AMDGPU::getNamedOperandIdx(
MI.getOpcode(), OperandName);
10214 return &
MI.getOperand(Idx);
10228 if (ST.isAmdHsaOS()) {
10231 RsrcDataFormat |= (1ULL << 56);
10236 RsrcDataFormat |= (2ULL << 59);
10239 return RsrcDataFormat;
10249 uint64_t EltSizeValue =
Log2_32(ST.getMaxPrivateElementSize(
true)) - 1;
10254 uint64_t IndexStride = ST.isWave64() ? 3 : 2;
10261 Rsrc23 &=
~AMDGPU::RSRC_DATA_FORMAT;
10267 unsigned Opc =
MI.getOpcode();
10273 return get(
Opc).mayLoad() &&
10280 if (!Addr || !Addr->
isFI())
10289 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::vdata);
10291 return MI.getOperand(VDataIdx).getReg();
10301 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::data);
10303 return MI.getOperand(DataIdx).getReg();
10324 if (!
MI.mayStore())
10337 unsigned Opc =
MI.getOpcode();
10339 unsigned DescSize =
Desc.getSize();
10344 unsigned Size = DescSize;
10348 if (
MI.isBranch() && ST.hasOffset3fBug())
10359 bool HasLiteral =
false;
10360 unsigned LiteralSize = 4;
10361 for (
int I = 0, E =
MI.getNumExplicitOperands();
I != E; ++
I) {
10366 if (ST.has64BitLiterals()) {
10367 switch (OpInfo.OperandType) {
10392 return HasLiteral ? DescSize + LiteralSize : DescSize;
10397 int VAddr0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr0);
10401 int RSrcIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::srsrc);
10402 return 8 + 4 * ((RSrcIdx - VAddr0Idx + 2) / 4);
10406 case TargetOpcode::BUNDLE:
10407 return getInstBundleSize(
MI);
10408 case TargetOpcode::INLINEASM:
10409 case TargetOpcode::INLINEASM_BR: {
10411 const char *AsmStr =
MI.getOperand(0).getSymbolName();
10415 if (
MI.isMetaInstruction())
10419 const auto *D16Info = AMDGPU::getT16D16Helper(
Opc);
10422 unsigned LoInstOpcode = D16Info->LoOp;
10424 DescSize =
Desc.getSize();
10428 if (
Opc == AMDGPU::V_FMA_MIX_F16_t16 ||
Opc == AMDGPU::V_FMA_MIX_BF16_t16) {
10431 DescSize =
Desc.getSize();
10440 if (
MI.isBranch() && ST.hasOffset3fBug())
10441 return InstSizeVerifyMode::NoVerify;
10442 return InstSizeVerifyMode::ExactSize;
10449 if (
MI.memoperands_empty())
10461 static const std::pair<int, const char *> TargetIndices[] = {
10501std::pair<unsigned, unsigned>
10508 static const std::pair<unsigned, const char *> TargetFlags[] = {
10526 static const std::pair<MachineMemOperand::Flags, const char *> TargetFlags[] =
10542 return AMDGPU::WWM_COPY;
10544 return AMDGPU::COPY;
10561 if (!IsLRSplitInst && Opcode != AMDGPU::IMPLICIT_DEF)
10565 if (RI.isSGPRClass(RI.getRegClassForReg(MRI, Reg)))
10566 return IsLRSplitInst;
10579 bool IsNullOrVectorRegister =
true;
10583 IsNullOrVectorRegister = !RI.isSGPRClass(RI.getRegClassForReg(MRI, Reg));
10586 return IsNullOrVectorRegister &&
10588 (!
MI.isTerminator() &&
MI.getOpcode() != AMDGPU::COPY &&
10589 MI.modifiesRegister(AMDGPU::EXEC, &RI)));
10597 if (ST.hasAddNoCarryInsts())
10613 if (ST.hasAddNoCarryInsts())
10617 Register UnusedCarry = !RS.isRegUsed(AMDGPU::VCC)
10619 : RS.scavengeRegisterBackwards(
10620 *RI.getBoolRC(),
I,
false,
10633 case AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR:
10634 case AMDGPU::SI_KILL_I1_TERMINATOR:
10643 case AMDGPU::SI_KILL_F32_COND_IMM_PSEUDO:
10644 return get(AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR);
10645 case AMDGPU::SI_KILL_I1_PSEUDO:
10646 return get(AMDGPU::SI_KILL_I1_TERMINATOR);
10658 const unsigned OffsetBits =
10660 return (1 << OffsetBits) - 1;
10664 if (!ST.isWave32())
10667 if (
MI.isInlineAsm())
10670 if (
MI.getNumOperands() <
MI.getDesc().getNumOperands())
10673 for (
auto &
Op :
MI.implicit_operands()) {
10674 if (
Op.isReg() &&
Op.getReg() == AMDGPU::VCC)
10675 Op.setReg(AMDGPU::VCC_LO);
10684 int Idx = AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::sbase);
10688 const int16_t RCID = getOpRegClassID(
MI.getDesc().operands()[Idx]);
10689 return RI.getRegClass(RCID)->hasSubClassEq(&AMDGPU::SGPR_128RegClass);
10705 if (
Imm > MaxImm) {
10706 if (
Imm <= MaxImm + 64) {
10708 Overflow =
Imm - MaxImm;
10723 Overflow =
High - Alignment.value();
10727 if (Overflow > 0) {
10735 if (ST.hasRestrictedSOffset())
10740 SOffset = Overflow;
10778 if (!ST.hasFlatInstOffsets())
10782 if (ST.hasFlatSegmentOffsetBug() && FlatVariant == FlatAddrSpace::FLAT &&
10787 if (ST.hasNegativeUnalignedScratchOffsetBug() &&
10788 FlatVariant == FlatAddrSpace::FlatScratch &&
Offset < 0 &&
10799std::pair<int64_t, int64_t>
10802 int64_t RemainderOffset = COffsetVal;
10803 int64_t ImmField = 0;
10808 if (AllowNegative) {
10810 int64_t
D = 1LL << NumBits;
10811 RemainderOffset = (COffsetVal /
D) *
D;
10812 ImmField = COffsetVal - RemainderOffset;
10814 if (ST.hasNegativeUnalignedScratchOffsetBug() &&
10816 (ImmField % 4) != 0) {
10818 RemainderOffset += ImmField % 4;
10819 ImmField -= ImmField % 4;
10821 }
else if (COffsetVal >= 0) {
10823 RemainderOffset = COffsetVal - ImmField;
10827 assert(RemainderOffset + ImmField == COffsetVal);
10828 return {ImmField, RemainderOffset};
10833 if (ST.hasNegativeScratchOffsetBug() &&
10841 switch (ST.getGeneration()) {
10875 case AMDGPU::V_MOVRELS_B32_dpp_gfx10:
10876 case AMDGPU::V_MOVRELS_B32_sdwa_gfx10:
10877 case AMDGPU::V_MOVRELD_B32_dpp_gfx10:
10878 case AMDGPU::V_MOVRELD_B32_sdwa_gfx10:
10879 case AMDGPU::V_MOVRELSD_B32_dpp_gfx10:
10880 case AMDGPU::V_MOVRELSD_B32_sdwa_gfx10:
10881 case AMDGPU::V_MOVRELSD_2_B32_dpp_gfx10:
10882 case AMDGPU::V_MOVRELSD_2_B32_sdwa_gfx10:
10889#define GENERATE_RENAMED_GFX9_CASES(OPCODE) \
10890 case OPCODE##_dpp: \
10891 case OPCODE##_e32: \
10892 case OPCODE##_e64: \
10893 case OPCODE##_e64_dpp: \
10894 case OPCODE##_sdwa:
10908 case AMDGPU::V_DIV_FIXUP_F16_gfx9_e64:
10909 case AMDGPU::V_DIV_FIXUP_F16_gfx9_fake16_e64:
10910 case AMDGPU::V_FMA_F16_gfx9_e64:
10911 case AMDGPU::V_FMA_F16_gfx9_fake16_e64:
10912 case AMDGPU::V_INTERP_P2_F16:
10913 case AMDGPU::V_MAD_F16_e64:
10914 case AMDGPU::V_MAD_U16_e64:
10915 case AMDGPU::V_MAD_I16_e64:
10924 "SIInsertWaitcnts should have promoted soft waitcnt instructions!");
10932 switch (ST.getGeneration()) {
10945 if (
isMAI(Opcode)) {
10959 if (MCOp == AMDGPU::INSTRUCTION_LIST_END && ST.hasGFX11_7Insts())
10962 if (MCOp == AMDGPU::INSTRUCTION_LIST_END && ST.hasGFX1250Insts())
10969 if (ST.hasGFX90AInsts()) {
10970 uint32_t NMCOp = AMDGPU::INSTRUCTION_LIST_END;
10971 if (ST.hasGFX940Insts())
10973 if (NMCOp == AMDGPU::INSTRUCTION_LIST_END)
10975 if (NMCOp == AMDGPU::INSTRUCTION_LIST_END)
10977 if (NMCOp != AMDGPU::INSTRUCTION_LIST_END)
10983 if (MCOp == AMDGPU::INSTRUCTION_LIST_END)
11002 for (
unsigned I = 0, E = (
MI.getNumOperands() - 1)/ 2;
I < E; ++
I)
11003 if (
MI.getOperand(1 + 2 *
I + 1).getImm() == SubReg) {
11004 auto &RegOp =
MI.getOperand(1 + 2 *
I);
11016 switch (
MI.getOpcode()) {
11018 case AMDGPU::REG_SEQUENCE:
11022 case AMDGPU::INSERT_SUBREG:
11023 if (RSR.
SubReg == (
unsigned)
MI.getOperand(3).getImm())
11040 if (!
P.Reg.isVirtual())
11045 while (
auto *
MI = DefInst) {
11047 switch (
MI->getOpcode()) {
11049 case AMDGPU::V_MOV_B32_e32: {
11050 auto &Op1 =
MI->getOperand(1);
11079 auto *DefBB =
DefMI.getParent();
11083 if (
UseMI.getParent() != DefBB)
11086 const int MaxInstScan = 20;
11090 auto E =
UseMI.getIterator();
11091 for (
auto I = std::next(
DefMI.getIterator());
I != E; ++
I) {
11092 if (
I->isDebugInstr())
11095 if (++NumInst > MaxInstScan)
11098 if (
I->modifiesRegister(AMDGPU::EXEC,
TRI))
11111 auto *DefBB =
DefMI.getParent();
11113 const int MaxUseScan = 10;
11117 auto &UseInst = *
Use.getParent();
11120 if (UseInst.getParent() != DefBB || UseInst.isPHI())
11123 if (++NumUse > MaxUseScan)
11130 const int MaxInstScan = 20;
11134 for (
auto I = std::next(
DefMI.getIterator()); ; ++
I) {
11137 if (
I->isDebugInstr())
11140 if (++NumInst > MaxInstScan)
11153 if (Reg == VReg && --NumUse == 0)
11155 }
else if (
TRI->regsOverlap(Reg, AMDGPU::EXEC))
11164 auto Cur =
MBB.begin();
11165 if (Cur !=
MBB.end())
11167 if (!Cur->isPHI() && Cur->readsRegister(Dst,
nullptr))
11170 }
while (Cur !=
MBB.end() && Cur != LastPHIIt);
11179 if (InsPt !=
MBB.end() &&
11180 (InsPt->getOpcode() == AMDGPU::SI_IF ||
11181 InsPt->getOpcode() == AMDGPU::SI_ELSE ||
11182 InsPt->getOpcode() == AMDGPU::SI_IF_BREAK) &&
11183 InsPt->definesRegister(Src,
nullptr)) {
11187 .
addReg(Src, {}, SrcSubReg)
11230 if (isFullCopyInstr(
MI)) {
11231 Register DstReg =
MI.getOperand(0).getReg();
11232 Register SrcReg =
MI.getOperand(1).getReg();
11254 unsigned *PredCost)
const {
11255 if (
MI.isBundle()) {
11258 unsigned Lat = 0,
Count = 0;
11259 for (++
I;
I != E &&
I->isBundledWithPred(); ++
I) {
11261 Lat = std::max(Lat, SchedModel.computeInstrLatency(&*
I));
11263 return Lat +
Count - 1;
11266 return SchedModel.computeInstrLatency(&
MI);
11270 if (!ST.hasGFX1250VALUBlockingCycles())
11277 if (
const auto *Entry = AMDGPU::getGFX1250BlockingCyclesInfo(
MI.getOpcode()))
11278 return Entry->GFX1250BlockingCycles;
11286 return *CallAddrOp;
11293 unsigned Opcode =
MI.getOpcode();
11295 auto HandleAddrSpaceCast = [
this, &MRI](
const MachineInstr &
MI) {
11301 unsigned SrcAS = SrcTy.getAddressSpace();
11304 ST.hasGloballyAddressableScratch()
11312 if (Opcode == TargetOpcode::G_ADDRSPACE_CAST)
11313 return HandleAddrSpaceCast(
MI);
11316 auto IID = GI->getIntrinsicID();
11323 case Intrinsic::amdgcn_if:
11324 case Intrinsic::amdgcn_else:
11338 if (Opcode == AMDGPU::G_LOAD || Opcode == AMDGPU::G_ZEXTLOAD ||
11339 Opcode == AMDGPU::G_SEXTLOAD) {
11340 if (
MI.memoperands_empty())
11344 return mmo->getAddrSpace() == AMDGPUAS::PRIVATE_ADDRESS ||
11345 mmo->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS;
11353 if (SIInstrInfo::isGenericAtomicRMWOpcode(Opcode) ||
11354 Opcode == AMDGPU::G_ATOMIC_CMPXCHG ||
11355 Opcode == AMDGPU::G_ATOMIC_CMPXCHG_WITH_SUCCESS ||
11361 if (Opcode == TargetOpcode::G_DYN_STACKALLOC)
11364 if (Opcode == AMDGPU::G_AMDGPU_WHOLE_WAVE_FUNC_SETUP)
11372 Formatter = std::make_unique<AMDGPUMIRFormatter>(ST);
11373 return Formatter.get();
11381 unsigned opcode =
MI.getOpcode();
11382 if (opcode == AMDGPU::V_READLANE_B32 ||
11383 opcode == AMDGPU::V_READFIRSTLANE_B32 ||
11384 opcode == AMDGPU::SI_RESTORE_S32_FROM_VGPR)
11389 if (
MI.isInlineAsm()) {
11395 if (!RC || !RI.isSGPRClass(RC))
11400 if (isCopyInstr(
MI)) {
11404 RI.getPhysRegBaseClass(srcOp.
getReg());
11412 if (
MI.isPreISelOpcode())
11427 if (
MI.memoperands_empty())
11431 return mmo->getAddrSpace() == AMDGPUAS::PRIVATE_ADDRESS ||
11432 mmo->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS;
11447 for (
unsigned I = 0, E =
MI.getNumOperands();
I != E; ++
I) {
11449 if (!
SrcOp.isReg())
11453 if (!Reg || !
SrcOp.readsReg())
11459 if (RegBank && RegBank->
getID() != AMDGPU::SGPRRegBankID)
11486 F,
"ds_ordered_count unsupported for this calling conv"));
11500 Register &SrcReg2, int64_t &CmpMask,
11501 int64_t &CmpValue)
const {
11502 if (!
MI.getOperand(0).isReg() ||
MI.getOperand(0).getSubReg())
11505 switch (
MI.getOpcode()) {
11508 case AMDGPU::S_CMP_EQ_U32:
11509 case AMDGPU::S_CMP_EQ_I32:
11510 case AMDGPU::S_CMP_LG_U32:
11511 case AMDGPU::S_CMP_LG_I32:
11512 case AMDGPU::S_CMP_LT_U32:
11513 case AMDGPU::S_CMP_LT_I32:
11514 case AMDGPU::S_CMP_GT_U32:
11515 case AMDGPU::S_CMP_GT_I32:
11516 case AMDGPU::S_CMP_LE_U32:
11517 case AMDGPU::S_CMP_LE_I32:
11518 case AMDGPU::S_CMP_GE_U32:
11519 case AMDGPU::S_CMP_GE_I32:
11520 case AMDGPU::S_CMP_EQ_U64:
11521 case AMDGPU::S_CMP_LG_U64:
11522 SrcReg =
MI.getOperand(0).getReg();
11523 if (
MI.getOperand(1).isReg()) {
11524 if (
MI.getOperand(1).getSubReg())
11526 SrcReg2 =
MI.getOperand(1).getReg();
11528 }
else if (
MI.getOperand(1).isImm()) {
11530 CmpValue =
MI.getOperand(1).getImm();
11536 case AMDGPU::S_CMPK_EQ_U32:
11537 case AMDGPU::S_CMPK_EQ_I32:
11538 case AMDGPU::S_CMPK_LG_U32:
11539 case AMDGPU::S_CMPK_LG_I32:
11540 case AMDGPU::S_CMPK_LT_U32:
11541 case AMDGPU::S_CMPK_LT_I32:
11542 case AMDGPU::S_CMPK_GT_U32:
11543 case AMDGPU::S_CMPK_GT_I32:
11544 case AMDGPU::S_CMPK_LE_U32:
11545 case AMDGPU::S_CMPK_LE_I32:
11546 case AMDGPU::S_CMPK_GE_U32:
11547 case AMDGPU::S_CMPK_GE_I32:
11548 SrcReg =
MI.getOperand(0).getReg();
11550 CmpValue =
MI.getOperand(1).getImm();
11560 if (S->isLiveIn(AMDGPU::SCC))
11569bool SIInstrInfo::invertSCCUse(
MachineInstr *SCCDef)
const {
11572 bool SCCIsDead =
false;
11575 constexpr unsigned ScanLimit = 12;
11576 unsigned Count = 0;
11577 for (MachineInstr &
MI :
11579 if (++
Count > ScanLimit)
11581 if (
MI.readsRegister(AMDGPU::SCC, &RI)) {
11582 if (
MI.getOpcode() == AMDGPU::S_CSELECT_B32 ||
11583 MI.getOpcode() == AMDGPU::S_CSELECT_B64 ||
11584 MI.getOpcode() == AMDGPU::S_CBRANCH_SCC0 ||
11585 MI.getOpcode() == AMDGPU::S_CBRANCH_SCC1)
11590 if (
MI.definesRegister(AMDGPU::SCC, &RI)) {
11603 for (MachineInstr *
MI : InvertInstr) {
11604 if (
MI->getOpcode() == AMDGPU::S_CSELECT_B32 ||
11605 MI->getOpcode() == AMDGPU::S_CSELECT_B64) {
11607 }
else if (
MI->getOpcode() == AMDGPU::S_CBRANCH_SCC0 ||
11608 MI->getOpcode() == AMDGPU::S_CBRANCH_SCC1) {
11609 MI->setDesc(
get(
MI->getOpcode() == AMDGPU::S_CBRANCH_SCC0
11610 ? AMDGPU::S_CBRANCH_SCC1
11611 : AMDGPU::S_CBRANCH_SCC0));
11624 bool NeedInversion)
const {
11625 MachineInstr *KillsSCC =
nullptr;
11630 if (
MI.modifiesRegister(AMDGPU::SCC, &RI))
11632 if (
MI.killsRegister(AMDGPU::SCC, &RI))
11635 if (NeedInversion && !invertSCCUse(SCCRedefine))
11637 if (MachineOperand *SccDef =
11639 SccDef->setIsDead(
false);
11648static std::optional<std::pair<int64_t, int64_t>>
11652 if (
Opc != AMDGPU::S_CSELECT_B32 &&
Opc != AMDGPU::S_CSELECT_B64)
11654 std::optional<int64_t>
A =
11658 std::optional<int64_t>
B =
11662 if (
Opc == AMDGPU::S_CSELECT_B32) {
11668 return std::pair(*
A, *
B);
11672 unsigned &NewDefOpc) {
11675 if (Def.getOpcode() != AMDGPU::S_ADD_I32 &&
11676 Def.getOpcode() != AMDGPU::S_ADD_U32)
11682 Def.getMF()->getSubtarget().getInstrInfo());
11684 auto Imm1 =
TII->getImmOrMaterializedImm(MRI, AddSrc1);
11685 auto Imm2 =
TII->getImmOrMaterializedImm(MRI, AddSrc2);
11686 if ((!Imm1 || *Imm1 != 1) && (!Imm2 || *Imm2 != 1))
11689 if (Def.getOpcode() == AMDGPU::S_ADD_I32) {
11691 Def.findRegisterDefOperand(AMDGPU::SCC,
nullptr);
11694 NewDefOpc = AMDGPU::S_ADD_U32;
11696 NeedInversion = !NeedInversion;
11701 Register SrcReg2, int64_t CmpMask,
11711 CmpValue = *ImmOpt;
11714 const auto optimizeCmpSelect = [&CmpInstr, SrcReg, CmpValue, MRI,
11715 this](
bool NeedInversion) ->
bool {
11720 unsigned NewDefOpc = Def->getOpcode();
11727 auto [
A,
B] = *Consts;
11728 int64_t
C = Def->getOpcode() == AMDGPU::S_CSELECT_B32 ?
Lo_32(CmpValue)
11731 NeedInversion = !NeedInversion;
11743 if (CmpValue != 0 ||
11749 if (!optimizeSCC(Def, &CmpInstr, NeedInversion))
11752 if (NewDefOpc != Def->getOpcode())
11753 Def->setDesc(
get(NewDefOpc));
11762 if (Def->getOpcode() == AMDGPU::S_OR_B32 &&
11769 if (Def1 && Def1->
getOpcode() == AMDGPU::COPY && Def2 &&
11778 auto [
A,
B] = *Consts;
11779 if (
A == 0 ||
B == 0)
11780 optimizeSCC(
Select, Def,
A == 0);
11789 const auto optimizeCmpAnd = [&CmpInstr, SrcReg, CmpValue, MRI,
11790 this](int64_t ExpectedValue,
unsigned SrcSize,
11791 bool IsReversible,
bool IsSigned) ->
bool {
11819 if (Def->getOpcode() != AMDGPU::S_AND_B32 &&
11820 Def->getOpcode() != AMDGPU::S_AND_B64)
11824 const auto isMask = [&Mask, SrcSize, MRI,
11836 SrcOp = &Def->getOperand(2);
11837 else if (isMask(&Def->getOperand(2)))
11838 SrcOp = &Def->getOperand(1);
11846 if (IsSigned && BitNo == SrcSize - 1)
11849 ExpectedValue <<= BitNo;
11851 bool IsReversedCC =
false;
11852 if (CmpValue != ExpectedValue) {
11855 IsReversedCC = CmpValue == (ExpectedValue ^ Mask);
11860 Register DefReg = Def->getOperand(0).getReg();
11861 if (IsReversedCC && !MRI->hasOneNonDBGUse(DefReg))
11864 if (!optimizeSCC(Def, &CmpInstr,
false))
11867 if (!MRI->use_nodbg_empty(DefReg)) {
11875 unsigned NewOpc = (SrcSize == 32) ? IsReversedCC ? AMDGPU::S_BITCMP0_B32
11876 : AMDGPU::S_BITCMP1_B32
11877 : IsReversedCC ? AMDGPU::S_BITCMP0_B64
11878 : AMDGPU::S_BITCMP1_B64;
11883 Def->eraseFromParent();
11891 case AMDGPU::S_CMP_EQ_U32:
11892 case AMDGPU::S_CMP_EQ_I32:
11893 case AMDGPU::S_CMPK_EQ_U32:
11894 case AMDGPU::S_CMPK_EQ_I32:
11895 return optimizeCmpAnd(1, 32,
true,
false) ||
11896 optimizeCmpSelect(
true);
11897 case AMDGPU::S_CMP_GE_U32:
11898 case AMDGPU::S_CMPK_GE_U32:
11899 return optimizeCmpAnd(1, 32,
false,
false);
11900 case AMDGPU::S_CMP_GE_I32:
11901 case AMDGPU::S_CMPK_GE_I32:
11902 return optimizeCmpAnd(1, 32,
false,
true);
11903 case AMDGPU::S_CMP_EQ_U64:
11904 return optimizeCmpAnd(1, 64,
true,
false) ||
11905 optimizeCmpSelect(
true);
11906 case AMDGPU::S_CMP_LG_U32:
11907 case AMDGPU::S_CMP_LG_I32:
11908 case AMDGPU::S_CMPK_LG_U32:
11909 case AMDGPU::S_CMPK_LG_I32:
11910 return optimizeCmpAnd(0, 32,
true,
false) ||
11911 optimizeCmpSelect(
false);
11912 case AMDGPU::S_CMP_GT_U32:
11913 case AMDGPU::S_CMPK_GT_U32:
11914 return optimizeCmpAnd(0, 32,
false,
false);
11915 case AMDGPU::S_CMP_GT_I32:
11916 case AMDGPU::S_CMPK_GT_I32:
11917 return optimizeCmpAnd(0, 32,
false,
true);
11918 case AMDGPU::S_CMP_LG_U64:
11919 return optimizeCmpAnd(0, 64,
true,
false) ||
11920 optimizeCmpSelect(
false);
11927 AMDGPU::OpName
OpName)
const {
11928 if (!ST.needsAlignedVGPRs())
11931 int OpNo = AMDGPU::getNamedOperandIdx(
MI.getOpcode(),
OpName);
11943 bool IsAGPR = RI.isAGPR(MRI, DataReg);
11945 IsAGPR ? &AMDGPU::AGPR_32RegClass : &AMDGPU::VGPR_32RegClass);
11949 : &AMDGPU::VReg_64_Align2RegClass);
11951 .
addReg(DataReg, {},
Op.getSubReg())
11956 Op.setSubReg(AMDGPU::sub0);
11961 if (!SchedModel.hasInstrSchedModel())
11967 unsigned RepeatRate = 0;
11969 PI = SchedModel.getWriteProcResBegin(SCDesc),
11970 PE = SchedModel.getWriteProcResEnd(SCDesc);
11972 RepeatRate = std::max(RepeatRate, (
unsigned)PI->ReleaseAtCycle);
11989 if (ST.hasGFX1250Insts())
11996 unsigned Opcode =
MI.getOpcode();
12002 Opcode == AMDGPU::V_ACCVGPR_WRITE_B32_e64 ||
12003 Opcode == AMDGPU::V_ACCVGPR_READ_B32_e64)
12006 if (!ST.hasGFX940Insts())
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
static const TargetRegisterClass * getRegClass(const MachineInstr &MI, Register Reg)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
Contains the definition of a TargetInstrInfo class that is common to all AMD GPUs.
AMDGPU Register Bank Select
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
AMD GCN specific subclass of TargetSubtarget.
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
const HexagonInstrInfo * TII
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static bool isUndef(const MachineInstr &MI)
TargetInstrInfo::RegSubRegPair RegSubRegPair
Register const TargetRegisterInfo * TRI
Promote Memory to Register
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
uint64_t IntrinsicInst * II
const SmallVectorImpl< MachineOperand > MachineBasicBlock * TBB
const SmallVectorImpl< MachineOperand > & Cond
This file declares the machine register scavenger class.
static cl::opt< bool > Fix16BitCopies("amdgpu-fix-16-bit-physreg-copies", cl::desc("Fix copies between 32 and 16 bit registers by extending to 32 bit"), cl::init(true), cl::ReallyHidden)
static void expandSGPRCopy(const SIInstrInfo &TII, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg, bool KillSrc, const TargetRegisterClass *RC, bool Forward)
static std::optional< std::pair< int64_t, int64_t > > getSelectConstants(const SIInstrInfo &TII, const MachineRegisterInfo &MRI, const MachineInstr &Sel)
If Sel is an S_CSELECT* of two different constants A and B, return them, truncated to the width of th...
static unsigned getNewFMAInst(const GCNSubtarget &ST, unsigned Opc)
static unsigned getIndirectSGPRWriteMovRelPseudo32(unsigned VecSize)
static bool compareMachineOp(const MachineOperand &Op0, const MachineOperand &Op1)
static bool isStride64(unsigned Opc)
static MachineBasicBlock * generateWaterFallLoop(const SIInstrInfo &TII, MachineInstr &MI, ArrayRef< MachineOperand * > ScalarOps, MachineDominatorTree *MDT, MachineBasicBlock::iterator Begin=nullptr, MachineBasicBlock::iterator End=nullptr, ArrayRef< Register > PhySGPRs={})
#define GENERATE_RENAMED_GFX9_CASES(OPCODE)
static std::tuple< unsigned, unsigned > extractRsrcPtr(const SIInstrInfo &TII, MachineInstr &MI, MachineOperand &Rsrc)
static unsigned VOP3OpIdxToSrcN(const MachineInstr &MI, unsigned OpIdx)
static bool followSubRegDef(MachineInstr &MI, TargetInstrInfo::RegSubRegPair &RSR)
static unsigned getIndirectSGPRWriteMovRelPseudo64(unsigned VecSize)
static MachineInstr * swapImmOperands(MachineInstr &MI, MachineOperand &NonRegOp1, MachineOperand &NonRegOp2)
static void copyFlagsToImplicitVCC(MachineInstr &MI, const MachineOperand &Orig)
static bool offsetsDoNotOverlap(LocationSize WidthA, int OffsetA, LocationSize WidthB, int OffsetB)
static void indirectCopyToAGPR(const SIInstrInfo &TII, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg, bool KillSrc, RegScavenger &RS, bool RegsOverlap, Register ImpUseSuperReg=Register())
Handle copying from SGPR to AGPR, or from AGPR to AGPR on GFX908.
static unsigned getWWMRegSpillSaveOpcode(unsigned Size, bool IsVectorSuperClass)
static bool memOpsHaveSameBaseOperands(ArrayRef< const MachineOperand * > BaseOps1, ArrayRef< const MachineOperand * > BaseOps2)
static unsigned getWWMRegSpillRestoreOpcode(unsigned Size, bool IsVectorSuperClass)
static unsigned getSGPRSpillSaveOpcode(unsigned Size, bool NeedsCFI)
static bool setsSCCIfResultIsZero(const MachineInstr &Def, bool &NeedInversion, unsigned &NewDefOpc)
static bool isSCCDeadOnExit(MachineBasicBlock *MBB)
static unsigned getIndirectVGPRWriteMovRelPseudoOpc(unsigned VecSize)
static unsigned subtargetEncodingFamily(const GCNSubtarget &ST)
static void preserveCondRegFlags(MachineOperand &CondReg, const MachineOperand &OrigCond)
static Register findImplicitSGPRRead(const MachineInstr &MI)
static unsigned getNewFMAAKInst(const GCNSubtarget &ST, unsigned Opc)
static cl::opt< unsigned > BranchOffsetBits("amdgpu-s-branch-bits", cl::ReallyHidden, cl::init(16), cl::desc("Restrict range of branch instructions (DEBUG)"))
static unsigned getAVSpillSaveOpcode(unsigned Size, bool NeedsCFI)
static bool memOpsHaveSameBasePtr(const MachineInstr &MI1, ArrayRef< const MachineOperand * > BaseOps1, const MachineInstr &MI2, ArrayRef< const MachineOperand * > BaseOps2)
static unsigned getSGPRSpillRestoreOpcode(unsigned Size)
static bool isRegOrFI(const MachineOperand &MO)
static bool isVCmp(const SIInstrInfo &TII, const MachineInstr &MI)
Return true if MI is a VALU comparison, i.e.
static unsigned getVGPRSpillSaveOpcode(unsigned Size, bool NeedsCFI)
static constexpr AMDGPU::OpName ModifierOpNames[]
static void reportIllegalCopy(const SIInstrInfo *TII, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg, bool KillSrc, const char *Msg="illegal VGPR to SGPR copy")
static MachineInstr * swapRegAndNonRegOperand(MachineInstr &MI, MachineOperand &RegOp, MachineOperand &NonRegOp)
static bool shouldReadExec(const MachineInstr &MI)
static unsigned getNewFMAMKInst(const GCNSubtarget &ST, unsigned Opc)
static bool isRenamedInGFX9(int Opcode)
static TargetInstrInfo::RegSubRegPair getRegOrUndef(const MachineOperand &RegOpnd)
static std::tuple< unsigned, unsigned, unsigned > splitGlobalAddressRelocFlags(const GCNSubtarget &ST, const MachineOperand &SrcOp)
static bool changesVGPRIndexingMode(const MachineInstr &MI)
static bool isSubRegOf(const SIRegisterInfo &TRI, const MachineOperand &SuperVec, const MachineOperand &SubReg)
static bool nodesHaveSameOperandValue(SDNode *N0, SDNode *N1, AMDGPU::OpName OpName)
Returns true if both nodes have the same value for the given operand Op, or if both nodes do not have...
static unsigned getNumOperandsNoGlue(SDNode *Node)
static bool canRemat(const MachineInstr &MI)
static unsigned getAVSpillRestoreOpcode(unsigned Size)
static void emitLoadScalarOpsFromVGPRLoop(const SIInstrInfo &TII, MachineRegisterInfo &MRI, MachineBasicBlock &PredBB, MachineBasicBlock &LoopBB, MachineBasicBlock &BodyBB, const DebugLoc &DL, ArrayRef< MachineOperand * > ScalarOps, ArrayRef< Register > PhySGPRs={})
static unsigned getVGPRSpillRestoreOpcode(unsigned Size)
Interface definition for SIInstrInfo.
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
const unsigned AndN2WrExecOpc
static const LaneMaskConstants & get(const GCNSubtarget &ST)
const unsigned XorTermOpc
const unsigned MovTermOpc
const unsigned OrSaveExecOpc
const unsigned AndSaveExecOpc
static LLVM_ABI Semantics SemanticsToEnum(const llvm::fltSemantics &Sem)
Class for arbitrary precision integers.
int64_t getSExtValue() const
Get sign extended value.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & front() const
Get the first element.
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
This class is the base class for the comparison instructions.
uint64_t getZExtValue() const
Opaque handle to a cycle within a GenericCycleInfo that wraps the cycle's preorder index.
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
Diagnostic information for unsupported feature in backend.
void changeImmediateDominator(DomTreeNodeBase< NodeT > *N, DomTreeNodeBase< NodeT > *NewIDom)
changeImmediateDominator - This method is used to update the dominator tree information when a node's...
DomTreeNodeBase< NodeT > * addNewBlock(NodeT *BB, NodeT *DomBB)
Add a new node to the dominator tree information.
bool properlyDominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
properlyDominates - Returns true iff A dominates B and A != B.
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
void getExitingBlocks(CycleRef C, SmallVectorImpl< BlockT * > &TmpStorage) const
Return all blocks of C that have a successor outside of C.
CycleRef getParentCycle(CycleRef C) const
bool contains(CycleRef Outer, CycleRef Inner) const
Returns true iff Outer contains Inner. O(1). Non-strict.
CycleRef getCycle(const BlockT *Block) const
Find the innermost cycle containing Block.
Itinerary data supplied by a subtarget to be used by a target.
constexpr unsigned getAddressSpace() const
This is an important class for using LLVM in a threaded context.
LiveInterval - This class represents the liveness of a register, or stack slot.
bool hasInterval(Register Reg) const
SlotIndex getInstructionIndex(const MachineInstr &Instr) const
Returns the base index of the given instruction.
LiveInterval & getInterval(Register Reg)
LLVM_ABI bool shrinkToUses(LiveInterval *li, SmallVectorImpl< MachineInstr * > *dead=nullptr)
After removing some uses of a register, shrink its live range to just the remaining uses.
SlotIndex ReplaceMachineInstrInMaps(MachineInstr &MI, MachineInstr &NewMI)
This class represents the liveness of a register, stack slot, etc.
static LocationSize precise(uint64_t Value)
TypeSize getValue() const
static const MCBinaryExpr * createAnd(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
static const MCBinaryExpr * createAShr(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
static const MCBinaryExpr * createSub(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
static LLVM_ABI const MCConstantExpr * create(int64_t Value, MCContext &Ctx, bool PrintInHex=false, unsigned SizeInBytes=0)
Describe properties that are true of each instruction in the target description file.
unsigned getNumOperands() const
Return the number of declared MachineOperands for this MachineInstruction.
ArrayRef< MCOperandInfo > operands() const
unsigned getNumDefs() const
Return the number of MachineOperands that are register definitions.
unsigned getSize() const
Return the number of bytes in the encoding of this instruction, or zero if the encoding size cannot b...
ArrayRef< MCPhysReg > implicit_uses() const
Return a list of registers that are potentially read by any instance of this machine instruction.
unsigned getOpcode() const
Return the opcode number for this descriptor.
This holds information about one operand of a machine instruction, indicating the register class for ...
uint8_t OperandType
Information about the type of the operand.
int16_t RegClass
This specifies the register class enumeration of the operand if the operand is a register.
bool hasSuperClassEq(const MCRegisterClass *RC) const
Returns true if RC is a super-class of or equal to this class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
Wrapper class representing physical registers. Should be passed by value.
static const MCSymbolRefExpr * create(const MCSymbol *Symbol, MCContext &Ctx, SMLoc Loc=SMLoc())
MCSymbol - Instances of this class represent a symbol name in the MC file, and MCSymbols are created ...
LLVM_ABI void setVariableValue(const MCExpr *Value)
Helper class for constructing bundles of MachineInstrs.
MachineBasicBlock::instr_iterator begin() const
Return an iterator to the first bundled instruction.
MIBundleBuilder & append(MachineInstr *MI)
Insert MI into MBB by appending it to the instructions in the bundle.
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
LLVM_ABI MCSymbol * getSymbol() const
Return the MCSymbol for this basic block.
void push_back(MachineInstr *MI)
LLVM_ABI LivenessQueryResult computeRegisterLiveness(const TargetRegisterInfo *TRI, MCRegister Reg, const_iterator Before, unsigned Neighborhood=10) const
Return whether (physical) register Reg has been defined and not killed as of just before Before.
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
MachineInstrBundleIterator< MachineInstr, true > reverse_iterator
Instructions::const_iterator const_instr_iterator
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
iterator_range< succ_iterator > successors()
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
MachineInstrBundleIterator< MachineInstr > iterator
@ LQR_Dead
Register is known to be fully dead.
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
bool isImmutableObjectIndex(int ObjectIdx) const
Returns true if the specified index corresponds to an immutable object.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
void push_back(MachineBasicBlock *MBB)
MCContext & getContext() const
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
MachineBasicBlock * CreateMachineBasicBlock(const BasicBlock *BB=nullptr, std::optional< UniqueBBID > BBID=std::nullopt)
CreateMachineInstr - Allocate a new MachineInstr.
void insert(iterator MBBI, MachineBasicBlock *MBB)
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addSym(MCSymbol *Sym, unsigned char TargetFlags=0) const
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addGlobalAddress(const GlobalValue *GV, int64_t Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
const MachineInstrBuilder & copyImplicitOps(const MachineInstr &OtherMI) const
Copy all the implicit operands from OtherMI onto this one.
const MachineInstrBuilder & addMemOperand(MachineMemOperand *MMO) const
MachineInstr * getInstr() const
If conversion operators fail, use this method to get the MachineInstr explicitly.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
bool mayLoadOrStore(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read or modify memory.
const MachineBasicBlock * getParent() const
LLVM_ABI void addImplicitDefUseOperands(MachineFunction &MF)
Add all implicit def and use operands to this instruction.
LLVM_ABI void addOperand(MachineFunction &MF, const MachineOperand &Op)
Add the specified operand to the instruction.
LLVM_ABI unsigned getNumExplicitOperands() const
Returns the number of non-implicit operands.
mop_range implicit_operands()
bool modifiesRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr modifies (fully define or partially define) the specified register.
bool mayLoad(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read memory.
LLVM_ABI bool hasUnmodeledSideEffects() const
Return true if this instruction has side effects that are not modeled by mayLoad / mayStore,...
void untieRegOperand(unsigned OpIdx)
Break any tie involving OpIdx.
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
LLVM_ABI void eraseFromBundle()
Unlink 'this' from its basic block and delete it.
bool hasOneMemOperand() const
Return true if this instruction has exactly one MachineMemOperand.
mop_range explicit_operands()
LLVM_ABI void tieOperands(unsigned DefIdx, unsigned UseIdx)
Add a tie between the register operands at DefIdx and UseIdx.
mmo_iterator memoperands_begin() const
Access to memory operands of the instruction.
LLVM_ABI bool hasOrderedMemoryRef() const
Return true if this instruction may have an ordered or volatile memory reference, or if the informati...
LLVM_ABI const MachineFunction * getMF() const
Return the function that contains the basic block that this instruction belongs to.
ArrayRef< MachineMemOperand * > memoperands() const
Access to memory operands of the instruction.
bool mayStore(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly modify memory.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
bool isMoveImmediate(QueryType Type=IgnoreBundle) const
Return true if this instruction is a move immediate (including conditional moves) instruction.
LLVM_ABI void removeOperand(unsigned OpNo)
Erase an operand from an instruction, leaving it with one fewer operand than it started with.
filtered_mop_range all_uses()
Returns an iterator range over all operands that are (explicit or implicit) register uses.
LLVM_ABI void setPostInstrSymbol(MachineFunction &MF, MCSymbol *Symbol)
Set a symbol that will be emitted just after the instruction itself.
LLVM_ABI void clearRegisterKills(Register Reg, const TargetRegisterInfo *RegInfo)
Clear all kill flags affecting Reg.
const MachineOperand & getOperand(unsigned i) const
uint32_t getFlags() const
Return the MI flags bitvector.
LLVM_ABI int findRegisterDefOperandIdx(Register Reg, const TargetRegisterInfo *TRI, bool isDead=false, bool Overlap=false) const
Returns the operand index that is a def of the specified register or -1 if it is not found.
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
MachineOperand * findRegisterDefOperand(Register Reg, const TargetRegisterInfo *TRI, bool isDead=false, bool Overlap=false)
Wrapper for findRegisterDefOperandIdx, it returns a pointer to the MachineOperand rather than an inde...
A description of a memory reference used in the backend.
unsigned getAddrSpace() const
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
MachineOperand class - Representation of each machine instruction operand.
void setSubReg(unsigned subReg)
unsigned getSubReg() const
LLVM_ABI unsigned getOperandNo() const
Returns the index of this operand in the instruction that it belongs to.
const GlobalValue * getGlobal() const
LLVM_ABI void ChangeToFrameIndex(int Idx, unsigned TargetFlags=0)
Replace this operand with a frame index.
void setImm(int64_t immVal)
bool isReg() const
isReg - Tests if this is a MO_Register operand.
void setIsDead(bool Val=true)
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
LLVM_ABI void ChangeToImmediate(int64_t ImmVal, unsigned TargetFlags=0)
ChangeToImmediate - Replace this operand with a new immediate operand of the specified value.
LLVM_ABI void ChangeToGA(const GlobalValue *GV, int64_t Offset, unsigned TargetFlags=0)
ChangeToGA - Replace this operand with a new global address operand.
void setIsKill(bool Val=true)
LLVM_ABI void ChangeToRegister(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isDebug=false)
ChangeToRegister - Replace this operand with a new register operand of the specified value.
void setOffset(int64_t Offset)
unsigned getTargetFlags() const
static MachineOperand CreateImm(int64_t Val)
bool isGlobal() const
isGlobal - Tests if this is a MO_GlobalAddress operand.
MachineOperandType getType() const
getType - Returns the MachineOperandType for this operand.
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
bool isTargetIndex() const
isTargetIndex - Tests if this is a MO_TargetIndex operand.
void setTargetFlags(unsigned F)
bool isFI() const
isFI - Tests if this is a MO_FrameIndex operand.
LLVM_ABI bool isIdenticalTo(const MachineOperand &Other) const
Returns true if this operand is identical to the specified operand except for liveness related flags ...
@ MO_Immediate
Immediate operand.
@ MO_Register
Register operand.
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
int64_t getOffset() const
Return the offset from the symbol in this operand.
bool isFPImm() const
isFPImm - Tests if this is a MO_FPImmediate operand.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
iterator_range< use_nodbg_iterator > use_nodbg_operands(Register Reg) const
bool use_nodbg_empty(Register RegNo) const
use_nodbg_empty - Return true if there are no non-Debug instructions using the specified register.
LLVM_ABI void moveOperands(MachineOperand *Dst, MachineOperand *Src, unsigned NumOps)
Move NumOps operands from Src to Dst, updating use-def lists as needed.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
bool reservedRegsFrozen() const
reservedRegsFrozen - Returns true after freezeReservedRegs() was called to ensure the set of reserved...
LLVM_ABI void clearVirtRegs()
clearVirtRegs - Remove all virtual registers (after physreg assignment).
void setRegAllocationHint(Register VReg, unsigned Type, Register PrefReg)
setRegAllocationHint - Specify a register allocation hint for the specified virtual register.
const MachineFunction & getMF() const
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
void setSimpleHint(Register VReg, Register PrefReg)
Specify the preferred (target independent) register allocation hint for the specified virtual registe...
const TargetRegisterInfo * getTargetRegisterInfo() const
LLVM_ABI bool isConstantPhysReg(MCRegister PhysReg) const
Returns true if PhysReg is unallocatable and constant throughout the function.
LLVM_ABI Register cloneVirtualRegister(Register VReg, StringRef Name="")
Create and return a new virtual register in the function with the same attributes as the given regist...
LLVM_ABI const TargetRegisterClass * constrainRegClass(Register Reg, const TargetRegisterClass *RC, unsigned MinNumRegs=0)
constrainRegClass - Constrain the register class of the specified virtual register to be a common sub...
iterator_range< use_iterator > use_operands(Register Reg) const
LLVM_ABI void removeRegOperandFromUseList(MachineOperand *MO)
Remove MO from its use-def list.
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
LLVM_ABI void addRegOperandToUseList(MachineOperand *MO)
Add MO to the linked list of operands for its register.
LLVM_ABI LLVM_READONLY MachineInstr * getUniqueVRegDef(Register Reg) const
getUniqueVRegDef - Return the unique machine instr that defines the specified virtual register or nul...
const RegisterBank & getRegBank(unsigned ID)
Get the register bank identified by ID.
This class implements the register bank concept.
unsigned getID() const
Get the identifier of this register bank.
Wrapper class representing virtual and physical registers.
MCRegister asMCReg() const
Utility to check-convert this value to a MCRegister.
constexpr bool isValid() const
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Represents one node in the SelectionDAG.
bool isMachineOpcode() const
Test if this node has a post-isel opcode, directly corresponding to a MachineInstr opcode.
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
unsigned getMachineOpcode() const
This may only be called if isMachineOpcode returns true.
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
bool isLegalMUBUFImmOffset(unsigned Imm) const
bool isInlineConstant(const APInt &Imm) const
void legalizeOperandsVOP3(MachineRegisterInfo &MRI, MachineInstr &MI) const
Fix operands in MI to satisfy constant bus requirements.
bool canAddToBBProlog(const MachineInstr &MI) const
static bool isDS(const MachineInstr &MI)
MachineBasicBlock * legalizeOperands(MachineInstr &MI, MachineDominatorTree *MDT=nullptr) const
Legalize all operands in this instruction.
bool areLoadsFromSameBasePtr(SDNode *Load0, SDNode *Load1, int64_t &Offset0, int64_t &Offset1) const override
unsigned getLiveRangeSplitOpcode(Register Reg, const MachineFunction &MF) const override
bool getMemOperandsWithOffsetWidth(const MachineInstr &LdSt, SmallVectorImpl< const MachineOperand * > &BaseOps, int64_t &Offset, bool &OffsetIsScalable, LocationSize &Width, const TargetRegisterInfo *TRI) const final
unsigned getInstSizeInBytes(const MachineInstr &MI) const override
static bool isNeverUniform(const MachineInstr &MI)
bool isXDLWMMA(const MachineInstr &MI) const
bool isBasicBlockPrologue(const MachineInstr &MI, Register Reg=Register()) const override
uint64_t getDefaultRsrcDataFormat() const
static bool isSOPP(const MachineInstr &MI)
bool mayAccessScratch(const MachineInstr &MI) const
bool isIGLP(unsigned Opcode) const
static bool isFLATScratch(const MachineInstr &MI)
bool isLegalFLATOffset(int64_t Offset, unsigned AddrSpace, AMDGPU::FlatAddrSpace FlatVariant) const
Returns if Offset is legal for the subtarget as the offset to a FLAT encoded instruction with the giv...
const MCInstrDesc & getIndirectRegWriteMovRelPseudo(unsigned VecSize, unsigned EltSize, bool IsSGPR) const
MachineInstrBuilder getAddNoCarry(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register DestReg) const
Return a partially built integer add instruction without carry.
bool mayAccessFlatAddressSpace(const MachineInstr &MI) const
bool shouldScheduleLoadsNear(SDNode *Load0, SDNode *Load1, int64_t Offset0, int64_t Offset1, unsigned NumLoads) const override
bool splitMUBUFOffset(uint32_t Imm, uint32_t &SOffset, uint32_t &ImmOffset, Align Alignment=Align(4)) const
bool isIgnorableUse(const MachineInstr &MI, unsigned OpIdx) const override
ArrayRef< std::pair< unsigned, const char * > > getSerializableDirectMachineOperandTargetFlags() const override
void moveToVALU(SIInstrWorklist &Worklist, MachineDominatorTree *MDT) const
Replace the instructions opcode with the equivalent VALU opcode.
static bool isSMRD(const MachineInstr &MI)
unsigned getGFX1250BlockingCyclesTable(const MachineInstr &MI) const
GFX1250 blocking-cycles table lookup with no occupancy subtarget gate.
void restoreExec(MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, Register Reg, SlotIndexes *Indexes=nullptr) const
void storeRegToStackSlotCFI(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register SrcReg, bool isKill, int FrameIndex, const TargetRegisterClass *RC) const
bool usesConstantBus(const MachineRegisterInfo &MRI, const MachineOperand &MO, const MCOperandInfo &OpInfo) const
Returns true if this operand uses the constant bus.
static unsigned getMaxMUBUFImmOffset(const GCNSubtarget &ST)
static unsigned getFoldableCopySrcIdx(const MachineInstr &MI)
unsigned getOpSize(uint32_t Opcode, unsigned OpNo) const
Return the size in bytes of the operand OpNo on the given.
void legalizeOperandsFLAT(MachineRegisterInfo &MRI, MachineInstr &MI) const
bool optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg, Register SrcReg2, int64_t CmpMask, int64_t CmpValue, const MachineRegisterInfo *MRI) const override
static std::optional< int64_t > extractSubregFromImm(int64_t ImmVal, unsigned SubRegIndex)
Return the extracted immediate value in a subregister use from a constant materialized in a super reg...
Register isStoreToStackSlot(const MachineInstr &MI, int &FrameIndex) const override
static bool isMTBUF(const MachineInstr &MI)
const MCInstrDesc & getIndirectGPRIDXPseudo(unsigned VecSize, bool IsIndirectSrc) const
static bool isDGEMM(unsigned Opcode)
static bool isEXP(const MachineInstr &MI)
static bool isSALU(const MachineInstr &MI)
static bool setsSCCIfResultIsNonZero(const MachineInstr &MI)
const MIRFormatter * getMIRFormatter() const override
static bool isXcntDrain(const MachineInstr &MI)
True if MI implicitly drains XCNT.
void legalizeGenericOperand(MachineBasicBlock &InsertMBB, MachineBasicBlock::iterator I, const TargetRegisterClass *DstRC, MachineOperand &Op, MachineRegisterInfo &MRI, const DebugLoc &DL) const
MachineInstr * buildShrunkInst(MachineInstr &MI, unsigned NewOpcode) const
static bool isVOP2(const MachineInstr &MI)
bool analyzeBranch(MachineBasicBlock &MBB, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, SmallVectorImpl< MachineOperand > &Cond, bool AllowModify=false) const override
static bool isSDWA(const MachineInstr &MI)
const MCInstrDesc & getKillTerminatorFromPseudo(unsigned Opcode) const
void insertNoops(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, unsigned Quantity) const override
static bool isGather4(const MachineInstr &MI)
MachineInstr * getWholeWaveFunctionSetup(MachineFunction &MF) const
bool isLegalVSrcOperand(const MachineRegisterInfo &MRI, const MCOperandInfo &OpInfo, const MachineOperand &MO) const
Check if MO would be a valid operand for the given operand definition OpInfo.
static bool isDOT(const MachineInstr &MI)
std::unique_ptr< PipelinerLoopInfo > analyzeLoopForPipelining(MachineBasicBlock *LoopBB) const override
InstSizeVerifyMode getInstSizeVerifyMode(const MachineInstr &MI) const override
MachineInstr * createPHISourceCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, unsigned SrcSubReg, Register Dst) const override
bool hasModifiers(unsigned Opcode) const
Return true if this instruction has any modifiers.
bool shouldClusterMemOps(ArrayRef< const MachineOperand * > BaseOps1, int64_t Offset1, bool OffsetIsScalable1, ArrayRef< const MachineOperand * > BaseOps2, int64_t Offset2, bool OffsetIsScalable2, unsigned ClusterSize, unsigned NumBytes) const override
static bool isSWMMAC(const MachineInstr &MI)
ScheduleHazardRecognizer * CreateTargetMIHazardRecognizer(const InstrItineraryData *II, const ScheduleDAGMI *DAG) const override
bool isHighLatencyDef(int Opc) const override
void legalizeOpWithMove(MachineInstr &MI, unsigned OpIdx) const
Legalize the OpIndex operand of this instruction by inserting a MOV.
bool reverseBranchCondition(SmallVectorImpl< MachineOperand > &Cond) const override
static bool isVOPC(const MachineInstr &MI)
void removeModOperands(MachineInstr &MI) const
unsigned getRepeatRate(const MachineInstr &MI) const
Get the repeat rate for a VALU instruction from the scheduling model.
unsigned getVectorRegSpillRestoreOpcode(Register Reg, const TargetRegisterClass *RC, unsigned Size, const SIMachineFunctionInfo &MFI) const
bool isLegalSingleSGPRReadInstOperand(const MachineRegisterInfo &MRI, const MachineInstr &MI, unsigned SrcN, const MachineOperand *MO=nullptr) const
Check if MO would be a legal operand for a single-SGPR-read instruction.
bool isXDL(const MachineInstr &MI) const
Register isStackAccess(const MachineInstr &MI, int &FrameIndex, TypeSize &MemBytes) const
static bool isVIMAGE(const MachineInstr &MI)
void enforceOperandRCAlignment(MachineInstr &MI, AMDGPU::OpName OpName) const
static bool isSOP2(const MachineInstr &MI)
static bool isGWS(const MachineInstr &MI)
bool hasRAWDependency(const MachineInstr &FirstMI, const MachineInstr &SecondMI) const
bool isLegalAV64PseudoImm(uint64_t Imm) const
Check if this immediate value can be used for AV_MOV_B64_IMM_PSEUDO.
bool isNeverCoissue(MachineInstr &MI) const
static bool isBUF(const MachineInstr &MI)
bool isNonCommutableDPP(const MachineInstr &MI) const
void handleCopyToPhysHelper(SIInstrWorklist &Worklist, Register DstReg, MachineInstr &Inst, MachineRegisterInfo &MRI, DenseMap< MachineInstr *, V2PhysSCopyInfo > &WaterFalls, DenseMap< MachineInstr *, bool > &V2SPhyCopiesToErase) const
bool hasModifiersSet(const MachineInstr &MI, AMDGPU::OpName OpName) const
bool isLegalToSwap(const MachineInstr &MI, unsigned fromIdx, unsigned toIdx) const
static bool isFLATGlobal(const MachineInstr &MI)
MachineInstr * foldMemoryOperandImpl(MachineFunction &MF, MachineInstr &MI, ArrayRef< unsigned > Ops, int FrameIndex, MachineInstr *&CopyMI, LiveIntervals *LIS=nullptr, VirtRegMap *VRM=nullptr) const override
bool isGlobalMemoryObject(const MachineInstr *MI) const override
static bool isVSAMPLE(const MachineInstr &MI)
bool isBufferSMRD(const MachineInstr &MI) const
static bool isKillTerminator(unsigned Opcode)
bool isVOPDAntidependencyAllowed(const MachineInstr &MI) const
If OpX is multicycle, anti-dependencies are not allowed.
bool findCommutedOpIndices(const MachineInstr &MI, unsigned &SrcOpIdx0, unsigned &SrcOpIdx1) const override
void insertScratchExecCopy(MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, Register Reg, bool IsSCCLive, SlotIndexes *Indexes=nullptr) const
bool hasVALU32BitEncoding(unsigned Opcode) const
Return true if this 64-bit VALU instruction has a 32-bit encoding.
unsigned getBlockingCycles(const MachineInstr &MI) const
unsigned getMovOpcode(const TargetRegisterClass *DstRC) const
Register isSGPRStackAccess(const MachineInstr &MI, int &FrameIndex, TypeSize &MemBytes) const
unsigned buildExtractSubReg(MachineBasicBlock::iterator MI, MachineRegisterInfo &MRI, const MachineOperand &SuperReg, const TargetRegisterClass *SuperRC, unsigned SubIdx, const TargetRegisterClass *SubRC) const
void legalizeOperandsVOP2(MachineRegisterInfo &MRI, MachineInstr &MI) const
Legalize operands in MI by either commuting it or inserting a copy of src1.
static bool isVPermPk16(unsigned Opcode)
static bool isVALU(const MachineInstr &MI, bool AllowLDSDMA)
bool foldImmediate(MachineInstr &UseMI, MachineInstr &DefMI, Register Reg, MachineRegisterInfo *MRI) const final
static bool isTRANS(const MachineInstr &MI)
static bool isImage(const MachineInstr &MI)
static bool isSOPK(const MachineInstr &MI)
const TargetRegisterClass * getOpRegClass(const MachineInstr &MI, unsigned OpNo) const
Return the correct register class for OpNo.
MachineBasicBlock * insertSimulatedTrap(MachineRegisterInfo &MRI, MachineBasicBlock &MBB, MachineInstr &MI, const DebugLoc &DL) const
Build instructions that simulate the behavior of a s_trap 2 instructions for hardware (namely,...
static unsigned getNonSoftWaitcntOpcode(unsigned Opcode)
static unsigned getDSShaderTypeValue(const MachineFunction &MF)
static bool isFoldableCopy(const MachineInstr &MI)
static bool isMUBUF(const MachineInstr &MI)
bool expandPostRAPseudo(MachineInstr &MI) const override
bool analyzeCompare(const MachineInstr &MI, Register &SrcReg, Register &SrcReg2, int64_t &CmpMask, int64_t &CmpValue) const override
void createWaterFallForSiCall(MachineInstr *MI, MachineDominatorTree *MDT, ArrayRef< MachineOperand * > ScalarOps, ArrayRef< Register > PhySGPRs={}) const
Wrapper function for generating waterfall for instruction MI This function take into consideration of...
void loadRegFromStackSlot(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg, int FrameIndex, const TargetRegisterClass *RC, Register VReg, unsigned SubReg=0, MachineInstr::MIFlag Flags=MachineInstr::NoFlags) const override
static bool isSegmentSpecificFLAT(const MachineInstr &MI)
bool isReMaterializableImpl(const MachineInstr &MI) const override
static bool isVOP3(const MCInstrDesc &Desc)
Register isLoadFromStackSlot(const MachineInstr &MI, int &FrameIndex) const override
bool physRegUsesConstantBus(const MachineOperand &Reg) const
void insertSelect(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register DstReg, ArrayRef< MachineOperand > Cond, Register TrueReg, Register FalseReg) const override
bool mayAccessVMEMThroughFlat(const MachineInstr &MI) const
static bool isDPP(const MachineInstr &MI)
bool analyzeBranchImpl(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, SmallVectorImpl< MachineOperand > &Cond, bool AllowModify) const
static bool isMFMA(const MachineInstr &MI)
bool isLowLatencyInstruction(const MachineInstr &MI) const
std::optional< DestSourcePair > isCopyInstrImpl(const MachineInstr &MI) const override
If the specific machine instruction is a instruction that moves/copies value from one register to ano...
void mutateAndCleanupImplicit(MachineInstr &MI, const MCInstrDesc &NewDesc) const
ValueUniformity getGenericValueUniformity(const MachineInstr &MI) const
static bool isMAI(const MCInstrDesc &Desc)
static bool isSrc1DPPRevOpcode(const GCNSubtarget &ST, uint32_t Opcode)
void reMaterialize(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg, unsigned SubIdx, const MachineInstr &Orig, LaneBitmask UsedLanes=LaneBitmask::getAll()) const override
static bool usesLGKM_CNT(const MachineInstr &MI)
void legalizeOperandsVALUt16(MachineInstr &Inst, MachineRegisterInfo &MRI) const
Fix operands in Inst to fix 16bit SALU to VALU lowering.
bool isImmOperandLegal(const MCInstrDesc &InstDesc, unsigned OpNo, const MachineOperand &MO) const
bool canShrink(const MachineInstr &MI, const MachineRegisterInfo &MRI) const
const MachineOperand & getCalleeOperand(const MachineInstr &MI) const override
bool isAsmOnlyOpcode(int MCOp) const
Check if this instruction should only be used by assembler.
bool isAlwaysGDS(uint32_t Opcode) const
static bool isVGPRSpill(const MachineInstr &MI)
ScheduleHazardRecognizer * CreateTargetPostRAHazardRecognizer(const InstrItineraryData *II, const ScheduleDAG *DAG) const override
This is used by the post-RA scheduler (SchedulePostRAList.cpp).
bool verifyInstruction(const MachineInstr &MI, StringRef &ErrInfo) const override
unsigned getInstrLatency(const InstrItineraryData *ItinData, const MachineInstr &MI, unsigned *PredCost=nullptr) const override
unsigned getVectorRegSpillSaveOpcode(Register Reg, const TargetRegisterClass *RC, unsigned Size, const SIMachineFunctionInfo &MFI, bool NeedsCFI) const
int64_t getNamedImmOperand(const MachineInstr &MI, AMDGPU::OpName OperandName) const
Get required immediate operand.
ArrayRef< std::pair< int, const char * > > getSerializableTargetIndices() const override
bool regUsesConstantBus(const MachineOperand &Reg, const MachineRegisterInfo &MRI) const
static bool isMIMG(const MachineInstr &MI)
MachineOperand buildExtractSubRegOrImm(MachineBasicBlock::iterator MI, MachineRegisterInfo &MRI, const MachineOperand &SuperReg, const TargetRegisterClass *SuperRC, unsigned SubIdx, const TargetRegisterClass *SubRC) const
bool isSchedulingBoundary(const MachineInstr &MI, const MachineBasicBlock *MBB, const MachineFunction &MF) const override
bool isLegalRegOperand(const MachineRegisterInfo &MRI, const MCOperandInfo &OpInfo, const MachineOperand &MO) const
Check if MO (a register operand) is a legal register for the given operand description or operand ind...
static unsigned getNumWaitStates(const MachineInstr &MI)
Return the number of wait states that result from executing this instruction.
unsigned getVALUOp(const MachineInstr &MI) const
static bool modifiesModeRegister(const MachineInstr &MI)
Return true if the instruction modifies the mode register.q.
Register readlaneVGPRToSGPR(Register SrcReg, MachineInstr &UseMI, MachineRegisterInfo &MRI, const TargetRegisterClass *DstRC=nullptr) const
Copy a value from a VGPR (SrcReg) to SGPR.
bool hasDivergentBranch(const MachineBasicBlock *MBB) const
Return whether the block terminate with divergent branch.
std::pair< int64_t, int64_t > splitFlatOffset(int64_t COffsetVal, unsigned AddrSpace, AMDGPU::FlatAddrSpace FlatVariant) const
Split COffsetVal into {immediate offset field, remainder offset} values.
unsigned removeBranch(MachineBasicBlock &MBB, int *BytesRemoved=nullptr) const override
void fixImplicitOperands(MachineInstr &MI) const
bool moveFlatAddrToVGPR(MachineInstr &Inst) const
Change SADDR form of a FLAT Inst to its VADDR form if saddr operand was moved to VGPR.
void copyPhysReg(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, Register DestReg, Register SrcReg, bool KillSrc, bool RenamableDest=false, bool RenamableSrc=false) const override
bool isMaskedByExec(Register Reg, const MachineInstr &Use, const MachineRegisterInfo &MRI, unsigned Depth=0) const
Return true if Reg is a lane mask that already has 0 in every bit corresponding to a lane that is ina...
void createReadFirstLaneFromCopyToPhysReg(MachineRegisterInfo &MRI, Register DstReg, MachineInstr &Inst) const
bool swapSourceModifiers(MachineInstr &MI, MachineOperand &Src0, AMDGPU::OpName Src0OpName, MachineOperand &Src1, AMDGPU::OpName Src1OpName) const
MachineBasicBlock * getBranchDestBlock(const MachineInstr &MI) const override
bool hasUnwantedEffectsWhenEXECEmpty(const MachineInstr &MI) const
This function is used to determine if an instruction can be safely executed under EXEC = 0 without ha...
bool getConstValDefinedInReg(const MachineInstr &MI, const Register Reg, int64_t &ImmVal) const override
static bool isAtomic(const MachineInstr &MI)
bool canInsertSelect(const MachineBasicBlock &MBB, ArrayRef< MachineOperand > Cond, Register DstReg, Register TrueReg, Register FalseReg, int &CondCycles, int &TrueCycles, int &FalseCycles) const override
bool isLiteralOperandLegal(const MCInstrDesc &InstDesc, const MCOperandInfo &OpInfo) const
static bool isWWMRegSpillOpcode(uint32_t Opcode)
static bool sopkIsZext(unsigned Opcode)
static bool isSGPRSpill(const MachineInstr &MI)
static bool isWMMA(const MachineInstr &MI)
ArrayRef< std::pair< MachineMemOperand::Flags, const char * > > getSerializableMachineMemOperandTargetFlags() const override
bool mayReadEXEC(const MachineRegisterInfo &MRI, const MachineInstr &MI) const
Returns true if the instruction could potentially depend on the value of exec.
void legalizeOperandsSMRD(MachineRegisterInfo &MRI, MachineInstr &MI) const
bool isBranchOffsetInRange(unsigned BranchOpc, int64_t BrOffset) const override
unsigned insertBranch(MachineBasicBlock &MBB, MachineBasicBlock *TBB, MachineBasicBlock *FBB, ArrayRef< MachineOperand > Cond, const DebugLoc &DL, int *BytesAdded=nullptr) const override
void insertNoop(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI) const override
std::pair< MachineInstr *, MachineInstr * > expandMovDPP64(MachineInstr &MI) const
static bool isSOPC(const MachineInstr &MI)
static bool isFLAT(const MachineInstr &MI)
bool isBarrier(unsigned Opcode) const
MachineInstr * commuteInstructionImpl(MachineInstr &MI, bool NewMI, unsigned OpIdx0, unsigned OpIdx1) const override
bool mayAccessLDSThroughFlat(const MachineInstr &MI, bool TgSplit) const
int pseudoToMCOpcode(int Opcode) const
Return a target-specific opcode if Opcode is a pseudo instruction.
const MCInstrDesc & getMCOpcodeFromPseudo(unsigned Opcode) const
Return the descriptor of the target-specific machine instruction that corresponds to the specified ps...
static bool usesVM_CNT(const MachineInstr &MI)
MachineInstr * createPHIDestinationCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, Register Dst) const override
static bool isFixedSize(const MachineInstr &MI)
bool isSafeToSink(MachineInstr &MI, MachineBasicBlock *SuccToSinkTo, MachineCycleInfo *CI) const override
LLVM_READONLY int commuteOpcode(unsigned Opc) const
ValueUniformity getValueUniformity(const MachineInstr &MI) const final
uint64_t getScratchRsrcWords23() const
LLVM_READONLY MachineOperand * getNamedOperand(MachineInstr &MI, AMDGPU::OpName OperandName) const
Returns the operand named Op.
MachineInstr * convertToThreeAddress(MachineInstr &MI, LiveIntervals *LIS) const override
std::pair< unsigned, unsigned > decomposeMachineOperandsTargetFlags(unsigned TF) const override
bool areMemAccessesTriviallyDisjoint(const MachineInstr &MIa, const MachineInstr &MIb) const override
bool isOperandLegal(const MachineInstr &MI, unsigned OpIdx, const MachineOperand *MO=nullptr) const
Check if MO is a legal operand if it was the OpIdx Operand for MI.
void storeRegToStackSlot(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register SrcReg, bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg, MachineInstr::MIFlag Flags=MachineInstr::NoFlags) const override
bool allowNegativeFlatOffset(AMDGPU::FlatAddrSpace FlatVariant) const
Returns true if negative offsets are allowed for the given FlatVariant.
void moveToVALUImpl(SIInstrWorklist &Worklist, MachineDominatorTree *MDT, MachineInstr &Inst, DenseMap< MachineInstr *, V2PhysSCopyInfo > &WaterFalls, DenseMap< MachineInstr *, bool > &V2SPhyCopiesToErase) const
static bool isLDSDMA(const MachineInstr &MI)
static bool isVOP1(const MachineInstr &MI)
SIInstrInfo(const GCNSubtarget &ST)
std::optional< int64_t > getImmOrMaterializedImm(const MachineRegisterInfo &MRI, const MachineOperand &Op, MachineInstr **DefMI=nullptr) const
void insertIndirectBranch(MachineBasicBlock &MBB, MachineBasicBlock &NewDestBB, MachineBasicBlock &RestoreBB, const DebugLoc &DL, int64_t BrOffset, RegScavenger *RS) const override
bool hasAnyModifiersSet(const MachineInstr &MI) const
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
Register getLongBranchReservedReg() const
bool isWholeWaveFunction() const
Register getStackPtrOffsetReg() const
unsigned getMaxMemoryClusterDWords() const
void setHasSpilledVGPRs(bool Spill=true)
bool isWWMReg(Register Reg) const
bool checkFlag(Register Reg, uint8_t Flag) const
void setHasSpilledSGPRs(bool Spill=true)
unsigned getScratchReservedForDynamicVGPRs() const
static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs=1)
ArrayRef< int16_t > getRegSplitParts(const TargetRegisterClass *RC, unsigned EltSize) const
unsigned getHWRegIndex(MCRegister Reg) const
bool isSGPRReg(const MachineRegisterInfo &MRI, Register Reg) const
unsigned getRegPressureLimit(const TargetRegisterClass *RC, MachineFunction &MF) const override
unsigned getChannelFromSubReg(unsigned SubReg) const
static bool isSGPRClass(const TargetRegisterClass *RC)
static bool isAGPRClass(const TargetRegisterClass *RC)
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
virtual bool hasVRegLiveness() const
Return true if this DAG supports VReg liveness and RegPressure.
MachineFunction & MF
Machine function.
HazardRecognizer - This determines whether or not an instruction can be issued this cycle,...
SlotIndex - An opaque wrapper around machine indexes.
SlotIndex getRegSlot(bool EC=false) const
Returns the register use/def slot in the current instruction for a normal or early-clobber def.
SlotIndex insertMachineInstrInMaps(MachineInstr &MI, bool Late=false)
Insert the given machine instruction into the mapping.
Implements a dense probed hash-table based set with some number of buckets stored inline.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Object returned by analyzeLoopForPipelining.
virtual ScheduleHazardRecognizer * CreateTargetMIHazardRecognizer(const InstrItineraryData *, const ScheduleDAGMI *DAG) const
Allocate and return a hazard recognizer to use for this target when scheduling the machine instructio...
virtual MachineInstr * createPHIDestinationCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, Register Dst) const
During PHI eleimination lets target to make necessary checks and insert the copy to the PHI destinati...
virtual const MachineOperand & getCalleeOperand(const MachineInstr &MI) const
Returns the callee operand from the given MI.
virtual void reMaterialize(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg, unsigned SubIdx, const MachineInstr &Orig, LaneBitmask UsedLanes=LaneBitmask::getAll()) const
Re-issue the specified 'original' instruction at the specific location targeting a new destination re...
virtual MachineInstr * createPHISourceCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, unsigned SrcSubReg, Register Dst) const
During PHI eleimination lets target to make necessary checks and insert the copy to the PHI destinati...
virtual MachineInstr * commuteInstructionImpl(MachineInstr &MI, bool NewMI, unsigned OpIdx1, unsigned OpIdx2) const
This method commutes the operands of the given machine instruction MI.
virtual bool isGlobalMemoryObject(const MachineInstr *MI) const
Returns true if MI is an instruction we are unable to reason about (like a call or something with unm...
virtual bool expandPostRAPseudo(MachineInstr &MI) const
This function is called for all pseudo instructions that remain after register allocation.
const MCAsmInfo & getMCAsmInfo() const
Return target specific asm information.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
const MCWriteProcResEntry * ProcResIter
static constexpr TypeSize getFixed(ScalarTy ExactSize)
A Use represents the edge between a Value definition and its users.
std::pair< iterator, bool > insert(const ValueT &V)
size_type count(const_arg_type_t< ValueT > V) const
Return 1 if the specified key is in the set, 0 otherwise.
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ REGION_ADDRESS
Address space for region memory. (GDS)
@ LOCAL_ADDRESS
Address space for local memory.
@ FLAT_ADDRESS
Address space for flat memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
@ PRIVATE_ADDRESS
Address space for private memory.
unsigned encodeFieldSaSdst(unsigned Encoded, unsigned SaSdst)
bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi)
const uint64_t RSRC_DATA_FORMAT
bool isPKFMACF16InlineConstant(uint32_t Literal, bool IsGFX11Plus)
LLVM_READONLY const MIMGInfo * getMIMGInfo(unsigned Opc)
bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi)
bool getWMMAIsXDL(unsigned Opc)
unsigned mapWMMA2AddrTo3AddrOpcode(unsigned Opc)
bool isInlinableLiteralV2I16(uint32_t Literal)
bool isDPMACCInstruction(unsigned Opc)
bool isHi16Reg(MCRegister Reg, const MCRegisterInfo &MRI)
bool isInlinableLiteralV2BF16(uint32_t Literal)
LLVM_READONLY int32_t getCommuteRev(uint32_t Opcode)
LLVM_READONLY int32_t getCommuteOrig(uint32_t Opcode)
unsigned getNumFlatOffsetBits(const MCSubtargetInfo &ST)
For pre-GFX12 FLAT instructions the offset must be positive; MSB is ignored and forced to zero.
bool isGFX12Plus(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2F16(uint32_t Literal)
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
bool isValid32BitLiteral(uint64_t Val, bool IsFP64)
LLVM_READONLY int32_t getGlobalVaddrOp(uint32_t Opcode)
LLVM_READNONE bool isLegalDPALU_DPPControl(const MCSubtargetInfo &ST, unsigned DC)
LLVM_READONLY int32_t getMFMAEarlyClobberOp(uint32_t Opcode)
bool getMAIIsGFX940XDL(unsigned Opc)
const uint64_t RSRC_ELEMENT_SIZE_SHIFT
bool isIntrinsicAlwaysUniform(unsigned IntrID)
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
bool isPackedSingleSGPR64BitInst(unsigned Opc)
The opcode is a packed 64-bit instruction which only reads low 64 bits of a scalar operand and propag...
LLVM_READONLY int32_t getIfAddr64Inst(uint32_t Opcode)
Check if Opcode is an Addr64 opcode.
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfoByEncoding(uint8_t DimEnc)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
const uint64_t RSRC_TID_ENABLE
LLVM_READONLY int32_t getVOPe32(uint32_t Opcode)
bool isIntrinsicSourceOfDivergence(unsigned IntrID)
constexpr bool isSISrcOperand(const MCOperandInfo &OpInfo)
Is this an AMDGPU specific source operand?
bool isGenericAtomic(unsigned Opc)
LLVM_READNONE bool isInlinableIntLiteral(int64_t Literal)
Is this literal inlinable, and not one of the values intended for floating point values.
unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode, const MIMGDimInfo *Dim, bool IsA16, bool IsG16Supported)
LLVM_READONLY int32_t getAddr64Inst(uint32_t Opcode)
int32_t getMCOpcode(uint32_t Opcode, unsigned Gen)
@ OPERAND_KIMM32
Operand with 32-bit immediate that uses the constant bus.
@ OPERAND_REG_INLINE_C_FP64
@ OPERAND_REG_IMM_NOINLINE_FP16
@ OPERAND_REG_INLINE_C_BF16
@ OPERAND_REG_INLINE_C_V2BF16
@ OPERAND_REG_IMM_V2INT64
@ OPERAND_REG_IMM_V2INT16
@ OPERAND_REG_IMM_INT32
Operands with register, 32-bit, or 64-bit immediate.
@ OPERAND_REG_IMM_V2FP16_SPLAT
@ OPERAND_REG_INLINE_C_INT64
@ OPERAND_REG_INLINE_C_INT16
Operands with register or inline constant.
@ OPERAND_REG_IMM_NOINLINE_V2FP16
@ OPERAND_REG_INLINE_C_V2FP16
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
@ OPERAND_REG_INLINE_AC_FP32
@ OPERAND_REG_IMM_V2INT32
@ OPERAND_REG_INLINE_C_FP32
@ OPERAND_REG_INLINE_C_INT32
@ OPERAND_REG_INLINE_C_V2INT16
@ OPERAND_INLINE_C_AV64_PSEUDO
@ OPERAND_REG_INLINE_AC_FP64
@ OPERAND_REG_INLINE_C_FP16
@ OPERAND_INLINE_SPLIT_BARRIER_INT32
LLVM_READONLY int32_t getBasicFromSDWAOp(uint32_t Opcode)
bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII, const MCSubtargetInfo &ST)
bool isSingleSGPRReadInst(unsigned Opc)
Packed instructions that read a single SGPR for SGPR operands, except for 64-bit elements which read ...
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
const uint64_t RSRC_INDEX_STRIDE_SHIFT
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
LLVM_READONLY int32_t getFlatScratchInstSVfromSS(uint32_t Opcode)
bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi)
LLVM_READNONE constexpr bool isGraphics(CallingConv::ID CC)
bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi)
Is this literal inlinable.
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
@ AMDGPU_VS
Used for Mesa vertex shaders, or AMDPAL last shader stage before rasterization (vertex shader if tess...
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ AMDGPU_HS
Used for Mesa/AMDPAL hull shaders (= tessellation control shaders).
@ AMDGPU_GS
Used for Mesa/AMDPAL geometry shaders.
@ AMDGPU_PS
Used for Mesa/AMDPAL pixel shaders.
@ Fast
Attempts to make calls as fast as possible (e.g.
@ AMDGPU_ES
Used for AMDPAL shader stage before geometry shader if geometry is in use.
@ AMDGPU_LS
Used for AMDPAL vertex shader if tessellation is in use.
@ C
The default llvm calling convention, compatible with C.
Not(const Pred &P) -> Not< Pred >
constexpr bool isSDWA(const T &...O)
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
@ Low
Lower the current thread's priority such that it does not affect foreground tasks significantly.
LLVM_ABI void finalizeBundle(MachineBasicBlock &MBB, MachineBasicBlock::instr_iterator FirstMI, MachineBasicBlock::instr_iterator LastMI)
finalizeBundle - Finalize a machine instruction bundle which includes a sequence of instructions star...
TargetInstrInfo::RegSubRegPair getRegSubRegPair(const MachineOperand &O)
Create RegSubRegPair from a register MachineOperand.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
constexpr uint64_t maxUIntN(uint64_t N)
Gets the maximum value for a N-bit unsigned integer.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
bool execMayBeModifiedBeforeUse(const MachineRegisterInfo &MRI, Register VReg, const MachineInstr &DefMI, const MachineInstr &UseMI)
Return false if EXEC is not changed between the def of VReg at DefMI and the use at UseMI.
RegState
Flags to represent properties of register accesses.
@ Implicit
Not emitted register (e.g. carry, or temporary result).
@ Kill
The last use of a register.
@ Undef
Value of the register doesn't matter.
@ Define
Register definition.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
constexpr RegState getKillRegState(bool B)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
TargetInstrInfo::RegSubRegPair getRegSequenceSubReg(MachineInstr &MI, unsigned SubReg)
Return the SubReg component from REG_SEQUENCE.
static const MachineMemOperand::Flags MONoClobber
Mark the MMO of a uniform load if there are no potentially clobbering stores on any path from the sta...
constexpr bool has_single_bit(T Value) noexcept
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
auto reverse(ContainerTy &&C)
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
MachineInstr * getVRegSubRegDef(const TargetInstrInfo::RegSubRegPair &P, const MachineRegisterInfo &MRI)
Return the defining instruction for a given reg:subreg pair skipping copy like instructions and subre...
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth, bool MustPreserveProvenance=false)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
LLVM_ABI VirtRegInfo AnalyzeVirtRegInBundle(MachineInstr &MI, Register Reg, SmallVectorImpl< std::pair< MachineInstr *, unsigned > > *Ops=nullptr)
AnalyzeVirtRegInBundle - Analyze how the current instruction or bundle uses a virtual register.
static const MachineMemOperand::Flags MOCooperative
Mark the MMO of cooperative load/store atomics.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
@ Xor
Bitwise or logical XOR of integers.
@ Sub
Subtraction of integers.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
bool isTargetSpecificOpcode(unsigned Opcode)
Check whether the given Opcode is a target-specific opcode.
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
constexpr unsigned DefaultMemoryClusterDWordsLimit
constexpr unsigned BitWidth
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
constexpr bool isIntN(unsigned N, int64_t x)
Checks if an signed integer fits into the given (dynamic) bit width.
static const MachineMemOperand::Flags MOLastUse
Mark the MMO of a load as the last use.
constexpr T reverseBits(T Val)
Reverse the bits in Val.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
constexpr RegState getUndefRegState(bool B)
ValueUniformity
Enum describing how values behave with respect to uniformity and divergence, to answer the question: ...
@ AlwaysUniform
The result value is always uniform.
@ NeverUniform
The result value can never be assumed to be uniform.
@ Default
The result value is uniform if and only if all operands are uniform.
static const MachineMemOperand::Flags MOThreadPrivate
Mark the MMO of accesses to memory locations that are never written to by other threads.
bool execMayBeModifiedBeforeAnyUse(const MachineRegisterInfo &MRI, Register VReg, const MachineInstr &DefMI)
Return false if EXEC is not changed between the def of VReg at DefMI and all its uses.
MCRegisterClass TargetRegisterClass
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Helper struct for the implementation of 3-address conversion to communicate updates made to instructi...
MachineInstr * RemoveMIUse
Other instruction whose def is no longer used by the converted instruction.
uint8_t GFX1250BlockingCycles
static constexpr uint64_t encode(Fields... Values)
This struct is a compact representation of a valid (non-zero power of two) alignment.
constexpr bool all() const
Summarize the scheduling resources required for an instruction of a particular scheduling class.
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
Utility to store machine instructions worklist.
MachineInstr * top() const
bool isDeferred(MachineInstr *MI)
SetVector< MachineInstr * > & getDeferredList()
void insert(MachineInstr *MI)
A pair composed of a register and a sub-register index.
VirtRegInfo - Information about a virtual register used by a set of operands.
bool Reads
Reads - One of the operands read the virtual register.
bool Writes
Writes - One of the operands writes the virtual register.