24#define DEBUG_TYPE "si-fold-operands"
45 unsigned DefSubReg = AMDGPU::NoSubRegister;
50 FoldableDef() =
delete;
52 unsigned DefSubReg = AMDGPU::NoSubRegister)
53 : DefRC(DefRC), DefSubReg(DefSubReg), Kind(FoldOp.
getType()) {
56 ImmToFold = FoldOp.
getImm();
57 }
else if (FoldOp.
isFI()) {
58 FrameIndexToFold = FoldOp.
getIndex();
68 unsigned DefSubReg = AMDGPU::NoSubRegister)
69 : ImmToFold(FoldImm), DefRC(DefRC), DefSubReg(DefSubReg),
74 FoldableDef Copy(*
this);
75 Copy.DefSubReg =
TRI.composeSubRegIndices(DefSubReg, SubReg);
83 return OpToFold->getReg();
86 unsigned getSubReg()
const {
88 return OpToFold->getSubReg();
99 return FrameIndexToFold;
107 std::optional<int64_t> getEffectiveImmVal()
const {
115 unsigned OpIdx)
const {
118 std::optional<int64_t> ImmToFold = getEffectiveImmVal();
125 return TII.isOperandLegal(
MI, OpIdx, &TmpOp);
128 if (DefSubReg != AMDGPU::NoSubRegister)
131 return TII.isOperandLegal(
MI, OpIdx, &TmpOp);
136 if (DefSubReg != AMDGPU::NoSubRegister)
138 return TII.isOperandLegal(
MI, OpIdx, OpToFold);
145struct FoldCandidate {
153 bool Commuted =
false,
int ShrinkOp = -1)
154 :
UseMI(
MI), Def(Def), ShrinkOpcode(ShrinkOp), UseOpNo(OpNo),
155 Commuted(Commuted) {}
157 bool isFI()
const {
return Def.isFI(); }
161 return Def.FrameIndexToFold;
164 bool isImm()
const {
return Def.isImm(); }
166 bool isReg()
const {
return Def.isReg(); }
170 bool isGlobal()
const {
return Def.isGlobal(); }
172 bool needsShrink()
const {
return ShrinkOpcode != -1; }
175class SIFoldOperandsImpl {
186 const FoldableDef &OpToFold)
const;
189 unsigned convertToVALUOp(
unsigned Opc,
bool UseVOP3 =
false)
const {
191 case AMDGPU::S_ADD_I32: {
192 if (ST->hasAddNoCarryInsts())
193 return UseVOP3 ? AMDGPU::V_ADD_U32_e64 : AMDGPU::V_ADD_U32_e32;
194 return UseVOP3 ? AMDGPU::V_ADD_CO_U32_e64 : AMDGPU::V_ADD_CO_U32_e32;
196 case AMDGPU::S_OR_B32:
197 return UseVOP3 ? AMDGPU::V_OR_B32_e64 : AMDGPU::V_OR_B32_e32;
198 case AMDGPU::S_AND_B32:
199 return UseVOP3 ? AMDGPU::V_AND_B32_e64 : AMDGPU::V_AND_B32_e32;
200 case AMDGPU::S_MUL_I32:
201 return AMDGPU::V_MUL_LO_U32_e64;
203 return AMDGPU::INSTRUCTION_LIST_END;
207 bool foldCopyToVGPROfScalarAddOfFrameIndex(
Register DstReg,
Register SrcReg,
213 int64_t ImmVal)
const;
217 int64_t ImmVal)
const;
221 const FoldableDef &OpToFold)
const;
224 bool isTemporallyDivergentUse(
const FoldableDef &OpToFold,
232 getRegSeqInit(
SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs,
235 std::pair<int64_t, const TargetRegisterClass *>
249 struct ANDMaskResult {
255 std::optional<ANDMaskResult> getANDMaskRegOperand(
MachineInstr &AndMI)
const;
260 bool foldInstOperand(
MachineInstr &
MI,
const FoldableDef &OpToFold)
const;
262 bool foldCopyToAGPRRegSequence(
MachineInstr *CopyMI)
const;
269 std::pair<const MachineOperand *, int> isOMod(
const MachineInstr &
MI)
const;
279 SIFoldOperandsImpl() =
default;
294 &getAnalysis<MachineLoopInfoWrapperPass>().getLI();
295 return SIFoldOperandsImpl().run(MF, MLI);
298 StringRef getPassName()
const override {
return "SI Fold Operands"; }
320char SIFoldOperandsLegacy::ID = 0;
329 TRI.getSubRegisterClass(RC, MO.getSubReg()))
337 case AMDGPU::V_MAC_F32_e64:
338 return AMDGPU::V_MAD_F32_e64;
339 case AMDGPU::V_MAC_F16_e64:
340 return AMDGPU::V_MAD_F16_e64;
341 case AMDGPU::V_FMAC_F32_e64:
342 return AMDGPU::V_FMA_F32_e64;
343 case AMDGPU::V_FMAC_F16_e64:
344 return AMDGPU::V_FMA_F16_gfx9_e64;
345 case AMDGPU::V_FMAC_F16_t16_e64:
346 return AMDGPU::V_FMA_F16_gfx9_t16_e64;
347 case AMDGPU::V_FMAC_F16_fake16_e64:
348 return AMDGPU::V_FMA_F16_gfx9_fake16_e64;
349 case AMDGPU::V_FMAC_LEGACY_F32_e64:
350 return AMDGPU::V_FMA_LEGACY_F32_e64;
351 case AMDGPU::V_FMAC_F64_e64:
352 return AMDGPU::V_FMA_F64_e64;
354 return AMDGPU::INSTRUCTION_LIST_END;
360 const FoldableDef &OpToFold)
const {
361 if (!OpToFold.isFI())
364 const unsigned Opc =
UseMI.getOpcode();
366 case AMDGPU::S_ADD_I32:
367 case AMDGPU::S_ADD_U32:
368 case AMDGPU::V_ADD_U32_e32:
369 case AMDGPU::V_ADD_CO_U32_e32:
373 return UseMI.getOperand(OpNo == 1 ? 2 : 1).isImm() &&
375 case AMDGPU::V_ADD_U32_e64:
376 case AMDGPU::V_ADD_CO_U32_e64:
377 return UseMI.getOperand(OpNo == 2 ? 3 : 2).isImm() &&
384 return OpNo == AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr);
388 int SIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::saddr);
392 int VIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr);
393 return OpNo == VIdx && SIdx == -1;
399bool SIFoldOperandsImpl::foldCopyToVGPROfScalarAddOfFrameIndex(
404 if (
TRI->isVGPR(*MRI, DstReg) &&
TRI->isSGPRReg(*MRI, SrcReg) &&
407 if (!Def ||
Def->getNumOperands() != 4)
410 MachineOperand *Src0 = &
Def->getOperand(1);
411 MachineOperand *Src1 = &
Def->getOperand(2);
422 const bool UseVOP3 = !Src0->
isImm() ||
TII->isInlineConstant(*Src0);
423 unsigned NewOp = convertToVALUOp(
Def->getOpcode(), UseVOP3);
424 if (NewOp == AMDGPU::INSTRUCTION_LIST_END ||
425 !
Def->getOperand(3).isDead())
428 MachineBasicBlock *
MBB =
Def->getParent();
430 if (NewOp != AMDGPU::V_ADD_CO_U32_e32) {
431 MachineInstrBuilder
Add =
434 if (
Add->getDesc().getNumDefs() == 2) {
436 Add.addDef(CarryOutReg, RegState::Dead);
440 Add.add(*Src0).add(*Src1).setMIFlags(
Def->getFlags());
444 Def->eraseFromParent();
445 MI.eraseFromParent();
449 assert(NewOp == AMDGPU::V_ADD_CO_U32_e32);
460 Def->eraseFromParent();
461 MI.eraseFromParent();
470 return new SIFoldOperandsLegacy();
473bool SIFoldOperandsImpl::canUseImmWithOpSel(
const MachineInstr *
MI,
475 int64_t ImmVal)
const {
482 int OpNo =
MI->getOperandNo(&Old);
484 unsigned Opcode =
MI->getOpcode();
485 uint8_t OpType =
TII->get(Opcode).operands()[OpNo].OperandType;
507bool SIFoldOperandsImpl::tryFoldImmWithOpSel(MachineInstr *
MI,
unsigned UseOpNo,
508 int64_t ImmVal)
const {
509 MachineOperand &Old =
MI->getOperand(UseOpNo);
510 unsigned Opcode =
MI->getOpcode();
511 int OpNo =
MI->getOperandNo(&Old);
512 uint8_t OpType =
TII->get(Opcode).operands()[OpNo].OperandType;
524 AMDGPU::OpName ModName = AMDGPU::OpName::NUM_OPERAND_NAMES;
525 unsigned SrcIdx = ~0;
526 if (OpNo == AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0)) {
527 ModName = AMDGPU::OpName::src0_modifiers;
529 }
else if (OpNo == AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src1)) {
530 ModName = AMDGPU::OpName::src1_modifiers;
532 }
else if (OpNo == AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src2)) {
533 ModName = AMDGPU::OpName::src2_modifiers;
536 assert(ModName != AMDGPU::OpName::NUM_OPERAND_NAMES);
537 int ModIdx = AMDGPU::getNamedOperandIdx(Opcode, ModName);
538 MachineOperand &
Mod =
MI->getOperand(ModIdx);
539 unsigned ModVal =
Mod.getImm();
545 uint32_t
Imm = (
static_cast<uint32_t
>(ImmHi) << 16) | ImmLo;
550 auto tryFoldToInline = [&](uint32_t
Imm) ->
bool {
559 uint16_t
Lo =
static_cast<uint16_t
>(
Imm);
560 uint16_t
Hi =
static_cast<uint16_t
>(
Imm >> 16);
566 if (ST->hasBF16InlineConstFromUpperFP32() &&
570 Mod.setImm(NewModVal);
575 if (
static_cast<int16_t
>(
Lo) < 0) {
576 int32_t SExt =
static_cast<int16_t
>(
Lo);
578 Mod.setImm(NewModVal);
593 uint32_t Swapped = (
static_cast<uint32_t
>(
Lo) << 16) |
Hi;
604 if (tryFoldToInline(
Imm))
613 bool IsUAdd = Opcode == AMDGPU::V_PK_ADD_U16;
614 bool IsUSub = Opcode == AMDGPU::V_PK_SUB_U16;
615 if (SrcIdx == 1 && (IsUAdd || IsUSub)) {
617 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::clamp);
618 bool Clamp =
MI->getOperand(ClampIdx).getImm() != 0;
621 uint16_t NegLo = -
static_cast<uint16_t
>(
Imm);
622 uint16_t NegHi = -
static_cast<uint16_t
>(
Imm >> 16);
623 uint32_t NegImm = (
static_cast<uint32_t
>(NegHi) << 16) | NegLo;
625 if (tryFoldToInline(NegImm)) {
627 IsUAdd ? AMDGPU::V_PK_SUB_U16 : AMDGPU::V_PK_ADD_U16;
628 MI->setDesc(
TII->get(NegOpcode));
637bool SIFoldOperandsImpl::updateOperand(FoldCandidate &Fold)
const {
638 MachineInstr *
MI = Fold.UseMI;
639 MachineOperand &Old =
MI->getOperand(Fold.UseOpNo);
642 std::optional<int64_t> ImmVal;
644 ImmVal = Fold.Def.getEffectiveImmVal();
646 if (ImmVal && canUseImmWithOpSel(Fold.UseMI, Fold.UseOpNo, *ImmVal)) {
647 if (tryFoldImmWithOpSel(Fold.UseMI, Fold.UseOpNo, *ImmVal))
653 int OpNo =
MI->getOperandNo(&Old);
654 if (!
TII->isOperandLegal(*
MI, OpNo, &New))
660 if ((Fold.isImm() || Fold.isFI() || Fold.isGlobal()) && Fold.needsShrink()) {
668 int Op32 = Fold.ShrinkOpcode;
669 MachineOperand &Dst0 =
MI->getOperand(0);
670 MachineOperand &Dst1 =
MI->getOperand(1);
678 MachineInstr *Inst32 =
TII->buildShrunkInst(*
MI, Op32);
680 if (HaveNonDbgCarryUse) {
683 .
addReg(AMDGPU::VCC, RegState::Kill);
693 for (
unsigned I =
MI->getNumOperands() - 1;
I > 0; --
I)
694 MI->removeOperand(
I);
695 MI->setDesc(
TII->get(AMDGPU::IMPLICIT_DEF));
698 TII->commuteInstruction(*Inst32,
false);
702 assert(!Fold.needsShrink() &&
"not handled");
707 if (NewMFMAOpc == -1)
709 MI->setDesc(
TII->get(NewMFMAOpc));
710 MI->untieRegOperand(0);
711 const MCInstrDesc &MCID =
MI->getDesc();
712 for (
unsigned I = 0;
I <
MI->getNumDefs(); ++
I)
714 MI->getOperand(
I).setIsEarlyClobber(
true);
719 int OpNo =
MI->getOperandNo(&Old);
720 if (!
TII->isOperandLegal(*
MI, OpNo, &New))
723 if (ST->hasBF16InlineConstFromUpperFP32() &&
725 AMDGPU::getNamedOperandIdx(
MI->getOpcode(), AMDGPU::OpName::src0)) {
726 unsigned Opcode =
MI->getOpcode();
727 uint8_t OpType =
TII->get(Opcode).operands()[OpNo].OperandType;
730 TII->isInlineConstant(*ImmVal, OpType)) {
733 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0_modifiers);
736 MachineOperand &ModOp =
MI->getOperand(Mod0);
747 if (Fold.isGlobal()) {
748 Old.
ChangeToGA(Fold.Def.OpToFold->getGlobal(),
749 Fold.Def.OpToFold->getOffset(),
750 Fold.Def.OpToFold->getTargetFlags());
759 MachineOperand *
New = Fold.Def.OpToFold;
763 TII->getRegClass(
MI->getDesc(), Fold.UseOpNo)) {
765 TRI->getRegClassForReg(*MRI,
New->getReg());
768 if (
New->getSubReg()) {
770 TRI->getMatchingSuperRegClass(NewRC, OpRC,
New->getSubReg());
776 if (
New->getReg().isVirtual() &&
779 <<
TRI->getRegClassName(ConstrainRC) <<
'\n');
786 if (Old.
getSubReg() == AMDGPU::lo16 &&
TRI->isSGPRReg(*MRI,
New->getReg()))
788 if (
New->getReg().isPhysical()) {
796 if (
MI->isBundledWithPred()) {
798 for (MachineOperand &MO : Header.operands()) {
799 if (MO.getReg() == OldReg) {
800 MO.setReg(
New->getReg());
801 MO.setSubReg(
New->getSubReg());
810 FoldCandidate &&Entry) {
812 for (FoldCandidate &Fold : FoldList)
813 if (Fold.UseMI == Entry.UseMI && Fold.UseOpNo == Entry.UseOpNo)
815 LLVM_DEBUG(
dbgs() <<
"Append " << (Entry.Commuted ?
"commuted" :
"normal")
816 <<
" operand " << Entry.UseOpNo <<
"\n " << *Entry.UseMI);
822 const FoldableDef &FoldOp,
823 bool Commuted =
false,
int ShrinkOp = -1) {
825 FoldCandidate(
MI, OpNo, FoldOp, Commuted, ShrinkOp));
833 if (!ST->hasPKF32InstsReplicatingLower32BitsOfScalarInput())
843 const FoldableDef &OpToFold) {
844 assert(OpToFold.isImm() &&
"Expected immediate operand");
845 uint64_t ImmVal = OpToFold.getEffectiveImmVal().value();
851bool SIFoldOperandsImpl::tryAddToFoldList(
852 SmallVectorImpl<FoldCandidate> &FoldList, MachineInstr *
MI,
unsigned OpNo,
853 const FoldableDef &OpToFold)
const {
854 const unsigned Opc =
MI->getOpcode();
856 auto tryToFoldAsFMAAKorMK = [&]() {
857 if (!OpToFold.isImm())
860 const bool TryAK = OpNo == 3;
861 const unsigned NewOpc = TryAK ? AMDGPU::S_FMAAK_F32 : AMDGPU::S_FMAMK_F32;
862 MI->setDesc(
TII->get(NewOpc));
865 bool FoldAsFMAAKorMK =
866 tryAddToFoldList(FoldList,
MI, TryAK ? 3 : 2, OpToFold);
867 if (FoldAsFMAAKorMK) {
869 MI->untieRegOperand(3);
872 MachineOperand &Op1 =
MI->getOperand(1);
873 MachineOperand &Op2 =
MI->getOperand(2);
890 bool IsLegal = OpToFold.isOperandLegal(*
TII, *
MI, OpNo);
891 if (!IsLegal && OpToFold.isImm()) {
892 if (std::optional<int64_t> ImmVal = OpToFold.getEffectiveImmVal())
893 IsLegal = canUseImmWithOpSel(
MI, OpNo, *ImmVal);
899 if (NewOpc != AMDGPU::INSTRUCTION_LIST_END) {
902 MI->setDesc(
TII->get(NewOpc));
907 bool FoldAsMAD = tryAddToFoldList(FoldList,
MI, OpNo, OpToFold);
909 MI->untieRegOperand(OpNo);
913 MI->removeOperand(
MI->getNumExplicitOperands() - 1);
919 if (
Opc == AMDGPU::S_FMAC_F32 && OpNo == 3) {
920 if (tryToFoldAsFMAAKorMK())
926 if ((
Opc == AMDGPU::S_FMAAK_F32 ||
Opc == AMDGPU::S_FMAMK_F32) &&
928 std::optional<int64_t> ImmVal = OpToFold.getEffectiveImmVal();
929 if (ImmVal && !
TII->isInlineConstant(*
MI, OpNo, *ImmVal)) {
930 unsigned ImmIdx =
Opc == AMDGPU::S_FMAAK_F32 ? 3 : 2;
931 MachineOperand &OpImm =
MI->getOperand(ImmIdx);
932 if (!OpImm.
isReg() &&
933 TII->isInlineConstant(*
MI,
MI->getOperand(OpNo), OpImm))
934 return tryToFoldAsFMAAKorMK();
939 if (OpToFold.isImm()) {
941 if (
Opc == AMDGPU::S_SETREG_B32)
942 ImmOpc = AMDGPU::S_SETREG_IMM32_B32;
943 else if (
Opc == AMDGPU::S_SETREG_B32_mode)
944 ImmOpc = AMDGPU::S_SETREG_IMM32_B32_mode;
946 MI->setDesc(
TII->get(ImmOpc));
955 bool CanCommute =
TII->findCommutedOpIndices(*
MI, OpNo, CommuteOpNo);
959 MachineOperand &
Op =
MI->getOperand(OpNo);
960 MachineOperand &CommutedOp =
MI->getOperand(CommuteOpNo);
966 if (!
Op.isReg() || !CommutedOp.
isReg())
971 if (
Op.isReg() && CommutedOp.
isReg() &&
972 (
Op.getReg() == CommutedOp.
getReg() &&
976 if (!
TII->commuteInstruction(*
MI,
false, OpNo, CommuteOpNo))
980 if (!OpToFold.isOperandLegal(*
TII, *
MI, CommuteOpNo)) {
981 if ((
Opc != AMDGPU::V_ADD_CO_U32_e64 &&
Opc != AMDGPU::V_SUB_CO_U32_e64 &&
982 Opc != AMDGPU::V_SUBREV_CO_U32_e64) ||
983 (!OpToFold.isImm() && !OpToFold.isFI() && !OpToFold.isGlobal())) {
984 TII->commuteInstruction(*
MI,
false, OpNo, CommuteOpNo);
990 MachineOperand &OtherOp =
MI->getOperand(OpNo);
991 if (!OtherOp.
isReg() ||
998 unsigned MaybeCommutedOpc =
MI->getOpcode();
1012 if (
Opc == AMDGPU::S_FMAC_F32 &&
1013 (OpNo != 1 || !
MI->getOperand(1).isIdenticalTo(
MI->getOperand(2)))) {
1014 if (tryToFoldAsFMAAKorMK())
1020 if (OpToFold.isImm() &&
1029bool SIFoldOperandsImpl::isUseSafeToFold(
const MachineInstr &
MI,
1030 const MachineOperand &UseMO)
const {
1032 return !
TII->isSDWA(
MI);
1039 if (
MI.modifiesRegister(
TRI.getExec(), &
TRI))
1047bool SIFoldOperandsImpl::isTemporallyDivergentUse(
1048 const FoldableDef &OpToFold,
const MachineInstr &
UseMI)
const {
1049 if (!OpToFold.isReg())
1051 const MachineInstr *
DefMI = OpToFold.DefMI;
1054 !
TRI->isSGPRReg(*MRI, OpToFold.getReg()))
1066 SubDef &&
TII.isFoldableCopy(*SubDef);
1068 unsigned SrcIdx =
TII.getFoldableCopySrcIdx(*SubDef);
1077 if (
SrcOp.getSubReg())
1085 MachineInstr &RegSeq,
1086 SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs)
const {
1102 else if (!
TRI->getCommonSubClass(RC, OpRC))
1107 Defs.emplace_back(&SrcOp, SubRegIdx);
1112 if (DefSrc && (DefSrc->
isReg() || DefSrc->
isImm())) {
1113 Defs.emplace_back(DefSrc, SubRegIdx);
1117 Defs.emplace_back(&SrcOp, SubRegIdx);
1127 SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs,
1130 if (!Def || !
Def->isRegSequence())
1133 return getRegSeqInit(*Def, Defs);
1136std::pair<int64_t, const TargetRegisterClass *>
1137SIFoldOperandsImpl::isRegSeqSplat(MachineInstr &RegSeq)
const {
1143 bool TryToMatchSplat64 =
false;
1145 std::optional<int64_t>
Imm;
1146 for (
unsigned I = 0,
E = Defs.
size();
I !=
E; ++
I) {
1147 const MachineOperand *
Op = Defs[
I].first;
1151 if (!Def ||
Def->isImplicitDef())
1157 int64_t SubImm =
Op->getImm();
1163 if (
Imm != SubImm) {
1164 if (
I == 1 && (
E & 1) == 0) {
1167 TryToMatchSplat64 =
true;
1175 if (!TryToMatchSplat64) {
1177 return {*
Imm, SrcRC};
1184 for (
unsigned I = 0,
E = Defs.
size();
I !=
E;
I += 2) {
1185 const MachineOperand *Op0 = Defs[
I].first;
1186 const MachineOperand *Op1 = Defs[
I + 1].first;
1191 unsigned SubReg0 = Defs[
I].second;
1192 unsigned SubReg1 = Defs[
I + 1].second;
1196 if (
TRI->getChannelFromSubReg(SubReg0) + 1 !=
1197 TRI->getChannelFromSubReg(SubReg1))
1200 if (
TRI->getSubRegIdxSize(SubReg0) != 32)
1205 SplatVal64 = MergedVal;
1206 else if (SplatVal64 != MergedVal)
1213 return {SplatVal64, RC64};
1216bool SIFoldOperandsImpl::tryFoldRegSeqSplat(
1217 MachineInstr *
UseMI,
unsigned UseOpIdx, int64_t SplatVal,
1220 if (UseOpIdx >=
Desc.getNumOperands())
1227 int16_t RCID =
TII->getOpRegClassID(
Desc.operands()[UseOpIdx]);
1236 if (SplatVal != 0 && SplatVal != -1) {
1240 uint8_t OpTy =
Desc.operands()[UseOpIdx].OperandType;
1247 OpRC =
TRI->getSubRegisterClass(OpRC, AMDGPU::sub0);
1254 OpRC =
TRI->getSubRegisterClass(OpRC, AMDGPU::sub0_sub1);
1260 if (!
TRI->getCommonSubClass(OpRC, SplatRC))
1265 if (!
TII->isOperandLegal(*
UseMI, UseOpIdx, &TmpOp))
1271bool SIFoldOperandsImpl::tryToFoldACImm(
1272 const FoldableDef &OpToFold, MachineInstr *
UseMI,
unsigned UseOpIdx,
1273 SmallVectorImpl<FoldCandidate> &FoldList)
const {
1275 if (UseOpIdx >=
Desc.getNumOperands())
1282 if (OpToFold.isImm() && OpToFold.isOperandLegal(*
TII, *
UseMI, UseOpIdx)) {
1293bool SIFoldOperandsImpl::foldOperand(
1294 FoldableDef OpToFold, MachineInstr *
UseMI,
int UseOpIdx,
1295 SmallVectorImpl<FoldCandidate> &FoldList,
1296 SmallVectorImpl<MachineInstr *> &CopiesToReplace)
const {
1300 if (!isUseSafeToFold(*
UseMI, *UseOp))
1303 if (isTemporallyDivergentUse(OpToFold, *
UseMI))
1307 if (UseOp->
isReg() && OpToFold.isReg()) {
1311 if (UseOp->
getSubReg() != AMDGPU::NoSubRegister &&
1313 !
TRI->isSGPRReg(*MRI, OpToFold.getReg())))
1326 std::tie(SplatVal, SplatRC) = isRegSeqSplat(*
UseMI);
1331 for (
unsigned I = 0;
I != UsesToProcess.size(); ++
I) {
1332 MachineOperand *RSUse = UsesToProcess[
I];
1333 MachineInstr *RSUseMI = RSUse->
getParent();
1343 if (tryFoldRegSeqSplat(RSUseMI, OpNo, SplatVal, SplatRC)) {
1344 FoldableDef SplatDef(SplatVal, SplatRC);
1352 if (RSUse->
getSubReg() != RegSeqDstSubReg)
1358 FoldList, CopiesToReplace);
1364 if (tryToFoldACImm(OpToFold,
UseMI, UseOpIdx, FoldList))
1367 if (frameIndexMayFold(*
UseMI, UseOpIdx, OpToFold)) {
1372 if (
TII->getNamedOperand(*
UseMI, AMDGPU::OpName::srsrc)->getReg() !=
1378 MachineOperand &SOff =
1379 *
TII->getNamedOperand(*
UseMI, AMDGPU::OpName::soffset);
1390 TII->getNamedOperand(*
UseMI, AMDGPU::OpName::cpol)->getImm();
1405 bool FoldingImmLike =
1406 OpToFold.isImm() || OpToFold.isFI() || OpToFold.isGlobal();
1425 for (
unsigned MovOp :
1426 {AMDGPU::S_MOV_B32, AMDGPU::V_MOV_B32_e32, AMDGPU::S_MOV_B64,
1427 AMDGPU::V_MOV_B64_PSEUDO, AMDGPU::V_MOV_B16_t16_e64,
1428 AMDGPU::V_ACCVGPR_WRITE_B32_e64, AMDGPU::AV_MOV_B32_IMM_PSEUDO,
1429 AMDGPU::AV_MOV_B64_IMM_PSEUDO}) {
1430 const MCInstrDesc &MovDesc =
TII->get(MovOp);
1440 const int SrcIdx = MovOp == AMDGPU::V_MOV_B16_t16_e64 ? 2 : 1;
1442 int16_t RegClassID =
TII->getOpRegClassID(MovDesc.
operands()[SrcIdx]);
1443 if (RegClassID != -1) {
1447 MovSrcRC =
TRI->getMatchingSuperRegClass(SrcRC, MovSrcRC, UseSubReg);
1451 if (MovOp == AMDGPU::AV_MOV_B32_IMM_PSEUDO &&
1452 (!OpToFold.isImm() ||
1453 !
TII->isImmOperandLegal(MovDesc, SrcIdx,
1454 *OpToFold.getEffectiveImmVal())))
1467 if (!OpToFold.isImm() ||
1468 !
TII->isImmOperandLegal(MovDesc, 1, *OpToFold.getEffectiveImmVal()))
1474 while (ImpOpI != ImpOpE) {
1481 if (MovOp == AMDGPU::V_MOV_B16_t16_e64) {
1483 MachineOperand NewSrcOp(SrcOp);
1505 LLVM_DEBUG(
dbgs() <<
"Folding " << *OpToFold.OpToFold <<
"\n into "
1510 unsigned SubRegIdx = OpToFold.getSubReg();
1524 static_assert(AMDGPU::sub1_hi16 == 12,
"Subregister layout has changed");
1529 if (SubRegIdx > AMDGPU::sub1) {
1530 LaneBitmask
M =
TRI->getSubRegIndexLaneMask(SubRegIdx);
1531 M |=
M.getLane(
M.getHighestLane() - 1);
1532 SmallVector<unsigned, 4> Indexes;
1533 TRI->getCoveringSubRegIndexes(
TRI->getRegClassForReg(*MRI,
UseReg), M,
1535 assert(Indexes.
size() == 1 &&
"Expected one 32-bit subreg to cover");
1536 SubRegIdx = Indexes[0];
1538 }
else if (
TII->getOpSize(*
UseMI, 1) == 4)
1541 SubRegIdx = AMDGPU::sub0;
1546 OpToFold.OpToFold->setIsKill(
false);
1551 if (foldCopyToAGPRRegSequence(
UseMI))
1556 if (UseOpc == AMDGPU::V_READFIRSTLANE_B32 ||
1557 (UseOpc == AMDGPU::V_READLANE_B32 &&
1559 AMDGPU::getNamedOperandIdx(UseOpc, AMDGPU::OpName::src0))) {
1564 if (FoldingImmLike) {
1567 *OpToFold.DefMI, *
UseMI))
1573 if (OpToFold.isImm()) {
1575 *OpToFold.getEffectiveImmVal());
1576 }
else if (OpToFold.isFI())
1579 assert(OpToFold.isGlobal());
1581 OpToFold.OpToFold->getOffset(),
1582 OpToFold.OpToFold->getTargetFlags());
1588 if (OpToFold.isReg() &&
TRI->isSGPRReg(*MRI, OpToFold.getReg())) {
1591 *OpToFold.DefMI, *
UseMI))
1613 UseDesc.
operands()[UseOpIdx].RegClass == -1)
1621 Changed |= tryAddToFoldList(FoldList,
UseMI, UseOpIdx, OpToFold);
1628 case AMDGPU::S_ADD_I32:
1629 case AMDGPU::S_ADD_U32:
1632 case AMDGPU::S_SUB_I32:
1633 case AMDGPU::S_SUB_U32:
1636 case AMDGPU::V_AND_B32_e64:
1637 case AMDGPU::V_AND_B32_e32:
1638 case AMDGPU::S_AND_B32:
1641 case AMDGPU::V_OR_B32_e64:
1642 case AMDGPU::V_OR_B32_e32:
1643 case AMDGPU::S_OR_B32:
1646 case AMDGPU::V_XOR_B32_e64:
1647 case AMDGPU::V_XOR_B32_e32:
1648 case AMDGPU::S_XOR_B32:
1651 case AMDGPU::S_XNOR_B32:
1654 case AMDGPU::S_NAND_B32:
1657 case AMDGPU::S_NOR_B32:
1660 case AMDGPU::S_ANDN2_B32:
1663 case AMDGPU::S_ORN2_B32:
1666 case AMDGPU::V_LSHL_B32_e64:
1667 case AMDGPU::V_LSHL_B32_e32:
1668 case AMDGPU::S_LSHL_B32:
1670 Result =
LHS << (
RHS & 31);
1672 case AMDGPU::V_LSHLREV_B32_e64:
1673 case AMDGPU::V_LSHLREV_B32_e32:
1674 Result =
RHS << (
LHS & 31);
1676 case AMDGPU::V_LSHR_B32_e64:
1677 case AMDGPU::V_LSHR_B32_e32:
1678 case AMDGPU::S_LSHR_B32:
1679 Result =
LHS >> (
RHS & 31);
1681 case AMDGPU::V_LSHRREV_B32_e64:
1682 case AMDGPU::V_LSHRREV_B32_e32:
1683 Result =
RHS >> (
LHS & 31);
1685 case AMDGPU::V_ASHR_I32_e64:
1686 case AMDGPU::V_ASHR_I32_e32:
1687 case AMDGPU::S_ASHR_I32:
1688 Result =
static_cast<int32_t
>(
LHS) >> (
RHS & 31);
1690 case AMDGPU::V_ASHRREV_I32_e64:
1691 case AMDGPU::V_ASHRREV_I32_e32:
1692 Result =
static_cast<int32_t
>(
RHS) >> (
LHS & 31);
1700 return IsScalar ? AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32;
1706bool SIFoldOperandsImpl::tryConstantFoldOp(MachineInstr *
MI)
const {
1707 if (!
MI->allImplicitDefsAreDead())
1710 unsigned Opc =
MI->getOpcode();
1712 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
1716 MachineOperand *Src0 = &
MI->getOperand(Src0Idx);
1717 std::optional<int64_t> Src0Imm =
TII->getImmOrMaterializedImm(*MRI, *Src0);
1719 if ((
Opc == AMDGPU::V_NOT_B32_e64 ||
Opc == AMDGPU::V_NOT_B32_e32 ||
1720 Opc == AMDGPU::S_NOT_B32) &&
1722 MI->getOperand(1).ChangeToImmediate(~*Src0Imm);
1723 TII->mutateAndCleanupImplicit(
1728 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
1732 MachineOperand *Src1 = &
MI->getOperand(Src1Idx);
1733 std::optional<int64_t> Src1Imm =
TII->getImmOrMaterializedImm(*MRI, *Src1);
1735 if (!Src0Imm && !Src1Imm)
1741 if (Src0Imm && Src1Imm) {
1746 bool IsSGPR =
TRI->isSGPRReg(*MRI,
MI->getOperand(0).getReg());
1750 MI->getOperand(Src0Idx).ChangeToImmediate(NewImm);
1751 MI->removeOperand(Src1Idx);
1758 if (
Opc == AMDGPU::S_SUB_I32 ||
Opc == AMDGPU::S_SUB_U32) {
1759 if (Src1Imm &&
static_cast<int32_t
>(*Src1Imm) == 0) {
1761 MI->removeOperand(Src1Idx);
1762 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1768 if (!
MI->isCommutable())
1771 if (Src0Imm && !Src1Imm) {
1777 int32_t Src1Val =
static_cast<int32_t
>(*Src1Imm);
1778 if (
Opc == AMDGPU::S_ADD_I32 ||
Opc == AMDGPU::S_ADD_U32) {
1781 MI->removeOperand(Src1Idx);
1782 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1788 if (
Opc == AMDGPU::V_OR_B32_e64 ||
1789 Opc == AMDGPU::V_OR_B32_e32 ||
1790 Opc == AMDGPU::S_OR_B32) {
1793 MI->removeOperand(Src1Idx);
1794 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1795 }
else if (Src1Val == -1) {
1797 MI->removeOperand(Src0Idx);
1798 TII->mutateAndCleanupImplicit(
1806 if (
Opc == AMDGPU::V_AND_B32_e64 ||
Opc == AMDGPU::V_AND_B32_e32 ||
1807 Opc == AMDGPU::S_AND_B32) {
1810 MI->removeOperand(Src0Idx);
1811 TII->mutateAndCleanupImplicit(
1813 }
else if (Src1Val == -1) {
1815 MI->removeOperand(Src1Idx);
1816 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1823 if (
Opc == AMDGPU::V_XOR_B32_e64 ||
Opc == AMDGPU::V_XOR_B32_e32 ||
1824 Opc == AMDGPU::S_XOR_B32) {
1827 MI->removeOperand(Src1Idx);
1828 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1837bool SIFoldOperandsImpl::tryFoldCndMask(MachineInstr &
MI)
const {
1838 unsigned Opc =
MI.getOpcode();
1839 if (
Opc != AMDGPU::V_CNDMASK_B32_e32 &&
Opc != AMDGPU::V_CNDMASK_B32_e64 &&
1840 Opc != AMDGPU::V_CNDMASK_B64_PSEUDO)
1843 MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
1844 MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
1846 std::optional<int64_t> Src1Imm =
TII->getImmOrMaterializedImm(*MRI, *Src1);
1850 std::optional<int64_t> Src0Imm =
TII->getImmOrMaterializedImm(*MRI, *Src0);
1851 if (!Src0Imm || *Src0Imm != *Src1Imm)
1856 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1_modifiers);
1858 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0_modifiers);
1859 if ((Src1ModIdx != -1 &&
MI.getOperand(Src1ModIdx).getImm() != 0) ||
1860 (Src0ModIdx != -1 &&
MI.getOperand(Src0ModIdx).getImm() != 0))
1866 int Src2Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2);
1868 MI.removeOperand(Src2Idx);
1869 MI.removeOperand(AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1));
1870 if (Src1ModIdx != -1)
1871 MI.removeOperand(Src1ModIdx);
1872 if (Src0ModIdx != -1)
1873 MI.removeOperand(Src0ModIdx);
1874 TII->mutateAndCleanupImplicit(
MI, NewDesc);
1881std::optional<SIFoldOperandsImpl::ANDMaskResult>
1882SIFoldOperandsImpl::getANDMaskRegOperand(MachineInstr &AndMI)
const {
1884 if (
Opc != AMDGPU::V_AND_B32_e64 &&
Opc != AMDGPU::V_AND_B32_e32 &&
1885 Opc != AMDGPU::S_AND_B32)
1886 return std::nullopt;
1888 std::optional<int64_t> MaskImm =
1893 MaskImm =
TII->getImmOrMaterializedImm(*MRI, AndMI.
getOperand(2));
1897 return std::nullopt;
1910bool SIFoldOperandsImpl::tryFoldRedundantAND(MachineInstr &ChildMI)
const {
1915 std::optional<ANDMaskResult> ChildResult = getANDMaskRegOperand(ChildMI);
1919 if (!ChildResult->Reg.isVirtual())
1922 MachineInstr *ParentMI = MRI->
getVRegDef(ChildResult->Reg);
1926 int64_t ParentMask = 0;
1927 std::optional<ANDMaskResult> ParentResult = getANDMaskRegOperand(*ParentMI);
1930 ParentMask = ParentResult->Mask;
1933 ParentMask = 0xffff;
1939 if ((ParentMask & ChildResult->Mask) != ParentMask)
1961bool SIFoldOperandsImpl::foldInstOperand(MachineInstr &
MI,
1962 const FoldableDef &OpToFold)
const {
1966 SmallVector<MachineInstr *, 4> CopiesToReplace;
1968 MachineOperand &Dst =
MI.getOperand(0);
1973 for (
auto *U : UsesToProcess) {
1974 MachineInstr *
UseMI =
U->getParent();
1976 FoldableDef SubOpToFold = OpToFold.getWithSubReg(*
TRI,
U->getSubReg());
1981 if (CopiesToReplace.
empty() && FoldList.
empty())
1985 for (MachineInstr *Copy : CopiesToReplace)
1986 Copy->addImplicitDefUseOperands(*MF);
1988 SetVector<MachineInstr *> ConstantFoldCandidates;
1989 for (FoldCandidate &Fold : FoldList) {
1990 assert(!Fold.isReg() || Fold.Def.OpToFold);
1991 if (Fold.isReg() && Fold.getReg().isVirtual()) {
1993 const MachineInstr *
DefMI = Fold.Def.DefMI;
2001 assert(Fold.Def.OpToFold && Fold.isReg());
2008 <<
static_cast<int>(Fold.UseOpNo) <<
" of "
2012 ConstantFoldCandidates.
insert(Fold.UseMI);
2014 }
else if (Fold.Commuted) {
2016 TII->commuteInstruction(*Fold.UseMI,
false);
2020 for (MachineInstr *
MI : ConstantFoldCandidates) {
2021 if (tryConstantFoldOp(
MI)) {
2031bool SIFoldOperandsImpl::foldCopyToAGPRRegSequence(MachineInstr *CopyMI)
const {
2038 if (!
TRI->isAGPRClass(DefRC))
2050 DenseMap<TargetInstrInfo::RegSubRegPair, Register> VGPRCopies;
2059 unsigned NumFoldable = 0;
2061 for (
unsigned I = 1;
I != NumRegSeqOperands;
I += 2) {
2078 DefRC, &AMDGPU::AGPR_32RegClass, SubRegIdx);
2098 TRI->getMatchingSuperRegClass(DefRC, InputRC, SubRegIdx);
2109 if (NumFoldable == 0)
2112 CopyMI->
setDesc(
TII->get(AMDGPU::REG_SEQUENCE));
2116 for (
auto [Def, DestSubIdx] : NewDefs) {
2117 if (!
Def->isReg()) {
2121 BuildMI(
MBB, CopyMI,
DL,
TII->get(AMDGPU::V_ACCVGPR_WRITE_B32_e64), Tmp)
2126 Def->setIsKill(
false);
2128 Register &VGPRCopy = VGPRCopies[Src];
2131 TRI->getSubRegisterClass(UseRC, DestSubIdx);
2156 B.addImm(DestSubIdx);
2163bool SIFoldOperandsImpl::tryFoldFoldableCopy(
2164 MachineInstr &
MI, MachineOperand *&CurrentKnownM0Val)
const {
2168 if (DstReg == AMDGPU::M0) {
2169 MachineOperand &NewM0Val =
MI.getOperand(1);
2170 if (CurrentKnownM0Val && CurrentKnownM0Val->
isIdenticalTo(NewM0Val)) {
2171 MI.eraseFromParent();
2182 MachineOperand *OpToFoldPtr;
2183 if (
MI.getOpcode() == AMDGPU::V_MOV_B16_t16_e64) {
2185 if (
TII->hasAnyModifiersSet(
MI))
2187 OpToFoldPtr = &
MI.getOperand(2);
2189 OpToFoldPtr = &
MI.getOperand(1);
2190 MachineOperand &OpToFold = *OpToFoldPtr;
2194 if (!FoldingImm && !OpToFold.
isReg())
2199 !
TRI->isConstantPhysReg(OpToFold.
getReg()))
2228 if (
MI.getOpcode() == AMDGPU::COPY && OpToFold.
isReg() &&
2230 if (DstRC == &AMDGPU::SReg_32RegClass &&
2232 if (!
TRI->getMatchingSuperRegClass(DstRC, &AMDGPU::SGPR_LO16RegClass,
2241 if (OpToFold.
isReg() &&
MI.isCopy() && !
MI.getOperand(1).getSubReg()) {
2242 if (foldCopyToAGPRRegSequence(&
MI))
2246 FoldableDef
Def(OpToFold, DstRC);
2247 bool Changed = foldInstOperand(
MI, Def);
2254 auto *InstToErase = &
MI;
2256 auto &SrcOp = InstToErase->getOperand(1);
2258 InstToErase->eraseFromParent();
2260 InstToErase =
nullptr;
2264 if (!InstToErase || !
TII->isFoldableCopy(*InstToErase))
2268 if (InstToErase && InstToErase->isRegSequence() &&
2270 InstToErase->eraseFromParent();
2280 return OpToFold.
isReg() &&
2281 foldCopyToVGPROfScalarAddOfFrameIndex(DstReg, OpToFold.
getReg(),
MI);
2286const MachineOperand *
2287SIFoldOperandsImpl::isClamp(
const MachineInstr &
MI)
const {
2288 unsigned Op =
MI.getOpcode();
2290 case AMDGPU::V_MAX_F32_e64:
2291 case AMDGPU::V_MAX_F16_e64:
2292 case AMDGPU::V_MAX_F16_t16_e64:
2293 case AMDGPU::V_MAX_F16_fake16_e64:
2294 case AMDGPU::V_MAX_F64_e64:
2295 case AMDGPU::V_MAX_NUM_F64_e64:
2296 case AMDGPU::V_PK_MAX_F16:
2297 case AMDGPU::V_MAX_BF16_PSEUDO_e64:
2298 case AMDGPU::V_PK_MAX_NUM_BF16: {
2299 if (
MI.mayRaiseFPException())
2302 if (!
TII->getNamedOperand(
MI, AMDGPU::OpName::clamp)->getImm())
2306 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2307 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2311 Src0->
getSubReg() != AMDGPU::NoSubRegister)
2315 if (
TII->hasModifiersSet(
MI, AMDGPU::OpName::omod))
2319 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0_modifiers)->getImm();
2321 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1_modifiers)->getImm();
2325 unsigned UnsetMods =
2326 (
Op == AMDGPU::V_PK_MAX_F16 ||
Op == AMDGPU::V_PK_MAX_NUM_BF16)
2329 if (Src0Mods != UnsetMods && Src1Mods != UnsetMods)
2339bool SIFoldOperandsImpl::tryFoldClamp(MachineInstr &
MI) {
2340 const MachineOperand *ClampSrc = isClamp(
MI);
2356 if (
Def->mayRaiseFPException())
2359 MachineOperand *DefClamp =
TII->getNamedOperand(*Def, AMDGPU::OpName::clamp);
2363 LLVM_DEBUG(
dbgs() <<
"Folding clamp " << *DefClamp <<
" into " << *Def);
2369 Register MIDstReg =
MI.getOperand(0).getReg();
2370 if (
TRI->isSGPRReg(*MRI, DefReg)) {
2379 MI.eraseFromParent();
2384 if (
TII->convertToThreeAddress(*Def,
nullptr,
nullptr))
2385 Def->eraseFromParent();
2392 case AMDGPU::V_MUL_F64_e64:
2393 case AMDGPU::V_MUL_F64_pseudo_e64: {
2395 case 0x3fe0000000000000:
2397 case 0x4000000000000000:
2399 case 0x4010000000000000:
2405 case AMDGPU::V_MUL_F32_e64: {
2406 switch (
static_cast<uint32_t>(Val)) {
2417 case AMDGPU::V_MUL_F16_e64:
2418 case AMDGPU::V_MUL_F16_t16_e64:
2419 case AMDGPU::V_MUL_F16_fake16_e64: {
2420 switch (
static_cast<uint16_t>(Val)) {
2431 case AMDGPU::V_PK_MUL_BF16: {
2432 switch (
static_cast<uint16_t>(Val)) {
2451std::pair<const MachineOperand *, int>
2452SIFoldOperandsImpl::isOMod(
const MachineInstr &
MI)
const {
2453 unsigned Op =
MI.getOpcode();
2455 case AMDGPU::V_MUL_F64_e64:
2456 case AMDGPU::V_MUL_F64_pseudo_e64:
2457 case AMDGPU::V_MUL_F32_e64:
2458 case AMDGPU::V_MUL_F16_t16_e64:
2459 case AMDGPU::V_MUL_F16_fake16_e64:
2460 case AMDGPU::V_MUL_F16_e64: {
2462 if ((
Op == AMDGPU::V_MUL_F32_e64 &&
2464 ((
Op == AMDGPU::V_MUL_F64_e64 ||
Op == AMDGPU::V_MUL_F64_pseudo_e64 ||
2465 Op == AMDGPU::V_MUL_F16_e64 ||
Op == AMDGPU::V_MUL_F16_t16_e64 ||
2466 Op == AMDGPU::V_MUL_F16_fake16_e64) &&
2469 MI.mayRaiseFPException())
2472 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2473 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2476 std::optional<int64_t> Src1Imm =
TII->getImmOrMaterializedImm(*MRI, *Src1);
2482 TII->hasModifiersSet(
MI, AMDGPU::OpName::src0_modifiers) ||
2483 TII->hasModifiersSet(
MI, AMDGPU::OpName::src1_modifiers) ||
2484 TII->hasModifiersSet(
MI, AMDGPU::OpName::omod) ||
2485 TII->hasModifiersSet(
MI, AMDGPU::OpName::clamp))
2488 return {Src0, OMod};
2490 case AMDGPU::V_ADD_F64_e64:
2491 case AMDGPU::V_ADD_F64_pseudo_e64:
2492 case AMDGPU::V_ADD_F32_e64:
2493 case AMDGPU::V_ADD_F16_e64:
2494 case AMDGPU::V_ADD_F16_t16_e64:
2495 case AMDGPU::V_ADD_F16_fake16_e64: {
2497 if ((
Op == AMDGPU::V_ADD_F32_e64 &&
2499 ((
Op == AMDGPU::V_ADD_F64_e64 ||
Op == AMDGPU::V_ADD_F64_pseudo_e64 ||
2500 Op == AMDGPU::V_ADD_F16_e64 ||
Op == AMDGPU::V_ADD_F16_t16_e64 ||
2501 Op == AMDGPU::V_ADD_F16_fake16_e64) &&
2506 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2507 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2511 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::src0_modifiers) &&
2512 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::src1_modifiers) &&
2513 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::clamp) &&
2514 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::omod))
2519 case AMDGPU::V_PK_MUL_BF16: {
2524 MI.mayRaiseFPException())
2527 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2528 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2531 std::optional<int64_t> Src1Imm =
TII->getImmOrMaterializedImm(*MRI, *Src1);
2535 int OMod =
getOModValue(AMDGPU::V_PK_MUL_BF16, *Src1Imm);
2540 const MachineOperand *Src0Mods =
2541 TII->getNamedOperand(
MI, AMDGPU::OpName::src0_modifiers);
2542 const MachineOperand *Src1Mods =
2543 TII->getNamedOperand(
MI, AMDGPU::OpName::src1_modifiers);
2546 TII->hasModifiersSet(
MI, AMDGPU::OpName::omod) ||
2547 TII->hasModifiersSet(
MI, AMDGPU::OpName::clamp))
2550 return {Src0, OMod};
2552 case AMDGPU::V_PK_ADD_BF16: {
2558 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2559 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2566 const MachineOperand *Src0Mods =
2567 TII->getNamedOperand(
MI, AMDGPU::OpName::src0_modifiers);
2568 const MachineOperand *Src1Mods =
2569 TII->getNamedOperand(
MI, AMDGPU::OpName::src1_modifiers);
2572 TII->hasModifiersSet(
MI, AMDGPU::OpName::omod) ||
2573 TII->hasModifiersSet(
MI, AMDGPU::OpName::clamp))
2584bool SIFoldOperandsImpl::tryFoldOMod(MachineInstr &
MI) {
2585 const MachineOperand *RegOp;
2587 std::tie(RegOp, OMod) = isOMod(
MI);
2589 RegOp->
getSubReg() != AMDGPU::NoSubRegister ||
2594 Register OModSrcReg =
Def->getOperand(0).getReg();
2598 if (
Def->isRegSequence() &&
Def->getNumOperands() == 5 &&
2599 Def->getOperand(2).getImm() == AMDGPU::lo16) {
2601 bool CanLookThrough =
true;
2602 MachineInstr *Hi16Def = MRI->
getVRegDef(
Def->getOperand(3).getReg());
2604 CanLookThrough =
false;
2606 if (CanLookThrough) {
2617 MachineOperand *DefOMod =
TII->getNamedOperand(*Def, AMDGPU::OpName::omod);
2621 if (
Def->mayRaiseFPException())
2626 if (
TII->hasModifiersSet(*Def, AMDGPU::OpName::clamp))
2636 MI.eraseFromParent();
2641 if (
TII->convertToThreeAddress(*Def,
nullptr,
nullptr))
2642 Def->eraseFromParent();
2649bool SIFoldOperandsImpl::tryFoldSGPRSplatRegSequence(MachineInstr &
MI) {
2652 if (!ST->hasPackedFP64SingleSGPROps() && !ST->hasPackedU64SingleSGPROps())
2659 if (!
TRI->isSGPRClass(RegClass) ||
TRI->getRegSizeInBits(*RegClass) != 128)
2663 if (!getRegSeqInit(Defs,
Reg))
2667 if (Defs.
size() <= 1)
2670 const auto &[FirstOp,
_] = Defs.
front();
2671 if (!FirstOp->isReg())
2674 Register FirstReg = FirstOp->getReg();
2675 unsigned FirstSubReg = FirstOp->getSubReg();
2678 if (!
TRI->isSGPRClass(FirstRegClass))
2683 const auto &[
Op,
_] =
Def;
2684 return Op->isReg() &&
Op->getReg() == FirstReg &&
2685 Op->getSubReg() == FirstSubReg;
2697 MachineInstrBuilder
RS =
BuildMI(*
MI.getParent(),
MI,
MI.getDebugLoc(),
2698 TII->get(AMDGPU::REG_SEQUENCE), NewDst);
2701 FirstOp->setIsKill(
false);
2703 RS.addImm(Defs[0].second);
2708 for (
unsigned i = 1; i < Defs.
size(); ++i) {
2709 RS.addReg(UndefReg, RegState::Undef);
2710 RS.addImm(Defs[i].second);
2719 MI.eraseFromParent();
2725bool SIFoldOperandsImpl::tryFoldRegSequence(MachineInstr &
MI) {
2729 if (tryFoldSGPRSplatRegSequence(
MI))
2732 auto Reg =
MI.getOperand(0).getReg();
2734 if (!ST->hasGFX90AInsts() || !
TRI->isVGPR(*MRI,
Reg) ||
2739 if (!getRegSeqInit(Defs,
Reg))
2742 for (
auto &[
Op, SubIdx] : Defs) {
2745 if (
TRI->isAGPR(*MRI,
Op->getReg()))
2748 const MachineInstr *SubDef = MRI->
getVRegDef(
Op->getReg());
2756 MachineInstr *
UseMI =
Op->getParent();
2765 if (
Op->getSubReg())
2771 if (!OpRC || !
TRI->isVectorSuperClass(OpRC))
2777 TII->get(AMDGPU::REG_SEQUENCE), Dst);
2779 for (
auto &[Def, SubIdx] : Defs) {
2780 Def->setIsKill(
false);
2781 if (
TRI->isAGPR(*MRI,
Def->getReg())) {
2792 if (!
TII->isOperandLegal(*
UseMI, OpIdx,
Op)) {
2794 RS->eraseFromParent();
2803 MI.eraseFromParent();
2811 Register &OutReg,
unsigned &OutSubReg) {
2821 if (
TRI.isAGPR(MRI, CopySrcReg)) {
2822 OutReg = CopySrcReg;
2831 if (!CopySrcDef || !CopySrcDef->
isCopy())
2838 OtherCopySrc.
getSubReg() != AMDGPU::NoSubRegister ||
2839 !
TRI.isAGPR(MRI, OtherCopySrcReg))
2842 OutReg = OtherCopySrcReg;
2876bool SIFoldOperandsImpl::tryFoldPhiAGPR(MachineInstr &
PHI) {
2880 if (!
TRI->isVGPR(*MRI, PhiOut))
2886 for (
unsigned K = 1;
K <
PHI.getNumExplicitOperands();
K += 2) {
2887 MachineOperand &MO =
PHI.getOperand(K);
2889 if (!Copy || !
Copy->isCopy())
2893 unsigned AGPRRegMask = AMDGPU::NoSubRegister;
2898 if (
const auto *SubRC =
TRI->getSubRegisterClass(CopyInRC, AGPRRegMask))
2909 bool IsAGPR32 = (ARC == &AMDGPU::AGPR_32RegClass);
2913 for (
unsigned K = 1;
K <
PHI.getNumExplicitOperands();
K += 2) {
2914 MachineOperand &MO =
PHI.getOperand(K);
2918 MachineBasicBlock *InsertMBB =
nullptr;
2921 unsigned CopyOpc = AMDGPU::COPY;
2926 if (
Def->isCopy()) {
2928 unsigned AGPRSubReg = AMDGPU::NoSubRegister;
2941 MachineOperand &CopyIn =
Def->getOperand(1);
2944 CopyOpc = AMDGPU::V_ACCVGPR_WRITE_B32_e64;
2947 InsertMBB =
Def->getParent();
2955 MachineInstr *
MI =
BuildMI(*InsertMBB, InsertPt,
PHI.getDebugLoc(),
2956 TII->get(CopyOpc), NewReg)
2966 PHI.getOperand(0).setReg(NewReg);
2972 TII->get(AMDGPU::COPY), PhiOut)
2980bool SIFoldOperandsImpl::tryFoldLoad(MachineInstr &
MI) {
2982 if (!ST->hasGFX90AInsts() ||
MI.getNumExplicitDefs() != 1)
2985 MachineOperand &
Def =
MI.getOperand(0);
3002 while (!
Users.empty()) {
3003 const MachineInstr *
I =
Users.pop_back_val();
3004 if (!
I->isCopy() && !
I->isRegSequence())
3006 Register DstReg =
I->getOperand(0).getReg();
3010 if (
TRI->isAGPR(*MRI, DstReg))
3014 Users.push_back(&U);
3019 if (!
TII->isOperandLegal(
MI, 0, &Def)) {
3024 while (!MoveRegs.
empty()) {
3066bool SIFoldOperandsImpl::tryOptimizeAGPRPhis(MachineBasicBlock &
MBB) {
3069 if (ST->hasGFX90AInsts())
3073 DenseMap<std::pair<Register, unsigned>, std::vector<MachineOperand *>>
3076 for (
auto &
MI :
MBB) {
3080 if (!
TRI->isAGPR(*MRI,
MI.getOperand(0).getReg()))
3083 for (
unsigned K = 1;
K <
MI.getNumOperands();
K += 2) {
3084 MachineOperand &PhiMO =
MI.getOperand(K);
3094 for (
const auto &[Entry, MOs] : RegToMO) {
3095 if (MOs.size() == 1)
3100 MachineBasicBlock *DefMBB =
Def->getParent();
3107 MachineInstr *VGPRCopy =
3109 TII->get(AMDGPU::V_ACCVGPR_READ_B32_e64), TempVGPR)
3115 TII->get(AMDGPU::COPY), TempAGPR)
3119 for (MachineOperand *MO : MOs) {
3131bool SIFoldOperandsImpl::run(
MachineFunction &MF,
const MachineLoopInfo *MLI) {
3137 MFI = MF.
getInfo<SIMachineFunctionInfo>();
3148 MachineOperand *CurrentKnownM0Val =
nullptr;
3156 if (tryConstantFoldOp(&
MI)) {
3161 if (tryFoldRedundantAND(
MI)) {
3166 if (
MI.isRegSequence() && tryFoldRegSequence(
MI)) {
3171 if (
MI.isPHI() && tryFoldPhiAGPR(
MI)) {
3176 if (
MI.mayLoad() && tryFoldLoad(
MI)) {
3181 if (
TII->isFoldableCopy(
MI)) {
3182 Changed |= tryFoldFoldableCopy(
MI, CurrentKnownM0Val);
3187 if (CurrentKnownM0Val &&
MI.modifiesRegister(AMDGPU::M0,
TRI))
3188 CurrentKnownM0Val =
nullptr;
3208 bool Changed = SIFoldOperandsImpl().run(MF, MLI);
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static bool updateOperand(Instruction *Inst, unsigned Idx, Instruction *Mat)
Updates the operand at Idx in instruction Inst with the result of instruction Mat.
This file builds on the ADT/GraphTraits.h file to build generic depth first graph iterator.
AMD GCN specific subclass of TargetSubtarget.
static Register UseReg(const MachineOperand &MO)
const HexagonInstrInfo * TII
iv Induction Variable Users
Register const TargetRegisterInfo * TRI
Promote Memory to Register
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
static bool isReg(const MCInst &MI, unsigned OpNo)
if(auto Err=PB.parsePassPipeline(MPM, Passes)) return wrap(std MPM run * Mod
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
static bool loopModifiesExec(const MachineLoop &L, const SIRegisterInfo &TRI)
static unsigned macToMad(unsigned Opc)
static bool isAGPRCopy(const SIRegisterInfo &TRI, const MachineRegisterInfo &MRI, const MachineInstr &Copy, Register &OutReg, unsigned &OutSubReg)
Checks whether Copy is a AGPR -> VGPR copy.
static void appendFoldCandidate(SmallVectorImpl< FoldCandidate > &FoldList, FoldCandidate &&Entry)
static const TargetRegisterClass * getRegOpRC(const MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI, const MachineOperand &MO)
static bool evalBinaryInstruction(unsigned Opcode, int32_t &Result, uint32_t LHS, uint32_t RHS)
static int getOModValue(unsigned Opc, int64_t Val)
static unsigned getMovOpc(bool IsScalar)
static MachineOperand * lookUpCopyChain(const SIInstrInfo &TII, const MachineRegisterInfo &MRI, Register SrcReg)
static bool checkImmOpForPKF32InstrReplicatesLower32BitsOfScalarOperand(const FoldableDef &OpToFold)
static bool isPKF32InstrReplicatesLower32BitsOfScalarOperand(const GCNSubtarget *ST, MachineInstr *MI, unsigned OpNo)
Interface definition for SIInstrInfo.
Interface definition for SIRegisterInfo.
static int Lookup(ArrayRef< TableEntry > Table, unsigned Opcode)
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Represents analyses that only rely on functions' control flow.
FunctionPass class - This class is used to implement most global optimizations.
const SIInstrInfo * getInstrInfo() const override
bool hasDOTOpSelHazard() const
bool zeroesHigh16BitsOfDest(unsigned Opcode) const
Returns if the result of this instruction with a 16-bit result returned in a 32-bit register implicit...
const HexagonRegisterInfo & getRegisterInfo() const
bool contains(const LoopT *L) const
Return true if the specified loop is contained within this loop.
LoopT * getLoopFor(const BlockT *BB) const
Return the inner most loop that BB lives in.
ArrayRef< MCOperandInfo > operands() const
int getOperandConstraint(unsigned OpNum, MCOI::OperandConstraint Constraint) const
Returns the value of the specified operand constraint if it is present.
bool isVariadic() const
Return true if this instruction can have a variable number of operands.
This holds information about one operand of a machine instruction, indicating the register class for ...
uint8_t OperandType
Information about the type of the operand.
bool hasSuperClassEq(const MCRegisterClass *RC) const
Returns true if RC is a super-class of or equal to this class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
bool hasSubClassEq(const MCRegisterClass *RC) const
Returns true if RC is a sub-class of or equal to this class.
An RAII based helper class to modify MachineFunctionProperties when running pass.
LLVM_ABI iterator SkipPHIsLabelsAndDebug(iterator I, Register Reg=Register(), bool SkipPseudoOp=true)
Return the first instruction in MBB after I that is not a PHI, label or debug.
LLVM_ABI LivenessQueryResult computeRegisterLiveness(const TargetRegisterInfo *TRI, MCRegister Reg, const_iterator Before, unsigned Neighborhood=10) const
Return whether (physical) register Reg has been defined and not killed as of just before Before.
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
LLVM_ABI iterator getFirstNonPHI()
Returns a pointer to the first instruction in this block that is not a PHINode instruction.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
MachineInstrBundleIterator< MachineInstr > iterator
LivenessQueryResult
Possible outcome of a register liveness query to computeRegisterLiveness()
@ LQR_Dead
Register is known to be fully dead.
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
Properties which a MachineFunction may have at a given point in time.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Register getReg(unsigned Idx) const
Get the register for the operand index.
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
bool isImplicitDef() const
const MachineBasicBlock * getParent() const
bool readsRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr reads the specified register.
LLVM_ABI bool allImplicitDefsAreDead() const
Return true if all the implicit defs of this instruction are dead.
unsigned getNumOperands() const
Retuns the total number of operands.
LLVM_ABI void addOperand(MachineFunction &MF, const MachineOperand &Op)
Add the specified operand to the instruction.
unsigned getOperandNo(const_mop_iterator I) const
Returns the number of the operand iterator I points to.
LLVM_ABI unsigned getNumExplicitOperands() const
Returns the number of non-implicit operands.
mop_range implicit_operands()
const MCInstrDesc & getDesc() const
Returns the target instruction descriptor of this MachineInstr.
void clearFlag(MIFlag Flag)
clearFlag - Clear a MI flag.
bool isRegSequence() const
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
MachineOperand * mop_iterator
iterator/begin/end - Iterate over all operands of a machine instruction.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
LLVM_ABI void removeOperand(unsigned OpNo)
Erase an operand from an instruction, leaving it with one fewer operand than it started with.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
Analysis pass that exposes the MachineLoopInfo for a machine function.
MachineOperand class - Representation of each machine instruction operand.
void setSubReg(unsigned subReg)
unsigned getSubReg() const
LLVM_ABI unsigned getOperandNo() const
Returns the index of this operand in the instruction that it belongs to.
LLVM_ABI void substVirtReg(Register Reg, unsigned SubIdx, const TargetRegisterInfo &)
substVirtReg - Substitute the current register with the virtual subregister Reg:SubReg.
LLVM_ABI void ChangeToFrameIndex(int Idx, unsigned TargetFlags=0)
Replace this operand with a frame index.
void setImm(int64_t immVal)
bool isReg() const
isReg - Tests if this is a MO_Register operand.
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
LLVM_ABI void ChangeToImmediate(int64_t ImmVal, unsigned TargetFlags=0)
ChangeToImmediate - Replace this operand with a new immediate operand of the specified value.
LLVM_ABI void ChangeToGA(const GlobalValue *GV, int64_t Offset, unsigned TargetFlags=0)
ChangeToGA - Replace this operand with a new global address operand.
void setIsKill(bool Val=true)
LLVM_ABI void ChangeToRegister(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isDebug=false)
ChangeToRegister - Replace this operand with a new register operand of the specified value.
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
LLVM_ABI void substPhysReg(MCRegister Reg, const TargetRegisterInfo &)
substPhysReg - Substitute the current register with the physical register Reg, taking any existing Su...
static MachineOperand CreateImm(int64_t Val)
bool isGlobal() const
isGlobal - Tests if this is a MO_GlobalAddress operand.
MachineOperandType getType() const
getType - Returns the MachineOperandType for this operand.
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
bool isFI() const
isFI - Tests if this is a MO_FrameIndex operand.
LLVM_ABI bool isIdenticalTo(const MachineOperand &Other) const
Returns true if this operand is identical to the specified operand except for liveness related flags ...
@ MO_Immediate
Immediate operand.
@ MO_GlobalAddress
Address of a global value.
@ MO_FrameIndex
Abstract Stack Frame Index.
@ MO_Register
Register operand.
static MachineOperand CreateFI(int Idx)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
use_nodbg_iterator use_nodbg_begin(Register RegNo) const
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
iterator_range< use_nodbg_iterator > use_nodbg_operands(Register Reg) const
bool use_nodbg_empty(Register RegNo) const
use_nodbg_empty - Return true if there are no non-Debug instructions using the specified register.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLVM_ABI bool hasOneNonDBGUser(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug instruction using the specified regis...
iterator_range< use_instr_nodbg_iterator > use_nodbg_instructions(Register Reg) const
void setRegAllocationHint(Register VReg, unsigned Type, Register PrefReg)
setRegAllocationHint - Specify a register allocation hint for the specified virtual register.
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
LLVM_ABI const TargetRegisterClass * constrainRegClass(Register Reg, const TargetRegisterClass *RC, unsigned MinNumRegs=0)
constrainRegClass - Constrain the register class of the specified virtual register to be a common sub...
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Wrapper class representing virtual and physical registers.
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
static bool hasSameClamp(const MachineInstr &A, const MachineInstr &B)
static std::optional< int64_t > extractSubregFromImm(int64_t ImmVal, unsigned SubRegIndex)
Return the extracted immediate value in a subregister use from a constant materialized in a super reg...
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
Register getScratchRSrcReg() const
Returns the physical register reserved for use as the resource descriptor for scratch accesses.
SIModeRegisterDefaults getMode() const
bool insert(const value_type &X)
Insert a new element into the SetVector.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
Represent a constant reference to a string, i.e.
static const unsigned CommuteAnyOperandIndex
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
bool isInlinableLiteralV216(uint32_t Literal, uint8_t OpType)
LLVM_READONLY int32_t getMFMAEarlyClobberOp(uint32_t Opcode)
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
bool isPackedSingleSGPR64BitInst(unsigned Opc)
The opcode is a packed 64-bit instruction which only reads low 64 bits of a scalar operand and propag...
LLVM_READONLY int32_t getVOPe32(uint32_t Opcode)
constexpr bool isSISrcOperand(const MCOperandInfo &OpInfo)
Is this an AMDGPU specific source operand?
@ OPERAND_REG_INLINE_C_FP64
@ OPERAND_REG_INLINE_C_BF16
@ OPERAND_REG_INLINE_C_V2BF16
@ OPERAND_REG_IMM_V2INT64
@ OPERAND_REG_IMM_V2INT16
@ OPERAND_REG_INLINE_C_INT64
@ OPERAND_REG_IMM_NOINLINE_V2FP16
@ OPERAND_REG_INLINE_C_V2FP16
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
@ OPERAND_REG_INLINE_AC_FP32
@ OPERAND_REG_INLINE_C_FP32
@ OPERAND_REG_INLINE_C_INT32
@ OPERAND_REG_INLINE_C_V2INT16
@ OPERAND_REG_INLINE_AC_FP64
LLVM_READONLY int32_t getFlatScratchInstSSfromSV(uint32_t Opcode)
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
constexpr bool isVOP3(const T &...O)
constexpr bool isMAI(const T &...O)
constexpr bool isSWMMAC(const T &...O)
constexpr bool isVOP3P(const T &...O)
constexpr bool isWMMA(const T &...O)
constexpr bool isDOT(const T &...O)
constexpr bool isPacked(const T &...O)
NodeAddr< DefNode * > Def
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
TargetInstrInfo::RegSubRegPair getRegSubRegPair(const MachineOperand &O)
Create RegSubRegPair from a register MachineOperand.
MachineBasicBlock::instr_iterator getBundleStart(MachineBasicBlock::instr_iterator I)
Returns an iterator to the first instruction in the bundle containing I.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
bool execMayBeModifiedBeforeUse(const MachineRegisterInfo &MRI, Register VReg, const MachineInstr &DefMI, const MachineInstr &UseMI)
Return false if EXEC is not changed between the def of VReg at DefMI and the use at UseMI.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
FunctionPass * createSIFoldOperandsLegacyPass()
char & SIFoldOperandsLegacyID
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
@ Sub
Subtraction of integers.
DWARFExpression::Operation Op
iterator_range< pointer_iterator< WrappedIteratorT > > make_pointer_range(RangeT &&Range)
iterator_range< df_iterator< T > > depth_first(const T &G)
LLVM_ABI Printable printReg(Register Reg, const TargetRegisterInfo *TRI=nullptr, unsigned SubIdx=0, const MachineRegisterInfo *MRI=nullptr)
Prints virtual and physical registers with or without a TRI instance.
constexpr uint64_t Make_64(uint32_t High, uint32_t Low)
Make a 64-bit integer from a high / low pair of 32-bit integers.
MCRegisterClass TargetRegisterClass
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
@ PreserveSign
The sign of a flushed-to-zero number is preserved in the sign of 0.
DenormalModeKind Output
Denormal flushing mode for floating point instruction results in the default floating point environme...
DenormalMode FP64FP16Denormals
If this is set, neither input or output denormals are flushed for both f64 and f16/v2f16 instructions...
bool IEEE
Floating point opcodes that support exception flag gathering quiet and propagate signaling NaN inputs...
DenormalMode FP32Denormals
If this is set, neither input or output denormals are flushed for most f32 instructions.