25#define DEBUG_TYPE "si-fold-operands"
46 unsigned DefSubReg = AMDGPU::NoSubRegister;
51 FoldableDef() =
delete;
53 unsigned DefSubReg = AMDGPU::NoSubRegister)
54 : DefRC(DefRC), DefSubReg(DefSubReg), Kind(FoldOp.
getType()) {
57 ImmToFold = FoldOp.
getImm();
58 }
else if (FoldOp.
isFI()) {
59 FrameIndexToFold = FoldOp.
getIndex();
69 unsigned DefSubReg = AMDGPU::NoSubRegister)
70 : ImmToFold(FoldImm), DefRC(DefRC), DefSubReg(DefSubReg),
75 FoldableDef Copy(*
this);
76 Copy.DefSubReg =
TRI.composeSubRegIndices(DefSubReg, SubReg);
84 return OpToFold->getReg();
87 unsigned getSubReg()
const {
89 return OpToFold->getSubReg();
100 return FrameIndexToFold;
108 std::optional<int64_t> getEffectiveImmVal()
const {
116 unsigned OpIdx)
const {
119 std::optional<int64_t> ImmToFold = getEffectiveImmVal();
126 return TII.isOperandLegal(
MI, OpIdx, &TmpOp);
129 if (DefSubReg != AMDGPU::NoSubRegister)
132 return TII.isOperandLegal(
MI, OpIdx, &TmpOp);
137 if (DefSubReg != AMDGPU::NoSubRegister)
139 return TII.isOperandLegal(
MI, OpIdx, OpToFold);
146struct FoldCandidate {
154 bool Commuted =
false,
int ShrinkOp = -1)
155 :
UseMI(
MI), Def(Def), ShrinkOpcode(ShrinkOp), UseOpNo(OpNo),
156 Commuted(Commuted) {}
158 bool isFI()
const {
return Def.isFI(); }
162 return Def.FrameIndexToFold;
165 bool isImm()
const {
return Def.isImm(); }
167 bool isReg()
const {
return Def.isReg(); }
171 bool isGlobal()
const {
return Def.isGlobal(); }
173 bool needsShrink()
const {
return ShrinkOpcode != -1; }
176class SIFoldOperandsImpl {
187 const FoldableDef &OpToFold)
const;
190 unsigned convertToVALUOp(
unsigned Opc,
bool UseVOP3 =
false)
const {
192 case AMDGPU::S_ADD_I32: {
193 if (ST->hasAddNoCarryInsts())
194 return UseVOP3 ? AMDGPU::V_ADD_U32_e64 : AMDGPU::V_ADD_U32_e32;
195 return UseVOP3 ? AMDGPU::V_ADD_CO_U32_e64 : AMDGPU::V_ADD_CO_U32_e32;
197 case AMDGPU::S_OR_B32:
198 return UseVOP3 ? AMDGPU::V_OR_B32_e64 : AMDGPU::V_OR_B32_e32;
199 case AMDGPU::S_AND_B32:
200 return UseVOP3 ? AMDGPU::V_AND_B32_e64 : AMDGPU::V_AND_B32_e32;
201 case AMDGPU::S_MUL_I32:
202 return AMDGPU::V_MUL_LO_U32_e64;
204 return AMDGPU::INSTRUCTION_LIST_END;
208 bool foldCopyToVGPROfScalarAddOfFrameIndex(
Register DstReg,
Register SrcReg,
214 int64_t ImmVal)
const;
218 int64_t ImmVal)
const;
222 const FoldableDef &OpToFold)
const;
225 bool isTemporallyDivergentUse(
const FoldableDef &OpToFold,
233 getRegSeqInit(
SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs,
236 std::pair<int64_t, const TargetRegisterClass *>
250 struct ANDMaskResult {
256 std::optional<ANDMaskResult> getANDMaskRegOperand(
MachineInstr &AndMI)
const;
262 bool foldInstOperand(
MachineInstr &
MI,
const FoldableDef &OpToFold)
const;
264 bool foldCopyToAGPRRegSequence(
MachineInstr *CopyMI)
const;
271 std::pair<const MachineOperand *, int> isOMod(
const MachineInstr &
MI)
const;
281 SIFoldOperandsImpl() =
default;
296 &getAnalysis<MachineLoopInfoWrapperPass>().getLI();
297 return SIFoldOperandsImpl().run(MF, MLI);
300 StringRef getPassName()
const override {
return "SI Fold Operands"; }
322char SIFoldOperandsLegacy::ID = 0;
331 TRI.getSubRegisterClass(RC, MO.getSubReg()))
339 case AMDGPU::V_MAC_F32_e64:
340 return AMDGPU::V_MAD_F32_e64;
341 case AMDGPU::V_MAC_F16_e64:
342 return AMDGPU::V_MAD_F16_e64;
343 case AMDGPU::V_FMAC_F32_e64:
344 return AMDGPU::V_FMA_F32_e64;
345 case AMDGPU::V_FMAC_F16_e64:
346 return AMDGPU::V_FMA_F16_gfx9_e64;
347 case AMDGPU::V_FMAC_F16_t16_e64:
348 return AMDGPU::V_FMA_F16_gfx9_t16_e64;
349 case AMDGPU::V_FMAC_F16_fake16_e64:
350 return AMDGPU::V_FMA_F16_gfx9_fake16_e64;
351 case AMDGPU::V_FMAC_LEGACY_F32_e64:
352 return AMDGPU::V_FMA_LEGACY_F32_e64;
353 case AMDGPU::V_FMAC_F64_e64:
354 return AMDGPU::V_FMA_F64_e64;
356 return AMDGPU::INSTRUCTION_LIST_END;
362 const FoldableDef &OpToFold)
const {
363 if (!OpToFold.isFI())
366 const unsigned Opc =
UseMI.getOpcode();
368 case AMDGPU::S_ADD_I32:
369 case AMDGPU::S_ADD_U32:
370 case AMDGPU::V_ADD_U32_e32:
371 case AMDGPU::V_ADD_CO_U32_e32:
375 return UseMI.getOperand(OpNo == 1 ? 2 : 1).isImm() &&
377 case AMDGPU::V_ADD_U32_e64:
378 case AMDGPU::V_ADD_CO_U32_e64:
379 return UseMI.getOperand(OpNo == 2 ? 3 : 2).isImm() &&
386 return OpNo == AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr);
390 int SIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::saddr);
394 int VIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr);
395 return OpNo == VIdx && SIdx == -1;
401bool SIFoldOperandsImpl::foldCopyToVGPROfScalarAddOfFrameIndex(
406 if (
TRI->isVGPR(*MRI, DstReg) &&
TRI->isSGPRReg(*MRI, SrcReg) &&
409 if (!Def ||
Def->getNumOperands() != 4)
412 MachineOperand *Src0 = &
Def->getOperand(1);
413 MachineOperand *Src1 = &
Def->getOperand(2);
424 const bool UseVOP3 = !Src0->
isImm() ||
TII->isInlineConstant(*Src0);
425 unsigned NewOp = convertToVALUOp(
Def->getOpcode(), UseVOP3);
426 if (NewOp == AMDGPU::INSTRUCTION_LIST_END ||
427 !
Def->getOperand(3).isDead())
430 MachineBasicBlock *
MBB =
Def->getParent();
432 if (NewOp != AMDGPU::V_ADD_CO_U32_e32) {
433 MachineInstrBuilder
Add =
436 if (
Add->getDesc().getNumDefs() == 2) {
438 Add.addDef(CarryOutReg, RegState::Dead);
442 Add.add(*Src0).add(*Src1).setMIFlags(
Def->getFlags());
446 Def->eraseFromParent();
447 MI.eraseFromParent();
451 assert(NewOp == AMDGPU::V_ADD_CO_U32_e32);
462 Def->eraseFromParent();
463 MI.eraseFromParent();
472 return new SIFoldOperandsLegacy();
475bool SIFoldOperandsImpl::canUseImmWithOpSel(
const MachineInstr *
MI,
477 int64_t ImmVal)
const {
484 int OpNo =
MI->getOperandNo(&Old);
486 unsigned Opcode =
MI->getOpcode();
487 uint8_t OpType =
TII->get(Opcode).operands()[OpNo].OperandType;
509bool SIFoldOperandsImpl::tryFoldImmWithOpSel(MachineInstr *
MI,
unsigned UseOpNo,
510 int64_t ImmVal)
const {
511 MachineOperand &Old =
MI->getOperand(UseOpNo);
512 unsigned Opcode =
MI->getOpcode();
513 int OpNo =
MI->getOperandNo(&Old);
514 uint8_t OpType =
TII->get(Opcode).operands()[OpNo].OperandType;
516 bool BF16FromUpperFP32 = ST->hasBF16InlineConstFromUpperFP32() &&
530 AMDGPU::OpName ModName = AMDGPU::OpName::NUM_OPERAND_NAMES;
531 unsigned SrcIdx = ~0;
532 if (OpNo == AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0)) {
533 ModName = AMDGPU::OpName::src0_modifiers;
535 }
else if (OpNo == AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src1)) {
536 ModName = AMDGPU::OpName::src1_modifiers;
538 }
else if (OpNo == AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src2)) {
539 ModName = AMDGPU::OpName::src2_modifiers;
542 assert(ModName != AMDGPU::OpName::NUM_OPERAND_NAMES);
543 int ModIdx = AMDGPU::getNamedOperandIdx(Opcode, ModName);
544 MachineOperand &
Mod =
MI->getOperand(ModIdx);
545 unsigned ModVal =
Mod.getImm();
551 uint32_t
Imm = (
static_cast<uint32_t
>(ImmHi) << 16) | ImmLo;
556 auto tryFoldToInline = [&](uint32_t
Imm) ->
bool {
565 uint16_t
Lo =
static_cast<uint16_t
>(
Imm);
566 uint16_t
Hi =
static_cast<uint16_t
>(
Imm >> 16);
572 if (BF16FromUpperFP32)
574 Mod.setImm(NewModVal);
579 if (!BF16FromUpperFP32 &&
static_cast<int16_t
>(
Lo) < 0) {
580 int32_t SExt =
static_cast<int16_t
>(
Lo);
582 Mod.setImm(NewModVal);
597 uint32_t Swapped = (
static_cast<uint32_t
>(
Lo) << 16) |
Hi;
598 if (!BF16FromUpperFP32 &&
609 if (tryFoldToInline(
Imm))
618 bool IsUAdd = Opcode == AMDGPU::V_PK_ADD_U16;
619 bool IsUSub = Opcode == AMDGPU::V_PK_SUB_U16;
620 if (SrcIdx == 1 && (IsUAdd || IsUSub)) {
622 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::clamp);
623 bool Clamp =
MI->getOperand(ClampIdx).getImm() != 0;
626 uint16_t NegLo = -
static_cast<uint16_t
>(
Imm);
627 uint16_t NegHi = -
static_cast<uint16_t
>(
Imm >> 16);
628 uint32_t NegImm = (
static_cast<uint32_t
>(NegHi) << 16) | NegLo;
630 if (tryFoldToInline(NegImm)) {
632 IsUAdd ? AMDGPU::V_PK_SUB_U16 : AMDGPU::V_PK_ADD_U16;
633 MI->setDesc(
TII->get(NegOpcode));
642bool SIFoldOperandsImpl::updateOperand(FoldCandidate &Fold)
const {
643 MachineInstr *
MI = Fold.UseMI;
644 MachineOperand &Old =
MI->getOperand(Fold.UseOpNo);
647 std::optional<int64_t> ImmVal;
649 ImmVal = Fold.Def.getEffectiveImmVal();
651 if (ImmVal && canUseImmWithOpSel(Fold.UseMI, Fold.UseOpNo, *ImmVal)) {
652 if (tryFoldImmWithOpSel(Fold.UseMI, Fold.UseOpNo, *ImmVal))
658 int OpNo =
MI->getOperandNo(&Old);
659 if (!
TII->isOperandLegal(*
MI, OpNo, &New))
666 if ((Fold.isImm() || Fold.isFI() || Fold.isGlobal()) && Fold.needsShrink()) {
674 int Op32 = Fold.ShrinkOpcode;
675 MachineOperand &Dst0 =
MI->getOperand(0);
676 MachineOperand &Dst1 =
MI->getOperand(1);
684 MachineInstr *Inst32 =
TII->buildShrunkInst(*
MI, Op32);
686 if (HaveNonDbgCarryUse) {
689 .
addReg(AMDGPU::VCC, RegState::Kill);
703 for (
unsigned I =
MI->getNumOperands() - 1;
I > 0; --
I)
704 MI->removeOperand(
I);
705 MI->setDesc(
TII->get(AMDGPU::IMPLICIT_DEF));
708 TII->commuteInstruction(*Inst32,
false);
712 assert(!Fold.needsShrink() &&
"not handled");
717 if (NewMFMAOpc == -1)
719 MI->setDesc(
TII->get(NewMFMAOpc));
720 MI->untieRegOperand(0);
721 const MCInstrDesc &MCID =
MI->getDesc();
722 for (
unsigned I = 0;
I <
MI->getNumDefs(); ++
I)
724 MI->getOperand(
I).setIsEarlyClobber(
true);
729 int OpNo =
MI->getOperandNo(&Old);
730 if (!
TII->isOperandLegal(*
MI, OpNo, &New))
733 if (ST->hasBF16InlineConstFromUpperFP32() &&
735 AMDGPU::getNamedOperandIdx(
MI->getOpcode(), AMDGPU::OpName::src0)) {
736 unsigned Opcode =
MI->getOpcode();
737 uint8_t OpType =
TII->get(Opcode).operands()[OpNo].OperandType;
740 TII->isInlineConstant(*ImmVal, OpType)) {
743 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0_modifiers);
746 MachineOperand &ModOp =
MI->getOperand(Mod0);
757 if (Fold.isGlobal()) {
758 Old.
ChangeToGA(Fold.Def.OpToFold->getGlobal(),
759 Fold.Def.OpToFold->getOffset(),
760 Fold.Def.OpToFold->getTargetFlags());
769 MachineOperand *
New = Fold.Def.OpToFold;
773 TII->getRegClass(
MI->getDesc(), Fold.UseOpNo)) {
775 TRI->getRegClassForReg(*MRI,
New->getReg());
778 if (
New->getSubReg()) {
780 TRI->getMatchingSuperRegClass(NewRC, OpRC,
New->getSubReg());
786 if (
New->getReg().isVirtual() &&
789 <<
TRI->getRegClassName(ConstrainRC) <<
'\n');
796 if (Old.
getSubReg() == AMDGPU::lo16 &&
TRI->isSGPRReg(*MRI,
New->getReg()))
798 if (
New->getReg().isPhysical()) {
806 if (
MI->isBundledWithPred()) {
808 for (MachineOperand &MO : Header.operands()) {
809 if (MO.getReg() == OldReg) {
810 MO.setReg(
New->getReg());
811 MO.setSubReg(
New->getSubReg());
820 FoldCandidate &&Entry) {
822 for (FoldCandidate &Fold : FoldList)
823 if (Fold.UseMI == Entry.UseMI && Fold.UseOpNo == Entry.UseOpNo)
825 LLVM_DEBUG(
dbgs() <<
"Append " << (Entry.Commuted ?
"commuted" :
"normal")
826 <<
" operand " << Entry.UseOpNo <<
"\n " << *Entry.UseMI);
832 const FoldableDef &FoldOp,
833 bool Commuted =
false,
int ShrinkOp = -1) {
835 FoldCandidate(
MI, OpNo, FoldOp, Commuted, ShrinkOp));
843 if (!ST->hasPKF32InstsReplicatingLower32BitsOfScalarInput())
853 const FoldableDef &OpToFold) {
854 assert(OpToFold.isImm() &&
"Expected immediate operand");
855 uint64_t ImmVal = OpToFold.getEffectiveImmVal().value();
861bool SIFoldOperandsImpl::tryAddToFoldList(
862 SmallVectorImpl<FoldCandidate> &FoldList, MachineInstr *
MI,
unsigned OpNo,
863 const FoldableDef &OpToFold)
const {
864 const unsigned Opc =
MI->getOpcode();
866 auto tryToFoldAsFMAAKorMK = [&]() {
867 if (!OpToFold.isImm())
870 const bool TryAK = OpNo == 3;
871 const unsigned NewOpc = TryAK ? AMDGPU::S_FMAAK_F32 : AMDGPU::S_FMAMK_F32;
872 MI->setDesc(
TII->get(NewOpc));
875 bool FoldAsFMAAKorMK =
876 tryAddToFoldList(FoldList,
MI, TryAK ? 3 : 2, OpToFold);
877 if (FoldAsFMAAKorMK) {
879 MI->untieRegOperand(3);
882 MachineOperand &Op1 =
MI->getOperand(1);
883 MachineOperand &Op2 =
MI->getOperand(2);
900 bool IsLegal = OpToFold.isOperandLegal(*
TII, *
MI, OpNo);
901 if (!IsLegal && OpToFold.isImm()) {
902 if (std::optional<int64_t> ImmVal = OpToFold.getEffectiveImmVal())
903 IsLegal = canUseImmWithOpSel(
MI, OpNo, *ImmVal);
909 if (NewOpc != AMDGPU::INSTRUCTION_LIST_END) {
912 MI->setDesc(
TII->get(NewOpc));
917 bool FoldAsMAD = tryAddToFoldList(FoldList,
MI, OpNo, OpToFold);
919 MI->untieRegOperand(OpNo);
923 MI->removeOperand(
MI->getNumExplicitOperands() - 1);
929 if (
Opc == AMDGPU::S_FMAC_F32 && OpNo == 3) {
930 if (tryToFoldAsFMAAKorMK())
936 if ((
Opc == AMDGPU::S_FMAAK_F32 ||
Opc == AMDGPU::S_FMAMK_F32) &&
938 std::optional<int64_t> ImmVal = OpToFold.getEffectiveImmVal();
939 if (ImmVal && !
TII->isInlineConstant(*
MI, OpNo, *ImmVal)) {
940 unsigned ImmIdx =
Opc == AMDGPU::S_FMAAK_F32 ? 3 : 2;
941 MachineOperand &OpImm =
MI->getOperand(ImmIdx);
942 if (!OpImm.
isReg() &&
943 TII->isInlineConstant(*
MI,
MI->getOperand(OpNo), OpImm))
944 return tryToFoldAsFMAAKorMK();
949 if (OpToFold.isImm()) {
951 if (
Opc == AMDGPU::S_SETREG_B32)
952 ImmOpc = AMDGPU::S_SETREG_IMM32_B32;
953 else if (
Opc == AMDGPU::S_SETREG_B32_mode)
954 ImmOpc = AMDGPU::S_SETREG_IMM32_B32_mode;
956 MI->setDesc(
TII->get(ImmOpc));
965 bool CanCommute =
TII->findCommutedOpIndices(*
MI, OpNo, CommuteOpNo);
969 MachineOperand &
Op =
MI->getOperand(OpNo);
970 MachineOperand &CommutedOp =
MI->getOperand(CommuteOpNo);
976 if (!
Op.isReg() || !CommutedOp.
isReg())
981 if (
Op.isReg() && CommutedOp.
isReg() &&
982 (
Op.getReg() == CommutedOp.
getReg() &&
986 if (!
TII->commuteInstruction(*
MI,
false, OpNo, CommuteOpNo))
990 if (!OpToFold.isOperandLegal(*
TII, *
MI, CommuteOpNo)) {
991 if ((
Opc != AMDGPU::V_ADD_CO_U32_e64 &&
Opc != AMDGPU::V_SUB_CO_U32_e64 &&
992 Opc != AMDGPU::V_SUBREV_CO_U32_e64) ||
993 (!OpToFold.isImm() && !OpToFold.isFI() && !OpToFold.isGlobal())) {
994 TII->commuteInstruction(*
MI,
false, OpNo, CommuteOpNo);
1000 MachineOperand &OtherOp =
MI->getOperand(OpNo);
1001 if (!OtherOp.
isReg() ||
1008 unsigned MaybeCommutedOpc =
MI->getOpcode();
1022 if (
Opc == AMDGPU::S_FMAC_F32 &&
1023 (OpNo != 1 || !
MI->getOperand(1).isIdenticalTo(
MI->getOperand(2)))) {
1024 if (tryToFoldAsFMAAKorMK())
1030 if (OpToFold.isImm() &&
1039bool SIFoldOperandsImpl::isUseSafeToFold(
const MachineInstr &
MI,
1040 const MachineOperand &UseMO)
const {
1042 return !
TII->isSDWA(
MI);
1049 if (
MI.modifiesRegister(
TRI.getExec(), &
TRI))
1057bool SIFoldOperandsImpl::isTemporallyDivergentUse(
1058 const FoldableDef &OpToFold,
const MachineInstr &
UseMI)
const {
1059 if (!OpToFold.isReg())
1061 const MachineInstr *
DefMI = OpToFold.DefMI;
1064 !
TRI->isSGPRReg(*MRI, OpToFold.getReg()))
1076 SubDef &&
TII.isFoldableCopy(*SubDef);
1078 unsigned SrcIdx =
TII.getFoldableCopySrcIdx(*SubDef);
1087 if (
SrcOp.getSubReg())
1095 MachineInstr &RegSeq,
1096 SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs)
const {
1112 else if (!
TRI->getCommonSubClass(RC, OpRC))
1117 Defs.emplace_back(&SrcOp, SubRegIdx);
1122 if (DefSrc && (DefSrc->
isReg() || DefSrc->
isImm())) {
1123 Defs.emplace_back(DefSrc, SubRegIdx);
1127 Defs.emplace_back(&SrcOp, SubRegIdx);
1137 SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs,
1140 if (!Def || !
Def->isRegSequence())
1143 return getRegSeqInit(*Def, Defs);
1146std::pair<int64_t, const TargetRegisterClass *>
1147SIFoldOperandsImpl::isRegSeqSplat(MachineInstr &RegSeq)
const {
1153 bool TryToMatchSplat64 =
false;
1155 std::optional<int64_t>
Imm;
1156 for (
unsigned I = 0,
E = Defs.
size();
I !=
E; ++
I) {
1157 const MachineOperand *
Op = Defs[
I].first;
1161 if (!Def ||
Def->isImplicitDef())
1167 int64_t SubImm =
Op->getImm();
1173 if (
Imm != SubImm) {
1174 if (
I == 1 && (
E & 1) == 0) {
1177 TryToMatchSplat64 =
true;
1185 if (!TryToMatchSplat64) {
1187 return {*
Imm, SrcRC};
1194 for (
unsigned I = 0,
E = Defs.
size();
I !=
E;
I += 2) {
1195 const MachineOperand *Op0 = Defs[
I].first;
1196 const MachineOperand *Op1 = Defs[
I + 1].first;
1201 unsigned SubReg0 = Defs[
I].second;
1202 unsigned SubReg1 = Defs[
I + 1].second;
1206 if (
TRI->getChannelFromSubReg(SubReg0) + 1 !=
1207 TRI->getChannelFromSubReg(SubReg1))
1210 if (
TRI->getSubRegIdxSize(SubReg0) != 32)
1215 SplatVal64 = MergedVal;
1216 else if (SplatVal64 != MergedVal)
1223 return {SplatVal64, RC64};
1226bool SIFoldOperandsImpl::tryFoldRegSeqSplat(
1227 MachineInstr *
UseMI,
unsigned UseOpIdx, int64_t SplatVal,
1230 if (UseOpIdx >=
Desc.getNumOperands())
1237 int16_t RCID =
TII->getOpRegClassID(
Desc.operands()[UseOpIdx]);
1246 if (SplatVal != 0 && SplatVal != -1) {
1250 uint8_t OpTy =
Desc.operands()[UseOpIdx].OperandType;
1257 OpRC =
TRI->getSubRegisterClass(OpRC, AMDGPU::sub0);
1264 OpRC =
TRI->getSubRegisterClass(OpRC, AMDGPU::sub0_sub1);
1270 if (!
TRI->getCommonSubClass(OpRC, SplatRC))
1275 if (!
TII->isOperandLegal(*
UseMI, UseOpIdx, &TmpOp))
1281bool SIFoldOperandsImpl::tryToFoldACImm(
1282 const FoldableDef &OpToFold, MachineInstr *
UseMI,
unsigned UseOpIdx,
1283 SmallVectorImpl<FoldCandidate> &FoldList)
const {
1285 if (UseOpIdx >=
Desc.getNumOperands())
1292 if (OpToFold.isImm() && OpToFold.isOperandLegal(*
TII, *
UseMI, UseOpIdx)) {
1303bool SIFoldOperandsImpl::foldOperand(
1304 FoldableDef OpToFold, MachineInstr *
UseMI,
int UseOpIdx,
1305 SmallVectorImpl<FoldCandidate> &FoldList,
1306 SmallVectorImpl<MachineInstr *> &CopiesToReplace)
const {
1310 if (!isUseSafeToFold(*
UseMI, *UseOp))
1313 if (isTemporallyDivergentUse(OpToFold, *
UseMI))
1317 if (UseOp->
isReg() && OpToFold.isReg()) {
1321 if (UseOp->
getSubReg() != AMDGPU::NoSubRegister &&
1323 !
TRI->isSGPRReg(*MRI, OpToFold.getReg())))
1336 std::tie(SplatVal, SplatRC) = isRegSeqSplat(*
UseMI);
1341 for (
unsigned I = 0;
I != UsesToProcess.size(); ++
I) {
1342 MachineOperand *RSUse = UsesToProcess[
I];
1343 MachineInstr *RSUseMI = RSUse->
getParent();
1353 if (tryFoldRegSeqSplat(RSUseMI, OpNo, SplatVal, SplatRC)) {
1354 FoldableDef SplatDef(SplatVal, SplatRC);
1362 if (RSUse->
getSubReg() != RegSeqDstSubReg)
1368 FoldList, CopiesToReplace);
1374 if (tryToFoldACImm(OpToFold,
UseMI, UseOpIdx, FoldList))
1377 if (frameIndexMayFold(*
UseMI, UseOpIdx, OpToFold)) {
1382 if (
TII->getNamedOperand(*
UseMI, AMDGPU::OpName::srsrc)->getReg() !=
1388 MachineOperand &SOff =
1389 *
TII->getNamedOperand(*
UseMI, AMDGPU::OpName::soffset);
1400 TII->getNamedOperand(*
UseMI, AMDGPU::OpName::cpol)->getImm();
1415 bool FoldingImmLike =
1416 OpToFold.isImm() || OpToFold.isFI() || OpToFold.isGlobal();
1435 for (
unsigned MovOp :
1436 {AMDGPU::S_MOV_B32, AMDGPU::V_MOV_B32_e32, AMDGPU::S_MOV_B64,
1437 AMDGPU::V_MOV_B64_PSEUDO, AMDGPU::V_MOV_B16_t16_e64,
1438 AMDGPU::V_ACCVGPR_WRITE_B32_e64, AMDGPU::AV_MOV_B32_IMM_PSEUDO,
1439 AMDGPU::AV_MOV_B64_IMM_PSEUDO}) {
1440 const MCInstrDesc &MovDesc =
TII->get(MovOp);
1450 const int SrcIdx = MovOp == AMDGPU::V_MOV_B16_t16_e64 ? 2 : 1;
1452 int16_t RegClassID =
TII->getOpRegClassID(MovDesc.
operands()[SrcIdx]);
1453 if (RegClassID != -1) {
1457 MovSrcRC =
TRI->getMatchingSuperRegClass(SrcRC, MovSrcRC, UseSubReg);
1461 if (MovOp == AMDGPU::AV_MOV_B32_IMM_PSEUDO &&
1462 (!OpToFold.isImm() ||
1463 !
TII->isImmOperandLegal(MovDesc, SrcIdx,
1464 *OpToFold.getEffectiveImmVal())))
1477 if (!OpToFold.isImm() ||
1478 !
TII->isImmOperandLegal(MovDesc, 1, *OpToFold.getEffectiveImmVal()))
1484 while (ImpOpI != ImpOpE) {
1491 if (MovOp == AMDGPU::V_MOV_B16_t16_e64) {
1493 MachineOperand NewSrcOp(SrcOp);
1515 LLVM_DEBUG(
dbgs() <<
"Folding " << *OpToFold.OpToFold <<
"\n into "
1520 unsigned SubRegIdx = OpToFold.getSubReg();
1525 TRI->isSGPRReg(*MRI,
UseReg) && SubRegIdx != AMDGPU::NoSubRegister) {
1528 unsigned Channel =
TRI->getChannelFromSubReg(SubRegIdx);
1530 SubRegIdx =
TRI->getRegSizeInBits(*UseRC) == 32
1531 ? AMDGPU::NoSubRegister
1537 OpToFold.OpToFold->setIsKill(
false);
1542 if (foldCopyToAGPRRegSequence(
UseMI))
1547 if (UseOpc == AMDGPU::V_READFIRSTLANE_B32 ||
1548 (UseOpc == AMDGPU::V_READLANE_B32 &&
1550 AMDGPU::getNamedOperandIdx(UseOpc, AMDGPU::OpName::src0))) {
1555 if (FoldingImmLike) {
1558 *OpToFold.DefMI, *
UseMI))
1564 if (OpToFold.isImm()) {
1566 *OpToFold.getEffectiveImmVal());
1567 }
else if (OpToFold.isFI())
1570 assert(OpToFold.isGlobal());
1572 OpToFold.OpToFold->getOffset(),
1573 OpToFold.OpToFold->getTargetFlags());
1579 if (OpToFold.isReg() &&
TRI->isSGPRReg(*MRI, OpToFold.getReg())) {
1582 *OpToFold.DefMI, *
UseMI))
1604 UseDesc.
operands()[UseOpIdx].RegClass == -1)
1612 Changed |= tryAddToFoldList(FoldList,
UseMI, UseOpIdx, OpToFold);
1619 case AMDGPU::S_ADD_I32:
1620 case AMDGPU::S_ADD_U32:
1623 case AMDGPU::S_SUB_I32:
1624 case AMDGPU::S_SUB_U32:
1627 case AMDGPU::V_AND_B32_e64:
1628 case AMDGPU::V_AND_B32_e32:
1629 case AMDGPU::S_AND_B32:
1632 case AMDGPU::V_OR_B32_e64:
1633 case AMDGPU::V_OR_B32_e32:
1634 case AMDGPU::S_OR_B32:
1637 case AMDGPU::V_XOR_B32_e64:
1638 case AMDGPU::V_XOR_B32_e32:
1639 case AMDGPU::S_XOR_B32:
1642 case AMDGPU::S_XNOR_B32:
1645 case AMDGPU::S_NAND_B32:
1648 case AMDGPU::S_NOR_B32:
1651 case AMDGPU::S_ANDN2_B32:
1654 case AMDGPU::S_ORN2_B32:
1657 case AMDGPU::V_LSHL_B32_e64:
1658 case AMDGPU::V_LSHL_B32_e32:
1659 case AMDGPU::S_LSHL_B32:
1661 Result =
LHS << (
RHS & 31);
1663 case AMDGPU::V_LSHLREV_B32_e64:
1664 case AMDGPU::V_LSHLREV_B32_e32:
1665 Result =
RHS << (
LHS & 31);
1667 case AMDGPU::V_LSHR_B32_e64:
1668 case AMDGPU::V_LSHR_B32_e32:
1669 case AMDGPU::S_LSHR_B32:
1670 Result =
LHS >> (
RHS & 31);
1672 case AMDGPU::V_LSHRREV_B32_e64:
1673 case AMDGPU::V_LSHRREV_B32_e32:
1674 Result =
RHS >> (
LHS & 31);
1676 case AMDGPU::V_ASHR_I32_e64:
1677 case AMDGPU::V_ASHR_I32_e32:
1678 case AMDGPU::S_ASHR_I32:
1679 Result =
static_cast<int32_t
>(
LHS) >> (
RHS & 31);
1681 case AMDGPU::V_ASHRREV_I32_e64:
1682 case AMDGPU::V_ASHRREV_I32_e32:
1683 Result =
static_cast<int32_t
>(
RHS) >> (
LHS & 31);
1691 return IsScalar ? AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32;
1697bool SIFoldOperandsImpl::tryConstantFoldOp(MachineInstr *
MI)
const {
1698 if (!
MI->allImplicitDefsAreDead())
1701 unsigned Opc =
MI->getOpcode();
1703 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
1707 MachineOperand *Src0 = &
MI->getOperand(Src0Idx);
1708 std::optional<int64_t> Src0Imm =
TII->getImmOrMaterializedImm(*MRI, *Src0);
1710 if ((
Opc == AMDGPU::V_NOT_B32_e64 ||
Opc == AMDGPU::V_NOT_B32_e32 ||
1711 Opc == AMDGPU::S_NOT_B32) &&
1713 MI->getOperand(1).ChangeToImmediate(~*Src0Imm);
1714 TII->mutateAndCleanupImplicit(
1719 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
1723 MachineOperand *Src1 = &
MI->getOperand(Src1Idx);
1724 std::optional<int64_t> Src1Imm =
TII->getImmOrMaterializedImm(*MRI, *Src1);
1726 if (!Src0Imm && !Src1Imm)
1732 if (Src0Imm && Src1Imm) {
1737 bool IsSGPR =
TRI->isSGPRReg(*MRI,
MI->getOperand(0).getReg());
1741 MI->getOperand(Src0Idx).ChangeToImmediate(NewImm);
1742 MI->removeOperand(Src1Idx);
1749 if (
Opc == AMDGPU::S_SUB_I32 ||
Opc == AMDGPU::S_SUB_U32) {
1750 if (Src1Imm &&
static_cast<int32_t
>(*Src1Imm) == 0) {
1752 MI->removeOperand(Src1Idx);
1753 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1759 if (!
MI->isCommutable())
1762 if (Src0Imm && !Src1Imm) {
1768 int32_t Src1Val =
static_cast<int32_t
>(*Src1Imm);
1769 if (
Opc == AMDGPU::S_ADD_I32 ||
Opc == AMDGPU::S_ADD_U32) {
1772 MI->removeOperand(Src1Idx);
1773 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1779 if (
Opc == AMDGPU::V_OR_B32_e64 ||
1780 Opc == AMDGPU::V_OR_B32_e32 ||
1781 Opc == AMDGPU::S_OR_B32) {
1784 MI->removeOperand(Src1Idx);
1785 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1786 }
else if (Src1Val == -1) {
1788 MI->removeOperand(Src0Idx);
1789 TII->mutateAndCleanupImplicit(
1797 if (
Opc == AMDGPU::V_AND_B32_e64 ||
Opc == AMDGPU::V_AND_B32_e32 ||
1798 Opc == AMDGPU::S_AND_B32) {
1801 MI->removeOperand(Src0Idx);
1802 TII->mutateAndCleanupImplicit(
1804 }
else if (Src1Val == -1) {
1806 MI->removeOperand(Src1Idx);
1807 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1814 if (
Opc == AMDGPU::V_XOR_B32_e64 ||
Opc == AMDGPU::V_XOR_B32_e32 ||
1815 Opc == AMDGPU::S_XOR_B32) {
1818 MI->removeOperand(Src1Idx);
1819 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1828bool SIFoldOperandsImpl::tryFoldCndMask(MachineInstr &
MI)
const {
1829 unsigned Opc =
MI.getOpcode();
1830 if (
Opc != AMDGPU::V_CNDMASK_B32_e32 &&
Opc != AMDGPU::V_CNDMASK_B32_e64 &&
1831 Opc != AMDGPU::V_CNDMASK_B64_PSEUDO)
1834 MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
1835 MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
1837 std::optional<int64_t> Src1Imm =
TII->getImmOrMaterializedImm(*MRI, *Src1);
1841 std::optional<int64_t> Src0Imm =
TII->getImmOrMaterializedImm(*MRI, *Src0);
1842 if (!Src0Imm || *Src0Imm != *Src1Imm)
1847 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1_modifiers);
1849 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0_modifiers);
1850 if ((Src1ModIdx != -1 &&
MI.getOperand(Src1ModIdx).getImm() != 0) ||
1851 (Src0ModIdx != -1 &&
MI.getOperand(Src0ModIdx).getImm() != 0))
1857 int Src2Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2);
1859 MI.removeOperand(Src2Idx);
1860 MI.removeOperand(AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1));
1861 if (Src1ModIdx != -1)
1862 MI.removeOperand(Src1ModIdx);
1863 if (Src0ModIdx != -1)
1864 MI.removeOperand(Src0ModIdx);
1865 TII->mutateAndCleanupImplicit(
MI, NewDesc);
1872std::optional<SIFoldOperandsImpl::ANDMaskResult>
1873SIFoldOperandsImpl::getANDMaskRegOperand(MachineInstr &AndMI)
const {
1875 if (
Opc != AMDGPU::V_AND_B32_e64 &&
Opc != AMDGPU::V_AND_B32_e32 &&
1876 Opc != AMDGPU::S_AND_B32)
1877 return std::nullopt;
1879 std::optional<int64_t> MaskImm =
1884 MaskImm =
TII->getImmOrMaterializedImm(*MRI, AndMI.
getOperand(2));
1888 return std::nullopt;
1901bool SIFoldOperandsImpl::tryFoldRedundantAND(MachineInstr &ChildMI)
const {
1906 std::optional<ANDMaskResult> ChildResult = getANDMaskRegOperand(ChildMI);
1910 if (!ChildResult->Reg.isVirtual())
1913 MachineInstr *ParentMI = MRI->
getVRegDef(ChildResult->Reg);
1917 int64_t ParentMask = 0;
1918 std::optional<ANDMaskResult> ParentResult = getANDMaskRegOperand(*ParentMI);
1921 ParentMask = ParentResult->Mask;
1924 ParentMask = 0xffff;
1930 if ((ParentMask & ChildResult->Mask) != ParentMask)
1957bool SIFoldOperandsImpl::tryFoldAndExec(MachineInstr &
MI)
const {
1963 if (!
MI.allImplicitDefsAreDead())
1967 unsigned ExecIdx = 0;
1968 for (
unsigned I : {1u, 2u}) {
1969 const MachineOperand &MO =
MI.getOperand(
I);
1975 MachineOperand &Src =
MI.getOperand(3 - ExecIdx);
1976 if (!Src.isReg() || !Src.getReg().isVirtual() || Src.getSubReg())
1980 if (!
TII->isMaskedByExec(SrcReg,
MI, *MRI))
2000 MI.eraseFromParent();
2004bool SIFoldOperandsImpl::foldInstOperand(MachineInstr &
MI,
2005 const FoldableDef &OpToFold)
const {
2009 SmallVector<MachineInstr *, 4> CopiesToReplace;
2011 MachineOperand &Dst =
MI.getOperand(0);
2016 for (
auto *U : UsesToProcess) {
2017 MachineInstr *
UseMI =
U->getParent();
2019 FoldableDef SubOpToFold = OpToFold.getWithSubReg(*
TRI,
U->getSubReg());
2024 if (CopiesToReplace.
empty() && FoldList.
empty())
2028 for (MachineInstr *Copy : CopiesToReplace)
2029 Copy->addImplicitDefUseOperands(*MF);
2031 SetVector<MachineInstr *> ConstantFoldCandidates;
2032 for (FoldCandidate &Fold : FoldList) {
2033 assert(!Fold.isReg() || Fold.Def.OpToFold);
2034 if (Fold.isReg() && Fold.getReg().isVirtual()) {
2036 const MachineInstr *
DefMI = Fold.Def.DefMI;
2044 assert(Fold.Def.OpToFold && Fold.isReg());
2051 <<
static_cast<int>(Fold.UseOpNo) <<
" of "
2055 ConstantFoldCandidates.
insert(Fold.UseMI);
2057 }
else if (Fold.Commuted) {
2059 TII->commuteInstruction(*Fold.UseMI,
false);
2063 for (MachineInstr *
MI : ConstantFoldCandidates) {
2064 if (tryConstantFoldOp(
MI)) {
2074bool SIFoldOperandsImpl::foldCopyToAGPRRegSequence(MachineInstr *CopyMI)
const {
2081 if (!
TRI->isAGPRClass(DefRC))
2093 DenseMap<TargetInstrInfo::RegSubRegPair, Register> VGPRCopies;
2102 unsigned NumFoldable = 0;
2104 for (
unsigned I = 1;
I != NumRegSeqOperands;
I += 2) {
2121 DefRC, &AMDGPU::AGPR_32RegClass, SubRegIdx);
2141 TRI->getMatchingSuperRegClass(DefRC, InputRC, SubRegIdx);
2152 if (NumFoldable == 0)
2155 CopyMI->
setDesc(
TII->get(AMDGPU::REG_SEQUENCE));
2159 for (
auto [Def, DestSubIdx] : NewDefs) {
2160 if (!
Def->isReg()) {
2164 BuildMI(
MBB, CopyMI,
DL,
TII->get(AMDGPU::V_ACCVGPR_WRITE_B32_e64), Tmp)
2169 Def->setIsKill(
false);
2171 Register &VGPRCopy = VGPRCopies[Src];
2174 TRI->getSubRegisterClass(UseRC, DestSubIdx);
2199 B.addImm(DestSubIdx);
2206bool SIFoldOperandsImpl::tryFoldFoldableCopy(
2207 MachineInstr &
MI, MachineOperand *&CurrentKnownM0Val)
const {
2211 if (DstReg == AMDGPU::M0) {
2212 MachineOperand &NewM0Val =
MI.getOperand(1);
2213 if (CurrentKnownM0Val && CurrentKnownM0Val->
isIdenticalTo(NewM0Val)) {
2214 MI.eraseFromParent();
2225 MachineOperand *OpToFoldPtr;
2226 if (
MI.getOpcode() == AMDGPU::V_MOV_B16_t16_e64) {
2228 if (
TII->hasAnyModifiersSet(
MI))
2230 OpToFoldPtr = &
MI.getOperand(2);
2232 OpToFoldPtr = &
MI.getOperand(1);
2233 MachineOperand &OpToFold = *OpToFoldPtr;
2237 if (!FoldingImm && !OpToFold.
isReg())
2242 !
TRI->isConstantPhysReg(OpToFold.
getReg()))
2271 if (
MI.getOpcode() == AMDGPU::COPY && OpToFold.
isReg() &&
2273 if (DstRC == &AMDGPU::SReg_32RegClass &&
2275 if (!
TRI->getMatchingSuperRegClass(DstRC, &AMDGPU::SGPR_LO16RegClass,
2284 if (OpToFold.
isReg() &&
MI.isCopy() && !
MI.getOperand(1).getSubReg()) {
2285 if (foldCopyToAGPRRegSequence(&
MI))
2289 FoldableDef
Def(OpToFold, DstRC);
2290 bool Changed = foldInstOperand(
MI, Def);
2297 auto *InstToErase = &
MI;
2299 auto &SrcOp = InstToErase->getOperand(1);
2301 InstToErase->eraseFromParent();
2303 InstToErase =
nullptr;
2307 if (!InstToErase || !
TII->isFoldableCopy(*InstToErase))
2311 if (InstToErase && InstToErase->isRegSequence() &&
2313 InstToErase->eraseFromParent();
2323 return OpToFold.
isReg() &&
2324 foldCopyToVGPROfScalarAddOfFrameIndex(DstReg, OpToFold.
getReg(),
MI);
2329const MachineOperand *
2330SIFoldOperandsImpl::isClamp(
const MachineInstr &
MI)
const {
2331 unsigned Op =
MI.getOpcode();
2333 case AMDGPU::V_MAX_F32_e64:
2334 case AMDGPU::V_MAX_F16_e64:
2335 case AMDGPU::V_MAX_F16_t16_e64:
2336 case AMDGPU::V_MAX_F16_fake16_e64:
2337 case AMDGPU::V_MAX_F64_e64:
2338 case AMDGPU::V_MAX_NUM_F64_e64:
2339 case AMDGPU::V_PK_MAX_F16:
2340 case AMDGPU::V_MAX_BF16_PSEUDO_e64:
2341 case AMDGPU::V_PK_MAX_NUM_BF16: {
2342 if (
MI.mayRaiseFPException())
2345 if (!
TII->getNamedOperand(
MI, AMDGPU::OpName::clamp)->getImm())
2349 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2350 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2354 Src0->
getSubReg() != AMDGPU::NoSubRegister)
2358 if (
TII->hasModifiersSet(
MI, AMDGPU::OpName::omod))
2362 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0_modifiers)->getImm();
2364 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1_modifiers)->getImm();
2368 unsigned UnsetMods =
2369 (
Op == AMDGPU::V_PK_MAX_F16 ||
Op == AMDGPU::V_PK_MAX_NUM_BF16)
2372 if (Src0Mods != UnsetMods || Src1Mods != UnsetMods)
2382bool SIFoldOperandsImpl::tryFoldClamp(MachineInstr &
MI) {
2383 const MachineOperand *ClampSrc = isClamp(
MI);
2399 if (
Def->mayRaiseFPException())
2402 MachineOperand *DefClamp =
TII->getNamedOperand(*Def, AMDGPU::OpName::clamp);
2406 LLVM_DEBUG(
dbgs() <<
"Folding clamp " << *DefClamp <<
" into " << *Def);
2412 Register MIDstReg =
MI.getOperand(0).getReg();
2413 if (
TRI->isSGPRReg(*MRI, DefReg)) {
2422 MI.eraseFromParent();
2427 if (
TII->convertToThreeAddress(*Def,
nullptr))
2428 Def->eraseFromParent();
2435 case AMDGPU::V_MUL_F64_e64:
2436 case AMDGPU::V_MUL_F64_pseudo_e64: {
2438 case 0x3fe0000000000000:
2440 case 0x4000000000000000:
2442 case 0x4010000000000000:
2448 case AMDGPU::V_MUL_F32_e64: {
2449 switch (
static_cast<uint32_t>(Val)) {
2460 case AMDGPU::V_MUL_F16_e64:
2461 case AMDGPU::V_MUL_F16_t16_e64:
2462 case AMDGPU::V_MUL_F16_fake16_e64: {
2463 switch (
static_cast<uint16_t>(Val)) {
2474 case AMDGPU::V_PK_MUL_BF16: {
2475 switch (
static_cast<uint16_t>(Val)) {
2494std::pair<const MachineOperand *, int>
2495SIFoldOperandsImpl::isOMod(
const MachineInstr &
MI)
const {
2496 unsigned Op =
MI.getOpcode();
2498 case AMDGPU::V_MUL_F64_e64:
2499 case AMDGPU::V_MUL_F64_pseudo_e64:
2500 case AMDGPU::V_MUL_F32_e64:
2501 case AMDGPU::V_MUL_F16_t16_e64:
2502 case AMDGPU::V_MUL_F16_fake16_e64:
2503 case AMDGPU::V_MUL_F16_e64: {
2505 if ((
Op == AMDGPU::V_MUL_F32_e64 &&
2507 ((
Op == AMDGPU::V_MUL_F64_e64 ||
Op == AMDGPU::V_MUL_F64_pseudo_e64 ||
2508 Op == AMDGPU::V_MUL_F16_e64 ||
Op == AMDGPU::V_MUL_F16_t16_e64 ||
2509 Op == AMDGPU::V_MUL_F16_fake16_e64) &&
2512 MI.mayRaiseFPException())
2515 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2516 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2519 std::optional<int64_t> Src1Imm =
TII->getImmOrMaterializedImm(*MRI, *Src1);
2525 TII->hasModifiersSet(
MI, AMDGPU::OpName::src0_modifiers) ||
2526 TII->hasModifiersSet(
MI, AMDGPU::OpName::src1_modifiers) ||
2527 TII->hasModifiersSet(
MI, AMDGPU::OpName::omod) ||
2528 TII->hasModifiersSet(
MI, AMDGPU::OpName::clamp))
2531 return {Src0, OMod};
2533 case AMDGPU::V_ADD_F64_e64:
2534 case AMDGPU::V_ADD_F64_pseudo_e64:
2535 case AMDGPU::V_ADD_F32_e64:
2536 case AMDGPU::V_ADD_F16_e64:
2537 case AMDGPU::V_ADD_F16_t16_e64:
2538 case AMDGPU::V_ADD_F16_fake16_e64: {
2540 if ((
Op == AMDGPU::V_ADD_F32_e64 &&
2542 ((
Op == AMDGPU::V_ADD_F64_e64 ||
Op == AMDGPU::V_ADD_F64_pseudo_e64 ||
2543 Op == AMDGPU::V_ADD_F16_e64 ||
Op == AMDGPU::V_ADD_F16_t16_e64 ||
2544 Op == AMDGPU::V_ADD_F16_fake16_e64) &&
2549 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2550 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2554 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::src0_modifiers) &&
2555 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::src1_modifiers) &&
2556 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::clamp) &&
2557 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::omod))
2562 case AMDGPU::V_PK_MUL_BF16: {
2567 MI.mayRaiseFPException())
2570 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2571 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2574 std::optional<int64_t> Src1Imm =
TII->getImmOrMaterializedImm(*MRI, *Src1);
2578 int OMod =
getOModValue(AMDGPU::V_PK_MUL_BF16, *Src1Imm);
2585 const MachineOperand *Src0Mods =
2586 TII->getNamedOperand(
MI, AMDGPU::OpName::src0_modifiers);
2587 const MachineOperand *Src1Mods =
2588 TII->getNamedOperand(
MI, AMDGPU::OpName::src1_modifiers);
2591 TII->hasModifiersSet(
MI, AMDGPU::OpName::omod) ||
2592 TII->hasModifiersSet(
MI, AMDGPU::OpName::clamp))
2595 return {Src0, OMod};
2597 case AMDGPU::V_PK_ADD_BF16: {
2603 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2604 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2611 const MachineOperand *Src0Mods =
2612 TII->getNamedOperand(
MI, AMDGPU::OpName::src0_modifiers);
2613 const MachineOperand *Src1Mods =
2614 TII->getNamedOperand(
MI, AMDGPU::OpName::src1_modifiers);
2617 TII->hasModifiersSet(
MI, AMDGPU::OpName::omod) ||
2618 TII->hasModifiersSet(
MI, AMDGPU::OpName::clamp))
2629bool SIFoldOperandsImpl::tryFoldOMod(MachineInstr &
MI) {
2630 const MachineOperand *RegOp;
2632 std::tie(RegOp, OMod) = isOMod(
MI);
2634 RegOp->
getSubReg() != AMDGPU::NoSubRegister ||
2639 Register OModSrcReg =
Def->getOperand(0).getReg();
2643 if (
Def->isRegSequence() &&
Def->getNumOperands() == 5 &&
2644 Def->getOperand(2).getImm() == AMDGPU::lo16) {
2646 bool CanLookThrough =
true;
2647 MachineInstr *Hi16Def = MRI->
getVRegDef(
Def->getOperand(3).getReg());
2649 CanLookThrough =
false;
2651 if (CanLookThrough) {
2662 MachineOperand *DefOMod =
TII->getNamedOperand(*Def, AMDGPU::OpName::omod);
2666 if (
Def->mayRaiseFPException())
2671 if (
TII->hasModifiersSet(*Def, AMDGPU::OpName::clamp))
2681 MI.eraseFromParent();
2686 if (
TII->convertToThreeAddress(*Def,
nullptr))
2687 Def->eraseFromParent();
2694bool SIFoldOperandsImpl::tryFoldSGPRSplatRegSequence(MachineInstr &
MI) {
2697 if (!ST->hasPackedFP64SingleSGPROps() && !ST->hasPackedU64SingleSGPROps())
2704 if (!
TRI->isSGPRClass(RegClass) ||
TRI->getRegSizeInBits(*RegClass) != 128)
2708 if (!getRegSeqInit(Defs,
Reg))
2712 if (Defs.
size() <= 1)
2715 const auto &[FirstOp,
_] = Defs.
front();
2716 if (!FirstOp->isReg())
2719 Register FirstReg = FirstOp->getReg();
2720 unsigned FirstSubReg = FirstOp->getSubReg();
2723 if (!
TRI->isSGPRClass(FirstRegClass))
2728 const auto &[
Op,
_] =
Def;
2729 return Op->isReg() &&
Op->getReg() == FirstReg &&
2730 Op->getSubReg() == FirstSubReg;
2742 MachineInstrBuilder
RS =
BuildMI(*
MI.getParent(),
MI,
MI.getDebugLoc(),
2743 TII->get(AMDGPU::REG_SEQUENCE), NewDst);
2746 FirstOp->setIsKill(
false);
2748 RS.addImm(Defs[0].second);
2753 for (
unsigned i = 1; i < Defs.
size(); ++i) {
2754 RS.addReg(UndefReg, RegState::Undef);
2755 RS.addImm(Defs[i].second);
2764 MI.eraseFromParent();
2770bool SIFoldOperandsImpl::tryFoldRegSequence(MachineInstr &
MI) {
2774 if (tryFoldSGPRSplatRegSequence(
MI))
2777 auto Reg =
MI.getOperand(0).getReg();
2779 if (!ST->hasGFX90AInsts() || !
TRI->isVGPR(*MRI,
Reg) ||
2784 if (!getRegSeqInit(Defs,
Reg))
2787 for (
auto &[
Op, SubIdx] : Defs) {
2790 if (
TRI->isAGPR(*MRI,
Op->getReg()))
2793 const MachineInstr *SubDef = MRI->
getVRegDef(
Op->getReg());
2801 MachineInstr *
UseMI =
Op->getParent();
2810 if (
Op->getSubReg())
2816 if (!OpRC || !
TRI->isVectorSuperClass(OpRC))
2822 TII->get(AMDGPU::REG_SEQUENCE), Dst);
2824 for (
auto &[Def, SubIdx] : Defs) {
2825 Def->setIsKill(
false);
2826 if (
TRI->isAGPR(*MRI,
Def->getReg())) {
2837 if (!
TII->isOperandLegal(*
UseMI, OpIdx,
Op)) {
2839 RS->eraseFromParent();
2848 MI.eraseFromParent();
2856 Register &OutReg,
unsigned &OutSubReg) {
2866 if (
TRI.isAGPR(MRI, CopySrcReg)) {
2867 OutReg = CopySrcReg;
2876 if (!CopySrcDef || !CopySrcDef->
isCopy())
2883 OtherCopySrc.
getSubReg() != AMDGPU::NoSubRegister ||
2884 !
TRI.isAGPR(MRI, OtherCopySrcReg))
2887 OutReg = OtherCopySrcReg;
2921bool SIFoldOperandsImpl::tryFoldPhiAGPR(MachineInstr &
PHI) {
2925 if (!
TRI->isVGPR(*MRI, PhiOut))
2931 for (
unsigned K = 1;
K <
PHI.getNumExplicitOperands();
K += 2) {
2932 MachineOperand &MO =
PHI.getOperand(K);
2934 if (!Copy || !
Copy->isCopy())
2938 unsigned AGPRRegMask = AMDGPU::NoSubRegister;
2943 if (
const auto *SubRC =
TRI->getSubRegisterClass(CopyInRC, AGPRRegMask))
2954 bool IsAGPR32 = (ARC == &AMDGPU::AGPR_32RegClass);
2958 for (
unsigned K = 1;
K <
PHI.getNumExplicitOperands();
K += 2) {
2959 MachineOperand &MO =
PHI.getOperand(K);
2963 MachineBasicBlock *InsertMBB =
nullptr;
2966 unsigned CopyOpc = AMDGPU::COPY;
2971 if (
Def->isCopy()) {
2973 unsigned AGPRSubReg = AMDGPU::NoSubRegister;
2986 MachineOperand &CopyIn =
Def->getOperand(1);
2989 CopyOpc = AMDGPU::V_ACCVGPR_WRITE_B32_e64;
2992 InsertMBB =
Def->getParent();
3000 MachineInstr *
MI =
BuildMI(*InsertMBB, InsertPt,
PHI.getDebugLoc(),
3001 TII->get(CopyOpc), NewReg)
3011 PHI.getOperand(0).setReg(NewReg);
3017 TII->get(AMDGPU::COPY), PhiOut)
3025bool SIFoldOperandsImpl::tryFoldLoad(MachineInstr &
MI) {
3027 if (!ST->hasGFX90AInsts() ||
MI.getNumExplicitDefs() != 1)
3030 MachineOperand &
Def =
MI.getOperand(0);
3047 while (!
Users.empty()) {
3048 const MachineInstr *
I =
Users.pop_back_val();
3049 if (!
I->isCopy() && !
I->isRegSequence())
3051 Register DstReg =
I->getOperand(0).getReg();
3055 if (
TRI->isAGPR(*MRI, DstReg))
3059 Users.push_back(&U);
3064 if (!
TII->isOperandLegal(
MI, 0, &Def)) {
3069 while (!MoveRegs.
empty()) {
3111bool SIFoldOperandsImpl::tryOptimizeAGPRPhis(MachineBasicBlock &
MBB) {
3114 if (ST->hasGFX90AInsts())
3118 DenseMap<std::pair<Register, unsigned>, std::vector<MachineOperand *>>
3121 for (
auto &
MI :
MBB) {
3125 if (!
TRI->isAGPR(*MRI,
MI.getOperand(0).getReg()))
3128 for (
unsigned K = 1;
K <
MI.getNumOperands();
K += 2) {
3129 MachineOperand &PhiMO =
MI.getOperand(K);
3139 for (
const auto &[Entry, MOs] : RegToMO) {
3140 if (MOs.size() == 1)
3145 MachineBasicBlock *DefMBB =
Def->getParent();
3152 MachineInstr *VGPRCopy =
3154 TII->get(AMDGPU::V_ACCVGPR_READ_B32_e64), TempVGPR)
3160 TII->get(AMDGPU::COPY), TempAGPR)
3164 for (MachineOperand *MO : MOs) {
3176bool SIFoldOperandsImpl::run(
MachineFunction &MF,
const MachineLoopInfo *MLI) {
3182 MFI = MF.
getInfo<SIMachineFunctionInfo>();
3193 MachineOperand *CurrentKnownM0Val =
nullptr;
3201 if (tryConstantFoldOp(&
MI)) {
3206 if (tryFoldRedundantAND(
MI)) {
3211 if (tryFoldAndExec(
MI)) {
3216 if (
MI.isRegSequence() && tryFoldRegSequence(
MI)) {
3221 if (
MI.isPHI() && tryFoldPhiAGPR(
MI)) {
3226 if (
MI.mayLoad() && tryFoldLoad(
MI)) {
3231 if (
TII->isFoldableCopy(
MI)) {
3232 Changed |= tryFoldFoldableCopy(
MI, CurrentKnownM0Val);
3237 if (CurrentKnownM0Val &&
MI.modifiesRegister(AMDGPU::M0,
TRI))
3238 CurrentKnownM0Val =
nullptr;
3258 bool Changed = SIFoldOperandsImpl().run(MF, MLI);
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static bool updateOperand(Instruction *Inst, unsigned Idx, Instruction *Mat)
Updates the operand at Idx in instruction Inst with the result of instruction Mat.
This file builds on the ADT/GraphTraits.h file to build generic depth first graph iterator.
AMD GCN specific subclass of TargetSubtarget.
static Register UseReg(const MachineOperand &MO)
const HexagonInstrInfo * TII
iv Induction Variable Users
Register const TargetRegisterInfo * TRI
Promote Memory to Register
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
static bool isReg(const MCInst &MI, unsigned OpNo)
if(auto Err=PB.parsePassPipeline(MPM, Passes)) return wrap(std MPM run * Mod
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
static bool loopModifiesExec(const MachineLoop &L, const SIRegisterInfo &TRI)
static unsigned macToMad(unsigned Opc)
static bool isAGPRCopy(const SIRegisterInfo &TRI, const MachineRegisterInfo &MRI, const MachineInstr &Copy, Register &OutReg, unsigned &OutSubReg)
Checks whether Copy is a AGPR -> VGPR copy.
static void appendFoldCandidate(SmallVectorImpl< FoldCandidate > &FoldList, FoldCandidate &&Entry)
static const TargetRegisterClass * getRegOpRC(const MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI, const MachineOperand &MO)
static bool evalBinaryInstruction(unsigned Opcode, int32_t &Result, uint32_t LHS, uint32_t RHS)
static int getOModValue(unsigned Opc, int64_t Val)
static unsigned getMovOpc(bool IsScalar)
static MachineOperand * lookUpCopyChain(const SIInstrInfo &TII, const MachineRegisterInfo &MRI, Register SrcReg)
static bool checkImmOpForPKF32InstrReplicatesLower32BitsOfScalarOperand(const FoldableDef &OpToFold)
static bool isPKF32InstrReplicatesLower32BitsOfScalarOperand(const GCNSubtarget *ST, MachineInstr *MI, unsigned OpNo)
Interface definition for SIInstrInfo.
Interface definition for SIRegisterInfo.
static int Lookup(ArrayRef< TableEntry > Table, unsigned Opcode)
static const LaneMaskConstants & get(const GCNSubtarget &ST)
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Represents analyses that only rely on functions' control flow.
FunctionPass class - This class is used to implement most global optimizations.
const SIInstrInfo * getInstrInfo() const override
bool hasDOTOpSelHazard() const
bool zeroesHigh16BitsOfDest(unsigned Opcode) const
Returns if the result of this instruction with a 16-bit result returned in a 32-bit register implicit...
const HexagonRegisterInfo & getRegisterInfo() const
bool contains(const LoopT *L) const
Return true if the specified loop is contained within this loop.
LoopT * getLoopFor(const BlockT *BB) const
Return the inner most loop that BB lives in.
ArrayRef< MCOperandInfo > operands() const
int getOperandConstraint(unsigned OpNum, MCOI::OperandConstraint Constraint) const
Returns the value of the specified operand constraint if it is present.
bool isVariadic() const
Return true if this instruction can have a variable number of operands.
This holds information about one operand of a machine instruction, indicating the register class for ...
uint8_t OperandType
Information about the type of the operand.
bool hasSuperClassEq(const MCRegisterClass *RC) const
Returns true if RC is a super-class of or equal to this class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
bool hasSubClassEq(const MCRegisterClass *RC) const
Returns true if RC is a sub-class of or equal to this class.
An RAII based helper class to modify MachineFunctionProperties when running pass.
LLVM_ABI iterator SkipPHIsLabelsAndDebug(iterator I, Register Reg=Register(), bool SkipPseudoOp=true)
Return the first instruction in MBB after I that is not a PHI, label or debug.
LLVM_ABI LivenessQueryResult computeRegisterLiveness(const TargetRegisterInfo *TRI, MCRegister Reg, const_iterator Before, unsigned Neighborhood=10) const
Return whether (physical) register Reg has been defined and not killed as of just before Before.
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
LLVM_ABI iterator getFirstNonPHI()
Returns a pointer to the first instruction in this block that is not a PHINode instruction.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
MachineInstrBundleIterator< MachineInstr > iterator
LivenessQueryResult
Possible outcome of a register liveness query to computeRegisterLiveness()
@ LQR_Dead
Register is known to be fully dead.
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
Properties which a MachineFunction may have at a given point in time.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Register getReg(unsigned Idx) const
Get the register for the operand index.
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
bool isImplicitDef() const
const MachineBasicBlock * getParent() const
bool readsRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr reads the specified register.
LLVM_ABI bool allImplicitDefsAreDead() const
Return true if all the implicit defs of this instruction are dead.
unsigned getNumOperands() const
Retuns the total number of operands.
LLVM_ABI void addOperand(MachineFunction &MF, const MachineOperand &Op)
Add the specified operand to the instruction.
unsigned getOperandNo(const_mop_iterator I) const
Returns the number of the operand iterator I points to.
LLVM_ABI unsigned getNumExplicitOperands() const
Returns the number of non-implicit operands.
mop_range implicit_operands()
const MCInstrDesc & getDesc() const
Returns the target instruction descriptor of this MachineInstr.
void clearFlag(MIFlag Flag)
clearFlag - Clear a MI flag.
bool isRegSequence() const
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
MachineOperand * mop_iterator
iterator/begin/end - Iterate over all operands of a machine instruction.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
LLVM_ABI void removeOperand(unsigned OpNo)
Erase an operand from an instruction, leaving it with one fewer operand than it started with.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
Analysis pass that exposes the MachineLoopInfo for a machine function.
MachineOperand class - Representation of each machine instruction operand.
void setSubReg(unsigned subReg)
unsigned getSubReg() const
LLVM_ABI unsigned getOperandNo() const
Returns the index of this operand in the instruction that it belongs to.
LLVM_ABI void substVirtReg(Register Reg, unsigned SubIdx, const TargetRegisterInfo &)
substVirtReg - Substitute the current register with the virtual subregister Reg:SubReg.
LLVM_ABI void ChangeToFrameIndex(int Idx, unsigned TargetFlags=0)
Replace this operand with a frame index.
void setImm(int64_t immVal)
bool isReg() const
isReg - Tests if this is a MO_Register operand.
void setIsDead(bool Val=true)
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
LLVM_ABI void ChangeToImmediate(int64_t ImmVal, unsigned TargetFlags=0)
ChangeToImmediate - Replace this operand with a new immediate operand of the specified value.
LLVM_ABI void ChangeToGA(const GlobalValue *GV, int64_t Offset, unsigned TargetFlags=0)
ChangeToGA - Replace this operand with a new global address operand.
void setIsKill(bool Val=true)
LLVM_ABI void ChangeToRegister(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isDebug=false)
ChangeToRegister - Replace this operand with a new register operand of the specified value.
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
LLVM_ABI void substPhysReg(MCRegister Reg, const TargetRegisterInfo &)
substPhysReg - Substitute the current register with the physical register Reg, taking any existing Su...
static MachineOperand CreateImm(int64_t Val)
bool isGlobal() const
isGlobal - Tests if this is a MO_GlobalAddress operand.
MachineOperandType getType() const
getType - Returns the MachineOperandType for this operand.
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
bool isFI() const
isFI - Tests if this is a MO_FrameIndex operand.
LLVM_ABI bool isIdenticalTo(const MachineOperand &Other) const
Returns true if this operand is identical to the specified operand except for liveness related flags ...
@ MO_Immediate
Immediate operand.
@ MO_GlobalAddress
Address of a global value.
@ MO_FrameIndex
Abstract Stack Frame Index.
@ MO_Register
Register operand.
static MachineOperand CreateFI(int Idx)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
use_nodbg_iterator use_nodbg_begin(Register RegNo) const
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
iterator_range< use_nodbg_iterator > use_nodbg_operands(Register Reg) const
bool use_nodbg_empty(Register RegNo) const
use_nodbg_empty - Return true if there are no non-Debug instructions using the specified register.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLVM_ABI bool hasOneNonDBGUser(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug instruction using the specified regis...
iterator_range< use_instr_nodbg_iterator > use_nodbg_instructions(Register Reg) const
void setRegAllocationHint(Register VReg, unsigned Type, Register PrefReg)
setRegAllocationHint - Specify a register allocation hint for the specified virtual register.
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
LLVM_ABI const TargetRegisterClass * constrainRegClass(Register Reg, const TargetRegisterClass *RC, unsigned MinNumRegs=0)
constrainRegClass - Constrain the register class of the specified virtual register to be a common sub...
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Wrapper class representing virtual and physical registers.
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
static bool hasSameClamp(const MachineInstr &A, const MachineInstr &B)
static std::optional< int64_t > extractSubregFromImm(int64_t ImmVal, unsigned SubRegIndex)
Return the extracted immediate value in a subregister use from a constant materialized in a super reg...
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
Register getScratchRSrcReg() const
Returns the physical register reserved for use as the resource descriptor for scratch accesses.
SIModeRegisterDefaults getMode() const
static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs=1)
bool insert(const value_type &X)
Insert a new element into the SetVector.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
Represent a constant reference to a string, i.e.
static const unsigned CommuteAnyOperandIndex
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
bool isInlinableLiteralV216(uint32_t Literal, uint8_t OpType)
LLVM_READONLY int32_t getMFMAEarlyClobberOp(uint32_t Opcode)
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
bool isPackedSingleSGPR64BitInst(unsigned Opc)
The opcode is a packed 64-bit instruction which only reads low 64 bits of a scalar operand and propag...
LLVM_READONLY int32_t getVOPe32(uint32_t Opcode)
constexpr bool isSISrcOperand(const MCOperandInfo &OpInfo)
Is this an AMDGPU specific source operand?
@ OPERAND_REG_INLINE_C_FP64
@ OPERAND_REG_INLINE_C_BF16
@ OPERAND_REG_INLINE_C_V2BF16
@ OPERAND_REG_IMM_V2INT64
@ OPERAND_REG_IMM_V2INT16
@ OPERAND_REG_INLINE_C_INT64
@ OPERAND_REG_IMM_NOINLINE_V2FP16
@ OPERAND_REG_INLINE_C_V2FP16
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
@ OPERAND_REG_INLINE_AC_FP32
@ OPERAND_REG_INLINE_C_FP32
@ OPERAND_REG_INLINE_C_INT32
@ OPERAND_REG_INLINE_C_V2INT16
@ OPERAND_REG_INLINE_AC_FP64
LLVM_READONLY int32_t getFlatScratchInstSSfromSV(uint32_t Opcode)
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
constexpr bool isVOP3(const T &...O)
constexpr bool isMAI(const T &...O)
constexpr bool isSWMMAC(const T &...O)
constexpr bool isVOP3P(const T &...O)
constexpr bool isWMMA(const T &...O)
constexpr bool isDOT(const T &...O)
constexpr bool isPacked(const T &...O)
NodeAddr< DefNode * > Def
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
TargetInstrInfo::RegSubRegPair getRegSubRegPair(const MachineOperand &O)
Create RegSubRegPair from a register MachineOperand.
MachineBasicBlock::instr_iterator getBundleStart(MachineBasicBlock::instr_iterator I)
Returns an iterator to the first instruction in the bundle containing I.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
bool execMayBeModifiedBeforeUse(const MachineRegisterInfo &MRI, Register VReg, const MachineInstr &DefMI, const MachineInstr &UseMI)
Return false if EXEC is not changed between the def of VReg at DefMI and the use at UseMI.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
FunctionPass * createSIFoldOperandsLegacyPass()
char & SIFoldOperandsLegacyID
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
@ Sub
Subtraction of integers.
DWARFExpression::Operation Op
iterator_range< pointer_iterator< WrappedIteratorT > > make_pointer_range(RangeT &&Range)
iterator_range< df_iterator< T > > depth_first(const T &G)
LLVM_ABI Printable printReg(Register Reg, const TargetRegisterInfo *TRI=nullptr, unsigned SubIdx=0, const MachineRegisterInfo *MRI=nullptr)
Prints virtual and physical registers with or without a TRI instance.
constexpr uint64_t Make_64(uint32_t High, uint32_t Low)
Make a 64-bit integer from a high / low pair of 32-bit integers.
MCRegisterClass TargetRegisterClass
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
@ PreserveSign
The sign of a flushed-to-zero number is preserved in the sign of 0.
DenormalModeKind Output
Denormal flushing mode for floating point instruction results in the default floating point environme...
DenormalMode FP64FP16Denormals
If this is set, neither input or output denormals are flushed for both f64 and f16/v2f16 instructions...
bool IEEE
Floating point opcodes that support exception flag gathering quiet and propagate signaling NaN inputs...
DenormalMode FP32Denormals
If this is set, neither input or output denormals are flushed for most f32 instructions.