77#define DEBUG_TYPE "si-fix-sgpr-copies"
80 "amdgpu-enable-merge-m0",
81 cl::desc(
"Merge and hoist M0 initializations"),
94 unsigned NumSVCopies = 0;
99 unsigned NumReadfirstlanes = 0;
101 bool NeedToBeConvertedToVALU =
false;
109 unsigned SiblingPenalty = 0;
111 V2SCopyInfo() : Copy(nullptr), ID(0){};
112 V2SCopyInfo(
unsigned Id, MachineInstr *
C,
unsigned Width)
113 : Copy(
C), NumReadfirstlanes(Width / 32), ID(
Id){};
114#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
116 dbgs() << ID <<
" : " << *Copy <<
"\n\tS:" << SChain.size()
117 <<
"\n\tSV:" << NumSVCopies <<
"\n\tSP: " << SiblingPenalty
118 <<
"\nScore: " << Score <<
"\n";
123class SIFixSGPRCopies {
124 MachineDominatorTree *MDT;
125 SmallVector<MachineInstr*, 4> SCCCopies;
126 SmallVector<MachineInstr*, 4> RegSequences;
127 SmallVector<MachineInstr*, 4> PHINodes;
128 SmallVector<MachineInstr*, 4> S2VCopies;
129 unsigned NextVGPRToSGPRCopyID = 0;
130 MapVector<unsigned, V2SCopyInfo> V2SCopies;
131 DenseMap<MachineInstr *, SetVector<unsigned>> SiblingPenalty;
132 DenseSet<MachineInstr *> PHISources;
135 MachineRegisterInfo *MRI;
136 const SIRegisterInfo *TRI;
137 const SIInstrInfo *TII;
139 SIFixSGPRCopies(MachineDominatorTree *MDT) : MDT(MDT) {}
143 unsigned getNextVGPRToSGPRCopyId() {
return ++NextVGPRToSGPRCopyID; }
144 bool needToBeConvertedToVALU(V2SCopyInfo *
I);
145 void analyzeVGPRToSGPRCopy(MachineInstr *
MI);
153 void processPHINode(MachineInstr &
MI);
158 bool tryMoveVGPRConstToSGPR(MachineOperand &MO,
Register NewDst,
159 MachineBasicBlock *BlockToInsertTo,
168 SIFixSGPRCopiesLegacy() : MachineFunctionPass(ID) {}
171 MachineDominatorTree *MDT =
172 &getAnalysis<MachineDominatorTreeWrapperPass>().getDomTree();
173 SIFixSGPRCopies Impl(MDT);
177 StringRef getPassName()
const override {
return "SI Fix SGPR copies"; }
179 void getAnalysisUsage(AnalysisUsage &AU)
const override {
187 MachineFunctionProperties getClearedProperties()
const override {
188 return MachineFunctionProperties().setNoPHIs();
200char SIFixSGPRCopiesLegacy::ID = 0;
205 return new SIFixSGPRCopiesLegacy();
208static std::pair<const TargetRegisterClass *, const TargetRegisterClass *>
212 Register DstReg = Copy.getOperand(0).getReg();
213 Register SrcReg = Copy.getOperand(1).getReg();
217 :
TRI.getPhysRegBaseClass(SrcReg);
224 :
TRI.getPhysRegBaseClass(DstReg);
226 return std::pair(SrcRC, DstRC);
232 return SrcRC != &AMDGPU::VReg_1RegClass &&
TRI.isSGPRClass(DstRC) &&
233 TRI.hasVectorRegisters(SrcRC);
239 return DstRC != &AMDGPU::VReg_1RegClass &&
TRI.isSGPRClass(SrcRC) &&
240 TRI.hasVectorRegisters(DstRC);
247 auto &Src =
MI.getOperand(1);
254 const auto *
UseMI = MO.getParent();
257 if (MO.isDef() ||
UseMI->getParent() !=
MI.getParent() ||
258 UseMI->getOpcode() <= TargetOpcode::GENERIC_OP_END)
261 unsigned OpIdx = MO.getOperandNo();
262 if (OpIdx >=
UseMI->getDesc().getNumOperands() ||
263 !
TII->isOperandLegal(*
UseMI, OpIdx, &Src))
316 if (SubReg != AMDGPU::NoSubRegister)
330 bool IsAGPR =
TRI->isAGPRClass(DstRC);
332 for (
unsigned I = 1,
N =
MI.getNumOperands();
I !=
N;
I += 2) {
334 TRI->getRegClassForOperandReg(MRI,
MI.getOperand(
I));
336 "Expected SGPR REG_SEQUENCE to only have SGPR inputs");
348 unsigned Opc = NewSrcRC == &AMDGPU::AGPR_32RegClass ?
349 AMDGPU::V_ACCVGPR_WRITE_B32_e64 : AMDGPU::COPY;
356 MI.getOperand(
I).setReg(TmpReg);
368 if (Copy->getOpcode() != AMDGPU::COPY)
371 if (!MoveImm || !MoveImm->isMoveImmediate())
375 TII->getNamedOperand(*MoveImm, AMDGPU::OpName::src0);
380 if (Copy->getOperand(1).getSubReg())
383 switch (MoveImm->getOpcode()) {
386 case AMDGPU::V_MOV_B32_e32:
387 case AMDGPU::AV_MOV_B32_IMM_PSEUDO:
388 SMovOp = AMDGPU::S_MOV_B32;
390 case AMDGPU::V_MOV_B64_e32:
391 case AMDGPU::V_MOV_B64_PSEUDO:
392 SMovOp = AMDGPU::S_MOV_B64_IMM_PSEUDO;
399template <
class UnaryPredicate>
409 while (!Worklist.
empty()) {
449 while (
I !=
MBB->end() &&
TII->isBasicBlockPrologue(*
I))
466 using InitListMap = std::map<unsigned, std::list<MachineInstr *>>;
477 for (
auto &MO :
MI.operands()) {
478 if ((MO.isReg() && ((MO.isDef() && MO.getReg() !=
Reg) || !MO.isDef())) ||
479 (!MO.isImm() && !MO.isReg()) || (MO.isImm() &&
Imm)) {
487 Inits[
Imm->getImm()].push_front(&
MI);
498 for (
auto &
Init : Inits) {
499 auto &Defs =
Init.second;
501 for (
auto I1 = Defs.begin(),
E = Defs.end(); I1 !=
E; ) {
504 for (
auto I2 = std::next(I1); I2 !=
E; ) {
513 auto interferes = [&MDT, From, To](
MachineInstr* &Clobber) ->
bool {
516 bool MayClobberFrom =
isReachable(Clobber, &*From, MBBTo, MDT);
517 bool MayClobberTo =
isReachable(Clobber, &*To, MBBTo, MDT);
518 if (!MayClobberFrom && !MayClobberTo)
520 if ((MayClobberFrom && !MayClobberTo) ||
521 (!MayClobberFrom && MayClobberTo))
527 return !((MBBFrom == MBBTo &&
535 return C.first !=
Init.first &&
541 if (!interferes(MI2, MI1)) {
551 if (!interferes(MI1, MI2)) {
569 if (!interferes(MI1,
I) && !interferes(MI2,
I)) {
573 <<
"and moving from "
590 for (
auto &
Init : Inits) {
591 auto &Defs =
Init.second;
592 auto I = Defs.begin();
593 while (
I != Defs.end()) {
594 if (MergedInstrs.
count(*
I)) {
595 (*I)->eraseFromParent();
603 for (
auto &
Init : Inits) {
604 auto &Defs =
Init.second;
605 for (
auto *
MI : Defs) {
606 auto *
MBB =
MI->getParent();
611 if (!
TII->isBasicBlockPrologue(*
B))
614 auto R = std::next(
MI->getReverseIterator());
615 const unsigned Threshold = 50;
617 for (
unsigned I = 0; R !=
B &&
I < Threshold; ++R, ++
I)
618 if (R->readsRegister(
Reg,
TRI) || R->modifiesRegister(
Reg,
TRI) ||
619 TII->isSchedulingBoundary(*R,
MBB, *
MBB->getParent()))
641 TRI =
ST.getRegisterInfo();
642 TII =
ST.getInstrInfo();
645 SmallVector<MachineInstr *, 8> Relegalize;
646 SmallVector<MachineInstr *, 4> RegMaskInstrs;
648 for (MachineBasicBlock &
MBB : MF) {
651 MachineInstr &
MI = *
I;
656 [](
const MachineOperand &MO) { return MO.isRegMask(); }))
659 switch (
MI.getOpcode()) {
663 if (
TII->isWMMA(
MI) &&
684 if (lowerSpecialCase(
MI,
I))
687 analyzeVGPRToSGPRCopy(&
MI);
692 case AMDGPU::STRICT_WQM:
693 case AMDGPU::SOFT_WQM:
694 case AMDGPU::STRICT_WWM:
695 case AMDGPU::INSERT_SUBREG:
697 case AMDGPU::REG_SEQUENCE: {
698 if (
TRI->isSGPRClass(
TII->getOpRegClass(
MI, 0))) {
699 for (MachineOperand &MO :
MI.operands()) {
703 if (SrcRC == &AMDGPU::VReg_1RegClass)
706 if (
TRI->hasVectorRegisters(SrcRC)) {
708 TRI->getEquivalentSGPRClass(SrcRC);
709 Register NewDst = MRI->createVirtualRegister(DestRC);
710 MachineBasicBlock *BlockToInsertCopy =
717 if (!tryMoveVGPRConstToSGPR(MO, NewDst, BlockToInsertCopy,
718 PointToInsertCopy,
DL)) {
719 MachineInstr *NewCopy =
720 BuildMI(*BlockToInsertCopy, PointToInsertCopy,
DL,
721 TII->get(AMDGPU::COPY), NewDst)
724 analyzeVGPRToSGPRCopy(NewCopy);
725 PHISources.
insert(NewCopy);
733 else if (
MI.isRegSequence())
738 case AMDGPU::V_WRITELANE_B32: {
741 if (
ST.getConstantBusLimit(
MI.getOpcode()) != 1)
751 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::src0);
753 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::src1);
754 MachineOperand &Src0 =
MI.getOperand(Src0Idx);
755 MachineOperand &Src1 =
MI.getOperand(Src1Idx);
759 Src0.
getReg() != AMDGPU::M0) &&
761 Src1.
getReg() != AMDGPU::M0)) {
768 for (MachineOperand *MO : {&Src0, &Src1}) {
770 MachineInstr *
DefMI = MRI->getVRegDef(MO->
getReg());
777 if (Copied.
isImm() &&
778 TII->isInlineConstant(APInt(64, Copied.
getImm(),
true))) {
792 TII->get(AMDGPU::COPY), AMDGPU::M0)
803 lowerVGPR2SGPRCopies(MF);
806 for (
auto *
MI : S2VCopies) {
815 for (
auto *
MI : RegSequences) {
817 if (
MI->isRegSequence())
820 for (
auto *
MI : PHINodes) {
823 while (!Relegalize.
empty())
826 if (MF.getTarget().getOptLevel() > CodeGenOptLevel::None &&
EnableM0Merge)
829 SiblingPenalty.clear();
832 RegSequences.clear();
840void SIFixSGPRCopies::processPHINode(MachineInstr &
MI) {
841 bool AllAGPRUses =
true;
842 SetVector<const MachineInstr *> worklist;
843 SmallPtrSet<const MachineInstr *, 4> Visited;
844 SetVector<MachineInstr *> PHIOperands;
848 bool HasUses =
false;
849 while (!worklist.
empty()) {
852 for (
const auto &Use : MRI->use_operands(
Reg)) {
854 const MachineInstr *
UseMI =
Use.getParent();
857 TRI->isAGPR(*MRI,
Use.getReg());
869 if (HasUses && AllAGPRUses && !
TRI->isAGPRClass(RC0)) {
871 MRI->setRegClass(PHIRes,
TRI->getEquivalentAGPRClass(RC0));
872 for (
unsigned I = 1,
N =
MI.getNumOperands();
I !=
N;
I += 2) {
873 MachineInstr *
DefMI = MRI->getVRegDef(
MI.getOperand(
I).getReg());
879 if (
TRI->hasVectorRegisters(MRI->getRegClass(PHIRes)) ||
880 RC0 == &AMDGPU::VReg_1RegClass) {
882 TII->legalizeOperands(
MI, MDT);
886 while (!PHIOperands.
empty()) {
891bool SIFixSGPRCopies::tryMoveVGPRConstToSGPR(
892 MachineOperand &MaybeVGPRConstMO,
Register DstReg,
893 MachineBasicBlock *BlockToInsertTo,
896 MachineInstr *
DefMI = MRI->getVRegDef(MaybeVGPRConstMO.
getReg());
900 MachineOperand *SrcConst =
TII->getNamedOperand(*
DefMI, AMDGPU::OpName::src0);
901 if (SrcConst->
isReg())
905 MRI->getRegClass(MaybeVGPRConstMO.
getReg());
906 unsigned MoveSize =
TRI->getRegSizeInBits(*SrcRC);
908 MoveSize == 64 ? AMDGPU::S_MOV_B64_IMM_PSEUDO : AMDGPU::S_MOV_B32;
909 BuildMI(*BlockToInsertTo, PointToInsertTo,
DL,
TII->get(MoveOp), DstReg)
911 if (MRI->hasOneUse(MaybeVGPRConstMO.
getReg()))
913 MaybeVGPRConstMO.
setReg(DstReg);
917bool SIFixSGPRCopies::lowerSpecialCase(MachineInstr &
MI,
927 if (DstReg == AMDGPU::M0 &&
TRI->hasVectorRegisters(SrcRC)) {
929 MRI->createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
931 const MCInstrDesc &ReadFirstLaneDesc =
932 TII->get(AMDGPU::V_READFIRSTLANE_B32);
933 BuildMI(*
MI.getParent(),
MI,
MI.getDebugLoc(), ReadFirstLaneDesc, TmpReg)
934 .
add(
MI.getOperand(1));
936 unsigned SubReg =
MI.getOperand(1).getSubReg();
937 MI.getOperand(1).setReg(TmpReg);
938 MI.getOperand(1).setSubReg(AMDGPU::NoSubRegister);
942 SubReg == AMDGPU::NoSubRegister
944 :
TRI->getMatchingSuperRegClass(SrcRC, OpRC, SubReg);
946 if (!MRI->constrainRegClass(SrcReg, ConstrainRC))
951 if (tryMoveVGPRConstToSGPR(
MI.getOperand(1), DstReg,
MI.getParent(),
MI,
953 I =
MI.eraseFromParent();
961 SIInstrWorklist worklist;
963 TII->moveToVALU(worklist, MDT);
972 MI.getOperand(1).ChangeToImmediate(
Imm);
973 MI.addImplicitDefUseOperands(*
MI.getMF());
974 MI.setDesc(
TII->get(SMovOp));
980void SIFixSGPRCopies::analyzeVGPRToSGPRCopy(MachineInstr*
MI) {
986 V2SCopyInfo
Info(getNextVGPRToSGPRCopyId(),
MI,
987 TRI->getRegSizeInBits(*DstRC));
988 SmallVector<MachineInstr *, 8> AnalysisWorklist;
991 DenseSet<MachineInstr *> Visited;
993 while (!AnalysisWorklist.
empty()) {
997 if (!Visited.
insert(Inst).second)
1017 SiblingPenalty[Inst].insert(
Info.ID);
1019 SmallVector<MachineInstr *, 4>
Users;
1025 !
I->findRegisterDefOperand(AMDGPU::SCC,
nullptr)) {
1026 if (
I->readsRegister(AMDGPU::SCC,
nullptr))
1032 !
TII->isVALU(*Inst,
true)) {
1033 for (
auto &U : MRI->use_instructions(
Reg))
1034 Users.push_back(&U);
1037 for (
auto *U :
Users) {
1038 if (
TII->isSALU(*U))
1039 Info.SChain.insert(U);
1043 V2SCopies[
Info.ID] = std::move(Info);
1048bool SIFixSGPRCopies::needToBeConvertedToVALU(V2SCopyInfo *Info) {
1049 if (
Info->SChain.empty()) {
1054 Info->SChain, [&](MachineInstr *
A, MachineInstr *
B) ->
bool {
1055 return SiblingPenalty[A].size() < SiblingPenalty[B].size();
1057 Info->Siblings.remove_if([&](
unsigned ID) {
return ID ==
Info->ID; });
1063 SmallSet<std::pair<Register, unsigned>, 4> SrcRegs;
1064 for (
auto J :
Info->Siblings) {
1065 auto *InfoIt = V2SCopies.find(J);
1066 if (InfoIt != V2SCopies.end()) {
1067 MachineInstr *SiblingCopy = InfoIt->second.Copy;
1076 Info->SiblingPenalty = SrcRegs.
size();
1079 Info->NumSVCopies +
Info->SiblingPenalty +
Info->NumReadfirstlanes;
1080 unsigned Profit =
Info->SChain.size();
1081 Info->Score = Penalty > Profit ? 0 : Profit - Penalty;
1082 Info->NeedToBeConvertedToVALU =
Info->Score < 3;
1083 return Info->NeedToBeConvertedToVALU;
1088 SmallVector<unsigned, 8> LoweringWorklist;
1089 for (
auto &
C : V2SCopies) {
1090 if (needToBeConvertedToVALU(&
C.second))
1098 while (!LoweringWorklist.
empty()) {
1100 auto *CurInfoIt = V2SCopies.find(CurID);
1101 if (CurInfoIt != V2SCopies.end() && !CurInfoIt->second.Erased) {
1102 V2SCopyInfo &
C = CurInfoIt->second;
1104 for (
auto S :
C.Siblings) {
1105 auto *SibInfoIt = V2SCopies.find(S);
1106 if (SibInfoIt != V2SCopies.end() && !SibInfoIt->second.Erased) {
1107 V2SCopyInfo &
SI = SibInfoIt->second;
1109 if (!
SI.NeedToBeConvertedToVALU) {
1110 SI.SChain.set_subtract(
C.SChain);
1111 if (needToBeConvertedToVALU(&SI))
1114 SI.Siblings.remove_if([&](
unsigned ID) {
return ID ==
C.ID; });
1118 <<
" is being turned to VALU\n");
1123 V2SCopies.remove_if([](
const auto &
P) {
return P.second.Erased; });
1129 for (
auto C : V2SCopies) {
1130 MachineInstr *
MI =
C.second.Copy;
1135 <<
" is being turned to v_readfirstlane_b32"
1136 <<
" Score: " <<
C.second.Score <<
"\n");
1137 Register DstReg =
MI->getOperand(0).getReg();
1138 MRI->constrainRegClass(DstReg, &AMDGPU::SReg_32_XM0RegClass);
1140 Register SrcReg =
MI->getOperand(1).getReg();
1141 unsigned SubReg =
MI->getOperand(1).getSubReg();
1143 TRI->getRegClassForOperandReg(*MRI,
MI->getOperand(1));
1144 size_t SrcSize =
TRI->getRegSizeInBits(*SrcRC);
1145 if (SrcSize == 16) {
1147 "We do not expect to see 16-bit copies from VGPR to SGPR unless "
1148 "we have 16-bit VGPRs");
1149 assert(MRI->getRegClass(DstReg) == &AMDGPU::SReg_32RegClass ||
1150 MRI->getRegClass(DstReg) == &AMDGPU::SReg_32_XM0RegClass);
1152 MRI->setRegClass(DstReg, &AMDGPU::SReg_32_XM0RegClass);
1153 Register VReg32 = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
1155 Register Undef = MRI->createVirtualRegister(&AMDGPU::VGPR_16RegClass);
1158 .
addReg(SrcReg, {}, SubReg)
1159 .addImm(AMDGPU::lo16)
1164 }
else if (SrcSize == 32) {
1165 const MCInstrDesc &ReadFirstLaneDesc =
1166 TII->get(AMDGPU::V_READFIRSTLANE_B32);
1169 .
addReg(SrcReg, {}, SubReg);
1172 SubReg == AMDGPU::NoSubRegister
1174 :
TRI->getMatchingSuperRegClass(MRI->getRegClass(SrcReg), OpRC,
1177 if (!MRI->constrainRegClass(SrcReg, ConstrainRC))
1181 TII->get(AMDGPU::REG_SEQUENCE), DstReg);
1182 int N =
TRI->getRegSizeInBits(*SrcRC) / 32;
1183 for (
int i = 0; i <
N; i++) {
1185 Result, *MRI,
MI->getOperand(1), SrcRC,
1186 TRI->getSubRegFromChannel(i), &AMDGPU::VGPR_32RegClass);
1188 MRI->createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
1190 TII->get(AMDGPU::V_READFIRSTLANE_B32), PartialDst)
1192 Result.addReg(PartialDst).addImm(
TRI->getSubRegFromChannel(i));
1195 MI->eraseFromParent();
1204 bool HasCmp =
ST.isWave32() ||
ST.hasScalarCompareEq64();
1205 for (MachineBasicBlock &
MBB : MF) {
1208 MachineInstr &
MI = *
I;
1212 const MachineOperand &Src =
MI.getOperand(1);
1215 if (SrcReg == AMDGPU::SCC) {
1217 MRI->createVirtualRegister(
TRI->getWaveMaskRegClass());
1219 MI.getDebugLoc(),
TII->get(LMC.CSelectOpc), SCCCopy)
1222 I =
BuildMI(*
MI.getParent(), std::next(
I),
I->getDebugLoc(),
1223 TII->get(AMDGPU::COPY), DstReg)
1225 MI.eraseFromParent();
1228 if (DstReg == AMDGPU::SCC) {
1231 if (HasCmp && !Src.getSubReg() &&
1232 TII->isMaskedByExec(SrcReg,
MI, *MRI)) {
1237 TII->get(LMC.CmpLgOpc))
1241 Register Tmp = MRI->createVirtualRegister(
TRI->getBoolRC());
1243 TII->get(LMC.AndOpc))
1248 MI.eraseFromParent();
1258 SIFixSGPRCopies Impl(&MDT);
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
AMD GCN specific subclass of TargetSubtarget.
const HexagonInstrInfo * TII
iv Induction Variable Users
Register const TargetRegisterInfo * TRI
Promote Memory to Register
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
static std::pair< const TargetRegisterClass *, const TargetRegisterClass * > getCopyRegClasses(const MachineInstr &Copy, const SIRegisterInfo &TRI, const MachineRegisterInfo &MRI)
static cl::opt< bool > EnableM0Merge("amdgpu-enable-merge-m0", cl::desc("Merge and hoist M0 initializations"), cl::init(true))
static bool foldVGPRCopyIntoRegSequence(MachineInstr &MI, const SIRegisterInfo *TRI, const SIInstrInfo *TII, MachineRegisterInfo &MRI)
bool searchPredecessors(const MachineBasicBlock *MBB, const MachineBasicBlock *CutOff, UnaryPredicate Predicate)
static bool isReachable(const MachineInstr *From, const MachineInstr *To, const MachineBasicBlock *CutOff, MachineDominatorTree &MDT)
static bool isVGPRToSGPRCopy(const TargetRegisterClass *SrcRC, const TargetRegisterClass *DstRC, const SIRegisterInfo &TRI)
static bool tryChangeVGPRtoSGPRinCopy(MachineInstr &MI, const SIRegisterInfo *TRI, const SIInstrInfo *TII)
static bool isSGPRToVGPRCopy(const TargetRegisterClass *SrcRC, const TargetRegisterClass *DstRC, const SIRegisterInfo &TRI)
static bool hoistAndMergeSGPRInits(unsigned Reg, ArrayRef< MachineInstr * > RegMaskInstrs, const MachineRegisterInfo &MRI, const TargetRegisterInfo *TRI, MachineDominatorTree &MDT, const TargetInstrInfo *TII)
static bool isSafeToFoldImmIntoCopy(const MachineInstr *Copy, const MachineInstr *MoveImm, const SIInstrInfo *TII, unsigned &SMovOp, int64_t &Imm)
static MachineBasicBlock::iterator getFirstNonPrologue(MachineBasicBlock *MBB, const TargetInstrInfo *TII)
static const LaneMaskConstants & get(const GCNSubtarget &ST)
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
AnalysisUsage & addRequired()
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Implements a dense probed hash-table based set.
NodeT * findNearestCommonDominator(NodeT *A, NodeT *B) const
Find nearest common dominator basic block for basic block A and B.
bool properlyDominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
properlyDominates - Returns true iff A dominates B and A != B.
FunctionPass class - This class is used to implement most global optimizations.
MachineInstrBundleIterator< MachineInstr, true > reverse_iterator
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
LLVM_ABI instr_iterator getFirstInstrTerminator()
Same getFirstTerminator but it ignores bundles and return an instr_iterator instead.
MachineInstrBundleIterator< MachineInstr > iterator
Analysis pass which computes a MachineDominatorTree.
Analysis pass which computes a MachineDominatorTree.
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
bool dominates(const MachineInstr *A, const MachineInstr *B) const
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
const MachineFunctionProperties & getProperties() const
Get the function properties.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
Representation of each machine instruction.
bool isImplicitDef() const
const MachineBasicBlock * getParent() const
bool isCompare(QueryType Type=IgnoreBundle) const
Return true if this instruction is a comparison.
bool isRegSequence() const
LLVM_ABI unsigned getNumExplicitDefs() const
Returns the number of non-implicit definitions.
bool isMoveImmediate(QueryType Type=IgnoreBundle) const
Return true if this instruction is a move immediate (including conditional moves) instruction.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
MachineOperand class - Representation of each machine instruction operand.
unsigned getSubReg() const
LLVM_ABI unsigned getOperandNo() const
Returns the index of this operand in the instruction that it belongs to.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
LLVM_ABI void ChangeToImmediate(int64_t ImmVal, unsigned TargetFlags=0)
ChangeToImmediate - Replace this operand with a new immediate operand of the specified value.
LLVM_ABI void ChangeToRegister(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isDebug=false)
ChangeToRegister - Replace this operand with a new register operand of the specified value.
Register getReg() const
getReg - Returns the register number.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
iterator_range< def_instr_iterator > def_instructions(Register Reg) const
use_instr_iterator use_instr_begin(Register RegNo) const
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
bool hasOneUse(Register RegNo) const
hasOneUse - Return true if there is exactly one instruction using the specified register.
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
iterator_range< reg_nodbg_iterator > reg_nodbg_operands(Register Reg) const
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Wrapper class representing virtual and physical registers.
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
A vector that has set insertion semantics.
bool empty() const
Determine if the SetVector is empty or not.
bool insert(const value_type &X)
Insert a new element into the SetVector.
value_type pop_back_val()
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
TargetInstrInfo - Interface to description of machine instruction set.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
std::pair< iterator, bool > insert(const ValueT &V)
bool contains(const_arg_type_t< ValueT > V) const
Check if the set contains the given element.
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
initializer< Ty > init(const Ty &Val)
PointerTypeMap run(const Module &M)
Compute the PointerTypeMap for the module M.
@ Resolved
Queried, materialization begun.
NodeAddr< DefNode * > Def
NodeAddr< InstrNode * > Instr
NodeAddr< UseNode * > Use
This is an optimization pass for GlobalISel generic memory operations.
void dump(const SparseBitVector< ElementSize > &LHS, raw_ostream &out)
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
@ Kill
The last use of a register.
@ Undef
Value of the register doesn't matter.
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
constexpr RegState getDefRegState(bool B)
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
char & SIFixSGPRCopiesLegacyID
LLVM_ABI Printable printMBBReference(const MachineBasicBlock &MBB)
Prints a machine basic block reference.
FunctionPass * createSIFixSGPRCopiesLegacyPass()
MCRegisterClass TargetRegisterClass
void insert(MachineInstr *MI)