78#define DEBUG_TYPE "si-fix-sgpr-copies"
81 "amdgpu-enable-merge-m0",
82 cl::desc(
"Merge and hoist M0 initializations"),
95 unsigned NumSVCopies = 0;
100 unsigned NumReadfirstlanes = 0;
102 bool NeedToBeConvertedToVALU =
false;
110 unsigned SiblingPenalty = 0;
112 V2SCopyInfo() : Copy(nullptr), ID(0){};
113 V2SCopyInfo(
unsigned Id, MachineInstr *
C,
unsigned Width)
114 : Copy(
C), NumReadfirstlanes(Width / 32), ID(
Id){};
115#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
117 dbgs() << ID <<
" : " << *Copy <<
"\n\tS:" << SChain.size()
118 <<
"\n\tSV:" << NumSVCopies <<
"\n\tSP: " << SiblingPenalty
119 <<
"\nScore: " << Score <<
"\n";
124class SIFixSGPRCopies {
125 MachineDominatorTree *MDT;
126 SmallVector<MachineInstr*, 4> SCCCopies;
127 SmallVector<MachineInstr*, 4> RegSequences;
128 SmallVector<MachineInstr*, 4> PHINodes;
129 SmallVector<MachineInstr*, 4> S2VCopies;
130 unsigned NextVGPRToSGPRCopyID = 0;
131 MapVector<unsigned, V2SCopyInfo> V2SCopies;
132 DenseMap<MachineInstr *, SetVector<unsigned>> SiblingPenalty;
133 DenseSet<MachineInstr *> PHISources;
136 MachineRegisterInfo *MRI;
137 const SIRegisterInfo *TRI;
138 const SIInstrInfo *TII;
140 SIFixSGPRCopies(MachineDominatorTree *MDT) : MDT(MDT) {}
142 bool run(MachineFunction &MF);
143 void fixSCCCopies(MachineFunction &MF);
144 void prepareRegSequenceAndPHIs(MachineFunction &MF);
145 unsigned getNextVGPRToSGPRCopyId() {
return ++NextVGPRToSGPRCopyID; }
146 bool needToBeConvertedToVALU(V2SCopyInfo *
I);
147 void analyzeVGPRToSGPRCopy(MachineInstr *
MI);
148 void lowerVGPR2SGPRCopies(MachineFunction &MF);
155 void processPHINode(MachineInstr &
MI);
160 bool tryMoveVGPRConstToSGPR(MachineOperand &MO,
Register NewDst,
161 MachineBasicBlock *BlockToInsertTo,
170 SIFixSGPRCopiesLegacy() : MachineFunctionPass(ID) {}
172 bool runOnMachineFunction(MachineFunction &MF)
override {
173 MachineDominatorTree *MDT =
174 &getAnalysis<MachineDominatorTreeWrapperPass>().getDomTree();
175 SIFixSGPRCopies Impl(MDT);
179 StringRef getPassName()
const override {
return "SI Fix SGPR copies"; }
181 void getAnalysisUsage(AnalysisUsage &AU)
const override {
189 MachineFunctionProperties getClearedProperties()
const override {
190 return MachineFunctionProperties().setNoPHIs();
202char SIFixSGPRCopiesLegacy::ID = 0;
207 return new SIFixSGPRCopiesLegacy();
210static std::pair<const TargetRegisterClass *, const TargetRegisterClass *>
214 Register DstReg = Copy.getOperand(0).getReg();
215 Register SrcReg = Copy.getOperand(1).getReg();
219 :
TRI.getPhysRegBaseClass(SrcReg);
226 :
TRI.getPhysRegBaseClass(DstReg);
228 return std::pair(SrcRC, DstRC);
234 return SrcRC != &AMDGPU::VReg_1RegClass &&
TRI.isSGPRClass(DstRC) &&
235 TRI.hasVectorRegisters(SrcRC);
241 return DstRC != &AMDGPU::VReg_1RegClass &&
TRI.isSGPRClass(SrcRC) &&
242 TRI.hasVectorRegisters(DstRC);
249 auto &Src =
MI.getOperand(1);
256 const auto *
UseMI = MO.getParent();
259 if (MO.isDef() ||
UseMI->getParent() !=
MI.getParent() ||
260 UseMI->getOpcode() <= TargetOpcode::GENERIC_OP_END)
263 unsigned OpIdx = MO.getOperandNo();
264 if (
OpIdx >=
UseMI->getDesc().getNumOperands() ||
318 if (SubReg != AMDGPU::NoSubRegister)
332 bool IsAGPR =
TRI->isAGPRClass(DstRC);
334 for (
unsigned I = 1,
N =
MI.getNumOperands();
I !=
N;
I += 2) {
336 TRI->getRegClassForOperandReg(MRI,
MI.getOperand(
I));
338 "Expected SGPR REG_SEQUENCE to only have SGPR inputs");
350 unsigned Opc = NewSrcRC == &AMDGPU::AGPR_32RegClass ?
351 AMDGPU::V_ACCVGPR_WRITE_B32_e64 : AMDGPU::COPY;
358 MI.getOperand(
I).setReg(TmpReg);
370 if (Copy->getOpcode() != AMDGPU::COPY)
373 if (!MoveImm->isMoveImmediate())
377 TII->getNamedOperand(*MoveImm, AMDGPU::OpName::src0);
382 if (Copy->getOperand(1).getSubReg())
385 switch (MoveImm->getOpcode()) {
388 case AMDGPU::V_MOV_B32_e32:
389 case AMDGPU::AV_MOV_B32_IMM_PSEUDO:
390 SMovOp = AMDGPU::S_MOV_B32;
392 case AMDGPU::V_MOV_B64_e32:
393 case AMDGPU::V_MOV_B64_PSEUDO:
394 SMovOp = AMDGPU::S_MOV_B64_IMM_PSEUDO;
401template <
class UnaryPredicate>
411 while (!Worklist.
empty()) {
451 while (
I !=
MBB->end() &&
TII->isBasicBlockPrologue(*
I))
467 using InitListMap = std::map<unsigned, std::list<MachineInstr *>>;
478 for (
auto &MO :
MI.operands()) {
479 if ((MO.isReg() && ((MO.isDef() && MO.getReg() !=
Reg) || !MO.isDef())) ||
480 (!MO.isImm() && !MO.isReg()) || (MO.isImm() && Imm)) {
488 Inits[Imm->getImm()].push_front(&
MI);
493 for (
auto &
Init : Inits) {
494 auto &Defs =
Init.second;
496 for (
auto I1 = Defs.begin(),
E = Defs.end(); I1 !=
E; ) {
499 for (
auto I2 = std::next(I1); I2 !=
E; ) {
508 auto interferes = [&MDT, From, To](
MachineInstr* &Clobber) ->
bool {
511 bool MayClobberFrom =
isReachable(Clobber, &*From, MBBTo, MDT);
512 bool MayClobberTo =
isReachable(Clobber, &*To, MBBTo, MDT);
513 if (!MayClobberFrom && !MayClobberTo)
515 if ((MayClobberFrom && !MayClobberTo) ||
516 (!MayClobberFrom && MayClobberTo))
522 return !((MBBFrom == MBBTo &&
530 return C.first !=
Init.first &&
536 if (!interferes(MI2, MI1)) {
546 if (!interferes(MI1, MI2)) {
564 if (!interferes(MI1,
I) && !interferes(MI2,
I)) {
568 <<
"and moving from "
585 for (
auto &
Init : Inits) {
586 auto &Defs =
Init.second;
587 auto I = Defs.begin();
588 while (
I != Defs.end()) {
589 if (MergedInstrs.
count(*
I)) {
590 (*I)->eraseFromParent();
598 for (
auto &
Init : Inits) {
599 auto &Defs =
Init.second;
600 for (
auto *
MI : Defs) {
601 auto *
MBB =
MI->getParent();
606 if (!
TII->isBasicBlockPrologue(*
B))
609 auto R = std::next(
MI->getReverseIterator());
610 const unsigned Threshold = 50;
612 for (
unsigned I = 0; R !=
B &&
I < Threshold; ++R, ++
I)
613 if (R->readsRegister(
Reg,
TRI) || R->definesRegister(
Reg,
TRI) ||
614 TII->isSchedulingBoundary(*R,
MBB, *
MBB->getParent()))
636 TRI =
ST.getRegisterInfo();
637 TII =
ST.getInstrInfo();
640 SmallVector<MachineInstr *, 8> Relegalize;
642 for (MachineBasicBlock &
MBB : MF) {
645 MachineInstr &
MI = *
I;
647 switch (
MI.getOpcode()) {
651 if (
TII->isWMMA(
MI) &&
672 if (lowerSpecialCase(
MI,
I))
675 analyzeVGPRToSGPRCopy(&
MI);
680 case AMDGPU::STRICT_WQM:
681 case AMDGPU::SOFT_WQM:
682 case AMDGPU::STRICT_WWM:
683 case AMDGPU::INSERT_SUBREG:
685 case AMDGPU::REG_SEQUENCE: {
686 if (
TRI->isSGPRClass(
TII->getOpRegClass(
MI, 0))) {
687 for (MachineOperand &MO :
MI.operands()) {
688 if (!MO.isReg() || !MO.getReg().isVirtual())
691 if (SrcRC == &AMDGPU::VReg_1RegClass)
694 if (
TRI->hasVectorRegisters(SrcRC)) {
696 TRI->getEquivalentSGPRClass(SrcRC);
697 Register NewDst = MRI->createVirtualRegister(DestRC);
698 MachineBasicBlock *BlockToInsertCopy =
699 MI.isPHI() ?
MI.getOperand(MO.getOperandNo() + 1).getMBB()
705 if (!tryMoveVGPRConstToSGPR(MO, NewDst, BlockToInsertCopy,
706 PointToInsertCopy,
DL)) {
707 MachineInstr *NewCopy =
708 BuildMI(*BlockToInsertCopy, PointToInsertCopy,
DL,
709 TII->get(AMDGPU::COPY), NewDst)
712 analyzeVGPRToSGPRCopy(NewCopy);
713 PHISources.
insert(NewCopy);
721 else if (
MI.isRegSequence())
726 case AMDGPU::V_WRITELANE_B32: {
729 if (
ST.getConstantBusLimit(
MI.getOpcode()) != 1)
739 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::src0);
741 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::src1);
742 MachineOperand &Src0 =
MI.getOperand(Src0Idx);
743 MachineOperand &Src1 =
MI.getOperand(Src1Idx);
747 Src0.
getReg() != AMDGPU::M0) &&
749 Src1.
getReg() != AMDGPU::M0)) {
756 for (MachineOperand *MO : {&Src0, &Src1}) {
757 if (MO->getReg().isVirtual()) {
758 MachineInstr *
DefMI = MRI->getVRegDef(MO->getReg());
762 MO->getReg() ==
Def.getReg() &&
763 MO->getSubReg() ==
Def.getSubReg()) {
765 if (Copied.
isImm() &&
766 TII->isInlineConstant(APInt(64, Copied.
getImm(),
true))) {
767 MO->ChangeToImmediate(Copied.
getImm());
780 TII->get(AMDGPU::COPY), AMDGPU::M0)
791 lowerVGPR2SGPRCopies(MF);
794 for (
auto *
MI : S2VCopies) {
803 for (
auto *
MI : RegSequences) {
805 if (
MI->isRegSequence())
808 for (
auto *
MI : PHINodes) {
811 while (!Relegalize.
empty())
814 if (MF.getTarget().getOptLevel() > CodeGenOptLevel::None &&
EnableM0Merge)
817 SiblingPenalty.clear();
820 RegSequences.clear();
828void SIFixSGPRCopies::processPHINode(MachineInstr &
MI) {
829 bool AllAGPRUses =
true;
830 SetVector<const MachineInstr *> worklist;
831 SmallPtrSet<const MachineInstr *, 4> Visited;
832 SetVector<MachineInstr *> PHIOperands;
836 bool HasUses =
false;
837 while (!worklist.
empty()) {
840 for (
const auto &Use : MRI->use_operands(
Reg)) {
842 const MachineInstr *
UseMI =
Use.getParent();
845 TRI->isAGPR(*MRI,
Use.getReg());
857 if (HasUses && AllAGPRUses && !
TRI->isAGPRClass(RC0)) {
859 MRI->setRegClass(PHIRes,
TRI->getEquivalentAGPRClass(RC0));
860 for (
unsigned I = 1,
N =
MI.getNumOperands();
I !=
N;
I += 2) {
861 MachineInstr *
DefMI = MRI->getVRegDef(
MI.getOperand(
I).getReg());
867 if (
TRI->hasVectorRegisters(MRI->getRegClass(PHIRes)) ||
868 RC0 == &AMDGPU::VReg_1RegClass) {
870 TII->legalizeOperands(
MI, MDT);
874 while (!PHIOperands.
empty()) {
879bool SIFixSGPRCopies::tryMoveVGPRConstToSGPR(
880 MachineOperand &MaybeVGPRConstMO,
Register DstReg,
881 MachineBasicBlock *BlockToInsertTo,
884 MachineInstr *
DefMI = MRI->getVRegDef(MaybeVGPRConstMO.
getReg());
888 MachineOperand *SrcConst =
TII->getNamedOperand(*
DefMI, AMDGPU::OpName::src0);
889 if (SrcConst->
isReg())
893 MRI->getRegClass(MaybeVGPRConstMO.
getReg());
894 unsigned MoveSize =
TRI->getRegSizeInBits(*SrcRC);
896 MoveSize == 64 ? AMDGPU::S_MOV_B64_IMM_PSEUDO : AMDGPU::S_MOV_B32;
897 BuildMI(*BlockToInsertTo, PointToInsertTo,
DL,
TII->get(MoveOp), DstReg)
899 if (MRI->hasOneUse(MaybeVGPRConstMO.
getReg()))
901 MaybeVGPRConstMO.
setReg(DstReg);
905bool SIFixSGPRCopies::lowerSpecialCase(MachineInstr &
MI,
915 if (DstReg == AMDGPU::M0 &&
TRI->hasVectorRegisters(SrcRC)) {
917 MRI->createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
919 const MCInstrDesc &ReadFirstLaneDesc =
920 TII->get(AMDGPU::V_READFIRSTLANE_B32);
921 BuildMI(*
MI.getParent(),
MI,
MI.getDebugLoc(), ReadFirstLaneDesc, TmpReg)
922 .
add(
MI.getOperand(1));
924 unsigned SubReg =
MI.getOperand(1).getSubReg();
925 MI.getOperand(1).setReg(TmpReg);
926 MI.getOperand(1).setSubReg(AMDGPU::NoSubRegister);
930 SubReg == AMDGPU::NoSubRegister
932 :
TRI->getMatchingSuperRegClass(SrcRC, OpRC, SubReg);
934 if (!MRI->constrainRegClass(SrcReg, ConstrainRC))
939 if (tryMoveVGPRConstToSGPR(
MI.getOperand(1), DstReg,
MI.getParent(),
MI,
941 I =
MI.eraseFromParent();
949 SIInstrWorklist worklist;
951 TII->moveToVALU(worklist, MDT);
960 MI.getOperand(1).ChangeToImmediate(Imm);
961 MI.addImplicitDefUseOperands(*
MI.getMF());
962 MI.setDesc(
TII->get(SMovOp));
968void SIFixSGPRCopies::analyzeVGPRToSGPRCopy(MachineInstr*
MI) {
974 V2SCopyInfo
Info(getNextVGPRToSGPRCopyId(),
MI,
975 TRI->getRegSizeInBits(*DstRC));
976 SmallVector<MachineInstr *, 8> AnalysisWorklist;
979 DenseSet<MachineInstr *> Visited;
981 while (!AnalysisWorklist.
empty()) {
985 if (!Visited.
insert(Inst).second)
1005 SiblingPenalty[Inst].insert(
Info.ID);
1007 SmallVector<MachineInstr *, 4>
Users;
1013 !
I->findRegisterDefOperand(AMDGPU::SCC,
nullptr)) {
1014 if (
I->readsRegister(AMDGPU::SCC,
nullptr))
1020 !
TII->isVALU(*Inst,
true)) {
1021 for (
auto &U : MRI->use_instructions(
Reg))
1022 Users.push_back(&U);
1025 for (
auto *U :
Users) {
1026 if (
TII->isSALU(*U))
1027 Info.SChain.insert(U);
1031 V2SCopies[
Info.ID] = std::move(Info);
1036bool SIFixSGPRCopies::needToBeConvertedToVALU(V2SCopyInfo *Info) {
1037 if (
Info->SChain.empty()) {
1042 Info->SChain, [&](MachineInstr *
A, MachineInstr *
B) ->
bool {
1043 return SiblingPenalty[A].size() < SiblingPenalty[B].size();
1045 Info->Siblings.remove_if([&](
unsigned ID) {
return ID ==
Info->ID; });
1051 SmallSet<std::pair<Register, unsigned>, 4> SrcRegs;
1052 for (
auto J :
Info->Siblings) {
1053 auto *InfoIt = V2SCopies.find(J);
1054 if (InfoIt != V2SCopies.end()) {
1055 MachineInstr *SiblingCopy = InfoIt->second.Copy;
1064 Info->SiblingPenalty = SrcRegs.
size();
1067 Info->NumSVCopies +
Info->SiblingPenalty +
Info->NumReadfirstlanes;
1068 unsigned Profit =
Info->SChain.size();
1069 Info->Score = Penalty > Profit ? 0 : Profit - Penalty;
1070 Info->NeedToBeConvertedToVALU =
Info->Score < 3;
1071 return Info->NeedToBeConvertedToVALU;
1074void SIFixSGPRCopies::lowerVGPR2SGPRCopies(MachineFunction &MF) {
1076 SmallVector<unsigned, 8> LoweringWorklist;
1077 for (
auto &
C : V2SCopies) {
1078 if (needToBeConvertedToVALU(&
C.second))
1086 while (!LoweringWorklist.
empty()) {
1088 auto *CurInfoIt = V2SCopies.find(CurID);
1089 if (CurInfoIt != V2SCopies.end() && !CurInfoIt->second.Erased) {
1090 V2SCopyInfo &
C = CurInfoIt->second;
1092 for (
auto S :
C.Siblings) {
1093 auto *SibInfoIt = V2SCopies.find(S);
1094 if (SibInfoIt != V2SCopies.end() && !SibInfoIt->second.Erased) {
1095 V2SCopyInfo &
SI = SibInfoIt->second;
1097 if (!
SI.NeedToBeConvertedToVALU) {
1098 SI.SChain.set_subtract(
C.SChain);
1099 if (needToBeConvertedToVALU(&SI))
1102 SI.Siblings.remove_if([&](
unsigned ID) {
return ID ==
C.ID; });
1106 <<
" is being turned to VALU\n");
1111 V2SCopies.remove_if([](
const auto &
P) {
return P.second.Erased; });
1117 for (
auto C : V2SCopies) {
1118 MachineInstr *
MI =
C.second.Copy;
1119 MachineBasicBlock *
MBB =
MI->getParent();
1123 <<
" is being turned to v_readfirstlane_b32"
1124 <<
" Score: " <<
C.second.Score <<
"\n");
1125 Register DstReg =
MI->getOperand(0).getReg();
1126 MRI->constrainRegClass(DstReg, &AMDGPU::SReg_32_XM0RegClass);
1128 Register SrcReg =
MI->getOperand(1).getReg();
1129 unsigned SubReg =
MI->getOperand(1).getSubReg();
1131 TRI->getRegClassForOperandReg(*MRI,
MI->getOperand(1));
1132 size_t SrcSize =
TRI->getRegSizeInBits(*SrcRC);
1133 if (SrcSize == 16) {
1135 "We do not expect to see 16-bit copies from VGPR to SGPR unless "
1136 "we have 16-bit VGPRs");
1137 assert(MRI->getRegClass(DstReg) == &AMDGPU::SReg_32RegClass ||
1138 MRI->getRegClass(DstReg) == &AMDGPU::SReg_32_XM0RegClass);
1140 MRI->setRegClass(DstReg, &AMDGPU::SReg_32_XM0RegClass);
1141 Register VReg32 = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
1143 Register Undef = MRI->createVirtualRegister(&AMDGPU::VGPR_16RegClass);
1146 .
addReg(SrcReg, {}, SubReg)
1147 .addImm(AMDGPU::lo16)
1152 }
else if (SrcSize == 32) {
1153 const MCInstrDesc &ReadFirstLaneDesc =
1154 TII->get(AMDGPU::V_READFIRSTLANE_B32);
1157 .
addReg(SrcReg, {}, SubReg);
1160 SubReg == AMDGPU::NoSubRegister
1162 :
TRI->getMatchingSuperRegClass(MRI->getRegClass(SrcReg), OpRC,
1165 if (!MRI->constrainRegClass(SrcReg, ConstrainRC))
1169 TII->get(AMDGPU::REG_SEQUENCE), DstReg);
1170 int N =
TRI->getRegSizeInBits(*SrcRC) / 32;
1171 for (
int i = 0; i <
N; i++) {
1173 Result, *MRI,
MI->getOperand(1), SrcRC,
1174 TRI->getSubRegFromChannel(i), &AMDGPU::VGPR_32RegClass);
1176 MRI->createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
1178 TII->get(AMDGPU::V_READFIRSTLANE_B32), PartialDst)
1180 Result.addReg(PartialDst).addImm(
TRI->getSubRegFromChannel(i));
1183 MI->eraseFromParent();
1187void SIFixSGPRCopies::fixSCCCopies(MachineFunction &MF) {
1188 const AMDGPU::LaneMaskConstants &LMC =
1190 for (MachineBasicBlock &
MBB : MF) {
1193 MachineInstr &
MI = *
I;
1199 if (SrcReg == AMDGPU::SCC) {
1201 MRI->createVirtualRegister(
TRI->getWaveMaskRegClass());
1206 I =
BuildMI(*
MI.getParent(), std::next(
I),
I->getDebugLoc(),
1207 TII->get(AMDGPU::COPY), DstReg)
1209 MI.eraseFromParent();
1212 if (DstReg == AMDGPU::SCC) {
1213 Register Tmp = MRI->createVirtualRegister(
TRI->getBoolRC());
1219 MI.eraseFromParent();
1229 SIFixSGPRCopies Impl(&MDT);
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
Provides AMDGPU specific target descriptions.
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
AMD GCN specific subclass of TargetSubtarget.
const HexagonInstrInfo * TII
iv Induction Variable Users
Register const TargetRegisterInfo * TRI
Promote Memory to Register
MachineInstr unsigned OpIdx
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
static std::pair< const TargetRegisterClass *, const TargetRegisterClass * > getCopyRegClasses(const MachineInstr &Copy, const SIRegisterInfo &TRI, const MachineRegisterInfo &MRI)
static cl::opt< bool > EnableM0Merge("amdgpu-enable-merge-m0", cl::desc("Merge and hoist M0 initializations"), cl::init(true))
static bool hoistAndMergeSGPRInits(unsigned Reg, const MachineRegisterInfo &MRI, const TargetRegisterInfo *TRI, MachineDominatorTree &MDT, const TargetInstrInfo *TII)
static bool foldVGPRCopyIntoRegSequence(MachineInstr &MI, const SIRegisterInfo *TRI, const SIInstrInfo *TII, MachineRegisterInfo &MRI)
bool searchPredecessors(const MachineBasicBlock *MBB, const MachineBasicBlock *CutOff, UnaryPredicate Predicate)
static bool isReachable(const MachineInstr *From, const MachineInstr *To, const MachineBasicBlock *CutOff, MachineDominatorTree &MDT)
static bool isVGPRToSGPRCopy(const TargetRegisterClass *SrcRC, const TargetRegisterClass *DstRC, const SIRegisterInfo &TRI)
static bool tryChangeVGPRtoSGPRinCopy(MachineInstr &MI, const SIRegisterInfo *TRI, const SIInstrInfo *TII)
static bool isSGPRToVGPRCopy(const TargetRegisterClass *SrcRC, const TargetRegisterClass *DstRC, const SIRegisterInfo &TRI)
static bool isSafeToFoldImmIntoCopy(const MachineInstr *Copy, const MachineInstr *MoveImm, const SIInstrInfo *TII, unsigned &SMovOp, int64_t &Imm)
static MachineBasicBlock::iterator getFirstNonPrologue(MachineBasicBlock *MBB, const TargetInstrInfo *TII)
const unsigned CSelectOpc
static const LaneMaskConstants & get(const GCNSubtarget &ST)
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
AnalysisUsage & addRequired()
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Implements a dense probed hash-table based set.
NodeT * findNearestCommonDominator(NodeT *A, NodeT *B) const
Find nearest common dominator basic block for basic block A and B.
bool properlyDominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
properlyDominates - Returns true iff A dominates B and A != B.
FunctionPass class - This class is used to implement most global optimizations.
MachineInstrBundleIterator< MachineInstr, true > reverse_iterator
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
LLVM_ABI instr_iterator getFirstInstrTerminator()
Same getFirstTerminator but it ignores bundles and return an instr_iterator instead.
MachineInstrBundleIterator< MachineInstr > iterator
Analysis pass which computes a MachineDominatorTree.
Analysis pass which computes a MachineDominatorTree.
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
bool dominates(const MachineInstr *A, const MachineInstr *B) const
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
const MachineFunctionProperties & getProperties() const
Get the function properties.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
Representation of each machine instruction.
bool isImplicitDef() const
const MachineBasicBlock * getParent() const
bool isCompare(QueryType Type=IgnoreBundle) const
Return true if this instruction is a comparison.
bool isRegSequence() const
LLVM_ABI unsigned getNumExplicitDefs() const
Returns the number of non-implicit definitions.
bool isMoveImmediate(QueryType Type=IgnoreBundle) const
Return true if this instruction is a move immediate (including conditional moves) instruction.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
MachineOperand class - Representation of each machine instruction operand.
unsigned getSubReg() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
LLVM_ABI void ChangeToRegister(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isDebug=false)
ChangeToRegister - Replace this operand with a new register operand of the specified value.
Register getReg() const
getReg - Returns the register number.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
iterator_range< def_instr_iterator > def_instructions(Register Reg) const
use_instr_iterator use_instr_begin(Register RegNo) const
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
bool hasOneUse(Register RegNo) const
hasOneUse - Return true if there is exactly one instruction using the specified register.
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
iterator_range< reg_nodbg_iterator > reg_nodbg_operands(Register Reg) const
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Wrapper class representing virtual and physical registers.
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
A vector that has set insertion semantics.
bool empty() const
Determine if the SetVector is empty or not.
bool insert(const value_type &X)
Insert a new element into the SetVector.
value_type pop_back_val()
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
TargetInstrInfo - Interface to description of machine instruction set.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
std::pair< iterator, bool > insert(const ValueT &V)
bool contains(const_arg_type_t< ValueT > V) const
Check if the set contains the given element.
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
initializer< Ty > init(const Ty &Val)
DXILDebugInfoMap run(Module &M)
@ Resolved
Queried, materialization begun.
NodeAddr< DefNode * > Def
NodeAddr< InstrNode * > Instr
NodeAddr< UseNode * > Use
This is an optimization pass for GlobalISel generic memory operations.
void dump(const SparseBitVector< ElementSize > &LHS, raw_ostream &out)
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
@ Kill
The last use of a register.
@ Undef
Value of the register doesn't matter.
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
constexpr RegState getDefRegState(bool B)
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
char & SIFixSGPRCopiesLegacyID
LLVM_ABI Printable printMBBReference(const MachineBasicBlock &MBB)
Prints a machine basic block reference.
FunctionPass * createSIFixSGPRCopiesLegacyPass()
MCRegisterClass TargetRegisterClass
void insert(MachineInstr *MI)