46#define DEBUG_TYPE "machine-scheduler"
51 "amdgpu-disable-unclustered-high-rp-reschedule",
cl::Hidden,
52 cl::desc(
"Disable unclustered high register pressure "
53 "reduction scheduling stage."),
57 "amdgpu-disable-clustered-low-occupancy-reschedule",
cl::Hidden,
58 cl::desc(
"Disable clustered low occupancy "
59 "rescheduling for ILP scheduling stage."),
65 "Sets the bias which adds weight to occupancy vs latency. Set it to "
66 "100 to chase the occupancy only."),
71 cl::desc(
"Relax occupancy targets for kernels which are memory "
72 "bound (amdgpu-membound-threshold), or "
73 "Wave Limited (amdgpu-limit-wave-threshold)."),
78 cl::desc(
"Use the AMDGPU specific RPTrackers during scheduling"),
82 "amdgpu-scheduler-pending-queue-limit",
cl::Hidden,
84 "Max (Available+Pending) size to inspect pending queue (0 disables)"),
87#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
88#define DUMP_MAX_REG_PRESSURE
90 "amdgpu-print-max-reg-pressure-regusage-before-scheduler",
cl::Hidden,
91 cl::desc(
"Print a list of live registers along with their def/uses at the "
92 "point of maximum register pressure before scheduling."),
96 "amdgpu-print-max-reg-pressure-regusage-after-scheduler",
cl::Hidden,
97 cl::desc(
"Print a list of live registers along with their def/uses at the "
98 "point of maximum register pressure after scheduling."),
103 "amdgpu-disable-rewrite-mfma-form-sched-stage",
cl::Hidden,
109 return O.error(
"'" + Arg +
"' value invalid for uint argument!");
112 return O.error(
"'" + Arg +
"' value must be in the range [0, 100]!");
119 cl::desc(
"Percent of VGPR limits that we should use as RP threshold "
120 "during scheduling. We have two limits relevant to scheduling: "
121 "Critical (avoid decreasing occupancy), Excess (avoid spilling). "
122 "This flag scales both limits back by an equal percent: (0 = use "
123 " default calculation, 1-100 = use percentage), default: 0"),
144 Context->RegClassInfo->getNumAllocatableRegs(&AMDGPU::SGPR_32RegClass);
146 Context->RegClassInfo->getNumAllocatableRegs(&AMDGPU::VGPR_32RegClass);
148 Context->RegClassInfo->getNumAllocatableRegs(&AMDGPU::AGPR_32RegClass);
170 "VGPRCriticalLimit calculation method.\n");
174 unsigned Addressable =
177 VGPRBudget = std::max(VGPRBudget, Granule);
193 <<
". VGPRCriticalLimit: " << OriginalVGPRCriticalLimit
239 if (!
Op.isReg() ||
Op.isImplicit())
241 if (
Op.getReg().isPhysical() ||
242 (
Op.isDef() &&
Op.getSubReg() != AMDGPU::NoSubRegister))
277 Pressure[AMDGPU::RegisterPressureSets::VGPR_32] =
286 unsigned SGPRPressure,
287 unsigned VGPRPressure,
288 unsigned AGPRPressure,
bool IsBottomUp) {
292 if (!
DAG->isTrackingPressure())
315 Pressure[AMDGPU::RegisterPressureSets::SReg_32] = SGPRPressure;
316 Pressure[AMDGPU::RegisterPressureSets::VGPR_32] = VGPRPressure;
317 Pressure[AMDGPU::RegisterPressureSets::AGPR_32] = AGPRPressure;
319 for (
const auto &Diff :
DAG->getPressureDiff(SU)) {
325 (IsBottomUp ? Diff.getUnitInc() : -Diff.getUnitInc());
328#ifdef EXPENSIVE_CHECKS
329 std::vector<unsigned> CheckPressure, CheckMaxPressure;
332 if (
Pressure[AMDGPU::RegisterPressureSets::SReg_32] !=
333 CheckPressure[AMDGPU::RegisterPressureSets::SReg_32] ||
334 Pressure[AMDGPU::RegisterPressureSets::VGPR_32] !=
335 CheckPressure[AMDGPU::RegisterPressureSets::VGPR_32] ||
336 Pressure[AMDGPU::RegisterPressureSets::AGPR_32] !=
337 CheckPressure[AMDGPU::RegisterPressureSets::AGPR_32]) {
338 errs() <<
"Register Pressure is inaccurate when calculated through "
340 <<
"SGPR got " <<
Pressure[AMDGPU::RegisterPressureSets::SReg_32]
342 << CheckPressure[AMDGPU::RegisterPressureSets::SReg_32] <<
"\n"
343 <<
"VGPR got " <<
Pressure[AMDGPU::RegisterPressureSets::VGPR_32]
345 << CheckPressure[AMDGPU::RegisterPressureSets::VGPR_32] <<
"\n"
346 <<
"AGPR got " <<
Pressure[AMDGPU::RegisterPressureSets::AGPR_32]
348 << CheckPressure[AMDGPU::RegisterPressureSets::AGPR_32] <<
"\n";
354 unsigned NewAGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::AGPR_32];
355 unsigned NewSGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::SReg_32];
356 unsigned NewVGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::VGPR_32];
366 const unsigned MaxVGPRPressureInc = 16;
367 bool ShouldTrackVGPRs = VGPRPressure + MaxVGPRPressureInc >=
VGPRExcessLimit;
370 bool ShouldTrackSGPRs =
371 !ShouldTrackVGPRs && !ShouldTrackAGPRs && SGPRPressure >=
SGPRExcessLimit;
406 : std::numeric_limits<int>::min();
408 if (SGPRDelta >= 0 || VGPRDelta >= 0 || AGPRDelta >= 0) {
411 if (VGPRDelta >= SGPRDelta && VGPRDelta >= AGPRDelta) {
415 }
else if (AGPRDelta >= SGPRDelta) {
429 bool HasBufferedModel =
448 dbgs() <<
"Prefer:\t\t";
449 DAG->dumpNode(*Preferred.
SU);
453 DAG->dumpNode(*Current.
SU);
456 dbgs() <<
"Reason:\t\t";
470 unsigned SGPRPressure = 0;
471 unsigned VGPRPressure = 0;
472 unsigned AGPRPressure = 0;
474 if (
DAG->isTrackingPressure()) {
476 SGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::SReg_32];
477 VGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::VGPR_32];
478 AGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::AGPR_32];
483 SGPRPressure =
T->getPressure().getSGPRNum();
484 VGPRPressure =
T->getPressure().getArchVGPRNum();
485 AGPRPressure =
T->getPressure().getAGPRNum();
490 for (
SUnit *SU : AQ) {
494 VGPRPressure, AGPRPressure, IsBottomUp);
514 for (
SUnit *SU : PQ) {
518 VGPRPressure, AGPRPressure, IsBottomUp);
538 bool &PickedPending) {
558 bool BotPending =
false;
578 "Last pick result should correspond to re-picking right now");
583 bool TopPending =
false;
603 "Last pick result should correspond to re-picking right now");
613 PickedPending = BotPending && TopPending;
616 if (BotPending || TopPending) {
623 Cand.setBest(TryCand);
628 IsTopNode = Cand.AtTop;
635 if (
DAG->top() ==
DAG->bottom()) {
637 Bot.Available.empty() &&
Bot.Pending.empty() &&
"ReadyQ garbage");
643 PickedPending =
false;
677 if (ReadyCycle > CurrentCycle)
748 if (
DAG->isTrackingPressure() &&
754 if (
DAG->isTrackingPressure() &&
759 bool SameBoundary = Zone !=
nullptr;
783 if (IsLegacyScheduler)
802 if (
DAG->isTrackingPressure() &&
812 bool SameBoundary = Zone !=
nullptr;
847 bool CandIsClusterSucc =
849 bool TryCandIsClusterSucc =
851 if (
tryGreater(TryCandIsClusterSucc, CandIsClusterSucc, TryCand, Cand,
856 if (
DAG->isTrackingPressure() &&
862 if (
DAG->isTrackingPressure() &&
908 if (
DAG->isTrackingPressure()) {
924 bool CandIsClusterSucc =
926 bool TryCandIsClusterSucc =
928 if (
tryGreater(TryCandIsClusterSucc, CandIsClusterSucc, TryCand, Cand,
937 bool SameBoundary = Zone !=
nullptr;
954 if (TryMayLoad || CandMayLoad) {
955 bool TryLongLatency =
957 bool CandLongLatency =
961 Zone->
isTop() ? CandLongLatency : TryLongLatency, TryCand,
979 if (
DAG->isTrackingPressure() &&
998 !
Rem.IsAcyclicLatencyLimited &&
tryLatency(TryCand, Cand, *Zone))
1016 StartingOccupancy(MFI.getOccupancy()), MinOccupancy(StartingOccupancy),
1017 RegionLiveOuts(this,
true) {
1023 LLVM_DEBUG(
dbgs() <<
"Starting occupancy is " << StartingOccupancy <<
".\n");
1025 MinOccupancy = std::min(MFI.getMinAllowedOccupancy(), StartingOccupancy);
1026 if (MinOccupancy != StartingOccupancy)
1027 LLVM_DEBUG(
dbgs() <<
"Allowing Occupancy drops to " << MinOccupancy
1032std::unique_ptr<GCNSchedStage>
1034 switch (SchedStageID) {
1036 return std::make_unique<OccInitialScheduleStage>(SchedStageID, *
this);
1038 return std::make_unique<RewriteMFMAFormStage>(SchedStageID, *
this);
1040 return std::make_unique<UnclusteredHighRPStage>(SchedStageID, *
this);
1042 return std::make_unique<ClusteredLowOccStage>(SchedStageID, *
this);
1044 return std::make_unique<PreRARematStage>(SchedStageID, *
this);
1046 return std::make_unique<ILPInitialScheduleStage>(SchedStageID, *
this);
1048 return std::make_unique<MemoryClauseInitialScheduleStage>(SchedStageID,
1051 return std::make_unique<LiveIntervalRPStage>(SchedStageID, *
this);
1064GCNScheduleDAGMILive::getRealRegPressure(
unsigned RegionIdx)
const {
1065 if (Regions[RegionIdx].first == Regions[RegionIdx].second)
1069 &LiveIns[RegionIdx]);
1075 assert(RegionBegin != RegionEnd &&
"Region must not be empty");
1079void GCNScheduleDAGMILive::computeBlockPressure(
unsigned RegionIdx,
1091 const MachineBasicBlock *OnlySucc =
nullptr;
1094 if (!Candidate->empty() && Candidate->pred_size() == 1) {
1095 SlotIndexes *Ind =
LIS->getSlotIndexes();
1097 OnlySucc = Candidate;
1102 size_t CurRegion = RegionIdx;
1103 for (
size_t E = Regions.size(); CurRegion !=
E; ++CurRegion)
1104 if (Regions[CurRegion].first->getParent() !=
MBB)
1109 auto LiveInIt = MBBLiveIns.find(
MBB);
1110 auto &Rgn = Regions[CurRegion];
1112 if (LiveInIt != MBBLiveIns.end()) {
1113 auto LiveIn = std::move(LiveInIt->second);
1115 MBBLiveIns.erase(LiveInIt);
1118 auto LRS = BBLiveInMap.lookup(NonDbgMI);
1119#ifdef EXPENSIVE_CHECKS
1128 if (Regions[CurRegion].first ==
I || NonDbgMI ==
I) {
1129 LiveIns[CurRegion] =
RPTracker.getLiveRegs();
1133 if (Regions[CurRegion].second ==
I) {
1134 Pressure[CurRegion] =
RPTracker.moveMaxPressure();
1135 if (CurRegion-- == RegionIdx)
1137 auto &Rgn = Regions[CurRegion];
1150 MBBLiveIns[OnlySucc] =
RPTracker.moveLiveRegs();
1155GCNScheduleDAGMILive::getRegionLiveInMap()
const {
1156 assert(!Regions.empty());
1157 std::vector<MachineInstr *> RegionFirstMIs;
1158 RegionFirstMIs.reserve(Regions.size());
1160 RegionFirstMIs.push_back(
1167GCNScheduleDAGMILive::getRegionLiveOutMap()
const {
1168 assert(!Regions.empty());
1169 std::vector<MachineInstr *> RegionLastMIs;
1170 RegionLastMIs.reserve(Regions.size());
1181 IdxToInstruction.clear();
1184 IsLiveOut ? DAG->getRegionLiveOutMap() : DAG->getRegionLiveInMap();
1185 for (
unsigned I = 0;
I < DAG->Regions.size();
I++) {
1186 auto &[RegionBegin, RegionEnd] = DAG->Regions[
I];
1188 if (RegionBegin == RegionEnd)
1192 IdxToInstruction[
I] = RegionKey;
1200 LiveIns.resize(Regions.size());
1201 Pressure.resize(Regions.size());
1202 RegionsWithHighRP.resize(Regions.size());
1203 RegionsWithExcessRP.resize(Regions.size());
1204 RegionsWithIGLPInstrs.resize(Regions.size());
1205 RegionsWithHighRP.reset();
1206 RegionsWithExcessRP.reset();
1207 RegionsWithIGLPInstrs.reset();
1212void GCNScheduleDAGMILive::runSchedStages() {
1213 LLVM_DEBUG(
dbgs() <<
"All regions recorded, starting actual scheduling.\n");
1216 if (!Regions.
empty()) {
1217 BBLiveInMap = getRegionLiveInMap();
1222#ifdef DUMP_MAX_REG_PRESSURE
1232 if (!Stage->initGCNSchedStage())
1235 for (
auto Region : Regions) {
1239 if (!Stage->initGCNRegion()) {
1240 Stage->advanceRegion();
1246 const unsigned RegionIdx = Stage->getRegionIdx();
1249 MRI, RegionLiveOuts.getLiveRegsForRegionIdx(RegionIdx));
1253 Stage->finalizeGCNRegion();
1254 Stage->advanceRegion();
1258 Stage->finalizeGCNSchedStage();
1261#ifdef DUMP_MAX_REG_PRESSURE
1274 OS <<
"Max Occupancy Initial Schedule";
1277 OS <<
"Instruction Rewriting Reschedule";
1280 OS <<
"Unclustered High Register Pressure Reschedule";
1283 OS <<
"Clustered Low Occupancy Reschedule";
1286 OS <<
"Pre-RA Rematerialize";
1289 OS <<
"Max ILP Initial Schedule";
1292 OS <<
"Max memory clause Initial Schedule";
1295 OS <<
"Live Interval RP Reschedule";
1315void RewriteMFMAFormStage::findReachingDefs(
1337 while (!Worklist.
empty()) {
1352 for (MachineBasicBlock *PredMBB : DefMBB->
predecessors()) {
1353 if (Visited.
insert(PredMBB).second)
1359void RewriteMFMAFormStage::findReachingUses(
1363 for (MachineOperand &UseMO :
1366 findReachingDefs(UseMO, LIS, ReachingDefIndexes);
1370 if (
any_of(ReachingDefIndexes, [DefIdx](SlotIndex RDIdx) {
1382 if (!
ST.hasGFX90AInsts() ||
MFI.getMinWavesPerEU() > 1)
1385 RegionsWithExcessArchVGPR.resize(
DAG.Regions.size());
1386 RegionsWithExcessArchVGPR.reset();
1390 RegionsWithExcessArchVGPR[
Region] =
true;
1393 if (RegionsWithExcessArchVGPR.none())
1396 TII =
ST.getInstrInfo();
1397 SRI =
ST.getRegisterInfo();
1399 std::vector<std::pair<MachineInstr *, unsigned>> RewriteCands;
1403 if (!initHeuristics(RewriteCands, CopyForUse, CopyForDef))
1406 int64_t
Cost = getRewriteCost(RewriteCands, CopyForUse, CopyForDef);
1413 return rewrite(RewriteCands);
1423 if (
DAG.RegionsWithHighRP.none() &&
DAG.RegionsWithExcessRP.none())
1430 InitialOccupancy =
DAG.MinOccupancy;
1433 TempTargetOccupancy =
MFI.getMaxWavesPerEU() >
DAG.MinOccupancy
1434 ? InitialOccupancy + 1
1436 IsAnyRegionScheduled =
false;
1437 S.SGPRLimitBias =
S.HighRPSGPRBias;
1438 S.VGPRLimitBias =
S.HighRPVGPRBias;
1442 <<
"Retrying function scheduling without clustering. "
1443 "Aggressively try to reduce register pressure to achieve occupancy "
1444 << TempTargetOccupancy <<
".\n");
1459 if (
DAG.StartingOccupancy <=
DAG.MinOccupancy)
1463 dbgs() <<
"Retrying function scheduling with lowest recorded occupancy "
1464 <<
DAG.MinOccupancy <<
".\n");
1469#define REMAT_PREFIX "[PreRARemat] "
1470#define REMAT_DEBUG(X) LLVM_DEBUG(dbgs() << REMAT_PREFIX; X;)
1472#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1473Printable PreRARematStage::ScoredRemat::print()
const {
1475 OS <<
'(' << MaxFreq <<
", " << FreqDiff <<
", " << RegionImpact <<
')';
1490 auto PrintTargetRegions = [&]() ->
void {
1491 if (TargetRegions.none()) {
1496 for (
unsigned I : TargetRegions.set_bits())
1503 dbgs() <<
"Analyzing ";
1504 MF.getFunction().printAsOperand(
dbgs(),
false);
1507 if (!setObjective()) {
1508 LLVM_DEBUG(
dbgs() <<
"no objective to achieve, occupancy is maximal at "
1509 <<
MFI.getMaxWavesPerEU() <<
'\n');
1514 dbgs() <<
"increase occupancy from " << *TargetOcc - 1 <<
'\n';
1516 dbgs() <<
"reduce spilling (minimum target occupancy is "
1517 <<
MFI.getMinWavesPerEU() <<
")\n";
1519 PrintTargetRegions();
1524 DAG.RegionLiveOuts.buildLiveRegMap();
1526 if (!Remater.analyze()) {
1542 DefRegToCandIdx.
resize(
DAG.MRI.getNumVirtRegs());
1543 const unsigned NumRegions =
DAG.Regions.size();
1545 for (
unsigned RegIdx = 0, E = Remater.getNumRegs(); RegIdx < E; ++RegIdx) {
1549 if (CandReg.
Uses.size() != 1)
1551 const auto [UseRegion,
Users] = *CandReg.
Uses.begin();
1570 "user must have at least one operand");
1577 assert(FirstUseMI &&
"there must be a user in the region");
1579 DAG.LIS->getInstructionIndex(*FirstUseMI).getRegSlot(
true);
1581 DAG.LIS->getInstructionIndex(*CandReg.
getLastDef()).getRegSlot(
true);
1583 const Rematerializer::Reg &DepReg = Remater.getReg(DepRegIdx);
1584 Register DepDefReg = DepReg.getDefReg();
1585 return MarkedRegs.contains(DepDefReg) ||
1586 !Remater.isRegIdenticalAtUses(DepDefReg, DepReg.Mask, RefIdx,
1591 [&](
const std::pair<Register, LaneBitmask> &RegAndMask) {
1592 const auto &[Reg, Mask] = RegAndMask;
1593 return !Remater.isRegIdenticalAtUses(Reg, Mask, RefIdx,
1598 Register DefReg = CandReg.getDefReg();
1599 MarkedRegs.
insert(DefReg);
1600 DefRegToCandIdx[DefReg] = Candidates.
size();
1608 for (
unsigned I = 0;
I < NumRegions; ++
I) {
1609 for (
const auto &[Reg, Mask] :
DAG.LiveIns[
I]) {
1612 unsigned CandIdx = DefRegToCandIdx[Reg];
1614 Candidates[CandIdx].LiveIn.set(
I);
1616 for (
const auto &[
Reg, Mask] :
1620 unsigned CandIdx = DefRegToCandIdx[
Reg];
1622 Candidates[CandIdx].LiveOut.set(
I);
1627 SmallVector<unsigned> CandidateOrder;
1628 for (
auto [CandIdx, Cand] :
enumerate(Candidates)) {
1629 Cand.init(FreqInfo, Remater,
DAG);
1630 Cand.update(TargetRegions, RPTargets, FreqInfo, !TargetOcc);
1631 if (!Cand.hasNullScore())
1642 Rollback = std::make_unique<RollbackSupport>(Remater);
1647 BitVector RecomputeRP(
DAG.Regions.size());
1649 RecomputeRP.reset();
1652 sort(CandidateOrder, [&](
unsigned LHSIndex,
unsigned RHSIndex) {
1653 return Candidates[LHSIndex] < Candidates[RHSIndex];
1657 dbgs() <<
"==== NEW REMAT ROUND ====\n"
1659 <<
"Candidates with non-null score, in rematerialization order:\n";
1660 for (
const ScoredRemat &Cand :
reverse(Candidates)) {
1662 << Remater.printRematReg(Cand.RegIdx) <<
'\n';
1664 PrintTargetRegions();
1670 while (!CandidateOrder.
empty()) {
1671 const ScoredRemat &Cand = Candidates[CandidateOrder.
back()];
1672 const Rematerializer::Reg &
Reg = Remater.getReg(Cand.RegIdx);
1680 if (!Cand.maybeBeneficial(TargetRegions, RPTargets)) {
1682 << Cand.print() <<
" | "
1683 << Remater.printRematReg(Cand.RegIdx));
1688#ifdef EXPENSIVE_CHECKS
1691 for (
const MachineInstr *
DefMI :
Reg.Defs) {
1696 if (!MO.isReg() || !MO.getReg() || !MO.readsReg() || MO.isDef())
1703 LiveInterval &LI =
DAG.LIS->getInterval(
UseReg);
1704 LaneBitmask LM =
DAG.MRI.getMaxLaneMaskForVReg(MO.getReg());
1706 LM =
DAG.TRI->getSubRegIndexLaneMask(MO.getSubReg());
1708 const unsigned UseRegion =
Reg.Uses.begin()->first;
1709 LaneBitmask LiveInMask =
DAG.LiveIns[UseRegion].at(
UseReg);
1710 LaneBitmask UncoveredLanes = LM & ~(LiveInMask & LM);
1714 if (UncoveredLanes.
any()) {
1716 for (LiveInterval::SubRange &SR : LI.
subranges())
1717 assert((SR.LaneMask & UncoveredLanes).none());
1725 REMAT_DEBUG(
dbgs() <<
"** REMAT " << Remater.printRematReg(Cand.RegIdx)
1727 removeFromLiveMaps(
Reg.getDefReg(), Cand.LiveIn, Cand.LiveOut);
1729 Rollback->LiveMapUpdates.emplace_back(Cand.RegIdx, Cand.LiveIn,
1732 Cand.rematerialize(Remater);
1737 updateRPTargets(Cand.Live, Cand.RPSave);
1738 RecomputeRP |= Cand.UnpredictableRPSave;
1739 RescheduleRegions |= Cand.Live;
1740 if (!TargetRegions.any()) {
1746 if (!updateAndVerifyRPTargets(RecomputeRP) && !TargetRegions.any()) {
1755 unsigned NumUsefulCandidates = 0;
1756 for (
unsigned CandIdx : CandidateOrder) {
1757 ScoredRemat &Candidate = Candidates[CandIdx];
1758 Candidate.update(TargetRegions, RPTargets, FreqInfo, !TargetOcc);
1759 if (!Candidate.hasNullScore())
1760 CandidateOrder[NumUsefulCandidates++] = CandIdx;
1762 if (NumUsefulCandidates == 0) {
1763 REMAT_DEBUG(
dbgs() <<
"Stop on exhausted rematerialization candidates\n");
1766 CandidateOrder.truncate(NumUsefulCandidates);
1769 if (RescheduleRegions.none())
1775 unsigned DynamicVGPRBlockSize =
MFI.getDynamicVGPRBlockSize();
1776 for (
unsigned I : RescheduleRegions.set_bits()) {
1777 DAG.Pressure[
I] = RPTargets[
I].getCurrentRP();
1779 <<
DAG.Pressure[
I].getOccupancy(
ST, DynamicVGPRBlockSize)
1780 <<
" (" << RPTargets[
I] <<
")\n");
1782 AchievedOcc =
MFI.getMaxWavesPerEU();
1783 for (
const GCNRegPressure &RP :
DAG.Pressure) {
1785 std::min(AchievedOcc,
RP.getOccupancy(
ST, DynamicVGPRBlockSize));
1789 dbgs() <<
"Retrying function scheduling with new min. occupancy of "
1790 << AchievedOcc <<
" from rematerializing (original was "
1791 <<
DAG.MinOccupancy;
1793 dbgs() <<
", target was " << *TargetOcc;
1797 DAG.setTargetOccupancy(getStageTargetOccupancy());
1808 S.SGPRLimitBias =
S.VGPRLimitBias = 0;
1809 if (
DAG.MinOccupancy > InitialOccupancy) {
1810 assert(IsAnyRegionScheduled);
1812 <<
" stage successfully increased occupancy to "
1813 <<
DAG.MinOccupancy <<
'\n');
1814 }
else if (!IsAnyRegionScheduled) {
1815 assert(
DAG.MinOccupancy == InitialOccupancy);
1817 <<
": No regions scheduled, min occupancy stays at "
1818 <<
DAG.MinOccupancy <<
", MFI occupancy stays at "
1819 <<
MFI.getOccupancy() <<
".\n");
1827 if (
DAG.begin() ==
DAG.end())
1834 unsigned NumRegionInstrs = std::distance(
DAG.begin(),
DAG.end());
1838 if (
DAG.begin() == std::prev(
DAG.end()))
1844 <<
"\n From: " << *
DAG.begin() <<
" To: ";
1846 else dbgs() <<
"End";
1847 dbgs() <<
" RegionInstrs: " << NumRegionInstrs <<
'\n');
1855 for (
auto &
I :
DAG) {
1868 dbgs() <<
"Pressure before scheduling:\nRegion live-ins:"
1870 <<
"Region live-in pressure: "
1874 S.HasHighPressure =
false;
1896 unsigned DynamicVGPRBlockSize =
DAG.MFI.getDynamicVGPRBlockSize();
1899 unsigned CurrentTargetOccupancy =
1900 IsAnyRegionScheduled ?
DAG.MinOccupancy : TempTargetOccupancy;
1902 (CurrentTargetOccupancy <= InitialOccupancy ||
1903 DAG.Pressure[
RegionIdx].getOccupancy(
ST, DynamicVGPRBlockSize) !=
1910 if (!IsAnyRegionScheduled && IsSchedulingThisRegion) {
1911 IsAnyRegionScheduled =
true;
1912 if (
MFI.getMaxWavesPerEU() >
DAG.MinOccupancy)
1913 DAG.setTargetOccupancy(TempTargetOccupancy);
1915 return IsSchedulingThisRegion;
1931 return !RevertAllRegions && RescheduleRegions[
RegionIdx] &&
1951 if (
S.HasHighPressure)
1972 if (
DAG.MinOccupancy < *TargetOcc) {
1974 <<
" cannot meet occupancy target, interrupting "
1975 "re-scheduling in all regions\n");
1976 RevertAllRegions =
true;
1987 unsigned DynamicVGPRBlockSize =
DAG.MFI.getDynamicVGPRBlockSize();
1998 unsigned TargetOccupancy = std::min(
1999 S.getTargetOccupancy(),
ST.getOccupancyWithWorkGroupSizes(
MF).second);
2000 unsigned WavesAfter = std::min(
2001 TargetOccupancy,
PressureAfter.getOccupancy(
ST, DynamicVGPRBlockSize));
2002 unsigned WavesBefore = std::min(
2004 LLVM_DEBUG(
dbgs() <<
"Occupancy before scheduling: " << WavesBefore
2005 <<
", after " << WavesAfter <<
".\n");
2011 unsigned NewOccupancy = std::max(WavesAfter, WavesBefore);
2015 if (WavesAfter < WavesBefore && WavesAfter <
DAG.MinOccupancy &&
2016 WavesAfter >=
MFI.getMinAllowedOccupancy()) {
2017 LLVM_DEBUG(
dbgs() <<
"Function is memory bound, allow occupancy drop up to "
2018 <<
MFI.getMinAllowedOccupancy() <<
" waves\n");
2019 NewOccupancy = WavesAfter;
2022 if (NewOccupancy <
DAG.MinOccupancy) {
2023 DAG.MinOccupancy = NewOccupancy;
2024 MFI.limitOccupancy(
DAG.MinOccupancy);
2026 <<
DAG.MinOccupancy <<
".\n");
2030 unsigned MaxVGPRs =
ST.getMaxNumVGPRs(
MF);
2033 unsigned MaxArchVGPRs = std::min(MaxVGPRs,
ST.getAddressableNumArchVGPRs());
2034 unsigned MaxSGPRs =
ST.getMaxNumSGPRs(
MF);
2058 unsigned ReadyCycle = CurrCycle;
2059 for (
auto &
D : SU.
Preds) {
2060 if (
D.isAssignedRegDep()) {
2063 unsigned DefReady = ReadyCycles[
DAG.getSUnit(
DefMI)->NodeNum];
2064 ReadyCycle = std::max(ReadyCycle, DefReady +
Latency);
2067 ReadyCycles[SU.
NodeNum] = ReadyCycle;
2074 std::pair<MachineInstr *, unsigned>
B)
const {
2075 return A.second <
B.second;
2081 if (ReadyCycles.empty())
2083 unsigned BBNum = ReadyCycles.begin()->first->getParent()->getNumber();
2084 dbgs() <<
"\n################## Schedule time ReadyCycles for MBB : " << BBNum
2085 <<
" ##################\n# Cycle #\t\t\tInstruction "
2089 for (
auto &
I : ReadyCycles) {
2090 if (
I.second > IPrev + 1)
2091 dbgs() <<
"****************************** BUBBLE OF " <<
I.second - IPrev
2092 <<
" CYCLES DETECTED ******************************\n\n";
2093 dbgs() <<
"[ " <<
I.second <<
" ] : " << *
I.first <<
"\n";
2106 unsigned SumBubbles = 0;
2108 unsigned CurrCycle = 0;
2109 for (
auto &SU : InputSchedule) {
2110 unsigned ReadyCycle =
2112 SumBubbles += ReadyCycle - CurrCycle;
2114 ReadyCyclesSorted.insert(std::make_pair(SU.getInstr(), ReadyCycle));
2116 CurrCycle = ++ReadyCycle;
2139 unsigned SumBubbles = 0;
2141 unsigned CurrCycle = 0;
2142 for (
auto &
MI :
DAG) {
2146 unsigned ReadyCycle =
2148 SumBubbles += ReadyCycle - CurrCycle;
2150 ReadyCyclesSorted.insert(std::make_pair(SU->
getInstr(), ReadyCycle));
2152 CurrCycle = ++ReadyCycle;
2169 if (WavesAfter <
DAG.MinOccupancy)
2173 if (
DAG.MFI.isDynamicVGPREnabled()) {
2176 DAG.MFI.getDynamicVGPRBlockSize());
2179 if (BlocksAfter > BlocksBefore)
2216 <<
"\n\t *** In shouldRevertScheduling ***\n"
2217 <<
" *********** BEFORE UnclusteredHighRPStage ***********\n");
2221 <<
"\n *********** AFTER UnclusteredHighRPStage ***********\n");
2223 unsigned OldMetric = MBefore.
getMetric();
2224 unsigned NewMetric = MAfter.
getMetric();
2225 unsigned WavesBefore = std::min(
2226 S.getTargetOccupancy(),
2233 LLVM_DEBUG(
dbgs() <<
"\tMetric before " << MBefore <<
"\tMetric after "
2234 << MAfter <<
"Profit: " << Profit <<
"\n");
2265 unsigned WavesAfter) {
2275 cl::desc(
"Percent increase of live interval RP over instant pressure to "
2276 "trigger rescheduling"),
2282 "Reduction factor (percent) for VGPR threshold during live interval RP "
2283 "reschedule stage"),
2287 "amdgpu-lirp-instant-lower-bound",
cl::Hidden,
2288 cl::desc(
"Lower bound (percent of the VGPR excess limit) on instant RP, "
2289 "below which a region is skipped"),
2299 if (!
S.VGPRThresholdPercent) {
2300 LLVM_DEBUG(
dbgs() <<
"LIRP: expected VGPRThresholdPercent to be enabled, "
2301 "not using live interval RP reschedule stage\n");
2309 unsigned InstantRP =
DAG.Pressure[
RegionIdx].getArchVGPRNum();
2310 auto [RegionBegin, RegionEnd] =
DAG.Regions[
RegionIdx];
2311 if (RegionBegin == RegionEnd)
2318 unsigned NewVGPRThresholdPercent =
2322 <<
", VGPRThresholdPercent: " <<
S.VGPRThresholdPercent
2323 <<
" -> " << NewVGPRThresholdPercent
2324 <<
", VGPRExcessLimit=" <<
S.VGPRExcessLimit
2325 <<
", VGPRCriticalLimit=" <<
S.VGPRCriticalLimit
2326 <<
", InstantRP=" << InstantRP <<
", LIRP=" << LIRP);
2328 bool DoRescheduling =
false;
2330 unsigned InstantRPLowerBound =
2332 if (LIRP >
S.VGPRExcessLimit) {
2333 LLVM_DEBUG(
dbgs() <<
" [LIRP exceeds the limit (" <<
S.VGPRExcessLimit
2334 <<
"), rescheduling]");
2335 DoRescheduling =
true;
2336 }
else if (LIRP > InstantRP && InstantRP > InstantRPLowerBound) {
2337 unsigned IncreasePercent = ((LIRP - InstantRP) * 100) / InstantRP;
2341 DoRescheduling =
true;
2347 SavedVGPRExcessLimit =
S.VGPRExcessLimit;
2348 SavedVGPRCriticalLimit =
S.VGPRCriticalLimit;
2349 SavedVGPRThresholdPercent =
S.VGPRThresholdPercent;
2350 S.VGPRThresholdPercent = NewVGPRThresholdPercent;
2358 S.VGPRExcessLimit = SavedVGPRExcessLimit;
2359 S.VGPRCriticalLimit = SavedVGPRCriticalLimit;
2360 S.VGPRThresholdPercent = SavedVGPRThresholdPercent;
2367 LLVM_DEBUG(
dbgs() <<
"New pressure will result in more spilling.\n");
2379 "instruction number mismatch");
2380 if (MIOrder.
empty())
2393 if (MII != RegionEnd) {
2395 bool NonDebugReordered =
2396 !
MI->isDebugInstr() &&
2402 if (NonDebugReordered)
2403 DAG.LIS->handleMove(*
MI,
true);
2410 if (!
MI->isDebugInstr()) {
2412 SlotIndex PrevIdx =
DAG.LIS->getSlotIndexes()->getIndexBefore(*
MI);
2413 if (PrevIdx >= MIIdx)
2414 DAG.LIS->handleMove(*
MI,
true);
2418 if (
MI->isDebugInstr()) {
2425 Op.setIsUndef(
false);
2428 if (
DAG.ShouldTrackLaneMasks) {
2453 if (RD->
getOpcode() == AMDGPU::AV_MOV_B32_IMM_PSEUDO ||
2454 RD->
getOpcode() == AMDGPU::AV_MOV_B64_IMM_PSEUDO)
2461bool RewriteMFMAFormStage::hasUseRequiringVGPR(
2463 const SmallPtrSetImpl<MachineInstr *> &RewriteSet) {
2464 for (SlotIndex RDIdx : Src2ReachingDefs) {
2465 const MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIdx);
2467 findReachingUses(RD,
DAG.LIS, ReachingUses);
2468 for (
const MachineOperand *UseMO : ReachingUses) {
2480void RewriteMFMAFormStage::resetRewriteCandsToVGPR(
2481 ArrayRef<std::pair<MachineInstr *, unsigned>> RewriteCands) {
2482 for (
auto [
MI, OriginalOpcode] : RewriteCands) {
2485 DAG.MRI.getRegClass(
MI->getOperand(0).getReg());
2487 DAG.MRI.setRegClass(
MI->getOperand(0).getReg(), VDefRC);
2488 MI->setDesc(
TII->get(OriginalOpcode));
2490 MachineOperand *Src2 =
TII->getNamedOperand(*
MI, AMDGPU::OpName::src2);
2499 DAG.MRI.setRegClass(Src2->
getReg(), VUseRC);
2503bool RewriteMFMAFormStage::isRewriteCandidate(MachineInstr *
MI)
const {
2504 if (!
static_cast<const SIInstrInfo *
>(
DAG.TII)->isMAI(*
MI))
2509 Register DstReg =
MI->getOperand(0).getReg();
2510 for (
const MachineInstr &
UseMI :
DAG.MRI.use_nodbg_instructions(DstReg)) {
2517bool RewriteMFMAFormStage::initHeuristics(
2518 std::vector<std::pair<MachineInstr *, unsigned>> &RewriteCands,
2519 DenseMap<MachineBasicBlock *, std::set<Register>> &CopyForUse,
2520 SmallPtrSetImpl<MachineInstr *> &CopyForDef) {
2525 SmallPtrSet<MachineInstr *, 16> RewriteSet;
2526 DenseSet<Register> CandSrc2Regs;
2527 for (MachineBasicBlock &
MBB :
MF) {
2528 for (MachineInstr &
MI :
MBB) {
2529 if (!isRewriteCandidate(&
MI))
2532 MachineOperand *Src2 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src2);
2533 if (Src2 && Src2->
isReg())
2539 for (MachineBasicBlock &
MBB :
MF) {
2540 for (MachineInstr &
MI :
MBB) {
2541 if (!isRewriteCandidate(&
MI))
2545 assert(ReplacementOp != -1);
2547 RewriteCands.push_back({&
MI,
MI.getOpcode()});
2548 MI.setDesc(
TII->get(ReplacementOp));
2550 MachineOperand *Src2 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src2);
2551 if (Src2->
isReg()) {
2553 findReachingDefs(*Src2,
DAG.LIS, Src2ReachingDefs);
2557 bool Src2NeedsVGPR = hasUseRequiringVGPR(Src2ReachingDefs, RewriteSet);
2558 Src2NeedsVGPRCache[&
MI] = Src2NeedsVGPR;
2560 for (SlotIndex RDIdx : Src2ReachingDefs) {
2561 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIdx);
2562 if (!Src2NeedsVGPR &&
2569 MachineOperand &Dst =
MI.getOperand(0);
2572 findReachingUses(&
MI,
DAG.LIS, DstReachingUses);
2574 for (MachineOperand *RUOp : DstReachingUses) {
2575 MachineInstr *UserMI = RUOp->getParent();
2577 if (
TII->isMAI(*UserMI) && RewriteSet.
contains(UserMI))
2583 CopyForUse[UserMI->
getParent()].insert(RUOp->getReg());
2585 if (
TII->isMAI(*UserMI))
2589 findReachingDefs(*RUOp,
DAG.LIS, DstUsesReachingDefs);
2591 for (SlotIndex RDIndex : DstUsesReachingDefs) {
2592 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIndex);
2593 if (
TII->isMAI(*RD))
2607 DAG.MRI.setRegClass(Dst.getReg(), ADefRC);
2608 if (Src2->
isReg()) {
2614 DAG.MRI.setRegClass(Src2->
getReg(), AUseRC);
2623int64_t RewriteMFMAFormStage::getRewriteCost(
2624 ArrayRef<std::pair<MachineInstr *, unsigned>> RewriteCands,
2625 const DenseMap<MachineBasicBlock *, std::set<Register>> &CopyForUse,
2626 const SmallPtrSetImpl<MachineInstr *> &CopyForDef) {
2627 MachineBlockFrequencyInfo *MBFI =
DAG.MBFI;
2629 int64_t BestSpillCost = 0;
2633 std::pair<unsigned, unsigned> MaxVectorRegs =
2634 ST.getMaxNumVectorRegs(
MF.getFunction());
2635 unsigned ArchVGPRThreshold = MaxVectorRegs.first;
2636 unsigned AGPRThreshold = MaxVectorRegs.second;
2637 unsigned CombinedThreshold =
ST.getMaxNumVGPRs(
MF);
2640 if (!RegionsWithExcessArchVGPR[Region])
2645 MF, ArchVGPRThreshold, AGPRThreshold, CombinedThreshold);
2653 MF, ArchVGPRThreshold, AGPRThreshold, CombinedThreshold);
2659 bool RelativeFreqIsDenom = EntryFreq > BlockFreq;
2660 uint64_t RelativeFreq = EntryFreq && BlockFreq
2661 ? (RelativeFreqIsDenom ? EntryFreq / BlockFreq
2662 : BlockFreq / EntryFreq)
2667 int64_t SpillCost = ((int)SpillCostAfter - (int)SpillCostBefore) * 2;
2670 if (RelativeFreqIsDenom)
2671 SpillCost /= (int64_t)RelativeFreq;
2673 SpillCost *= (int64_t)RelativeFreq;
2676 if (SpillCost > 0) {
2677 resetRewriteCandsToVGPR(RewriteCands);
2681 if (SpillCost < BestSpillCost)
2682 BestSpillCost = SpillCost;
2687 Cost = BestSpillCost;
2690 unsigned CopyCost = 0;
2694 for (MachineInstr *
DefMI : CopyForDef) {
2706 for (
auto &[UseBlock, UseRegs] : CopyForUse) {
2720 resetRewriteCandsToVGPR(RewriteCands);
2722 return Cost + CopyCost;
2725bool RewriteMFMAFormStage::rewrite(
2726 ArrayRef<std::pair<MachineInstr *, unsigned>> RewriteCands) {
2727 DenseMap<MachineInstr *, unsigned> FirstMIToRegion;
2728 DenseMap<MachineInstr *, unsigned> LastMIToRegion;
2736 if (
Entry.second !=
Entry.first->getParent()->end())
2779 DenseSet<Register> RewriteRegs;
2782 DenseMap<Register, Register> RedefMap;
2784 DenseMap<Register, DenseSet<MachineOperand *>>
ReplaceMap;
2786 DenseMap<Register, SmallPtrSet<MachineInstr *, 8>> ReachingDefCopyMap;
2789 DenseMap<unsigned, DenseMap<Register, SmallPtrSet<MachineOperand *, 8>>>
2794 SmallPtrSet<MachineInstr *, 16> RewriteCandsSet;
2795 DenseSet<Register> RewriteSrc2Regs;
2796 for (
auto &[
MI, OriginalOpcode] : RewriteCands) {
2798 MachineOperand *Src2 =
TII->getNamedOperand(*
MI, AMDGPU::OpName::src2);
2799 if (Src2 && Src2->
isReg())
2803 for (
auto &[
MI, OriginalOpcode] : RewriteCands) {
2805 if (ReplacementOp == -1)
2807 MI->setDesc(
TII->get(ReplacementOp));
2810 MachineOperand *Src2 =
TII->getNamedOperand(*
MI, AMDGPU::OpName::src2);
2811 if (Src2->
isReg()) {
2818 findReachingDefs(*Src2,
DAG.LIS, Src2ReachingDefs);
2819 SmallSetVector<MachineInstr *, 8> Src2DefsReplace;
2823 bool Src2NeedsVGPR = Src2NeedsVGPRCache.lookup(
MI);
2825 for (SlotIndex RDIndex : Src2ReachingDefs) {
2826 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIndex);
2827 if (!Src2NeedsVGPR &&
2831 Src2DefsReplace.
insert(RD);
2834 if (!Src2DefsReplace.
empty()) {
2835 auto RI = RedefMap.
find(Src2Reg);
2836 if (RI != RedefMap.
end()) {
2837 MappedReg = RI->second;
2842 SRI->getEquivalentVGPRClass(Src2RC);
2845 MappedReg =
DAG.MRI.createVirtualRegister(VGPRRC);
2846 RedefMap[Src2Reg] = MappedReg;
2851 for (MachineInstr *RD : Src2DefsReplace) {
2853 if (ReachingDefCopyMap[Src2Reg].insert(RD).second) {
2854 MachineInstrBuilder VGPRCopy =
2857 .
addDef(MappedReg, {}, 0)
2858 .addUse(Src2Reg, {}, 0);
2859 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
2864 unsigned UpdateRegion = LastMIToRegion[RD];
2865 DAG.Regions[UpdateRegion].second = VGPRCopy;
2866 LastMIToRegion.
erase(RD);
2873 RewriteRegs.
insert(Src2Reg);
2883 MachineOperand *Dst = &
MI->getOperand(0);
2892 SmallVector<MachineInstr *, 8> DstUseDefsReplace;
2894 findReachingUses(
MI,
DAG.LIS, DstReachingUses);
2896 for (MachineOperand *RUOp : DstReachingUses) {
2897 MachineInstr *UserMI = RUOp->
getParent();
2899 if (
TII->isMAI(*UserMI) && RewriteCandsSet.
contains(UserMI))
2903 if (
find(DstReachingUseCopies, RUOp) == DstReachingUseCopies.
end())
2907 if (
TII->isMAI(*UserMI))
2911 findReachingDefs(*RUOp,
DAG.LIS, DstUsesReachingDefs);
2913 for (SlotIndex RDIndex : DstUsesReachingDefs) {
2914 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIndex);
2915 if (
TII->isMAI(*RD))
2920 if (
find(DstUseDefsReplace, RD) == DstUseDefsReplace.
end())
2925 if (!DstUseDefsReplace.
empty()) {
2926 auto RI = RedefMap.
find(DstReg);
2927 if (RI != RedefMap.
end()) {
2928 MappedReg = RI->second;
2935 MappedReg =
DAG.MRI.createVirtualRegister(VGPRRC);
2936 RedefMap[DstReg] = MappedReg;
2941 for (MachineInstr *RD : DstUseDefsReplace) {
2943 if (ReachingDefCopyMap[DstReg].insert(RD).second) {
2944 MachineInstrBuilder VGPRCopy =
2947 .
addDef(MappedReg, {}, 0)
2948 .addUse(DstReg, {}, 0);
2949 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
2953 auto LMI = LastMIToRegion.
find(RD);
2954 if (LMI != LastMIToRegion.
end()) {
2955 unsigned UpdateRegion = LMI->second;
2956 DAG.Regions[UpdateRegion].second = VGPRCopy;
2957 LastMIToRegion.
erase(RD);
2963 DenseSet<MachineOperand *> &DstRegSet =
ReplaceMap[DstReg];
2966 MachineInstr *EarliestSameBlockUse =
nullptr;
2967 for (MachineOperand *RU : DstReachingUseCopies) {
2968 MachineBasicBlock *RUBlock = RU->getParent()->getParent();
2971 if (RUBlock !=
MI->getParent()) {
2977 if (!SameBlockCopyReg.
isValid()) {
2980 SameBlockCopyReg =
DAG.MRI.createVirtualRegister(VGPRRC);
2984 MachineInstr *UseInst = RU->getParent();
2985 if (!EarliestSameBlockUse ||
2987 DAG.LIS->getInstructionIndex(*UseInst),
2988 DAG.LIS->getInstructionIndex(*EarliestSameBlockUse)))
2989 EarliestSameBlockUse = UseInst;
2990 RU->setReg(SameBlockCopyReg);
2994 if (SameBlockCopyReg.
isValid()) {
2995 MachineInstrBuilder VGPRCopy =
2998 TII->get(TargetOpcode::COPY), SameBlockCopyReg)
3000 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
3005 RewriteRegs.
insert(DstReg);
3015 std::pair<unsigned, DenseMap<Register, SmallPtrSet<MachineOperand *, 8>>>;
3016 for (RUBType RUBlockEntry : ReachingUseTracker) {
3017 using RUDType = std::pair<Register, SmallPtrSet<MachineOperand *, 8>>;
3018 for (RUDType RUDst : RUBlockEntry.second) {
3019 MachineOperand *OpBegin = *RUDst.second.begin();
3020 SlotIndex InstPt =
DAG.LIS->getInstructionIndex(*OpBegin->
getParent());
3023 for (MachineOperand *User : RUDst.second) {
3024 SlotIndex NewInstPt =
DAG.LIS->getInstructionIndex(*
User->getParent());
3031 Register NewUseReg =
DAG.MRI.createVirtualRegister(VGPRRC);
3032 MachineInstr *UseInst =
DAG.LIS->getInstructionFromIndex(InstPt);
3034 MachineInstrBuilder VGPRCopy =
3037 .
addDef(NewUseReg, {}, 0)
3038 .addUse(RUDst.first, {}, 0);
3039 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
3043 auto FI = FirstMIToRegion.
find(UseInst);
3044 if (FI != FirstMIToRegion.
end()) {
3045 unsigned UpdateRegion = FI->second;
3046 DAG.Regions[UpdateRegion].first = VGPRCopy;
3047 FirstMIToRegion.
erase(UseInst);
3051 for (MachineOperand *User : RUDst.second) {
3052 User->setReg(NewUseReg);
3063 for (std::pair<Register, Register> NewDef : RedefMap) {
3068 for (MachineOperand *ReplaceOp :
ReplaceMap[OldReg])
3069 ReplaceOp->setReg(NewReg);
3073 for (
Register RewriteReg : RewriteRegs) {
3074 Register RegToRewrite = RewriteReg;
3077 auto RI = RedefMap.find(RewriteReg);
3078 if (RI != RedefMap.end())
3079 RegToRewrite = RI->second;
3084 DAG.MRI.setRegClass(RegToRewrite, AGPRRC);
3088 DAG.LIS->reanalyze(
DAG.MF);
3090 RegionPressureMap LiveInUpdater(&
DAG,
false);
3091 LiveInUpdater.buildLiveRegMap();
3094 DAG.LiveIns[Region] = LiveInUpdater.getLiveRegsForRegionIdx(Region);
3101unsigned PreRARematStage::getStageTargetOccupancy()
const {
3102 return TargetOcc ? *TargetOcc :
MFI.getMinWavesPerEU();
3105bool PreRARematStage::setObjective() {
3109 unsigned MaxSGPRs =
ST.getMaxNumSGPRs(
F);
3110 unsigned MaxVGPRs =
ST.getMaxNumVGPRs(
F);
3111 bool HasVectorRegisterExcess =
false;
3112 for (
unsigned I = 0,
E =
DAG.Regions.size();
I !=
E; ++
I) {
3113 const GCNRegPressure &
RP =
DAG.Pressure[
I];
3114 GCNRPTarget &
Target = RPTargets.emplace_back(MaxSGPRs, MaxVGPRs,
MF, RP);
3116 TargetRegions.set(
I);
3117 HasVectorRegisterExcess |=
Target.hasVectorRegisterExcess();
3120 if (HasVectorRegisterExcess ||
DAG.MinOccupancy >=
MFI.getMaxWavesPerEU()) {
3123 TargetOcc = std::nullopt;
3127 TargetOcc =
DAG.MinOccupancy + 1;
3128 const unsigned VGPRBlockSize =
MFI.getDynamicVGPRBlockSize();
3129 MaxSGPRs =
ST.getMaxNumSGPRs(*TargetOcc,
false);
3130 MaxVGPRs =
ST.getMaxNumVGPRs(*TargetOcc, VGPRBlockSize);
3131 for (
auto [
I, Target] :
enumerate(RPTargets)) {
3132 Target.setTarget(MaxSGPRs, MaxVGPRs);
3134 TargetRegions.set(
I);
3138 return TargetRegions.any();
3141bool PreRARematStage::ScoredRemat::maybeBeneficial(
3143 for (
unsigned I : TargetRegions.set_bits()) {
3144 if (Live[
I] && RPTargets[
I].isSaveBeneficial(RPSave))
3157 const unsigned NumRegions =
DAG.Regions.size();
3161 for (
unsigned I = 0;
I < NumRegions; ++
I) {
3165 if (BlockFreq && BlockFreq <
MinFreq)
3174 if (
MinFreq >= ScaleFactor * ScaleFactor) {
3175 for (uint64_t &Freq :
Regions)
3176 Freq /= ScaleFactor;
3182void PreRARematStage::ScoredRemat::init(
const FreqInfo &Freq,
3187 assert(Reg.Uses.size() == 1 &&
"expected users in single region");
3188 const unsigned UseRegion = Reg.Uses.begin()->first;
3193 for (
unsigned I : Live.set_bits()) {
3196 if (!LiveIn[
I] || !LiveOut[
I] ||
I == UseRegion)
3197 UnpredictableRPSave.set(
I);
3204 int64_t DefOrMin = std::max(Freq.
Regions[Reg.DefRegion], Freq.
MinFreq);
3205 int64_t UseOrMax = Freq.
Regions[UseRegion];
3208 FreqDiff = DefOrMin - UseOrMax;
3211void PreRARematStage::ScoredRemat::update(
const BitVector &TargetRegions,
3213 const FreqInfo &FreqInfo,
3217 for (
unsigned I : TargetRegions.
set_bits()) {
3226 if (!NumRegsBenefit)
3230 RegionImpact += (UnpredictableRPSave[
I] ? 1 : 2) * NumRegsBenefit;
3233 uint64_t Freq = FreqInfo.
Regions[
I];
3234 if (UnpredictableRPSave[
I]) {
3239 MaxFreq = std::max(MaxFreq, Freq);
3244void PreRARematStage::ScoredRemat::rematerialize(
3245 Rematerializer &Remater)
const {
3246 const Rematerializer::Reg &
Reg = Remater.getReg(RegIdx);
3247 Rematerializer::DependencyReuseInfo DRI;
3248 for (RegisterIdx DepRegIdx :
Reg.Dependencies)
3249 DRI.
reuse(DepRegIdx);
3250 unsigned UseRegion =
Reg.Uses.begin()->first;
3251 Remater.rematerializeToRegion(RegIdx, UseRegion, DRI);
3254void PreRARematStage::updateRPTargets(
const BitVector &Regions,
3255 const GCNRegPressure &RPSave) {
3257 RPTargets[
I].saveRP(RPSave);
3258 if (TargetRegions[
I] && RPTargets[
I].satisfied()) {
3260 TargetRegions.reset(
I);
3265bool PreRARematStage::updateAndVerifyRPTargets(
const BitVector &Regions) {
3266 bool TooOptimistic =
false;
3268 GCNRPTarget &
Target = RPTargets[
I];
3274 if (!TargetRegions[
I] && !
Target.satisfied()) {
3276 TooOptimistic =
true;
3277 TargetRegions.set(
I);
3280 return TooOptimistic;
3283void PreRARematStage::removeFromLiveMaps(
Register Reg,
const BitVector &LiveIn,
3284 const BitVector &LiveOut) {
3286 LiveOut.
size() ==
DAG.Regions.size() &&
"region num mismatch");
3290 DAG.RegionLiveOuts.getLiveRegsForRegionIdx(
I).erase(
Reg);
3293void PreRARematStage::addToLiveMaps(
Register Reg, LaneBitmask Mask,
3294 const BitVector &LiveIn,
3295 const BitVector &LiveOut) {
3297 LiveOut.
size() ==
DAG.Regions.size() &&
"region num mismatch");
3298 std::pair<Register, LaneBitmask> LiveReg(
Reg, Mask);
3300 DAG.LiveIns[
I].insert(LiveReg);
3302 DAG.RegionLiveOuts.getLiveRegsForRegionIdx(
I).insert(LiveReg);
3314 if (
DAG.MinOccupancy >= *TargetOcc)
3318 for (
const auto &[
RegionIdx, OrigMIOrder, MaxPressure] : RegionReverts) {
3328 if (AchievedOcc >= *TargetOcc) {
3329 DAG.setTargetOccupancy(AchievedOcc);
3334 DAG.setTargetOccupancy(*TargetOcc - 1);
3339 assert(Rollback &&
"rollbacker should be defined");
3340 Rollback->Listener.rollback(Remater);
3341 for (
const auto &[RegIdx, LiveIn, LiveOut] : Rollback->LiveMapUpdates) {
3342 const Rematerializer::Reg &
Reg = Remater.getReg(RegIdx);
3343 addToLiveMaps(
Reg.getDefReg(),
Reg.Mask, LiveIn, LiveOut);
3346#ifdef EXPENSIVE_CHECKS
3351 for (
unsigned I : RescheduleRegions.set_bits())
3352 DAG.Pressure[
I] =
DAG.getRealRegPressure(
I);
3357void GCNScheduleDAGMILive::setTargetOccupancy(
unsigned TargetOccupancy) {
3358 MinOccupancy = TargetOccupancy;
3359 if (
MFI.getOccupancy() < TargetOccupancy)
3360 MFI.increaseOccupancy(
MF, MinOccupancy);
3362 MFI.limitOccupancy(MinOccupancy);
3379 if (HasIGLPInstrs) {
3380 SavedMutations.clear();
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static SUnit * pickOnlyChoice(SchedBoundary &Zone)
This file implements the BitVector class.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file defines the GCNRegPressure class, which tracks registry pressure by bookkeeping number of S...
static cl::opt< bool > GCNTrackers("amdgpu-use-amdgpu-trackers", cl::Hidden, cl::desc("Use the AMDGPU specific RPTrackers during scheduling"), cl::init(false))
static cl::opt< bool > DisableClusteredLowOccupancy("amdgpu-disable-clustered-low-occupancy-reschedule", cl::Hidden, cl::desc("Disable clustered low occupancy " "rescheduling for ILP scheduling stage."), cl::init(false))
#define REMAT_PREFIX
Allows to easily filter for this stage's debug output.
static MachineInstr * getLastMIForRegion(MachineBasicBlock::iterator RegionBegin, MachineBasicBlock::iterator RegionEnd)
static bool shouldCheckPending(SchedBoundary &Zone, const TargetSchedModel *SchedModel)
static cl::opt< bool > EnableLiveIntervalRPReschedule("amdgpu-lirp-reschedule", cl::Hidden, cl::desc("Enable live interval RP reschedule stage"), cl::init(true))
static cl::opt< bool > RelaxedOcc("amdgpu-schedule-relaxed-occupancy", cl::Hidden, cl::desc("Relax occupancy targets for kernels which are memory " "bound (amdgpu-membound-threshold), or " "Wave Limited (amdgpu-limit-wave-threshold)."), cl::init(false))
static cl::opt< bool > DisableUnclusterHighRP("amdgpu-disable-unclustered-high-rp-reschedule", cl::Hidden, cl::desc("Disable unclustered high register pressure " "reduction scheduling stage."), cl::init(false))
static void printScheduleModel(std::set< std::pair< MachineInstr *, unsigned >, EarlierIssuingCycle > &ReadyCycles)
static bool isReachingDefAGPRForm(MachineInstr *RD, const SmallPtrSetImpl< MachineInstr * > &RewriteSet, const DenseSet< Register > &CandSrc2Regs, const SIInstrInfo &TII)
Returns true if reaching def RD will be in AGPR form after the rewrite and so needs no bridge copy: a...
static cl::opt< bool > PrintMaxRPRegUsageAfterScheduler("amdgpu-print-max-reg-pressure-regusage-after-scheduler", cl::Hidden, cl::desc("Print a list of live registers along with their def/uses at the " "point of maximum register pressure after scheduling."), cl::init(false))
static bool hasIGLPInstrs(ScheduleDAGInstrs *DAG)
static cl::opt< bool > DisableRewriteMFMAFormSchedStage("amdgpu-disable-rewrite-mfma-form-sched-stage", cl::Hidden, cl::desc("Disable rewrite mfma rewrite scheduling stage"), cl::init(true))
static bool canUsePressureDiffs(const SUnit &SU)
Checks whether SU can use the cached DAG pressure diffs to compute the current register pressure.
static cl::opt< unsigned > LiveIntervalRPVGPRReduction("amdgpu-lirp-vgpr-reduction", cl::Hidden, cl::desc("Reduction factor (percent) for VGPR threshold during live interval RP " "reschedule stage"), cl::init(90))
static cl::opt< unsigned > PendingQueueLimit("amdgpu-scheduler-pending-queue-limit", cl::Hidden, cl::desc("Max (Available+Pending) size to inspect pending queue (0 disables)"), cl::init(256))
static cl::opt< bool > PrintMaxRPRegUsageBeforeScheduler("amdgpu-print-max-reg-pressure-regusage-before-scheduler", cl::Hidden, cl::desc("Print a list of live registers along with their def/uses at the " "point of maximum register pressure before scheduling."), cl::init(false))
static cl::opt< unsigned > LiveIntervalRPInstantLowerBound("amdgpu-lirp-instant-lower-bound", cl::Hidden, cl::desc("Lower bound (percent of the VGPR excess limit) on instant RP, " "below which a region is skipped"), cl::init(10))
static cl::opt< unsigned > ScheduleMetricBias("amdgpu-schedule-metric-bias", cl::Hidden, cl::desc("Sets the bias which adds weight to occupancy vs latency. Set it to " "100 to chase the occupancy only."), cl::init(10))
static cl::opt< unsigned > LiveIntervalRPThreshold("amdgpu-lirp-threshold", cl::Hidden, cl::desc("Percent increase of live interval RP over instant pressure to " "trigger rescheduling"), cl::init(10))
static Register UseReg(const MachineOperand &MO)
const HexagonInstrInfo * TII
static constexpr std::pair< StringLiteral, StringLiteral > ReplaceMap[]
iv Induction Variable Users
A common definition of LaneBitmask for use in TableGen and CodeGen.
Promote Memory to Register
MIR-level target-independent rematerialization helpers.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & front() const
Get the first element.
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
iterator_range< const_set_bits_iterator > set_bits() const
size_type size() const
Returns the number of bits in this bitvector.
uint64_t getFrequency() const
Returns the frequency as a fixpoint number scaled by the entry frequency.
bool initGCNSchedStage() override
bool shouldRevertScheduling(unsigned WavesAfter) override
bool initGCNRegion() override
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
iterator find(const_arg_type_t< KeyT > Val)
bool erase(const KeyT &Val)
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Implements a dense probed hash-table based set.
bool reset(const MachineInstr &MI, MachineBasicBlock::const_iterator End, const LiveRegSet *LiveRegs=nullptr)
Reset tracker to the point before the MI filling LiveRegs upon this point using LIS.
GCNRegPressure bumpDownwardPressure(const MachineInstr *MI, const SIRegisterInfo *TRI) const
Mostly copy/paste from CodeGen/RegisterPressure.cpp Calculate the impact MI will have on CurPressure ...
GCNMaxILPSchedStrategy(const MachineSchedContext *C)
bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const override
Apply a set of heuristics to a new candidate.
bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const override
GCNMaxMemoryClauseSchedStrategy tries best to clause memory instructions as much as possible.
GCNMaxMemoryClauseSchedStrategy(const MachineSchedContext *C)
GCNMaxOccupancySchedStrategy(const MachineSchedContext *C, bool IsLegacyScheduler=false)
void finalizeSchedule() override
Allow targets to perform final scheduling actions at the level of the whole MachineFunction.
void schedule() override
Orders nodes according to selected style.
GCNPostScheduleDAGMILive(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S, bool RemoveKillFlags)
Models a register pressure target, allowing to evaluate and track register savings against that targe...
unsigned getNumRegsBenefit(const GCNRegPressure &SaveRP) const
Returns the benefit towards achieving the RP target that saving SaveRP represents,...
GCNRegPressure getPressure() const
virtual bool initGCNRegion()
GCNRegPressure PressureBefore
bool isRegionWithExcessRP() const
void modifyRegionSchedule(unsigned RegionIdx, ArrayRef< MachineInstr * > MIOrder)
Sets the schedule of region RegionIdx to MIOrder.
bool mayCauseSpilling(unsigned WavesAfter)
ScheduleMetrics getScheduleMetrics(const std::vector< SUnit > &InputSchedule)
GCNScheduleDAGMILive & DAG
const GCNSchedStageID StageID
std::vector< MachineInstr * > Unsched
GCNRegPressure PressureAfter
virtual void finalizeGCNRegion()
SIMachineFunctionInfo & MFI
unsigned computeSUnitReadyCycle(const SUnit &SU, unsigned CurrCycle, DenseMap< unsigned, unsigned > &ReadyCycles, const TargetSchedModel &SM)
virtual void finalizeGCNSchedStage()
virtual bool initGCNSchedStage()
virtual bool shouldRevertScheduling(unsigned WavesAfter)
std::vector< std::unique_ptr< ScheduleDAGMutation > > SavedMutations
GCNSchedStage(GCNSchedStageID StageID, GCNScheduleDAGMILive &DAG)
MachineBasicBlock * CurrentMBB
This is a minimal scheduler strategy.
GCNDownwardRPTracker DownwardTracker
bool useGCNTrackers() const
void getRegisterPressures(bool AtTop, const RegPressureTracker &RPTracker, SUnit *SU, std::vector< unsigned > &Pressure, std::vector< unsigned > &MaxPressure, GCNDownwardRPTracker &DownwardTracker, GCNUpwardRPTracker &UpwardTracker, ScheduleDAGMI *DAG, const SIRegisterInfo *SRI)
GCNSchedStrategy(const MachineSchedContext *C)
SmallVector< GCNSchedStageID, 4 > SchedStages
unsigned SGPRCriticalLimit
unsigned VGPRThresholdPercent
std::vector< unsigned > MaxPressure
bool hasNextStage() const
SUnit * pickNodeBidirectional(bool &IsTopNode, bool &PickedPending)
GCNSchedStageID getCurrentStage()
bool tryPendingCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const
Evaluates instructions in the pending queue using a subset of scheduling heuristics.
SmallVectorImpl< GCNSchedStageID >::iterator CurrentStage
unsigned VGPRCriticalLimit
void schedNode(SUnit *SU, bool IsTopNode) override
Notify MachineSchedStrategy that ScheduleDAGMI has scheduled an instruction and updated scheduled/rem...
std::optional< bool > GCNTrackersOverride
GCNDownwardRPTracker * getDownwardTracker()
unsigned AGPRCriticalLimit
std::vector< unsigned > Pressure
void initialize(ScheduleDAGMI *DAG) override
Initialize the strategy after building the DAG for a new region.
GCNUpwardRPTracker UpwardTracker
void printCandidateDecision(const SchedCandidate &Current, const SchedCandidate &Preferred)
void pickNodeFromQueue(SchedBoundary &Zone, const CandPolicy &ZonePolicy, const RegPressureTracker &RPTracker, SchedCandidate &Cand, bool &IsPending, bool IsBottomUp)
void initCandidate(SchedCandidate &Cand, SUnit *SU, bool AtTop, const RegPressureTracker &RPTracker, const SIRegisterInfo *SRI, unsigned SGPRPressure, unsigned VGPRPressure, unsigned AGPRPressure, bool IsBottomUp)
SUnit * pickNode(bool &IsTopNode) override
Pick the next node to schedule, or return NULL.
GCNUpwardRPTracker * getUpwardTracker()
GCNSchedStageID getNextStage() const
void finalizeSchedule() override
Allow targets to perform final scheduling actions at the level of the whole MachineFunction.
void schedule() override
Orders nodes according to selected style.
GCNScheduleDAGMILive(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S)
void recede(const MachineInstr &MI)
Move to the state of RP just before the MI .
void reset(const MachineInstr &MI)
Resets tracker to the point just after MI (in program order), which can be a debug instruction.
void compute(FunctionT &F)
Compute the cycle info for a function.
void traceCandidate(const SchedCandidate &Cand)
LLVM_ABI void setPolicy(CandPolicy &Policy, bool IsPostRA, SchedBoundary &CurrZone, SchedBoundary *OtherZone)
Set the CandPolicy given a scheduling zone given the current resources and latencies inside and outsi...
MachineSchedPolicy RegionPolicy
const TargetSchedModel * SchedModel
const MachineSchedContext * Context
const TargetRegisterInfo * TRI
SchedCandidate BotCand
Candidate last picked from Bot boundary.
SchedCandidate TopCand
Candidate last picked from Top boundary.
virtual bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const
Apply a set of heuristics to a new candidate.
void initialize(ScheduleDAGMI *dag) override
Initialize the strategy after building the DAG for a new region.
void schedNode(SUnit *SU, bool IsTopNode) override
Update the scheduler's state after scheduling a node.
GenericScheduler(const MachineSchedContext *C)
bool shouldRevertScheduling(unsigned WavesAfter) override
void resize(typename StorageT::size_type S)
void finalizeGCNRegion() override
bool initGCNRegion() override
bool initGCNSchedStage() override
LiveInterval - This class represents the liveness of a register, or stack slot.
bool hasSubRanges() const
Returns true if subregister liveness information is available.
iterator_range< subrange_iterator > subranges()
SlotIndex getInstructionIndex(const MachineInstr &Instr) const
Returns the base index of the given instruction.
SlotIndex getMBBEndIdx(const MachineBasicBlock *mbb) const
Return the last index in the given basic block.
LiveInterval & getInterval(Register Reg)
LLVM_ABI void dump() const
MachineBasicBlock * getMBBFromIndex(SlotIndex index) const
VNInfo * getVNInfoAt(SlotIndex Idx) const
getVNInfoAt - Return the VNInfo that is live at Idx, or NULL.
uint8_t getCopyCost() const
getCopyCost - Return the cost of copying a value between two registers in this class.
int getNumber() const
MachineBasicBlocks are uniquely numbered at the function level, unless they're not in a MachineFuncti...
succ_iterator succ_begin()
unsigned succ_size() const
iterator_range< pred_iterator > predecessors()
MachineInstrBundleIterator< MachineInstr > iterator
MachineBlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate machine basic b...
LLVM_ABI BlockFrequency getBlockFreq(const MachineBasicBlock *MBB) const
getblockFreq - Return block frequency.
LLVM_ABI BlockFrequency getEntryFreq() const
Divide a block's BlockFrequency::getFrequency() value by this value to obtain the entry block - relat...
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
unsigned getNumOperands() const
Retuns the total number of operands.
bool mayLoad(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read memory.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
const MachineOperand & getOperand(unsigned i) const
MachineOperand class - Representation of each machine instruction operand.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
Register getReg() const
getReg - Returns the register number.
bool shouldRevertScheduling(unsigned WavesAfter) override
bool shouldRevertScheduling(unsigned WavesAfter) override
bool shouldRevertScheduling(unsigned WavesAfter) override
void finalizeGCNRegion() override
bool initGCNRegion() override
bool initGCNSchedStage() override
Capture a change in pressure for a single pressure set.
Simple wrapper around std::function<void(raw_ostream&)>.
Helpers for implementing custom MachineSchedStrategy classes.
Track the current register pressure at some position in the instruction stream, and remember the high...
LLVM_ABI void advance()
Advance across the current instruction.
LLVM_ABI void getDownwardPressure(const MachineInstr *MI, std::vector< unsigned > &PressureResult, std::vector< unsigned > &MaxPressureResult)
Get the pressure of each PSet after traversing this instruction top-down.
const std::vector< unsigned > & getRegSetPressureAtPos() const
Get the register set pressure at the current position, which may be less than the pressure across the...
LLVM_ABI void getUpwardPressure(const MachineInstr *MI, std::vector< unsigned > &PressureResult, std::vector< unsigned > &MaxPressureResult)
Get the pressure of each PSet after traversing this instruction bottom-up.
GCNRPTracker::LiveRegSet & getLiveRegsForRegionIdx(unsigned RegionIdx)
List of registers defined and used by a machine instruction.
LLVM_ABI void detectDeadDefs(const MachineInstr &MI, const LiveIntervals &LIS, const MachineRegisterInfo &MRI)
Use liveness information to find dead defs at MI's dead slot not marked with a dead flag and move the...
LLVM_ABI void adjustLaneLiveness(const LiveIntervals &LIS, const MachineRegisterInfo &MRI, SlotIndex Pos)
Use liveness information to find out which uses/defs are partially undefined/dead at Pos and adjust t...
LLVM_ABI void collect(const MachineInstr &MI, const TargetRegisterInfo &TRI, const MachineRegisterInfo &MRI, bool TrackLaneMasks, bool IgnoreDead)
Analyze the given instruction MI and fill in the Uses, Defs and DeadDefs list based on the MachineOpe...
Wrapper class representing virtual and physical registers.
constexpr bool isValid() const
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
static constexpr bool isVirtualRegister(unsigned Reg)
Return true if the specified register number is in the virtual register namespace.
MIR-level target-independent rematerializer.
bool isIGLPMutationOnly(unsigned Opcode) const
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
unsigned getOccupancy() const
unsigned getDynamicVGPRBlockSize() const
unsigned getMinAllowedOccupancy() const
Scheduling unit. This is a node in the scheduling DAG.
bool isInstr() const
Returns true if this SUnit refers to a machine instruction as opposed to an SDNode.
unsigned TopReadyCycle
Cycle relative to start when node is ready.
unsigned NodeNum
Entry # of node in the node vector.
unsigned short Latency
Node latency.
bool isScheduled
True once scheduled.
unsigned ParentClusterIdx
The parent cluster id.
unsigned BotReadyCycle
Cycle relative to end when node is ready.
bool isBottomReady() const
SmallVector< SDep, 4 > Preds
All sunit predecessors.
MachineInstr * getInstr() const
Returns the representative MachineInstr for this SUnit.
Each Scheduling boundary is associated with ready queues.
LLVM_ABI void releasePending()
Release pending ready nodes in to the available queue.
LLVM_ABI unsigned getLatencyStallCycles(SUnit *SU)
Get the difference between the given SUnit's ready time and the current cycle.
LLVM_ABI SUnit * pickOnlyChoice()
Call this before applying any other heuristics to the Available queue.
LLVM_ABI void bumpCycle(unsigned NextCycle)
Move the boundary of scheduled code by one cycle.
unsigned getCurrMOps() const
Micro-ops issued in the current cycle.
unsigned getCurrCycle() const
Number of cycles to issue the instructions scheduled in this zone.
LLVM_ABI bool checkHazard(SUnit *SU)
Does this SU have a hazard within the current instruction group.
A ScheduleDAG for scheduling lists of MachineInstr.
bool ScheduleSingleMIRegions
True if regions with a single MI should be scheduled.
MachineBasicBlock::iterator RegionEnd
The end of the range to be scheduled.
virtual void finalizeSchedule()
Allow targets to perform final scheduling actions at the level of the whole MachineFunction.
virtual void exitRegion()
Called when the scheduler has finished scheduling the current region.
const MachineLoopInfo * MLI
bool RemoveKillFlags
True if the DAG builder should remove kill flags (in preparation for rescheduling).
MachineBasicBlock::iterator RegionBegin
The beginning of the range to be scheduled.
void schedule() override
Implement ScheduleDAGInstrs interface for scheduling a sequence of reorderable instructions.
ScheduleDAGMILive(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S)
RegPressureTracker RPTracker
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
void addMutation(std::unique_ptr< ScheduleDAGMutation > Mutation)
Add a postprocessing step to the DAG builder.
void schedule() override
Implement ScheduleDAGInstrs interface for scheduling a sequence of reorderable instructions.
ScheduleDAGMI(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S, bool RemoveKillFlags)
std::vector< std::unique_ptr< ScheduleDAGMutation > > Mutations
Ordered list of DAG postprocessing steps.
MachineRegisterInfo & MRI
Virtual/real register map.
const TargetInstrInfo * TII
Target instruction information.
MachineFunction & MF
Machine function.
static const unsigned ScaleFactor
unsigned getMetric() const
bool empty() const
Determine if the SetVector is empty or not.
bool insert(const value_type &X)
Insert a new element into the SetVector.
SlotIndex - An opaque wrapper around machine indexes.
static bool isSameInstr(SlotIndex A, SlotIndex B)
isSameInstr - Return true if A and B refer to the same instruction.
static bool isEarlierInstr(SlotIndex A, SlotIndex B)
isEarlierInstr - Return true if A refers to an instruction earlier than B.
SlotIndex getPrevSlot() const
Returns the previous slot in the index list.
SlotIndex getMBBStartIdx(const MachineBasicBlock *mbb) const
Returns the first index in the given basic block.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
SmallSet - This maintains a set of unique values, optimizing for the case when the set is small (less...
bool contains(const T &V) const
Check if the SmallSet contains the given element.
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Provide an instruction scheduling machine model to CodeGen passes.
LLVM_ABI bool hasInstrSchedModel() const
Return true if this machine model includes an instruction-level scheduling model.
unsigned getMicroOpBufferSize() const
Number of micro-ops that may be buffered for OOO execution.
bool initGCNSchedStage() override
bool initGCNRegion() override
void finalizeGCNSchedStage() override
bool shouldRevertScheduling(unsigned WavesAfter) override
VNInfo - Value Number Information.
SlotIndex def
The index of the defining instruction.
bool isPHIDef() const
Returns true if this value is defined by a PHI instruction (or was, PHI instructions may have been el...
LLVM Value Representation.
std::pair< iterator, bool > insert(const ValueT &V)
bool contains(const_arg_type_t< ValueT > V) const
Check if the set contains the given element.
self_iterator getIterator()
This class implements an extremely fast bulk output stream that can only output to a stream.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize)
unsigned getAllocatedNumVGPRBlocks(const MCSubtargetInfo &STI, unsigned NumVGPRs, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
LLVM_READONLY int32_t getAGPRFormOp(uint32_t Opcode)
initializer< Ty > init(const Ty &Val)
@ User
could "use" a pointer
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI int biasPhysReg(const SUnit *SU, bool isTop, bool BiasPRegsExtra=false)
Minimize physical register live ranges.
auto find(R &&Range, const T &Val)
Provide wrappers to std::find which take ranges instead of having to pass begin/end explicitly.
bool isEqual(const GCNRPTracker::LiveRegSet &S1, const GCNRPTracker::LiveRegSet &S2)
Printable print(const GCNRegPressure &RP, const GCNSubtarget *ST=nullptr, unsigned DynamicVGPRBlockSize=0)
LLVM_ABI unsigned getWeakLeft(const SUnit *SU, bool isTop)
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
GCNRegPressure getRegPressure(const MachineRegisterInfo &MRI, Range &&LiveRegs)
std::unique_ptr< ScheduleDAGMutation > createIGroupLPDAGMutation(AMDGPU::SchedulingPhase Phase)
Phase specifes whether or not this is a reentry into the IGroupLPDAGMutation.
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
std::pair< MachineBasicBlock::iterator, MachineBasicBlock::iterator > RegionBoundaries
A region's boundaries i.e.
IterT skipDebugInstructionsForward(IterT It, IterT End, bool SkipPseudoOp=true)
Increment It until it points to a non-debug instruction or to End and return the resulting iterator.
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI bool tryPressure(const PressureChange &TryP, const PressureChange &CandP, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason, const TargetRegisterInfo *TRI, const MachineFunction &MF)
@ UnclusteredHighRPReschedule
@ MemoryClauseInitialSchedule
@ LiveIntervalRPReschedule
@ ClusteredLowOccupancyReschedule
auto reverse(ContainerTy &&C)
void sort(IteratorTy Start, IteratorTy End)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
cl::opt< unsigned, false, VGPRThresholdParser > VGPRThresholdPercentOpt
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
LLVM_ABI cl::opt< bool > VerifyScheduling
LLVM_ABI bool tryLatency(GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, SchedBoundary &Zone)
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
IterT skipDebugInstructionsBackward(IterT It, IterT Begin, bool SkipPseudoOp=true)
Decrement It until it points to a non-debug instruction or to Begin and return the resulting iterator...
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
bool isTheSameCluster(unsigned A, unsigned B)
Return whether the input cluster ID's are the same and valid.
DWARFExpression::Operation Op
LLVM_ABI bool tryGreater(int TryVal, int CandVal, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason)
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
ArrayRef(const T &OneElt) -> ArrayRef< T >
OutputIt move(R &&Range, OutputIt Out)
Provide wrappers to std::move which take ranges instead of having to pass begin/end explicitly.
DenseMap< MachineInstr *, GCNRPTracker::LiveRegSet > getLiveRegMap(Range &&R, bool After, LiveIntervals &LIS)
creates a map MachineInstr -> LiveRegSet R - range of iterators on instructions After - upon entry or...
GCNRPTracker::LiveRegSet getLiveRegsBefore(const MachineInstr &MI, const LiveIntervals &LIS)
LLVM_ABI bool tryLess(int TryVal, int CandVal, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason)
Return true if this heuristic determines order.
LLVM_ABI void dumpMaxRegPressure(MachineFunction &MF, GCNRegPressure::RegKind Kind, LiveIntervals &LIS, const MachineLoopInfo *MLI)
unsigned estimateGreedyVGPRPressure(MachineBasicBlock::const_iterator RegionBegin, MachineBasicBlock::const_iterator RegionEnd, const GCNRPTracker::LiveRegSet &LiveIns, const LiveIntervals &LIS, const MachineRegisterInfo &MRI, const SIRegisterInfo &TRI)
Estimate VGPR pressure using greedy, non-splitting register allocation simulation,...
LLVM_ABI Printable printMBBReference(const MachineBasicBlock &MBB)
Prints a machine basic block reference.
MCRegisterClass TargetRegisterClass
Implement std::hash so that hash_code can be used in STL containers.
bool operator()(std::pair< MachineInstr *, unsigned > A, std::pair< MachineInstr *, unsigned > B) const
unsigned getArchVGPRNum() const
unsigned getAGPRNum() const
unsigned getSGPRNum() const
Policy for scheduling the next instruction in the candidate's zone.
Store the state used by GenericScheduler heuristics, required for the lifetime of one invocation of p...
void setBest(SchedCandidate &Best)
void reset(const CandPolicy &NewPolicy)
LLVM_ABI void initResourceDelta(const ScheduleDAGMI *DAG, const TargetSchedModel *SchedModel)
SchedResourceDelta ResDelta
Status of an instruction's critical resource consumption.
unsigned DemandedResources
constexpr bool any() const
static constexpr LaneBitmask getNone()
MachineSchedContext provides enough context from the MachineScheduler pass for the target to instanti...
Execution frequency information required by scoring heuristics.
SmallVector< uint64_t > Regions
Per-region execution frequencies. 0 when unknown.
uint64_t MinFreq
Minimum and maximum observed frequencies.
FreqInfo(MachineFunction &MF, const GCNScheduleDAGMILive &DAG)
PressureChange CriticalMax
PressureChange CurrentMax
DependencyReuseInfo & reuse(RegisterIdx DepIdx)
A rematerializable register, potentially defined by multiple instructions.
LLVM_ABI std::pair< MachineInstr *, MachineInstr * > getRegionUseBounds(unsigned UseRegion, const LiveIntervals &LIS) const
Returns the first and last user of the register in region UseRegion.
SmallVector< MachineInstr *, 1 > Defs
All instructions that define the register, in program order.
SmallDenseMap< unsigned, RegionUsers, 2 > Uses
Uses of the register, mapped by region.
MachineInstr * getLastDef() const
SmallVector< RegisterIdx, 2 > Dependencies
This register's rematerializable dependencies, one per unique rematerializable register operand over ...
bool parse(cl::Option &O, StringRef ArgName, StringRef Arg, unsigned &Value)