LLVM 24.0.0git
GCNHazardRecognizer.cpp
Go to the documentation of this file.
1//===-- GCNHazardRecognizers.cpp - GCN Hazard Recognizer Impls ------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements hazard recognizers for scheduling on GCN processors.
10//
11//===----------------------------------------------------------------------===//
12
13#include "GCNHazardRecognizer.h"
14#include "AMDGPUTargetMachine.h"
15#include "AMDGPUWaitcntUtils.h"
16#include "GCNSubtarget.h"
18#include "llvm/ADT/Statistic.h"
23#include "llvm/Support/Debug.h"
25
26using namespace llvm;
27
28#define DEBUG_TYPE "gcn-hazard-recognizer"
29// Opt-in debug type for the per-candidate co-execution slot traces, which are
30// far too noisy for the normal debug output. Pass both types to get everything.
31#define DEBUG_TYPE_VERBOSE "gcn-hazard-recognizer-verbose"
32
33STATISTIC(NumWMMANopsHoisted,
34 "Number of WMMA hazard V_NOPs hoisted from loops");
35STATISTIC(NumWMMAHoistingBailed,
36 "Number of WMMA hazards where V_NOP hoisting was not possible");
37
38namespace {
39
40struct MFMAPaddingRatioParser : public cl::parser<unsigned> {
41 MFMAPaddingRatioParser(cl::Option &O) : cl::parser<unsigned>(O) {}
42
43 bool parse(cl::Option &O, StringRef ArgName, StringRef Arg, unsigned &Value) {
44 if (Arg.getAsInteger(0, Value))
45 return O.error("'" + Arg + "' value invalid for uint argument!");
46
47 if (Value > 100)
48 return O.error("'" + Arg + "' value must be in the range [0, 100]!");
49
50 return false;
51 }
52};
53
54} // end anonymous namespace
55
57 MFMAPaddingRatio("amdgpu-mfma-padding-ratio", cl::init(0), cl::Hidden,
58 cl::desc("Fill a percentage of the latency between "
59 "neighboring MFMA with s_nops."));
60
61// This is intended for debugging purposes only.
63 NopPadding("amdgpu-snop-padding", cl::init(0), cl::Hidden,
64 cl::desc("Insert a s_nop x before every instruction"));
65
67 "amdgpu-wmma-vnop-hoisting", cl::init(true), cl::Hidden,
68 cl::desc("Hoist WMMA hazard V_NOPs from loops to preheaders"));
69
70//===----------------------------------------------------------------------===//
71// Hazard Recognizer Implementation
72//===----------------------------------------------------------------------===//
73
75 const GCNSubtarget &ST);
76
79 MachineLoopInfo *MLI)
80 : Mode(Mode), CurrCycleInstr(nullptr), MF(MF),
81 ST(MF.getSubtarget<GCNSubtarget>()), TII(*ST.getInstrInfo()),
82 TRI(TII.getRegisterInfo()), TSchedModel(TII.getSchedModel()), MLI(MLI),
83 ClauseUses(TRI.getNumRegUnits()), ClauseDefs(TRI.getNumRegUnits()) {
84 MaxLookAhead = MF.getRegInfo().isPhysRegUsed(AMDGPU::AGPR0) ? 19 : 5;
85 RunLdsBranchVmemWARHazardFixup = shouldRunLdsBranchVmemWARHazardFixup(MF, ST);
87 if (isPreRA())
88 dbgs() << " PreRA hazard recognizer: " << MF.getName() << "\n";
89 });
90}
91
95
97 // Dump any active co-execution window that did not complete naturally
98 // (e.g. region ended before the window expired).
100 if (CurrentCoExecStage.has_value()) {
101 unsigned Stage = *CurrentCoExecStage;
102 if (Stage < AMDGPU::MaxCoExecStages)
103 CoExecWindowLog[Stage] = ActiveCoExecInfo.Pattern[Stage];
104 dbgs() << " CoExec window ended at stage " << Stage << ":\n";
105 dumpCoExecWindow();
106 }
107 });
108}
109
111 EmittedInstrs.clear();
112 EmittedVALUInstrs.clear();
113 HasPendingWMMACoexecHazard = false;
114 if (isSchedulerMode())
115 schedulerReset();
116}
117
118void GCNHazardRecognizer::schedulerReset() {
119 LLVM_DEBUG({
120 if (CurrentCoExecStage.has_value() || CyclesUntilTRANS > 0 ||
121 CyclesUntilVALU > 0)
122 dbgs() << " Scheduler Reset: clearing co-exec window, TRANS="
123 << CyclesUntilTRANS << ", VALU=" << CyclesUntilVALU << "\n";
124 });
125 CurrentCoExecStage = std::nullopt;
126 CoExecWindowStartCycle = 0;
127 CyclesUntilTRANS = 0;
128 CyclesUntilVALU = 0;
129 ActiveCoExecInfo = AMDGPU::CoExecInfo();
130 CoExecWindowLog.fill('.');
131}
132
133void GCNHazardRecognizer::dumpCoExecWindow() const {
134 unsigned W = ActiveCoExecInfo.TotalWindow;
135 if (W == 0)
136 return;
137
138 // Print the stage numbers row.
139 dbgs() << " Stages: ";
140 for (unsigned I = 0; I < W; ++I)
141 dbgs() << I % 10 << ' ';
142 dbgs() << '\n';
143
144 // Print the pattern row.
145 dbgs() << " Slots: ";
146 for (unsigned I = 0; I < W; ++I)
147 dbgs() << ActiveCoExecInfo.Pattern[I] << ' ';
148 dbgs() << '\n';
149
150 // Print the scheduled row.
151 dbgs() << " Scheduled: ";
152 for (unsigned I = 0; I < W; ++I)
153 dbgs() << CoExecWindowLog[I] << ' ';
154 dbgs() << '\n';
155}
156
157void GCNHazardRecognizer::schedulerAdvanceCycle() {
158 // Record what happened at the current stage of the co-exec window.
159 if (CurrentCoExecStage.has_value()) {
160 unsigned Stage = *CurrentCoExecStage;
161 if (Stage < AMDGPU::MaxCoExecStages) {
162 if (CurrCycleInstr)
163 CoExecWindowLog[Stage] = ActiveCoExecInfo.Pattern[Stage];
164 else
165 CoExecWindowLog[Stage] = '-';
166 }
167 }
168
169 LLVM_DEBUG({
170 bool HasState = CurrentCoExecStage.has_value() || CyclesUntilTRANS > 0 ||
171 CyclesUntilVALU > 0;
172 if (HasState) {
173 dbgs() << " Scheduler AdvanceCycle:";
174 if (CurrentCoExecStage.has_value()) {
175 unsigned Stage = *CurrentCoExecStage;
176 unsigned Next = Stage + 1;
177 if (Next >= ActiveCoExecInfo.TotalWindow)
178 dbgs() << " stage " << Stage << "->expired";
179 else
180 dbgs() << " stage " << Stage << "->" << Next;
181 }
182 if (CyclesUntilTRANS > 0)
183 dbgs() << " TRANS=" << CyclesUntilTRANS << "->"
184 << (CyclesUntilTRANS - 1);
185 if (CyclesUntilVALU > 0)
186 dbgs() << " VALU=" << CyclesUntilVALU << "->" << (CyclesUntilVALU - 1);
187 dbgs() << "\n";
188 }
189 });
190
191 // Decrement hazard counters.
192 if (CyclesUntilTRANS > 0)
193 --CyclesUntilTRANS;
194 if (CyclesUntilVALU > 0)
195 --CyclesUntilVALU;
196
197 // Advance WMMA co-execution window.
198 if (CurrentCoExecStage.has_value()) {
199 unsigned Stage = *CurrentCoExecStage + 1;
200 if (Stage >= ActiveCoExecInfo.TotalWindow) {
201 // Window expired.
202 LLVM_DEBUG({
203 dbgs() << " CoExec window complete:\n";
204 dumpCoExecWindow();
205 });
206 CurrentCoExecStage = std::nullopt;
207 } else {
208 CurrentCoExecStage = Stage;
209 }
210 }
211}
212
213bool GCNHazardRecognizer::hasCoExecWindowModel() const {
214 // The co-execution slot patterns returned by getCoExecInfo() are derived from
215 // gfx1250 timings, so the window model is restricted to gfx1250 for now.
216 // gfx1251 and gfx12.5-generic report the same co-execution hazard features
217 // but have different WMMA latencies, so they need their own slot patterns
218 // before they can be modeled here.
219 if (ST.hasWMMACoexecutionHazards() && ST.hasTransCoexecutionHazard() &&
221 return true;
222
223 if (ST.hasGFX950Insts() &&
224 AMDGPU::getSchedStrategy(MF.getFunction()) == "coexec")
225 return true;
226
227 return false;
228}
229
230void GCNHazardRecognizer::updateWMMAWindowState(const MachineInstr &MI) {
231 if (!hasCoExecWindowModel())
232 return;
233
236 return;
237
238 // If a previous window was still active, dump it before starting a new one.
239 // Record the current stage (filled by this new WMMA) before dumping.
240 LLVM_DEBUG({
241 if (CurrentCoExecStage.has_value()) {
242 unsigned Stage = *CurrentCoExecStage;
243 if (Stage < AMDGPU::MaxCoExecStages)
244 CoExecWindowLog[Stage] = ActiveCoExecInfo.Pattern[Stage];
245 dbgs() << " CoExec window interrupted at stage " << Stage << ":\n";
246 dumpCoExecWindow();
247 }
248 });
249
250 // Start a new co-execution window.
251 ActiveCoExecInfo = AMDGPU::getCoExecInfo(MI, TII);
252 CurrentCoExecStage = 0;
253 CoExecWindowLog.fill('.');
254
255 LLVM_DEBUG(dbgs() << " WMMA window started: " << ActiveCoExecInfo.Pattern
256 << " (window=" << ActiveCoExecInfo.TotalWindow << ")\n"
257 << " " << MI);
258}
259
260void GCNHazardRecognizer::updateTRANSState(const MachineInstr &MI) {
261 if (!hasCoExecWindowModel())
262 return;
264 return;
265
266 // Back-to-back TRANS instructions have a 1-cycle hazard.
267 // This is checked via checkTRANSHazard() and does not create a co-exec
268 // window. The TRANS shadow slot allows anything except TRANS and
269 // multi-cycle VALU.
270 // Set to 2: bumpCycle advances to the next pick's cycle (decrementing
271 // by 1 via AdvanceCycle) before the next instruction's hazard check, so
272 // the counter is observed at 1 there. That 1-cycle stall lets the
273 // strategy pick a non-TRANS, non-multi-cycle-VALU candidate to fill the
274 // shadow slot.
275 CyclesUntilTRANS = 2;
276 LLVM_DEBUG(dbgs() << " TRANS hazard set: CyclesUntilTRANS=2\n");
277}
278
279void GCNHazardRecognizer::updateMultiCycleVALUState(const MachineInstr &MI) {
280 if (!hasCoExecWindowModel())
281 return;
282 // Multi-cycle VALU (CVT, etc.) blocks subsequent VALU for repeat rate cycles.
283 if (!SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true))
284 return;
285
286 // Skip WMMA, MFMA, and TRANS - they have their own tracking.
289 return;
290
291 unsigned RepeatRate = TII.getRepeatRate(MI);
292 if (RepeatRate > 1) {
293 // bumpCycle's AdvanceCycle decrements once before the next pick's
294 // hazard check (same convention as CyclesUntilTRANS), so to expose
295 // RepeatRate-1 cycles of shadow we must seed with RepeatRate.
296 CyclesUntilVALU = RepeatRate;
297 LLVM_DEBUG(dbgs() << " Multi-cycle VALU: repeat=" << RepeatRate
298 << ", CyclesUntilVALU=" << CyclesUntilVALU << "\n");
299 }
300}
301
307
308unsigned GCNHazardRecognizer::checkTRANSHazard(const MachineInstr &MI) const {
309 if (!CyclesUntilTRANS)
310 return 0;
311
312 // Only TRANS and multi-cycle VALU are blocked by the TRANS shadow.
314 return CyclesUntilTRANS;
315
316 if (SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true) &&
318 TII.getRepeatRate(MI) > 1)
319 return CyclesUntilTRANS;
320
321 return 0;
322}
323
324unsigned
325GCNHazardRecognizer::checkMultiCycleVALUHazard(const MachineInstr &MI) const {
326 if (!CyclesUntilVALU)
327 return 0;
328
329 // Multi-cycle VALU blocks anything on the VALU pipe - VALU, WMMA, SWMMAC,
330 // and TRANS - for RepeatRate-1 cycles. Only off-pipe instructions (MEM,
331 // SALU, control) can fill the shadow.
332 if (!SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true) &&
335 return 0;
336
337 return CyclesUntilVALU;
338}
339
340unsigned
341GCNHazardRecognizer::checkWMMACoexecSlot(const MachineInstr &MI) const {
342 // No hazard if not in a WMMA window.
343 if (!CurrentCoExecStage.has_value())
344 return 0;
345
346 unsigned Stage = *CurrentCoExecStage;
348 unsigned StallCycles = ActiveCoExecInfo.getStallCycles(InstMask, Stage);
349
350 // No stall required if the instruction can co-execute at the current stage.
351 if (StallCycles == 0)
352 return 0;
353
354 // Stall for the required number of cycles until the next allowed stage.
355 unsigned NextStage = Stage + StallCycles;
356 if (NextStage < ActiveCoExecInfo.TotalWindow) {
359 dbgs() << " CoExec stall: stage=" << Stage << "("
360 << AMDGPU::getStageTypeName(ActiveCoExecInfo.getType(Stage))
361 << ") mask=" << AMDGPU::getCoExecMaskName(InstMask)
362 << " -> stall " << StallCycles << " (next allowed=" << NextStage
363 << ")\n"
364 << " " << MI);
365 return StallCycles;
366 }
367
368 // No compatible slot in window - stall until window ends.
371 dbgs() << " CoExec stall: stage=" << Stage << "("
372 << AMDGPU::getStageTypeName(ActiveCoExecInfo.getType(Stage))
373 << ") mask=" << AMDGPU::getCoExecMaskName(InstMask) << " -> stall "
374 << StallCycles << " (window ends)\n"
375 << " " << MI);
376 return StallCycles;
377}
378
379unsigned
380GCNHazardRecognizer::checkMultiShadowHazard(const MachineInstr &MI) const {
381 // This models a VALU caught in both a WMMA and a TRANS shadow.
382 if (!hasCoExecWindowModel())
383 return 0;
384
385 // No hazard if not in a WMMA window.
386 if (!CurrentCoExecStage.has_value())
387 return 0;
388
389 if (!CyclesUntilTRANS)
390 return 0;
391
392 if (!SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true) ||
394 return 0;
395
396 // We have a VALU instruction that is under both a TRANS and WMMA shadow.
397 // We need to wait for at least one to clear.
398
399 unsigned LookAheadStage = *CurrentCoExecStage + CyclesUntilTRANS;
401 return CyclesUntilTRANS +
402 ActiveCoExecInfo.getStallCycles(InstMask, LookAheadStage);
403}
404
405void GCNHazardRecognizer::schedulerEmitInstruction(MachineInstr *MI) {
406 LLVM_DEBUG({
407 bool InWindow = CurrentCoExecStage.has_value();
408 bool HasActiveState =
409 InWindow || CyclesUntilTRANS > 0 || CyclesUntilVALU > 0;
410 if (HasActiveState) {
411 if (InWindow) {
412 unsigned Stage = *CurrentCoExecStage;
413 dbgs() << " Stage " << Stage << "("
414 << AMDGPU::getStageTypeName(ActiveCoExecInfo.getType(Stage))
415 << ") Emit ["
417 << "]: " << *MI;
418 } else {
419 dbgs() << " Emit ["
421 << "]: " << *MI;
422 }
423 }
424 });
426 bool HasActiveState = CurrentCoExecStage.has_value() ||
427 CyclesUntilTRANS > 0 || CyclesUntilVALU > 0;
428 if (!HasActiveState)
429 dbgs() << " Emit ["
431 << "]: " << *MI;
432 });
433 updateWMMAWindowState(*MI);
434 updateTRANSState(*MI);
435 updateMultiCycleVALUState(*MI);
436}
437
441
443 CurrCycleInstr = MI;
444 if (isSchedulerMode())
445 schedulerEmitInstruction(MI);
446}
447
448static bool isDivFMas(unsigned Opcode) {
449 return Opcode == AMDGPU::V_DIV_FMAS_F32_e64 || Opcode == AMDGPU::V_DIV_FMAS_F64_e64;
450}
451
452static bool isSGetReg(unsigned Opcode) {
453 return Opcode == AMDGPU::S_GETREG_B32 || Opcode == AMDGPU::S_GETREG_B32_const;
454}
455
456static bool isSSetReg(unsigned Opcode) {
457 switch (Opcode) {
458 case AMDGPU::S_SETREG_B32:
459 case AMDGPU::S_SETREG_B32_mode:
460 case AMDGPU::S_SETREG_IMM32_B32:
461 case AMDGPU::S_SETREG_IMM32_B32_mode:
462 return true;
463 }
464 return false;
465}
466
467static bool isRWLane(unsigned Opcode) {
468 return Opcode == AMDGPU::V_READLANE_B32 || Opcode == AMDGPU::V_WRITELANE_B32;
469}
470
471static bool isRFE(unsigned Opcode) {
472 return Opcode == AMDGPU::S_RFE_B64;
473}
474
475static bool isSMovRel(unsigned Opcode) {
476 switch (Opcode) {
477 case AMDGPU::S_MOVRELS_B32:
478 case AMDGPU::S_MOVRELS_B64:
479 case AMDGPU::S_MOVRELD_B32:
480 case AMDGPU::S_MOVRELD_B64:
481 return true;
482 default:
483 return false;
484 }
485}
486
488 const MachineInstr &MI) {
489 if (TII.isAlwaysGDS(MI.getOpcode()))
490 return true;
491
492 switch (MI.getOpcode()) {
493 case AMDGPU::S_SENDMSG:
494 case AMDGPU::S_SENDMSGHALT:
495 case AMDGPU::S_TTRACEDATA:
496 return true;
497 // These DS opcodes don't support GDS.
498 case AMDGPU::DS_NOP:
499 case AMDGPU::DS_PERMUTE_B32:
500 case AMDGPU::DS_BPERMUTE_B32:
501 return false;
502 default:
503 if (TII.isDS(MI.getOpcode())) {
504 int GDS = AMDGPU::getNamedOperandIdx(MI.getOpcode(),
505 AMDGPU::OpName::gds);
506 if (MI.getOperand(GDS).getImm())
507 return true;
508 }
509 return false;
510 }
511}
512
513static bool isPermlane(const MachineInstr &MI) {
514 unsigned Opcode = MI.getOpcode();
515 return Opcode == AMDGPU::V_PERMLANE16_B32_e64 ||
516 Opcode == AMDGPU::V_PERMLANE64_B32 ||
517 Opcode == AMDGPU::V_PERMLANEX16_B32_e64 ||
518 Opcode == AMDGPU::V_PERMLANE16_VAR_B32_e64 ||
519 Opcode == AMDGPU::V_PERMLANEX16_VAR_B32_e64 ||
520 Opcode == AMDGPU::V_PERMLANE16_SWAP_B32_e32 ||
521 Opcode == AMDGPU::V_PERMLANE16_SWAP_B32_e64 ||
522 Opcode == AMDGPU::V_PERMLANE32_SWAP_B32_e32 ||
523 Opcode == AMDGPU::V_PERMLANE32_SWAP_B32_e64 ||
524 Opcode == AMDGPU::V_PERMLANE_BCAST_B32_e64 ||
525 Opcode == AMDGPU::V_PERMLANE_UP_B32_e64 ||
526 Opcode == AMDGPU::V_PERMLANE_DOWN_B32_e64 ||
527 Opcode == AMDGPU::V_PERMLANE_XOR_B32_e64 ||
528 Opcode == AMDGPU::V_PERMLANE_IDX_GEN_B32_e64;
529}
530
531static bool isLdsDma(const MachineInstr &MI) {
533}
534
535static unsigned getHWReg(const SIInstrInfo *TII, const MachineInstr &RegInstr) {
536 const MachineOperand *RegOp = TII->getNamedOperand(RegInstr,
537 AMDGPU::OpName::simm16);
538 return std::get<0>(AMDGPU::Hwreg::HwregEncoding::decode(RegOp->getImm()));
539}
540
543 MachineInstr *MI = SU->getInstr();
544 // If we are not in "HazardRecognizerMode" and therefore not being run from
545 // the scheduler, track possible stalls from hazards but don't insert noops.
547
548 if (MI->isBundle())
549 return NoHazard;
550
551 // Check co-execution slot hazards and pipeline stalls in scheduler modes.
552 if (isSchedulerMode()) {
553 if (checkMultiShadowHazard(*MI) > 0)
554 return Hazard;
555 if (checkWMMACoexecSlot(*MI) > 0)
556 return Hazard;
557 if (checkTRANSHazard(*MI) > 0)
558 return Hazard;
559 if (checkMultiCycleVALUHazard(*MI) > 0)
560 return Hazard;
561 // The remaining checks are all defined by register dependences.
562 if (!hasPhysRegs())
563 return NoHazard;
564 }
565
566 if (SIInstrInfo::isSMRD(*MI) && checkSMRDHazards(MI) > 0)
567 return HazardType;
568
569 if (ST.hasNSAtoVMEMBug() && checkNSAtoVMEMHazard(MI) > 0)
570 return HazardType;
571
572 if (checkFPAtomicToDenormModeHazard(MI) > 0)
573 return HazardType;
574
575 // Hazards which cannot be mitigated with S_NOPs.
576 if (!isHazardRecognizerMode()) {
577 if (checkWMMACoexecutionHazards(MI) > 0) {
578 HasPendingWMMACoexecHazard = true;
579 return Hazard;
580 }
581 }
582
583 if (ST.hasNoDataDepHazard())
584 return NoHazard;
585
586 if (SIInstrInfo::isVMEM(*MI) && checkVMEMHazards(MI) > 0)
587 return HazardType;
588
589 if (SIInstrInfo::isVALU(*MI, /*AllowLDSDMA=*/true) &&
590 checkVALUHazards(MI) > 0)
591 return HazardType;
592
593 if (SIInstrInfo::isDPP(*MI) && checkDPPHazards(MI) > 0)
594 return HazardType;
595
596 if (isDivFMas(MI->getOpcode()) && checkDivFMasHazards(MI) > 0)
597 return HazardType;
598
599 if (isRWLane(MI->getOpcode()) && checkRWLaneHazards(MI) > 0)
600 return HazardType;
601
602 if ((SIInstrInfo::isVALU(*MI, /*AllowLDSDMA=*/true) ||
605 checkMAIVALUHazards(MI) > 0)
606 return HazardType;
607
608 if (isSGetReg(MI->getOpcode()) && checkGetRegHazards(MI) > 0)
609 return HazardType;
610
611 if (isSSetReg(MI->getOpcode()) && checkSetRegHazards(MI) > 0)
612 return HazardType;
613
614 if (isRFE(MI->getOpcode()) && checkRFEHazards(MI) > 0)
615 return HazardType;
616
617 if (((ST.hasReadM0MovRelInterpHazard() &&
618 (TII.isVINTRP(*MI) || isSMovRel(MI->getOpcode()) ||
619 MI->getOpcode() == AMDGPU::DS_WRITE_ADDTID_B32 ||
620 MI->getOpcode() == AMDGPU::DS_READ_ADDTID_B32)) ||
621 (ST.hasReadM0SendMsgHazard() && isSendMsgTraceDataOrGDS(TII, *MI)) ||
622 (ST.hasReadM0LdsDmaHazard() && isLdsDma(*MI)) ||
623 (ST.hasReadM0LdsDirectHazard() &&
624 MI->readsRegister(AMDGPU::LDS_DIRECT, /*TRI=*/nullptr))) &&
625 checkReadM0Hazards(MI) > 0)
626 return HazardType;
627
628 if (SIInstrInfo::isMAI(*MI) && checkMAIHazards(MI) > 0)
629 return HazardType;
630
632 checkMAILdStHazards(MI) > 0)
633 return HazardType;
634
635 if (MI->isInlineAsm() && checkInlineAsmHazards(MI) > 0)
636 return HazardType;
637
638 return NoHazard;
639}
640
642 unsigned Quantity) {
643 while (Quantity > 0) {
644 unsigned Arg = std::min(Quantity, 8u);
645 Quantity -= Arg;
646 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(), TII.get(AMDGPU::S_NOP))
647 .addImm(Arg - 1);
648 }
649}
650
651unsigned
652GCNHazardRecognizer::getMFMAPipelineWaitStates(const MachineInstr &MI) const {
653 const MCSchedClassDesc *SC = TSchedModel.resolveSchedClass(&MI);
654 assert(TSchedModel.getWriteProcResBegin(SC) !=
655 TSchedModel.getWriteProcResEnd(SC));
656 return TSchedModel.getWriteProcResBegin(SC)->ReleaseAtCycle;
657}
658
659void GCNHazardRecognizer::processBundle() {
660 MachineBasicBlock::instr_iterator MI = std::next(CurrCycleInstr->getIterator());
661 MachineBasicBlock::instr_iterator E = CurrCycleInstr->getParent()->instr_end();
662 // Check bundled MachineInstr's for hazards.
663 for (; MI != E && MI->isInsideBundle(); ++MI) {
664 CurrCycleInstr = &*MI;
665 unsigned WaitStates = PreEmitNoopsCommon(CurrCycleInstr);
666
668 fixHazards(CurrCycleInstr);
669
670 insertNoopsInBundle(CurrCycleInstr, TII, WaitStates);
671 }
672
673 // It’s unnecessary to track more than MaxLookAhead instructions. Since we
674 // include the bundled MI directly after, only add a maximum of
675 // (MaxLookAhead - 1) noops to EmittedInstrs.
676 for (unsigned i = 0, e = std::min(WaitStates, MaxLookAhead - 1); i < e; ++i)
677 EmittedInstrs.push_front(nullptr);
678
679 EmittedInstrs.push_front(CurrCycleInstr);
680 EmittedInstrs.resize(MaxLookAhead);
681 }
682 CurrCycleInstr = nullptr;
683}
684
685void GCNHazardRecognizer::runOnInstruction(MachineInstr *MI) {
687
688 unsigned NumPreNoops = PreEmitNoops(MI);
689 EmitNoops(NumPreNoops);
690 if (MI->isInsideBundle())
691 insertNoopsInBundle(MI, TII, NumPreNoops);
692 else
693 TII.insertNoops(*MI->getParent(), MachineBasicBlock::iterator(MI),
694 NumPreNoops);
696 AdvanceCycle();
697}
698
701 CurrCycleInstr = MI;
702 unsigned W = PreEmitNoopsCommon(MI);
703 fixHazards(MI);
704 CurrCycleInstr = nullptr;
705 return std::max(W, NopPadding.getValue());
706}
707
709 unsigned W = 0;
710
711 // Check co-execution slot hazards and pipeline stalls in scheduler modes.
712 if (isSchedulerMode()) {
713 W = checkWMMACoexecSlot(*MI);
714 W = std::max(W, checkTRANSHazard(*MI));
715 W = std::max(W, checkMultiCycleVALUHazard(*MI));
716 W = std::max(W, checkMultiShadowHazard(*MI));
717 // The remaining checks are all defined by register dependences.
718 if (!hasPhysRegs())
719 return W;
720 }
721
722 return std::max(W, PreEmitNoopsCommon(MI));
723}
724
726 if (MI->isBundle())
727 return 0;
728
729 int WaitStates = 0;
730
732 return std::max(WaitStates, checkSMRDHazards(MI));
733
734 if (ST.hasNSAtoVMEMBug())
735 WaitStates = std::max(WaitStates, checkNSAtoVMEMHazard(MI));
736
737 WaitStates = std::max(WaitStates, checkFPAtomicToDenormModeHazard(MI));
738
739 if (ST.hasNoDataDepHazard())
740 return WaitStates;
741
743 WaitStates = std::max(WaitStates, checkVMEMHazards(MI));
744
745 if (SIInstrInfo::isVALU(*MI, /*AllowLDSDMA=*/true))
746 WaitStates = std::max(WaitStates, checkVALUHazards(MI));
747
749 WaitStates = std::max(WaitStates, checkDPPHazards(MI));
750
751 if (isDivFMas(MI->getOpcode()))
752 WaitStates = std::max(WaitStates, checkDivFMasHazards(MI));
753
754 if (isRWLane(MI->getOpcode()))
755 WaitStates = std::max(WaitStates, checkRWLaneHazards(MI));
756
757 if ((SIInstrInfo::isVALU(*MI, /*AllowLDSDMA=*/true) ||
760 checkMAIVALUHazards(MI) > 0)
761 WaitStates = std::max(WaitStates, checkMAIVALUHazards(MI));
762
763 if (MI->isInlineAsm())
764 return std::max(WaitStates, checkInlineAsmHazards(MI));
765
766 if (isSGetReg(MI->getOpcode()))
767 return std::max(WaitStates, checkGetRegHazards(MI));
768
769 if (isSSetReg(MI->getOpcode()))
770 return std::max(WaitStates, checkSetRegHazards(MI));
771
772 if (isRFE(MI->getOpcode()))
773 return std::max(WaitStates, checkRFEHazards(MI));
774
775 if ((ST.hasReadM0MovRelInterpHazard() &&
776 (TII.isVINTRP(*MI) || isSMovRel(MI->getOpcode()) ||
777 MI->getOpcode() == AMDGPU::DS_WRITE_ADDTID_B32 ||
778 MI->getOpcode() == AMDGPU::DS_READ_ADDTID_B32)) ||
779 (ST.hasReadM0SendMsgHazard() && isSendMsgTraceDataOrGDS(TII, *MI)) ||
780 (ST.hasReadM0LdsDmaHazard() && isLdsDma(*MI)) ||
781 (ST.hasReadM0LdsDirectHazard() &&
782 MI->readsRegister(AMDGPU::LDS_DIRECT, /*TRI=*/nullptr)))
783 return std::max(WaitStates, checkReadM0Hazards(MI));
784
786 return std::max(WaitStates, checkMAIHazards(MI));
787
789 return std::max(WaitStates, checkMAILdStHazards(MI));
790
791 if (ST.hasGFX950Insts() && isPermlane(*MI))
792 return std::max(WaitStates, checkPermlaneHazards(MI));
793
794 return WaitStates;
795}
796
798 EmittedInstrs.push_front(nullptr);
799}
800
802 if (isSchedulerMode())
803 schedulerAdvanceCycle();
804
805 // When the scheduler detects a stall, it will call AdvanceCycle() without
806 // emitting any instructions.
807 if (!CurrCycleInstr) {
808 EmittedInstrs.push_front(nullptr);
809
810 if (HasPendingWMMACoexecHazard)
811 EmittedVALUInstrs.push_front(nullptr);
812 return;
813 }
814
815 HasPendingWMMACoexecHazard = false;
816
817 if (CurrCycleInstr->isBundle()) {
818 processBundle();
819 return;
820 }
821
822 unsigned NumWaitStates = TII.getNumWaitStates(*CurrCycleInstr);
823 if (!NumWaitStates) {
824 CurrCycleInstr = nullptr;
825 return;
826 }
827
828 // Keep track of emitted instructions
829 EmittedInstrs.push_front(CurrCycleInstr);
830
831 bool IsVALUOrWMMA =
832 SIInstrInfo::isVALU(*CurrCycleInstr, /*AllowLDSDMA=*/true) ||
833 SIInstrInfo::isWMMA(*CurrCycleInstr) ||
834 SIInstrInfo::isSWMMAC(*CurrCycleInstr);
835 if (IsVALUOrWMMA) {
836 EmittedVALUInstrs.push_front(CurrCycleInstr);
837 } else {
838 // A pending WMMA co-execution hazard optimistically records stall cycles as
839 // future V_NOPs. If the scheduler instead stalls for a different
840 // (S_NOP-resolvable) hazard and schedules a non-VALU into those cycles,
841 // they will not resolve the VALU-pipe hazard, so drop them here.
842 while (!EmittedVALUInstrs.empty() && EmittedVALUInstrs.front() == nullptr)
843 EmittedVALUInstrs.pop_front();
844 }
845
846 // Add a nullptr for each additional wait state after the first. Make sure
847 // not to add more than getMaxLookAhead() items to the list, since we
848 // truncate the list to that size right after this loop.
849 for (unsigned i = 1, e = std::min(NumWaitStates, getMaxLookAhead());
850 i < e; ++i) {
851 EmittedInstrs.push_front(nullptr);
852 }
853
854 // getMaxLookahead() is the largest number of wait states we will ever need
855 // to insert, so there is no point in keeping track of more than that many
856 // wait states.
857 EmittedInstrs.resize(getMaxLookAhead());
858 if (EmittedVALUInstrs.size() > MaxVALULookAhead)
859 EmittedVALUInstrs.resize(MaxVALULookAhead);
860
861 CurrCycleInstr = nullptr;
862}
863
866 "Bottom-up scheduling shouldn't run in hazard recognizer mode");
867}
868
869//===----------------------------------------------------------------------===//
870// Helper Functions
871//===----------------------------------------------------------------------===//
872
874
875// Search for a hazard in a block and its predecessors.
876template <typename StateT>
877static bool
878hasHazard(StateT InitialState,
879 function_ref<HazardFnResult(StateT &, const MachineInstr &)> IsHazard,
880 function_ref<void(StateT &, const MachineInstr &)> UpdateState,
881 const MachineBasicBlock *InitialMBB,
883 struct StateMapKey {
885 unsigned Idx;
886 static bool isEqual(const StateMapKey &LHS, const StateMapKey &RHS) {
887 return LHS.States == RHS.States && LHS.Idx == RHS.Idx;
888 }
889 };
890 struct StateMapKeyTraits : DenseMapInfo<StateMapKey> {
891 static unsigned getHashValue(const StateMapKey &Key) {
892 return StateT::getHashValue((*Key.States)[Key.Idx]);
893 }
894 static unsigned getHashValue(const StateT &State) {
895 return StateT::getHashValue(State);
896 }
897 static bool isEqual(const StateMapKey &LHS, const StateMapKey &RHS) {
898 return StateT::isEqual((*LHS.States)[LHS.Idx], (*RHS.States)[RHS.Idx]);
899 }
900 static bool isEqual(const StateT &LHS, const StateMapKey &RHS) {
901 return StateT::isEqual(LHS, (*RHS.States)[RHS.Idx]);
902 }
903 };
904
907
909 const MachineBasicBlock *MBB = InitialMBB;
910 StateT State = InitialState;
911
913 unsigned WorkIdx = 0;
914 for (;;) {
915 bool Expired = false;
916 for (auto E = MBB->instr_rend(); I != E; ++I) {
917 // No need to look at parent BUNDLE instructions.
918 if (I->isBundle())
919 continue;
920
921 auto Result = IsHazard(State, *I);
922 if (Result == HazardFound)
923 return true;
924 if (Result == HazardExpired) {
925 Expired = true;
926 break;
927 }
928
929 if (I->isInlineAsm() || I->isMetaInstruction())
930 continue;
931
932 UpdateState(State, *I);
933 }
934
935 if (!Expired) {
936 unsigned StateIdx = States.size();
937 StateMapKey Key = {&States, StateIdx};
938 auto Insertion = StateMap.insert_as(std::pair(Key, StateIdx), State);
939 if (Insertion.second) {
940 States.emplace_back(State);
941 } else {
942 StateIdx = Insertion.first->second;
943 }
944 for (MachineBasicBlock *Pred : MBB->predecessors())
945 Worklist.insert(std::pair(Pred, StateIdx));
946 }
947
948 if (WorkIdx == Worklist.size())
949 break;
950
951 unsigned StateIdx;
952 std::tie(MBB, StateIdx) = Worklist[WorkIdx++];
953 State = States[StateIdx];
954 I = MBB->instr_rbegin();
955 }
956
957 return false;
958}
959
960// Returns a minimum wait states since \p I walking all predecessors.
961// Only scans until \p IsExpired does not return true.
962// Can only be run in a hazard recognizer mode.
963static int
965 const MachineBasicBlock *MBB,
967 int WaitStates, GCNHazardRecognizer::IsExpiredFn IsExpired,
971 for (auto E = MBB->instr_rend(); I != E; ++I) {
972 // Don't add WaitStates for parent BUNDLE instructions.
973 if (I->isBundle())
974 continue;
975
976 if (IsHazard(*I))
977 return WaitStates;
978
979 if (I->isInlineAsm())
980 continue;
981
982 WaitStates += GetNumWaitStates(*I);
983
984 if (IsExpired(*I, WaitStates))
985 return std::numeric_limits<int>::max();
986 }
987
988 int MinWaitStates = std::numeric_limits<int>::max();
989 for (MachineBasicBlock *Pred : MBB->predecessors()) {
990 if (!Visited.insert(Pred).second)
991 continue;
992
993 int W = getWaitStatesSince(IsHazard, Pred, Pred->instr_rbegin(), WaitStates,
994 IsExpired, Visited, GetNumWaitStates);
995
996 MinWaitStates = std::min(MinWaitStates, W);
997 }
998
999 return MinWaitStates;
1000}
1001
1002static int
1004 const MachineInstr *MI,
1009 return getWaitStatesSince(IsHazard, MI->getParent(),
1010 std::next(MI->getReverseIterator()), 0, IsExpired,
1011 Visited, GetNumWaitStates);
1012}
1013
1014int GCNHazardRecognizer::getWaitStatesSince(
1015 IsHazardFn IsHazard, int Limit, GetNumWaitStatesFn GetNumWaitStates) const {
1016 if (isHazardRecognizerMode()) {
1017 auto IsExpiredFn = [Limit](const MachineInstr &, int WaitStates) {
1018 return WaitStates >= Limit;
1019 };
1020 return ::getWaitStatesSince(IsHazard, CurrCycleInstr, IsExpiredFn,
1021 GetNumWaitStates);
1022 }
1023
1024 int WaitStates = 0;
1025 for (MachineInstr *MI : EmittedInstrs) {
1026 if (MI) {
1027 if (IsHazard(*MI))
1028 return WaitStates;
1029
1030 if (MI->isInlineAsm())
1031 continue;
1032 }
1033 WaitStates += MI ? GetNumWaitStates(*MI) : 1;
1034
1035 if (WaitStates >= Limit)
1036 break;
1037 }
1038 return std::numeric_limits<int>::max();
1039}
1040
1041int GCNHazardRecognizer::getWaitStatesSince(IsHazardFn IsHazard,
1042 int Limit) const {
1043 return getWaitStatesSince(IsHazard, Limit, SIInstrInfo::getNumWaitStates);
1044}
1045
1046int GCNHazardRecognizer::getWaitStatesSinceVALU(IsHazardFn IsHazard,
1047 int Limit) const {
1048 if (isHazardRecognizerMode()) {
1049 auto GetVALUWaitStates = [](const MachineInstr &MI) -> unsigned {
1050 return SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true) ? 1 : 0;
1051 };
1052 return getWaitStatesSince(IsHazard, Limit, GetVALUWaitStates);
1053 }
1054
1055 // EmittedVALUInstrs is capped at MaxVALULookAhead, so a Limit beyond that
1056 // window could miss a hazard. Keep the cap in sync with the wait-state
1057 // tables.
1058 assert(Limit <= (int)MaxVALULookAhead &&
1059 "Limit exceeds the EmittedVALUInstrs lookahead window");
1060 int WaitStates = 0;
1061 for (MachineInstr *MI : EmittedVALUInstrs) {
1062 if (MI) {
1063 if (IsHazard(*MI))
1064 return WaitStates;
1065 }
1066
1067 ++WaitStates;
1068
1069 if (WaitStates >= Limit)
1070 break;
1071 }
1072 return std::numeric_limits<int>::max();
1073}
1074
1075int GCNHazardRecognizer::getWaitStatesSinceDef(unsigned Reg,
1076 IsHazardFn IsHazardDef,
1077 int Limit) const {
1078 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1079
1080 auto IsHazardFn = [IsHazardDef, TRI, Reg](const MachineInstr &MI) {
1081 return IsHazardDef(MI) && MI.modifiesRegister(Reg, TRI);
1082 };
1083
1084 return getWaitStatesSince(IsHazardFn, Limit);
1085}
1086
1087int GCNHazardRecognizer::getWaitStatesSinceSetReg(IsHazardFn IsHazard,
1088 int Limit) const {
1089 auto IsHazardFn = [IsHazard](const MachineInstr &MI) {
1090 return isSSetReg(MI.getOpcode()) && IsHazard(MI);
1091 };
1092
1093 return getWaitStatesSince(IsHazardFn, Limit);
1094}
1095
1096//===----------------------------------------------------------------------===//
1097// No-op Hazard Detection
1098//===----------------------------------------------------------------------===//
1099
1100static void addRegUnits(const SIRegisterInfo &TRI, BitVector &BV,
1101 MCRegister Reg) {
1102 for (MCRegUnit Unit : TRI.regunits(Reg))
1103 BV.set(static_cast<unsigned>(Unit));
1104}
1105
1108 BitVector &DefSet, BitVector &UseSet) {
1109 for (const MachineOperand &Op : Ops) {
1110 if (Op.isReg())
1111 addRegUnits(TRI, Op.isDef() ? DefSet : UseSet, Op.getReg().asMCReg());
1112 }
1113}
1114
1115void GCNHazardRecognizer::addClauseInst(const MachineInstr &MI) const {
1116 addRegsToSet(TRI, MI.operands(), ClauseDefs, ClauseUses);
1117}
1118
1120 return !SIInstrInfo::isSMRD(*MI);
1121}
1122
1124 return !SIInstrInfo::isVMEM(*MI);
1125}
1126
1127int GCNHazardRecognizer::checkSoftClauseHazards(MachineInstr *MEM) const {
1128 // SMEM soft clause are only present on VI+, and only matter if xnack is
1129 // enabled.
1130 if (!ST.isXNACKEnabled())
1131 return 0;
1132
1133 bool IsSMRD = TII.isSMRD(*MEM);
1134
1135 resetClause();
1136
1137 // A soft-clause is any group of consecutive SMEM instructions. The
1138 // instructions in this group may return out of order and/or may be
1139 // replayed (i.e. the same instruction issued more than once).
1140 //
1141 // In order to handle these situations correctly we need to make sure that
1142 // when a clause has more than one instruction, no instruction in the clause
1143 // writes to a register that is read by another instruction in the clause
1144 // (including itself). If we encounter this situation, we need to break the
1145 // clause by inserting a non SMEM instruction.
1146
1147 for (MachineInstr *MI : EmittedInstrs) {
1148 // When we hit a non-SMEM instruction then we have passed the start of the
1149 // clause and we can stop.
1150 if (!MI)
1151 break;
1152
1154 break;
1155
1156 addClauseInst(*MI);
1157 }
1158
1159 if (ClauseDefs.none())
1160 return 0;
1161
1162 // We need to make sure not to put loads and stores in the same clause if they
1163 // use the same address. For now, just start a new clause whenever we see a
1164 // store.
1165 if (MEM->mayStore())
1166 return 1;
1167
1168 addClauseInst(*MEM);
1169
1170 // If the set of defs and uses intersect then we cannot add this instruction
1171 // to the clause, so we have a hazard.
1172 return ClauseDefs.anyCommon(ClauseUses) ? 1 : 0;
1173}
1174
1175int GCNHazardRecognizer::checkSMRDHazards(MachineInstr *SMRD) const {
1176 int WaitStatesNeeded = 0;
1177
1178 WaitStatesNeeded = checkSoftClauseHazards(SMRD);
1179
1180 // This SMRD hazard only affects SI.
1181 if (!ST.hasSMRDReadVALUDefHazard())
1182 return WaitStatesNeeded;
1183
1184 // A read of an SGPR by SMRD instruction requires 4 wait states when the
1185 // SGPR was written by a VALU instruction.
1186 int SmrdSgprWaitStates = 4;
1187 auto IsHazardDefFn = [this](const MachineInstr &MI) {
1188 return TII.isVALU(MI, /*AllowLDSDMA=*/true);
1189 };
1190 auto IsBufferHazardDefFn = [this](const MachineInstr &MI) {
1191 return TII.isSALU(MI);
1192 };
1193
1194 bool IsBufferSMRD = TII.isBufferSMRD(*SMRD);
1195
1196 for (const MachineOperand &Use : SMRD->uses()) {
1197 if (!Use.isReg())
1198 continue;
1199 int WaitStatesNeededForUse =
1200 SmrdSgprWaitStates - getWaitStatesSinceDef(Use.getReg(), IsHazardDefFn,
1201 SmrdSgprWaitStates);
1202 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
1203
1204 // This fixes what appears to be undocumented hardware behavior in SI where
1205 // s_mov writing a descriptor and s_buffer_load_dword reading the descriptor
1206 // needs some number of nops in between. We don't know how many we need, but
1207 // let's use 4. This wasn't discovered before probably because the only
1208 // case when this happens is when we expand a 64-bit pointer into a full
1209 // descriptor and use s_buffer_load_dword instead of s_load_dword, which was
1210 // probably never encountered in the closed-source land.
1211 if (IsBufferSMRD) {
1212 int WaitStatesNeededForUse =
1213 SmrdSgprWaitStates - getWaitStatesSinceDef(Use.getReg(),
1214 IsBufferHazardDefFn,
1215 SmrdSgprWaitStates);
1216 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
1217 }
1218 }
1219
1220 return WaitStatesNeeded;
1221}
1222
1223int GCNHazardRecognizer::checkVMEMHazards(MachineInstr *VMEM) const {
1224 if (!ST.hasVMEMReadSGPRVALUDefHazard())
1225 return 0;
1226
1227 int WaitStatesNeeded = checkSoftClauseHazards(VMEM);
1228
1229 // A read of an SGPR by a VMEM instruction requires 5 wait states when the
1230 // SGPR was written by a VALU Instruction.
1231 const int VmemSgprWaitStates = 5;
1232 auto IsHazardDefFn = [this](const MachineInstr &MI) {
1233 return TII.isVALU(MI, /*AllowLDSDMA=*/true);
1234 };
1235 for (const MachineOperand &Use : VMEM->uses()) {
1236 if (!Use.isReg() || TRI.isVectorRegister(MF.getRegInfo(), Use.getReg()))
1237 continue;
1238
1239 int WaitStatesNeededForUse =
1240 VmemSgprWaitStates - getWaitStatesSinceDef(Use.getReg(), IsHazardDefFn,
1241 VmemSgprWaitStates);
1242 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
1243 }
1244 return WaitStatesNeeded;
1245}
1246
1247int GCNHazardRecognizer::checkDPPHazards(MachineInstr *DPP) const {
1248 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1249 const SIInstrInfo *TII = ST.getInstrInfo();
1250
1251 // Check for DPP VGPR read after VALU VGPR write and EXEC write.
1252 int DppVgprWaitStates = 2;
1253 int DppExecWaitStates = 5;
1254 int WaitStatesNeeded = 0;
1255 auto IsHazardDefFn = [TII](const MachineInstr &MI) {
1256 return TII->isVALU(MI, /*AllowLDSDMA=*/true);
1257 };
1258
1259 for (const MachineOperand &Use : DPP->uses()) {
1260 if (!Use.isReg() || !TRI->isVGPR(MF.getRegInfo(), Use.getReg()))
1261 continue;
1262 int WaitStatesNeededForUse =
1263 DppVgprWaitStates - getWaitStatesSinceDef(
1264 Use.getReg(),
1265 [](const MachineInstr &) { return true; },
1266 DppVgprWaitStates);
1267 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
1268 }
1269
1270 WaitStatesNeeded = std::max(
1271 WaitStatesNeeded,
1272 DppExecWaitStates - getWaitStatesSinceDef(AMDGPU::EXEC, IsHazardDefFn,
1273 DppExecWaitStates));
1274
1275 return WaitStatesNeeded;
1276}
1277
1278int GCNHazardRecognizer::checkDivFMasHazards(MachineInstr *DivFMas) const {
1279 const SIInstrInfo *TII = ST.getInstrInfo();
1280
1281 // v_div_fmas requires 4 wait states after a write to vcc from a VALU
1282 // instruction.
1283 const int DivFMasWaitStates = 4;
1284 auto IsHazardDefFn = [TII](const MachineInstr &MI) {
1285 return TII->isVALU(MI, /*AllowLDSDMA=*/true);
1286 };
1287 int WaitStatesNeeded = getWaitStatesSinceDef(AMDGPU::VCC, IsHazardDefFn,
1288 DivFMasWaitStates);
1289
1290 return DivFMasWaitStates - WaitStatesNeeded;
1291}
1292
1293int GCNHazardRecognizer::checkGetRegHazards(MachineInstr *GetRegInstr) const {
1294 const SIInstrInfo *TII = ST.getInstrInfo();
1295 unsigned GetRegHWReg = getHWReg(TII, *GetRegInstr);
1296
1297 const int GetRegWaitStates = 2;
1298 auto IsHazardFn = [TII, GetRegHWReg](const MachineInstr &MI) {
1299 return GetRegHWReg == getHWReg(TII, MI);
1300 };
1301 int WaitStatesNeeded = getWaitStatesSinceSetReg(IsHazardFn, GetRegWaitStates);
1302
1303 return GetRegWaitStates - WaitStatesNeeded;
1304}
1305
1306int GCNHazardRecognizer::checkSetRegHazards(MachineInstr *SetRegInstr) const {
1307 const SIInstrInfo *TII = ST.getInstrInfo();
1308 unsigned HWReg = getHWReg(TII, *SetRegInstr);
1309
1310 const int SetRegWaitStates = ST.getSetRegWaitStates();
1311 auto IsHazardFn = [TII, HWReg](const MachineInstr &MI) {
1312 return HWReg == getHWReg(TII, MI);
1313 };
1314 int WaitStatesNeeded = getWaitStatesSinceSetReg(IsHazardFn, SetRegWaitStates);
1315 return SetRegWaitStates - WaitStatesNeeded;
1316}
1317
1318int GCNHazardRecognizer::createsVALUHazard(const MachineInstr &MI) const {
1319 if (!MI.mayStore())
1320 return -1;
1321
1322 const SIInstrInfo *TII = ST.getInstrInfo();
1323 unsigned Opcode = MI.getOpcode();
1324 const MCInstrDesc &Desc = MI.getDesc();
1325
1326 int VDataIdx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vdata);
1327 int VDataRCID = -1;
1328 if (VDataIdx != -1)
1329 VDataRCID = TII->getOpRegClassID(Desc.operands()[VDataIdx]);
1330
1331 if (TII->isMUBUF(MI) || TII->isMTBUF(MI)) {
1332 // There is no hazard if the instruction does not use vector regs
1333 // (like wbinvl1)
1334 if (VDataIdx == -1)
1335 return -1;
1336 if (AMDGPU::getRegBitWidth(VDataRCID) > 64) {
1337 // When SOFFSET-dependent wide-store windows apply, the BUFFER_STORE
1338 // source-vgpr WAR hazard exists for every SOFFSET shape; the wait-state
1339 // count differs by SOFFSET and is computed in checkVALUHazardsHelper.
1340 // Otherwise the hazard only exists if soffset is not an SGPR.
1341 if (ST.hasVDecCoExecHazard())
1342 return VDataIdx;
1343 const MachineOperand *SOffset =
1344 TII->getNamedOperand(MI, AMDGPU::OpName::soffset);
1345 if (!SOffset || !SOffset->isReg())
1346 return VDataIdx;
1347 }
1348 }
1349
1350 // MIMG instructions create a hazard if they don't use a 256-bit T# and
1351 // the store size is greater than 8 bytes and they have more than two bits
1352 // of their dmask set.
1353 // All our MIMG definitions use a 256-bit T#, so we can skip checking for them.
1354 if (TII->isMIMG(MI)) {
1355 int SRsrcIdx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::srsrc);
1356 assert(SRsrcIdx != -1 && AMDGPU::getRegBitWidth(TII->getOpRegClassID(
1357 Desc.operands()[SRsrcIdx])) == 256);
1358 (void)SRsrcIdx;
1359 }
1360
1361 if (TII->isFLAT(MI)) {
1362 // There is no hazard if the instruction does not use vector regs
1363 if (VDataIdx == -1)
1364 return -1;
1365
1366 if (AMDGPU::getRegBitWidth(VDataRCID) > 64)
1367 return VDataIdx;
1368 }
1369
1370 return -1;
1371}
1372
1373int GCNHazardRecognizer::checkUniformWindowVALUHazardsHelper(
1374 Register Reg) const {
1375 // Wide stores need a single wait-state bubble before a VALU that overwrites
1376 // store data. createsVALUHazard already excludes MUBUF/MTBUF stores with an
1377 // SGPR SOFFSET.
1378 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1379
1380 auto IsHazard = [&](const MachineInstr &MI) {
1381 int DataIdx = createsVALUHazard(MI);
1382 return DataIdx >= 0 &&
1383 TRI->regsOverlap(MI.getOperand(DataIdx).getReg(), Reg);
1384 };
1385
1386 return std::max(0, 1 - getWaitStatesSince(IsHazard, /*Limit=*/1));
1387}
1388
1389int GCNHazardRecognizer::checkSOFFSETWindowVALUHazardsHelper(
1390 Register Reg) const {
1391 // The required wait-state window depends on the producer's SOFFSET shape:
1392 // - MUBUF/MTBUF wide store with sgpr SOFFSET: 1 wait state.
1393 // - MUBUF/MTBUF wide store with literal/absent SOFFSET, and FLAT wide
1394 // store: 2 wait states.
1395 // The 1-cycle sgpr-SOFFSET window was measured on gfx950.
1396 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1397 const SIInstrInfo *TII = ST.getInstrInfo();
1398
1399 int WaitStatesNeeded = 0;
1400
1401 // Scan each wait-state window separately and take the max padding needed.
1402 // getWaitStatesSince supplies the minimum distance to a producer over paths.
1403 for (int Window = 1; Window <= 2; ++Window) {
1404 auto IsHazard = [&](const MachineInstr &MI) {
1405 int DataIdx = createsVALUHazard(MI);
1406 if (DataIdx < 0 ||
1407 !TRI->regsOverlap(MI.getOperand(DataIdx).getReg(), Reg))
1408 return false;
1409
1410 // Window 1 matches every hazard producer. Window 2 excludes BUF stores
1411 // with an SGPR SOFFSET, which only require a single wait state.
1412 if (Window == 1 || !TII->isBUF(MI))
1413 return true;
1414
1415 const MachineOperand *SOffset =
1416 TII->getNamedOperand(MI, AMDGPU::OpName::soffset);
1417 return !SOffset || !SOffset->isReg();
1418 };
1419 WaitStatesNeeded = std::max(WaitStatesNeeded,
1420 Window - getWaitStatesSince(IsHazard, Window));
1421 }
1422
1423 return WaitStatesNeeded;
1424}
1425
1426int GCNHazardRecognizer::checkVALUHazardsHelper(
1427 const MachineOperand &Def, const MachineRegisterInfo &MRI) const {
1428 // Helper to check for the hazard where VMEM instructions that store more
1429 // than 8 bytes can have their store data overwritten by the next
1430 // instruction.
1431 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1432
1433 if (!TRI->isVectorRegister(MRI, Def.getReg()))
1434 return 0;
1435
1436 if (ST.hasVDecCoExecHazard())
1437 return checkSOFFSETWindowVALUHazardsHelper(Def.getReg());
1438
1439 return checkUniformWindowVALUHazardsHelper(Def.getReg());
1440}
1441
1442/// Dest sel forwarding issue occurs if additional logic is needed to swizzle /
1443/// pack the computed value into correct bit position of the dest register. This
1444/// occurs if we have SDWA with dst_sel != DWORD or if we have op_sel with
1445/// dst_sel that is not aligned to the register. This function analayzes the \p
1446/// MI and \returns an operand with dst forwarding issue, or nullptr if
1447/// none exists.
1448static const MachineOperand *
1450 if (!SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/false))
1451 return nullptr;
1452
1453 const SIInstrInfo *TII = ST.getInstrInfo();
1454
1455 unsigned Opcode = MI.getOpcode();
1456
1457 // There are three different types of instructions
1458 // which produce forwarded dest: 1. SDWA with dst_sel != DWORD, 2. VOP3
1459 // which write hi bits (e.g. op_sel[3] == 1), and 3. FP8DstSelInst
1460 // (instructions with dest byte sel, e.g. CVT_SR_BF8_F32) and
1461 // op_sel[3:2]
1462 // != 0
1463 if (SIInstrInfo::isSDWA(MI)) {
1464 // Type 1: SDWA with dst_sel != DWORD
1465 if (auto *DstSel = TII->getNamedOperand(MI, AMDGPU::OpName::dst_sel))
1466 if (DstSel->getImm() != AMDGPU::SDWA::DWORD)
1467 return TII->getNamedOperand(MI, AMDGPU::OpName::vdst);
1468 }
1469
1470 AMDGPU::FPType IsFP4OrFP8ConvOpc = AMDGPU::getFPDstSelType(Opcode);
1471 if (AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::op_sel)) {
1472 // Type 2: VOP3 which write the hi bits
1473 if (TII->getNamedImmOperand(MI, AMDGPU::OpName::src0_modifiers) &
1475 return TII->getNamedOperand(MI, AMDGPU::OpName::vdst);
1476
1477 // Type 3: FP8DstSelInst with op_sel[3:2] != 0)
1478 if (IsFP4OrFP8ConvOpc == AMDGPU::FPType::FP8 &&
1479 (TII->getNamedImmOperand(MI, AMDGPU::OpName::src2_modifiers) &
1481 return TII->getNamedOperand(MI, AMDGPU::OpName::vdst);
1482 }
1483
1484 // Special case: nop is required for all the opsel values for fp4 sr variant
1485 // cvt scale instructions
1486 if (IsFP4OrFP8ConvOpc == AMDGPU::FPType::FP4)
1487 return TII->getNamedOperand(MI, AMDGPU::OpName::vdst);
1488
1489 return nullptr;
1490}
1491
1492/// Checks whether the provided \p MI "consumes" the operand with a Dest sel
1493/// fowarding issue \p Dst . We may "consume" the Dst via a standard explicit
1494/// RAW, or through irregular ways (e.g implicit RAW, certain types of WAW)
1496 const MachineOperand *Dst,
1497 const SIRegisterInfo *TRI) {
1498 // We must consider implicit reads of the VALU. SDWA with dst_sel and
1499 // UNUSED_PRESERVE will implicitly read the result from forwarded dest,
1500 // and we must account for that hazard.
1501 // We also must account for WAW hazards. In particular, WAW with dest
1502 // preserve semantics (e.g. VOP3 with op_sel, VOP2 &&
1503 // !zeroesHigh16BitsOfDest) will read the forwarded dest for parity
1504 // check for ECC. Without accounting for this hazard, the ECC will be
1505 // wrong.
1506 // TODO: limit to RAW (including implicit reads) + problematic WAW (i.e.
1507 // complete zeroesHigh16BitsOfDest)
1508 for (auto &Operand : VALU->operands()) {
1509 if (Operand.isReg() && TRI->regsOverlap(Dst->getReg(), Operand.getReg())) {
1510 return true;
1511 }
1512 }
1513 return false;
1514}
1515
1516int GCNHazardRecognizer::checkVALUHazards(MachineInstr *VALU) const {
1517 int WaitStatesNeeded = 0;
1518
1519 if (ST.hasTransForwardingHazard() && !SIInstrInfo::isTRANS(*VALU)) {
1520 const int TransDefWaitstates = 1;
1521
1522 auto IsTransDefFn = [this, VALU](const MachineInstr &MI) {
1524 return false;
1525 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1526 const SIInstrInfo *TII = ST.getInstrInfo();
1527 Register Def = TII->getNamedOperand(MI, AMDGPU::OpName::vdst)->getReg();
1528
1529 for (const MachineOperand &Use : VALU->explicit_uses()) {
1530 if (Use.isReg() && TRI->regsOverlap(Def, Use.getReg()))
1531 return true;
1532 }
1533
1534 return false;
1535 };
1536
1537 int WaitStatesNeededForDef =
1538 TransDefWaitstates -
1539 getWaitStatesSince(IsTransDefFn, TransDefWaitstates);
1540 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForDef);
1541 }
1542
1543 if (ST.hasDstSelForwardingHazard() || ST.hasCvtScaleForwardingHazard()) {
1544 const int Shift16DefWaitstates = 1;
1545
1546 auto IsShift16BitDefFn = [this, VALU](const MachineInstr &ProducerMI) {
1547 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1548 const MachineOperand *ForwardedDst =
1549 getDstSelForwardingOperand(ProducerMI, ST);
1550 if (ForwardedDst) {
1551 return consumesDstSelForwardingOperand(VALU, ForwardedDst, TRI);
1552 }
1553
1554 if (ProducerMI.isInlineAsm()) {
1555 // Assume inline asm has dst forwarding hazard
1556 for (auto &Def : ProducerMI.all_defs()) {
1557 if (consumesDstSelForwardingOperand(VALU, &Def, TRI))
1558 return true;
1559 }
1560 }
1561
1562 return false;
1563 };
1564
1565 int WaitStatesNeededForDef =
1566 Shift16DefWaitstates -
1567 getWaitStatesSince(IsShift16BitDefFn, Shift16DefWaitstates);
1568 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForDef);
1569 }
1570
1571 if (ST.hasVDecCoExecHazard()) {
1572 const int VALUWriteSGPRVALUReadWaitstates = 2;
1573 const int VALUWriteEXECRWLane = 4;
1574 const int VALUWriteVGPRReadlaneRead = 1;
1575
1576 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1577 const MachineRegisterInfo &MRI = MF.getRegInfo();
1579 auto IsVALUDefSGPRFn = [&UseReg, TRI](const MachineInstr &MI) {
1580 if (!SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true))
1581 return false;
1582 return MI.modifiesRegister(UseReg, TRI);
1583 };
1584
1585 for (const MachineOperand &Use : VALU->explicit_uses()) {
1586 if (!Use.isReg())
1587 continue;
1588
1589 UseReg = Use.getReg();
1590 if (TRI->isSGPRReg(MRI, UseReg)) {
1591 int WaitStatesNeededForDef =
1592 VALUWriteSGPRVALUReadWaitstates -
1593 getWaitStatesSince(IsVALUDefSGPRFn,
1594 VALUWriteSGPRVALUReadWaitstates);
1595 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForDef);
1596 }
1597 }
1598
1599 if (VALU->readsRegister(AMDGPU::VCC, TRI)) {
1600 UseReg = AMDGPU::VCC;
1601 int WaitStatesNeededForDef =
1602 VALUWriteSGPRVALUReadWaitstates -
1603 getWaitStatesSince(IsVALUDefSGPRFn, VALUWriteSGPRVALUReadWaitstates);
1604 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForDef);
1605 }
1606
1607 switch (VALU->getOpcode()) {
1608 case AMDGPU::V_READLANE_B32:
1609 case AMDGPU::V_READFIRSTLANE_B32: {
1610 MachineOperand *Src = TII.getNamedOperand(*VALU, AMDGPU::OpName::src0);
1611 UseReg = Src->getReg();
1612 int WaitStatesNeededForDef =
1613 VALUWriteVGPRReadlaneRead -
1614 getWaitStatesSince(IsVALUDefSGPRFn, VALUWriteVGPRReadlaneRead);
1615 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForDef);
1616 }
1617 [[fallthrough]];
1618 case AMDGPU::V_WRITELANE_B32: {
1619 UseReg = AMDGPU::EXEC;
1620 int WaitStatesNeededForDef =
1621 VALUWriteEXECRWLane -
1622 getWaitStatesSince(IsVALUDefSGPRFn, VALUWriteEXECRWLane);
1623 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForDef);
1624 break;
1625 }
1626 default:
1627 break;
1628 }
1629 }
1630
1631 // This checks for the hazard where VMEM instructions that store more than
1632 // 8 bytes can have there store data over written by the next instruction.
1633 if (!ST.has12DWordStoreHazard())
1634 return WaitStatesNeeded;
1635
1636 const MachineRegisterInfo &MRI = MF.getRegInfo();
1637
1638 for (const MachineOperand &Def : VALU->defs()) {
1639 WaitStatesNeeded = std::max(WaitStatesNeeded, checkVALUHazardsHelper(Def, MRI));
1640 }
1641
1642 return WaitStatesNeeded;
1643}
1644
1645int GCNHazardRecognizer::checkInlineAsmHazards(MachineInstr *IA) const {
1646 // This checks for hazards associated with inline asm statements.
1647 // Since inline asms can contain just about anything, we use this
1648 // to call/leverage other check*Hazard routines. Note that
1649 // this function doesn't attempt to address all possible inline asm
1650 // hazards (good luck), but is a collection of what has been
1651 // problematic thus far.
1652
1653 // see checkVALUHazards()
1654 if (!ST.has12DWordStoreHazard() && !ST.hasDstSelForwardingHazard() &&
1655 !ST.hasCvtScaleForwardingHazard())
1656 return 0;
1657
1658 const MachineRegisterInfo &MRI = MF.getRegInfo();
1659 int WaitStatesNeeded = 0;
1660
1661 for (const MachineOperand &Op :
1663 if (Op.isReg() && Op.isDef()) {
1664 if (!TRI.isVectorRegister(MRI, Op.getReg()))
1665 continue;
1666
1667 if (ST.has12DWordStoreHazard()) {
1668 WaitStatesNeeded =
1669 std::max(WaitStatesNeeded, checkVALUHazardsHelper(Op, MRI));
1670 }
1671 }
1672 }
1673
1674 if (ST.hasDstSelForwardingHazard()) {
1675 const int Shift16DefWaitstates = 1;
1676
1677 auto IsShift16BitDefFn = [this, &IA](const MachineInstr &ProducerMI) {
1678 const MachineOperand *Dst = getDstSelForwardingOperand(ProducerMI, ST);
1679 // Assume inline asm reads the dst
1680 if (Dst)
1681 return IA->modifiesRegister(Dst->getReg(), &TRI) ||
1682 IA->readsRegister(Dst->getReg(), &TRI);
1683
1684 if (ProducerMI.isInlineAsm()) {
1685 // If MI is inline asm, assume it has dst forwarding hazard
1686 for (auto &Def : ProducerMI.all_defs()) {
1687 if (IA->modifiesRegister(Def.getReg(), &TRI) ||
1688 IA->readsRegister(Def.getReg(), &TRI)) {
1689 return true;
1690 }
1691 }
1692 }
1693
1694 return false;
1695 };
1696
1697 int WaitStatesNeededForDef =
1698 Shift16DefWaitstates -
1699 getWaitStatesSince(IsShift16BitDefFn, Shift16DefWaitstates);
1700 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForDef);
1701 }
1702
1703 return WaitStatesNeeded;
1704}
1705
1706int GCNHazardRecognizer::checkRWLaneHazards(MachineInstr *RWLane) const {
1707 const SIInstrInfo *TII = ST.getInstrInfo();
1708 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1709 const MachineRegisterInfo &MRI = MF.getRegInfo();
1710
1711 const MachineOperand *LaneSelectOp =
1712 TII->getNamedOperand(*RWLane, AMDGPU::OpName::src1);
1713
1714 if (!LaneSelectOp->isReg() || !TRI->isSGPRReg(MRI, LaneSelectOp->getReg()))
1715 return 0;
1716
1717 Register LaneSelectReg = LaneSelectOp->getReg();
1718 auto IsHazardFn = [TII](const MachineInstr &MI) {
1719 return TII->isVALU(MI, /*AllowLDSDMA=*/true);
1720 };
1721
1722 const int RWLaneWaitStates = 4;
1723 int WaitStatesSince = getWaitStatesSinceDef(LaneSelectReg, IsHazardFn,
1724 RWLaneWaitStates);
1725 return RWLaneWaitStates - WaitStatesSince;
1726}
1727
1728int GCNHazardRecognizer::checkRFEHazards(MachineInstr *RFE) const {
1729 if (!ST.hasRFEHazards())
1730 return 0;
1731
1732 const SIInstrInfo *TII = ST.getInstrInfo();
1733
1734 const int RFEWaitStates = 1;
1735
1736 auto IsHazardFn = [TII](const MachineInstr &MI) {
1737 return getHWReg(TII, MI) == AMDGPU::Hwreg::ID_TRAPSTS;
1738 };
1739 int WaitStatesNeeded = getWaitStatesSinceSetReg(IsHazardFn, RFEWaitStates);
1740 return RFEWaitStates - WaitStatesNeeded;
1741}
1742
1743int GCNHazardRecognizer::checkReadM0Hazards(MachineInstr *MI) const {
1744 const SIInstrInfo *TII = ST.getInstrInfo();
1745 const int ReadM0WaitStates = 1;
1746 auto IsHazardFn = [TII](const MachineInstr &MI) { return TII->isSALU(MI); };
1747 return ReadM0WaitStates -
1748 getWaitStatesSinceDef(AMDGPU::M0, IsHazardFn, ReadM0WaitStates);
1749}
1750
1751void GCNHazardRecognizer::emitVNops(MachineBasicBlock &MBB,
1753 int WaitStatesNeeded, bool IsHoisting) {
1754 const DebugLoc &DL = IsHoisting ? DebugLoc() : InsertPt->getDebugLoc();
1755 for (int I = 0; I < WaitStatesNeeded; ++I)
1756 BuildMI(MBB, InsertPt, DL, TII.get(AMDGPU::V_NOP_e32));
1757}
1758
1759void GCNHazardRecognizer::fixHazards(MachineInstr *MI) {
1760 fixVMEMtoScalarWriteHazards(MI);
1761 fixVcmpxPermlaneHazards(MI);
1762 fixSMEMtoVectorWriteHazards(MI);
1763 fixVcmpxExecWARHazard(MI);
1764 fixLdsBranchVmemWARHazard(MI);
1765 if (ST.hasLdsDirect()) {
1766 fixLdsDirectVALUHazard(MI);
1767 fixLdsDirectVMEMHazard(MI);
1768 }
1769 fixVALUPartialForwardingHazard(MI);
1770 fixVALUTransUseHazard(MI);
1771 fixVALUTransCoexecutionHazards(MI);
1772 fixWMMAHazards(MI); // fall-through if co-execution is enabled.
1773 fixWMMACoexecutionHazards(MI);
1774 fixShift64HighRegBug(MI);
1775 fixVALUMaskWriteHazard(MI);
1776 fixRequiredExportPriority(MI);
1777 if (ST.hasVPermPk16Hazard())
1778 fixVPermPk16Hazard(MI);
1779 if (ST.requiresWaitIdleBeforeGetReg())
1780 fixGetRegWaitIdle(MI);
1781 if (ST.hasDsAtomicAsyncBarrierArriveB64PipeBug())
1782 fixDsAtomicAsyncBarrierArriveB64(MI);
1783 if (ST.hasScratchBaseForwardingHazard())
1784 fixScratchBaseForwardingHazard(MI);
1785 if (ST.setRegModeNeedsVNOPs())
1786 fixSetRegMode(MI);
1787 if (ST.hasNeedsTDMDrain())
1788 fixTDM(MI);
1789}
1790
1792 const MachineInstr &MI) {
1793 return (TII.isVOPC(MI) ||
1794 (MI.isCompare() && (TII.isVOP3(MI) || TII.isSDWA(MI)))) &&
1795 MI.modifiesRegister(AMDGPU::EXEC, &TRI);
1796}
1797
1798bool GCNHazardRecognizer::fixVcmpxPermlaneHazards(MachineInstr *MI) {
1799 if (!ST.hasVcmpxPermlaneHazard() || !isPermlane(*MI))
1800 return false;
1801
1802 const SIInstrInfo *TII = ST.getInstrInfo();
1803 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1804 auto IsHazardFn = [TII, TRI](const MachineInstr &MI) {
1805 return isVCmpXWritesExec(*TII, *TRI, MI);
1806 };
1807
1808 auto IsExpiredFn = [](const MachineInstr &MI, int) {
1809 unsigned Opc = MI.getOpcode();
1810 return SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true) &&
1811 Opc != AMDGPU::V_NOP_e32 && Opc != AMDGPU::V_NOP_e64 &&
1812 Opc != AMDGPU::V_NOP_sdwa;
1813 };
1814
1815 if (::getWaitStatesSince(IsHazardFn, MI, IsExpiredFn) ==
1816 std::numeric_limits<int>::max())
1817 return false;
1818
1819 // V_NOP will be discarded by SQ.
1820 // Use V_MOV_B32 v?, v?. Register must be alive so use src0 of V_PERMLANE*
1821 // which is always a VGPR and available.
1822 auto *Src0 = TII->getNamedOperand(*MI, AMDGPU::OpName::src0);
1823 Register Reg = Src0->getReg();
1824 bool IsUndef = Src0->isUndef();
1825 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(),
1826 TII->get(AMDGPU::V_MOV_B32_e32))
1829
1830 return true;
1831}
1832
1833bool GCNHazardRecognizer::fixVMEMtoScalarWriteHazards(MachineInstr *MI) {
1834 if (!ST.hasVMEMtoScalarWriteHazard())
1835 return false;
1836 assert(!ST.hasExtendedWaitCounts());
1837
1839 return false;
1840
1841 if (MI->getNumDefs() == 0)
1842 return false;
1843
1844 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1845
1846 auto IsHazardFn = [TRI, MI](const MachineInstr &I) {
1848 return false;
1849
1850 for (const MachineOperand &Def : MI->defs()) {
1851 const MachineOperand *Op =
1852 I.findRegisterUseOperand(Def.getReg(), TRI, false);
1853 if (!Op)
1854 continue;
1855 return true;
1856 }
1857 return false;
1858 };
1859
1860 auto IsExpiredFn = [](const MachineInstr &MI, int) {
1861 return SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true) ||
1862 (MI.getOpcode() == AMDGPU::S_WAITCNT &&
1863 !MI.getOperand(0).getImm()) ||
1864 (MI.getOpcode() == AMDGPU::S_WAITCNT_DEPCTR &&
1865 AMDGPU::DepCtr::decodeFieldVmVsrc(MI.getOperand(0).getImm()) == 0);
1866 };
1867
1868 if (::getWaitStatesSince(IsHazardFn, MI, IsExpiredFn) ==
1869 std::numeric_limits<int>::max())
1870 return false;
1871
1872 const SIInstrInfo *TII = ST.getInstrInfo();
1873 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(),
1874 TII->get(AMDGPU::S_WAITCNT_DEPCTR))
1876 return true;
1877}
1878
1879bool GCNHazardRecognizer::fixSMEMtoVectorWriteHazards(MachineInstr *MI) {
1880 if (!ST.hasSMEMtoVectorWriteHazard())
1881 return false;
1882 assert(!ST.hasExtendedWaitCounts());
1883
1884 if (!SIInstrInfo::isVALU(*MI, /*AllowLDSDMA=*/true))
1885 return false;
1886
1887 AMDGPU::OpName SDSTName;
1888 switch (MI->getOpcode()) {
1889 case AMDGPU::V_READLANE_B32:
1890 case AMDGPU::V_READFIRSTLANE_B32:
1891 SDSTName = AMDGPU::OpName::vdst;
1892 break;
1893 default:
1894 SDSTName = AMDGPU::OpName::sdst;
1895 break;
1896 }
1897
1898 const SIInstrInfo *TII = ST.getInstrInfo();
1899 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1900 const AMDGPU::IsaVersion IV = AMDGPU::getIsaVersion(ST.getCPU());
1901 const MachineOperand *SDST = TII->getNamedOperand(*MI, SDSTName);
1902 if (!SDST) {
1903 for (const auto &MO : MI->implicit_operands()) {
1904 if (MO.isDef() && TRI->isSGPRClass(TRI->getPhysRegBaseClass(MO.getReg()))) {
1905 SDST = &MO;
1906 break;
1907 }
1908 }
1909 }
1910
1911 if (!SDST)
1912 return false;
1913
1914 const Register SDSTReg = SDST->getReg();
1915 auto IsHazardFn = [SDSTReg, TRI](const MachineInstr &I) {
1916 return SIInstrInfo::isSMRD(I) && I.readsRegister(SDSTReg, TRI);
1917 };
1918
1919 auto IsExpiredFn = [TII, IV](const MachineInstr &MI, int) {
1920 if (TII->isSALU(MI)) {
1921 switch (MI.getOpcode()) {
1922 case AMDGPU::S_SETVSKIP:
1923 case AMDGPU::S_VERSION:
1924 case AMDGPU::S_WAITCNT_VSCNT:
1925 case AMDGPU::S_WAITCNT_VMCNT:
1926 case AMDGPU::S_WAITCNT_EXPCNT:
1927 // These instructions cannot not mitigate the hazard.
1928 return false;
1929 case AMDGPU::S_WAITCNT_LGKMCNT:
1930 // Reducing lgkmcnt count to 0 always mitigates the hazard.
1931 return (MI.getOperand(1).getImm() == 0) &&
1932 (MI.getOperand(0).getReg() == AMDGPU::SGPR_NULL);
1933 case AMDGPU::S_WAITCNT: {
1934 const int64_t Imm = MI.getOperand(0).getImm();
1935 AMDGPU::Waitcnt Decoded = AMDGPU::decodeWaitcnt(IV, Imm);
1936 // DsCnt corresponds to LGKMCnt here.
1937 return Decoded.get(AMDGPU::DS_CNT) == 0;
1938 }
1939 default:
1940 assert((!SIInstrInfo::isWaitcnt(MI.getOpcode()) ||
1941 MI.getOpcode() == AMDGPU::S_WAIT_IDLE) &&
1942 "unexpected wait count instruction");
1943 // SOPP instructions cannot mitigate the hazard.
1944 if (TII->isSOPP(MI))
1945 return false;
1946 // At this point the SALU can be assumed to mitigate the hazard
1947 // because either:
1948 // (a) it is independent of the at risk SMEM (breaking chain),
1949 // or
1950 // (b) it is dependent on the SMEM, in which case an appropriate
1951 // s_waitcnt lgkmcnt _must_ exist between it and the at risk
1952 // SMEM instruction.
1953 return true;
1954 }
1955 }
1956 return false;
1957 };
1958
1959 if (::getWaitStatesSince(IsHazardFn, MI, IsExpiredFn) ==
1960 std::numeric_limits<int>::max())
1961 return false;
1962
1963 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(),
1964 TII->get(AMDGPU::S_MOV_B32), AMDGPU::SGPR_NULL)
1965 .addImm(0);
1966 return true;
1967}
1968
1969bool GCNHazardRecognizer::fixVcmpxExecWARHazard(MachineInstr *MI) {
1970 if (!ST.hasVcmpxExecWARHazard())
1971 return false;
1972 assert(!ST.hasExtendedWaitCounts());
1973
1974 if (!SIInstrInfo::isVALU(*MI, /*AllowLDSDMA=*/true))
1975 return false;
1976
1977 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1978 if (!MI->modifiesRegister(AMDGPU::EXEC, TRI))
1979 return false;
1980
1981 auto IsHazardFn = [TRI](const MachineInstr &I) {
1982 if (SIInstrInfo::isVALU(I, /*AllowLDSDMA=*/true))
1983 return false;
1984 return I.readsRegister(AMDGPU::EXEC, TRI);
1985 };
1986
1987 const SIInstrInfo *TII = ST.getInstrInfo();
1988 auto IsExpiredFn = [TII, TRI](const MachineInstr &MI, int) {
1989 if (SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true)) {
1990 if (TII->getNamedOperand(MI, AMDGPU::OpName::sdst))
1991 return true;
1992 for (auto MO : MI.implicit_operands())
1993 if (MO.isDef() && TRI->isSGPRClass(TRI->getPhysRegBaseClass(MO.getReg())))
1994 return true;
1995 }
1996 if (MI.getOpcode() == AMDGPU::S_WAITCNT_DEPCTR &&
1997 AMDGPU::DepCtr::decodeFieldSaSdst(MI.getOperand(0).getImm()) == 0)
1998 return true;
1999 return false;
2000 };
2001
2002 if (::getWaitStatesSince(IsHazardFn, MI, IsExpiredFn) ==
2003 std::numeric_limits<int>::max())
2004 return false;
2005
2006 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(),
2007 TII->get(AMDGPU::S_WAITCNT_DEPCTR))
2009 return true;
2010}
2011
2013 const GCNSubtarget &ST) {
2014 if (!ST.hasLdsBranchVmemWARHazard())
2015 return false;
2016
2017 // Check if the necessary condition for the hazard is met: both LDS and VMEM
2018 // instructions need to appear in the same function.
2019 bool HasLds = false;
2020 bool HasVmem = false;
2021 for (auto &MBB : MF) {
2022 for (auto &MI : MBB) {
2024 HasVmem |= SIInstrInfo::isVMEM(MI);
2025 if (HasLds && HasVmem)
2026 return true;
2027 }
2028 }
2029 return false;
2030}
2031
2033 return I.getOpcode() == AMDGPU::S_WAITCNT_VSCNT &&
2034 I.getOperand(0).getReg() == AMDGPU::SGPR_NULL &&
2035 !I.getOperand(1).getImm();
2036}
2037
2038bool GCNHazardRecognizer::fixLdsBranchVmemWARHazard(MachineInstr *MI) {
2039 if (!RunLdsBranchVmemWARHazardFixup)
2040 return false;
2041
2042 assert(ST.hasLdsBranchVmemWARHazard());
2043 assert(!ST.hasExtendedWaitCounts());
2044
2045 auto IsHazardInst = [](const MachineInstr &MI) {
2047 return 1;
2049 return 2;
2050 return 0;
2051 };
2052
2053 auto InstType = IsHazardInst(*MI);
2054 if (!InstType)
2055 return false;
2056
2057 auto IsExpiredFn = [&IsHazardInst](const MachineInstr &I, int) {
2058 return IsHazardInst(I) || isStoreCountWaitZero(I);
2059 };
2060
2061 auto IsHazardFn = [InstType, &IsHazardInst](const MachineInstr &I) {
2062 if (!I.isBranch())
2063 return false;
2064
2065 auto IsHazardFn = [InstType, IsHazardInst](const MachineInstr &I) {
2066 auto InstType2 = IsHazardInst(I);
2067 return InstType2 && InstType != InstType2;
2068 };
2069
2070 auto IsExpiredFn = [InstType, &IsHazardInst](const MachineInstr &I, int) {
2071 auto InstType2 = IsHazardInst(I);
2072 if (InstType == InstType2)
2073 return true;
2074
2075 return isStoreCountWaitZero(I);
2076 };
2077
2078 return ::getWaitStatesSince(IsHazardFn, &I, IsExpiredFn) !=
2079 std::numeric_limits<int>::max();
2080 };
2081
2082 if (::getWaitStatesSince(IsHazardFn, MI, IsExpiredFn) ==
2083 std::numeric_limits<int>::max())
2084 return false;
2085
2086 const SIInstrInfo *TII = ST.getInstrInfo();
2087 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(),
2088 TII->get(AMDGPU::S_WAITCNT_VSCNT))
2089 .addReg(AMDGPU::SGPR_NULL, RegState::Undef)
2090 .addImm(0);
2091
2092 return true;
2093}
2094
2095bool GCNHazardRecognizer::fixLdsDirectVALUHazard(MachineInstr *MI) {
2097 return false;
2098
2099 const int NoHazardWaitStates = 15;
2100 const MachineOperand *VDST = TII.getNamedOperand(*MI, AMDGPU::OpName::vdst);
2101 const Register VDSTReg = VDST->getReg();
2102
2103 bool VisitedTrans = false;
2104 auto IsHazardFn = [this, VDSTReg, &VisitedTrans](const MachineInstr &I) {
2105 if (!SIInstrInfo::isVALU(I, /*AllowLDSDMA=*/true))
2106 return false;
2107 VisitedTrans = VisitedTrans || SIInstrInfo::isTRANS(I);
2108 // Cover both WAR and WAW
2109 return I.readsRegister(VDSTReg, &TRI) || I.modifiesRegister(VDSTReg, &TRI);
2110 };
2111 auto IsExpiredFn = [&](const MachineInstr &I, int WaitStates) {
2112 if (WaitStates >= NoHazardWaitStates)
2113 return true;
2114 // Instructions which cause va_vdst==0 expire hazard
2117 };
2118 auto GetWaitStatesFn = [](const MachineInstr &MI) {
2119 return SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true) ? 1 : 0;
2120 };
2121
2122 DenseSet<const MachineBasicBlock *> Visited;
2123 auto Count = ::getWaitStatesSince(IsHazardFn, MI->getParent(),
2124 std::next(MI->getReverseIterator()), 0,
2125 IsExpiredFn, Visited, GetWaitStatesFn);
2126
2127 // Transcendentals can execute in parallel to other VALUs.
2128 // This makes va_vdst count unusable with a mixture of VALU and TRANS.
2129 if (VisitedTrans)
2130 Count = 0;
2131
2132 MachineOperand *WaitVdstOp =
2133 TII.getNamedOperand(*MI, AMDGPU::OpName::waitvdst);
2134 WaitVdstOp->setImm(std::min(Count, NoHazardWaitStates));
2135
2136 return true;
2137}
2138
2139bool GCNHazardRecognizer::fixLdsDirectVMEMHazard(MachineInstr *MI) {
2141 return false;
2142
2143 const MachineOperand *VDST = TII.getNamedOperand(*MI, AMDGPU::OpName::vdst);
2144 const Register VDSTReg = VDST->getReg();
2145
2146 auto IsHazardFn = [this, VDSTReg](const MachineInstr &I) {
2148 return false;
2149 return I.readsRegister(VDSTReg, &TRI) || I.modifiesRegister(VDSTReg, &TRI);
2150 };
2151 bool LdsdirCanWait = ST.hasLdsWaitVMSRC();
2152 // TODO: On GFX12 the hazard should expire on S_WAIT_LOADCNT/SAMPLECNT/BVHCNT
2153 // according to the type of VMEM instruction.
2154 auto IsExpiredFn = [this, LdsdirCanWait](const MachineInstr &I, int) {
2155 return SIInstrInfo::isVALU(I, /*AllowLDSDMA=*/true) ||
2157 (I.getOpcode() == AMDGPU::S_WAITCNT && !I.getOperand(0).getImm()) ||
2158 (I.getOpcode() == AMDGPU::S_WAITCNT_DEPCTR &&
2159 AMDGPU::DepCtr::decodeFieldVmVsrc(I.getOperand(0).getImm()) == 0) ||
2160 (LdsdirCanWait && SIInstrInfo::isLDSDIR(I) &&
2161 !TII.getNamedOperand(I, AMDGPU::OpName::waitvsrc)->getImm());
2162 };
2163
2164 if (::getWaitStatesSince(IsHazardFn, MI, IsExpiredFn) ==
2165 std::numeric_limits<int>::max())
2166 return false;
2167
2168 if (LdsdirCanWait) {
2169 TII.getNamedOperand(*MI, AMDGPU::OpName::waitvsrc)->setImm(0);
2170 } else {
2171 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(),
2172 TII.get(AMDGPU::S_WAITCNT_DEPCTR))
2174 }
2175
2176 return true;
2177}
2178
2179bool GCNHazardRecognizer::fixVALUPartialForwardingHazard(MachineInstr *MI) {
2180 if (!ST.hasVALUPartialForwardingHazard())
2181 return false;
2182 assert(!ST.hasExtendedWaitCounts());
2183
2184 if (!ST.isWave64() || !SIInstrInfo::isVALU(*MI, /*AllowLDSDMA=*/true))
2185 return false;
2186
2187 SmallSetVector<Register, 4> SrcVGPRs;
2188
2189 for (const MachineOperand &Use : MI->explicit_uses()) {
2190 if (Use.isReg() && TRI.isVGPR(MF.getRegInfo(), Use.getReg()))
2191 SrcVGPRs.insert(Use.getReg());
2192 }
2193
2194 // Only applies with >= 2 unique VGPR sources
2195 if (SrcVGPRs.size() <= 1)
2196 return false;
2197
2198 // Look for the following pattern:
2199 // Va <- VALU [PreExecPos]
2200 // intv1
2201 // Exec <- SALU [ExecPos]
2202 // intv2
2203 // Vb <- VALU [PostExecPos]
2204 // intv3
2205 // MI Va, Vb (WaitState = 0)
2206 //
2207 // Where:
2208 // intv1 + intv2 <= 2 VALUs
2209 // intv3 <= 4 VALUs
2210 //
2211 // If found, insert an appropriate S_WAITCNT_DEPCTR before MI.
2212
2213 const int Intv1plus2MaxVALUs = 2;
2214 const int Intv3MaxVALUs = 4;
2215 const int IntvMaxVALUs = 6;
2216 const int NoHazardVALUWaitStates = IntvMaxVALUs + 2;
2217
2218 struct StateType {
2219 SmallDenseMap<Register, int, 4> DefPos;
2220 int ExecPos = std::numeric_limits<int>::max();
2221 int VALUs = 0;
2222
2223 static unsigned getHashValue(const StateType &State) {
2224 hash_code H = hash_combine(State.ExecPos, State.VALUs);
2225 for (const auto &[Reg, Pos] : State.DefPos)
2226 H = hash_combine(H, Reg, Pos);
2227 return H;
2228 }
2229 static bool isEqual(const StateType &LHS, const StateType &RHS) {
2230 return LHS.DefPos == RHS.DefPos && LHS.ExecPos == RHS.ExecPos &&
2231 LHS.VALUs == RHS.VALUs;
2232 }
2233 };
2234
2235 StateType State;
2236
2237 // This overloads expiry testing with all the hazard detection
2238 auto IsHazardFn = [&, this](StateType &State, const MachineInstr &I) {
2239 // Too many VALU states have passed
2240 if (State.VALUs > NoHazardVALUWaitStates)
2241 return HazardExpired;
2242
2243 // Instructions which cause va_vdst==0 expire hazard
2246 (I.getOpcode() == AMDGPU::S_WAITCNT_DEPCTR &&
2247 AMDGPU::DepCtr::decodeFieldVaVdst(I.getOperand(0).getImm()) == 0))
2248 return HazardExpired;
2249
2250 // Track registers writes
2251 bool Changed = false;
2252 if (SIInstrInfo::isVALU(I, /*AllowLDSDMA=*/true)) {
2253 for (Register Src : SrcVGPRs) {
2254 if (!State.DefPos.count(Src) && I.modifiesRegister(Src, &TRI)) {
2255 State.DefPos[Src] = State.VALUs;
2256 Changed = true;
2257 }
2258 }
2259 } else if (SIInstrInfo::isSALU(I)) {
2260 if (State.ExecPos == std::numeric_limits<int>::max()) {
2261 if (!State.DefPos.empty() && I.modifiesRegister(AMDGPU::EXEC, &TRI)) {
2262 State.ExecPos = State.VALUs;
2263 Changed = true;
2264 }
2265 }
2266 }
2267
2268 // Early expiration: too many VALUs in intv3
2269 if (State.VALUs > Intv3MaxVALUs && State.DefPos.empty())
2270 return HazardExpired;
2271
2272 // Only evaluate state if something changed
2273 if (!Changed)
2274 return NoHazardFound;
2275
2276 // Determine positions of VALUs pre/post exec change
2277 if (State.ExecPos == std::numeric_limits<int>::max())
2278 return NoHazardFound;
2279
2280 int PreExecPos = std::numeric_limits<int>::max();
2281 int PostExecPos = std::numeric_limits<int>::max();
2282
2283 for (auto Entry : State.DefPos) {
2284 int DefVALUs = Entry.second;
2285 if (DefVALUs != std::numeric_limits<int>::max()) {
2286 if (DefVALUs >= State.ExecPos)
2287 PreExecPos = std::min(PreExecPos, DefVALUs);
2288 else
2289 PostExecPos = std::min(PostExecPos, DefVALUs);
2290 }
2291 }
2292
2293 // Need a VALUs post exec change
2294 if (PostExecPos == std::numeric_limits<int>::max())
2295 return NoHazardFound;
2296
2297 // Too many VALUs in intv3?
2298 int Intv3VALUs = PostExecPos;
2299 if (Intv3VALUs > Intv3MaxVALUs)
2300 return HazardExpired;
2301
2302 // Too many VALUs in intv2?
2303 int Intv2VALUs = (State.ExecPos - PostExecPos) - 1;
2304 if (Intv2VALUs > Intv1plus2MaxVALUs)
2305 return HazardExpired;
2306
2307 // Need a VALUs pre exec change
2308 if (PreExecPos == std::numeric_limits<int>::max())
2309 return NoHazardFound;
2310
2311 // Too many VALUs in intv1?
2312 int Intv1VALUs = PreExecPos - State.ExecPos;
2313 if (Intv1VALUs > Intv1plus2MaxVALUs)
2314 return HazardExpired;
2315
2316 // Too many VALUs in intv1 + intv2
2317 if (Intv1VALUs + Intv2VALUs > Intv1plus2MaxVALUs)
2318 return HazardExpired;
2319
2320 return HazardFound;
2321 };
2322 auto UpdateStateFn = [](StateType &State, const MachineInstr &MI) {
2323 if (SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true))
2324 State.VALUs += 1;
2325 };
2326
2327 if (!hasHazard<StateType>(State, IsHazardFn, UpdateStateFn, MI->getParent(),
2328 std::next(MI->getReverseIterator())))
2329 return false;
2330
2331 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(),
2332 TII.get(AMDGPU::S_WAITCNT_DEPCTR))
2334
2335 return true;
2336}
2337
2338bool GCNHazardRecognizer::fixVALUTransUseHazard(MachineInstr *MI) {
2339 if (!ST.hasVALUTransUseHazard())
2340 return false;
2341 assert(!ST.hasExtendedWaitCounts());
2342
2343 if (!SIInstrInfo::isVALU(*MI, /*AllowLDSDMA=*/true))
2344 return false;
2345
2346 SmallSet<Register, 4> SrcVGPRs;
2347
2348 for (const MachineOperand &Use : MI->explicit_uses()) {
2349 if (Use.isReg() && TRI.isVGPR(MF.getRegInfo(), Use.getReg()))
2350 SrcVGPRs.insert(Use.getReg());
2351 }
2352
2353 // Look for the following pattern:
2354 // Va <- TRANS VALU
2355 // intv
2356 // MI Va (WaitState = 0)
2357 //
2358 // Where:
2359 // intv <= 5 VALUs / 1 TRANS
2360 //
2361 // If found, insert an appropriate S_WAITCNT_DEPCTR before MI.
2362
2363 const int IntvMaxVALUs = 5;
2364 const int IntvMaxTRANS = 1;
2365
2366 struct StateType {
2367 int VALUs = 0;
2368 int TRANS = 0;
2369
2370 static unsigned getHashValue(const StateType &State) {
2371 return hash_combine(State.VALUs, State.TRANS);
2372 }
2373 static bool isEqual(const StateType &LHS, const StateType &RHS) {
2374 return LHS.VALUs == RHS.VALUs && LHS.TRANS == RHS.TRANS;
2375 }
2376 };
2377
2378 StateType State;
2379
2380 // This overloads expiry testing with all the hazard detection
2381 auto IsHazardFn = [&, this](StateType &State, const MachineInstr &I) {
2382 // Too many VALU states have passed
2383 if (State.VALUs > IntvMaxVALUs || State.TRANS > IntvMaxTRANS)
2384 return HazardExpired;
2385
2386 // Instructions which cause va_vdst==0 expire hazard
2389 (I.getOpcode() == AMDGPU::S_WAITCNT_DEPCTR &&
2390 AMDGPU::DepCtr::decodeFieldVaVdst(I.getOperand(0).getImm()) == 0))
2391 return HazardExpired;
2392
2393 // Track registers writes
2394 if (SIInstrInfo::isTRANS(I)) {
2395 for (Register Src : SrcVGPRs) {
2396 if (I.modifiesRegister(Src, &TRI)) {
2397 return HazardFound;
2398 }
2399 }
2400 }
2401
2402 return NoHazardFound;
2403 };
2404 auto UpdateStateFn = [](StateType &State, const MachineInstr &MI) {
2405 if (SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true))
2406 State.VALUs += 1;
2408 State.TRANS += 1;
2409 };
2410
2411 if (!hasHazard<StateType>(State, IsHazardFn, UpdateStateFn, MI->getParent(),
2412 std::next(MI->getReverseIterator())))
2413 return false;
2414
2415 // Hazard is observed - insert a wait on va_dst counter to ensure hazard is
2416 // avoided.
2417 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(),
2418 TII.get(AMDGPU::S_WAITCNT_DEPCTR))
2420
2421 return true;
2422}
2423
2424bool GCNHazardRecognizer::fixVALUTransCoexecutionHazards(MachineInstr *MI) {
2425 if (!ST.hasTransCoexecutionHazard() || // Coexecution disabled.
2426 !SIInstrInfo::isVALU(*MI, /*AllowLDSDMA=*/true) ||
2428 return false;
2429
2430 const SIInstrInfo *TII = ST.getInstrInfo();
2431 const SIRegisterInfo *TRI = ST.getRegisterInfo();
2432
2433 auto IsTransHazardFn = [MI, TII, TRI](const MachineInstr &I) {
2434 if (!SIInstrInfo::isTRANS(I))
2435 return false;
2436
2437 // RAW: Trans(I) writes, VALU(MI) reads.
2438 Register TransDef = TII->getNamedOperand(I, AMDGPU::OpName::vdst)->getReg();
2439 for (const MachineOperand &ValuUse : MI->explicit_uses()) {
2440 if (ValuUse.isReg() && TRI->regsOverlap(TransDef, ValuUse.getReg()))
2441 return true;
2442 }
2443
2444 auto *ValuDst = TII->getNamedOperand(*MI, AMDGPU::OpName::vdst);
2445 if (!ValuDst || !ValuDst->isReg())
2446 return false;
2447
2448 // WAR: Trans(I) reads, VALU(MI) writes.
2449 Register ValuDef = ValuDst->getReg();
2450 for (const MachineOperand &TransUse : I.explicit_uses()) {
2451 if (TransUse.isReg() && TRI->regsOverlap(ValuDef, TransUse.getReg()))
2452 return true;
2453 }
2454
2455 return false;
2456 };
2457
2458 auto IsExpiredFn = [](const MachineInstr &I, int) {
2459 return SIInstrInfo::isVALU(I, /*AllowLDSDMA=*/true);
2460 };
2461
2462 const int HasVALU = std::numeric_limits<int>::max();
2463 if (::getWaitStatesSince(IsTransHazardFn, MI, IsExpiredFn) == HasVALU)
2464 return false;
2465
2466 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(), TII->get(AMDGPU::V_NOP_e32));
2467 return true;
2468}
2469
2470bool GCNHazardRecognizer::fixWMMAHazards(MachineInstr *MI) {
2472 return false;
2473
2474 const SIInstrInfo *TII = ST.getInstrInfo();
2475 const SIRegisterInfo *TRI = ST.getRegisterInfo();
2476
2477 auto IsHazardFn = [MI, TII, TRI, this](const MachineInstr &I) {
2479 return false;
2480
2481 // Src0(matrix A) or Src1(matrix B) of the current wmma instruction overlaps
2482 // with the dest(matrix D) of the previous wmma.
2483 const Register CurSrc0Reg =
2484 TII->getNamedOperand(*MI, AMDGPU::OpName::src0)->getReg();
2485 const Register CurSrc1Reg =
2486 TII->getNamedOperand(*MI, AMDGPU::OpName::src1)->getReg();
2487
2488 const Register PrevDstReg =
2489 TII->getNamedOperand(I, AMDGPU::OpName::vdst)->getReg();
2490
2491 if (TRI->regsOverlap(PrevDstReg, CurSrc0Reg) ||
2492 TRI->regsOverlap(PrevDstReg, CurSrc1Reg)) {
2493 return true;
2494 }
2495
2496 // GFX12+ allows overlap of matrix C with PrevDstReg (hardware will stall)
2497 // but Index can't overlap with PrevDstReg.
2498 if (AMDGPU::isGFX12Plus(ST)) {
2499 if (SIInstrInfo::isSWMMAC(*MI)) {
2500 const Register CurIndex =
2501 TII->getNamedOperand(*MI, AMDGPU::OpName::src2)->getReg();
2502 if (TRI->regsOverlap(PrevDstReg, CurIndex))
2503 return true;
2504 }
2505 return false;
2506 }
2507
2508 return false;
2509 };
2510
2511 auto IsExpiredFn = [](const MachineInstr &I, int) {
2512 return SIInstrInfo::isVALU(I, /*AllowLDSDMA=*/true);
2513 };
2514
2515 if (::getWaitStatesSince(IsHazardFn, MI, IsExpiredFn) ==
2516 std::numeric_limits<int>::max())
2517 return false;
2518
2519 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(), TII->get(AMDGPU::V_NOP_e32));
2520
2521 return true;
2522}
2523
2525 return SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/false) &&
2527}
2528
2529// Classify XDL WMMA instructions into co-execution hazard categories
2530// (Refer to SPG 4.6.12.1), mainly based on instruction latency.
2531//
2532// Category 0: WMMA with Latency 8
2533// WMMA_*F16, WMMA_*BF16
2534// WMMA_*_16X16X128_{FP8,BF8}
2535// WMMA_*F8F6F4 if SRCA & SRCB are not both F4
2536//
2537// Category 1: WMMA Latency 16
2538// WMMA_IU8
2539//
2540// Category 2: SWMMAC with Latency 8
2541// SWMMAC_*F16, SWMMAC_*BF16,
2542// SWMMAC_*FP8FP8
2543// SWMMAC_*BF8FP8
2544// SWMMAC_*FP8BF8
2545// SWMMAC_*BF8BF8
2546//
2547// Category 3: SWMMAC with Latency 16
2548// SWMMAC_IU8
2549//
2550// Category 4: 16 Pass GFX1251 WMMA with latency 16
2551// V_WMMA_*_16X16X32_{F16,BF16}
2552// V_WMMA_{F32,F16}_16X16X64_{FP8,BF8}*
2553// V_WMMA_F32_16x16x128_F8F6F4 (F4 only)
2554// V_SWMMAC_*_16X16X64_{F16,BF16}
2555// V_SWMMAC_{F32,F16}_16X16X128_{FP8,BF8}*
2556//
2557// Category 5: 32 Pass GFX1251 WMMA with latency 32
2558// V_WMMA_F32_16x16x128_F8F6F4 (not all F4)
2559// V_WMMA_{F32,F16}_16X16X128_{FP8,BF8}*
2560// V_WMMA_F32_32X16X128_F4
2561// V_WMMA_I32_16X16X64_IU8
2562// V_WMMA_I32_16X16X64_IU8
2563//
2564// Category 6: gfx1250 WMMA with Latency 4 (one co-execution slot)
2565// WMMA_*_16X16X64_{FP8,BF8}
2566// WMMA_*F8F6F4 if SRCA & SRCB are both F4
2568 const SIInstrInfo *TII,
2569 const TargetSchedModel &SchedModel,
2570 const GCNSubtarget &ST) {
2571 assert(TII->isXDLWMMA(MI) && "must be xdl wmma");
2572 bool IsSWMMAC = SIInstrInfo::isSWMMAC(MI);
2573 bool IsLowestRateWMMA = ST.hasGFX125xLowestRateWMMA();
2574 unsigned Category = 0;
2575
2576 unsigned Latency = SchedModel.computeInstrLatency(&MI);
2577 switch (Latency) {
2578 case 4:
2579 // Dense 4-cycle WMMA (gfx1250 16x16x64 FP8/BF8 and f8f6f4 with both
2580 // inputs F4). One co-execution slot; there is no 4-cycle SWMMAC.
2581 assert(!IsSWMMAC && "no 4-cycle SWMMAC expected");
2582 Category = 6;
2583 break;
2584 case 8:
2585 Category = IsSWMMAC ? 2 : 0;
2586 break;
2587 case 16:
2588 Category = IsLowestRateWMMA ? 4 : (IsSWMMAC ? 3 : 1);
2589 break;
2590 case 32:
2591 assert(IsLowestRateWMMA && "latency 32 is not expected");
2592 Category = 5;
2593 break;
2594 default:
2595 llvm_unreachable("unexpected xdl wmma latency");
2596 } // end switch.
2597
2598 return Category;
2599}
2600
2601int GCNHazardRecognizer::checkWMMACoexecutionHazards(MachineInstr *MI) const {
2602 if (!ST.hasWMMACoexecutionHazards())
2603 return 0;
2604
2605 const SIInstrInfo *TII = ST.getInstrInfo();
2606 if (!TII->isXDLWMMA(*MI) && !isCoexecutableVALUInst(*MI))
2607 return 0;
2608
2609 // WaitStates here is the number of V_NOPs or unrelated VALU instructions must
2610 // be in between the first WMMA and the second instruction to cover the hazard
2611 // (WMMAWaitStates if the second is also a WMMA, VALUWaitStates if the second
2612 // is a VALU). Refer to SPG 4.6.12.1. "Requirements for WMMA data hazards" for
2613 // numbers, which depends on the category of the first WMMA.
2614 const int WMMAWaitStates[] = {5, 9, 3, 5, 9, 17, 2};
2615 const int VALUWaitStates[] = {4, 8, 2, 4, 8, 16, 1};
2616 unsigned Category = 0;
2617
2618 auto IsWMMAHazardFn = [MI, TII, &Category, this](const MachineInstr &I) {
2619 if (!TII->isXDLWMMA(I))
2620 return false;
2621
2622 Category = getWMMAHazardInstInCategory(I, TII, TSchedModel, ST);
2623 return hasWMMAToWMMARegOverlap(I, *MI);
2624 };
2625
2626 auto IsVALUHazardFn = [MI, TII, &Category, this](const MachineInstr &I) {
2627 if (!TII->isXDLWMMA(I))
2628 return false;
2629
2630 Category = getWMMAHazardInstInCategory(I, TII, TSchedModel, ST);
2631 return hasWMMAToVALURegOverlap(I, *MI);
2632 };
2633
2634 int WaitStatesNeeded = -1;
2635 int ExistingVALUs = 0; // Existing number of VALU ops in between.
2636 bool IsLowestRateWMMA = ST.hasGFX125xLowestRateWMMA();
2637
2638 // getWaitStatesSinceVALU checks for a hazard between instruction 'I' and
2639 // 'MI':
2640 // - If a hazard exists: returns the number of VALUs in between and sets
2641 // 'Category' via IsWMMAHazardFn/IsVALUHazardFn for instruction 'I'.
2642 // - If no hazard exists: returns INT_MAX, making WaitStatesNeeded negative,
2643 // so no V_NOP insertion is needed.
2644 if (TII->isXDLWMMA(*MI)) {
2645 // Maximum of MMAWaitStates.
2646 const int WMMAWaitsLimit = IsLowestRateWMMA ? 17 : 9;
2647 ExistingVALUs = getWaitStatesSinceVALU(IsWMMAHazardFn, WMMAWaitsLimit);
2648 WaitStatesNeeded = WMMAWaitStates[Category] - ExistingVALUs;
2649 } else { // Must be a co-executable VALU.
2650 // Maximum of VALUWaitStates.
2651 const int VALUWaitsLimit = IsLowestRateWMMA ? 16 : 8;
2652 ExistingVALUs = getWaitStatesSinceVALU(IsVALUHazardFn, VALUWaitsLimit);
2653 WaitStatesNeeded = VALUWaitStates[Category] - ExistingVALUs;
2654 }
2655
2656 return WaitStatesNeeded;
2657}
2658
2659bool GCNHazardRecognizer::hasWMMAToWMMARegOverlap(
2660 const MachineInstr &WMMA, const MachineInstr &MI) const {
2661 Register D0 = TII.getNamedOperand(WMMA, AMDGPU::OpName::vdst)->getReg();
2662 Register A1 = TII.getNamedOperand(MI, AMDGPU::OpName::src0)->getReg();
2663 Register B1 = TII.getNamedOperand(MI, AMDGPU::OpName::src1)->getReg();
2664
2665 // WMMA0 writes (D0), WMMA1 reads (A1/B1/Idx1).
2666 if (TRI.regsOverlap(D0, A1) || TRI.regsOverlap(D0, B1))
2667 return true;
2668
2670 Register Idx1 = TII.getNamedOperand(MI, AMDGPU::OpName::src2)->getReg();
2671 if (TRI.regsOverlap(D0, Idx1))
2672 return true;
2673 }
2674 return false;
2675}
2676
2677bool GCNHazardRecognizer::hasWMMAToVALURegOverlap(
2678 const MachineInstr &WMMA, const MachineInstr &MI) const {
2679 // WMMA writes, VALU reads.
2680 Register D0 = TII.getNamedOperand(WMMA, AMDGPU::OpName::vdst)->getReg();
2681 for (const MachineOperand &ValuUse : MI.explicit_uses()) {
2682 if (ValuUse.isReg() && TRI.regsOverlap(D0, ValuUse.getReg()))
2683 return true;
2684 }
2685
2686 // WMMA reads or writes, VALU writes.
2687 Register A0 = TII.getNamedOperand(WMMA, AMDGPU::OpName::src0)->getReg();
2688 Register B0 = TII.getNamedOperand(WMMA, AMDGPU::OpName::src1)->getReg();
2689 SmallVector<Register, 4> WMMARegs({D0, A0, B0});
2690
2691 if (SIInstrInfo::isSWMMAC(WMMA)) {
2692 Register Idx0 = TII.getNamedOperand(WMMA, AMDGPU::OpName::src2)->getReg();
2693 WMMARegs.push_back(Idx0);
2694 }
2695
2696 for (const MachineOperand &ValuDef : MI.defs()) {
2697 Register VDstReg = ValuDef.getReg();
2698 for (Register WMMAReg : WMMARegs) {
2699 if (TRI.regsOverlap(VDstReg, WMMAReg))
2700 return true;
2701 }
2702 }
2703 return false;
2704}
2705
2706bool GCNHazardRecognizer::isCoexecutionHazardFor(const MachineInstr &I,
2707 const MachineInstr &MI) const {
2708 // I is the potential WMMA hazard source, MI is the instruction being checked
2709 // for hazard.
2710 if (!TII.isXDLWMMA(I))
2711 return false;
2712
2713 // Dispatch based on MI type
2714 if (TII.isXDLWMMA(MI))
2715 return hasWMMAToWMMARegOverlap(I, MI);
2717 return hasWMMAToVALURegOverlap(I, MI);
2718
2719 return false;
2720}
2721
2722bool GCNHazardRecognizer::hasWMMAHazardInLoop(MachineLoop *L, MachineInstr *MI,
2723 bool IncludeSubloops) {
2724 // Scan loop for any WMMA that hazards MI.
2725 // TODO: Avoid full loop scan when WMMA is beyond VALU distance.
2726 for (MachineBasicBlock *MBB : L->getBlocks()) {
2727 if (!IncludeSubloops && MLI->getLoopFor(MBB) != L)
2728 continue;
2729 for (MachineInstr &I : *MBB) {
2730 if (&I == MI)
2731 continue;
2732 if (isCoexecutionHazardFor(I, *MI))
2733 return true;
2734 }
2735 }
2736 return false;
2737}
2738
2739bool GCNHazardRecognizer::tryHoistWMMAVnopsFromLoop(MachineInstr *MI,
2740 int WaitStatesNeeded) {
2741 if (!MLI)
2742 return false;
2743
2744 MachineLoop *L = MLI->getLoopFor(MI->getParent());
2745 if (!L) {
2746 ++NumWMMAHoistingBailed;
2747 return false;
2748 }
2749
2750 // If innermost loop has WMMA hazard, we can't hoist at all
2751 if (hasWMMAHazardInLoop(L, MI)) {
2752 ++NumWMMAHoistingBailed;
2753 return false;
2754 }
2755
2756 // Find outermost loop with no internal hazard
2757 MachineLoop *TargetLoop = L;
2758 while (MachineLoop *Parent = TargetLoop->getParentLoop()) {
2759 if (hasWMMAHazardInLoop(Parent, MI, false))
2760 break; // Parent has hazard in its own blocks, stop here
2761 TargetLoop = Parent; // Safe to hoist further out
2762 }
2763
2764 // Need valid preheader to insert V_NOPs
2765 MachineBasicBlock *Preheader = TargetLoop->getLoopPreheader();
2766 if (!Preheader) {
2767 ++NumWMMAHoistingBailed;
2768 return false;
2769 }
2770
2771 LLVM_DEBUG(dbgs() << "WMMA V_NOP Hoisting: Moving " << WaitStatesNeeded
2772 << " V_NOPs from loop to " << printMBBReference(*Preheader)
2773 << "\n");
2774
2775 emitVNops(*Preheader, Preheader->getFirstTerminator(), WaitStatesNeeded,
2776 /*IsHoisting=*/true);
2777 NumWMMANopsHoisted += WaitStatesNeeded;
2778 return true;
2779}
2780
2781bool GCNHazardRecognizer::fixWMMACoexecutionHazards(MachineInstr *MI) {
2782 int WaitStatesNeeded = checkWMMACoexecutionHazards(MI);
2783 if (WaitStatesNeeded <= 0)
2784 return false;
2785
2786 if (EnableWMMAVnopHoisting && tryHoistWMMAVnopsFromLoop(MI, WaitStatesNeeded))
2787 return true;
2788
2789 emitVNops(*MI->getParent(), MI->getIterator(), WaitStatesNeeded);
2790 return true;
2791}
2792
2793bool GCNHazardRecognizer::fixShift64HighRegBug(MachineInstr *MI) {
2794 if (!ST.hasShift64HighRegBug())
2795 return false;
2796 assert(!ST.hasExtendedWaitCounts());
2797
2798 switch (MI->getOpcode()) {
2799 default:
2800 return false;
2801 case AMDGPU::V_LSHLREV_B64_e64:
2802 case AMDGPU::V_LSHRREV_B64_e64:
2803 case AMDGPU::V_ASHRREV_I64_e64:
2804 break;
2805 }
2806
2807 MachineOperand *Amt = TII.getNamedOperand(*MI, AMDGPU::OpName::src0);
2808 if (!Amt->isReg())
2809 return false;
2810
2811 Register AmtReg = Amt->getReg();
2812 const MachineRegisterInfo &MRI = MF.getRegInfo();
2813 // Check if this is a last VGPR in the allocation block.
2814 if (!TRI.isVGPR(MRI, AmtReg) || ((AmtReg - AMDGPU::VGPR0) & 7) != 7)
2815 return false;
2816
2817 if (AmtReg != AMDGPU::VGPR255 && MRI.isPhysRegUsed(AmtReg + 1))
2818 return false;
2819
2820 assert(ST.needsAlignedVGPRs());
2821 static_assert(AMDGPU::VGPR0 + 1 == AMDGPU::VGPR1);
2822
2823 const DebugLoc &DL = MI->getDebugLoc();
2824 MachineBasicBlock *MBB = MI->getParent();
2825 MachineOperand *Src1 = TII.getNamedOperand(*MI, AMDGPU::OpName::src1);
2826
2827 // In:
2828 //
2829 // Dst = shiftrev64 Amt, Src1
2830 //
2831 // if Dst!=Src1 then avoid the bug with:
2832 //
2833 // Dst.sub0 = Amt
2834 // Dst = shift64 Dst.sub0, Src1
2835
2836 Register DstReg = MI->getOperand(0).getReg();
2837 if (!Src1->isReg() || Src1->getReg() != DstReg) {
2838 Register DstLo = TRI.getSubReg(DstReg, AMDGPU::sub0);
2839 runOnInstruction(
2840 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_MOV_B32_e32), DstLo).add(*Amt));
2841 Amt->setReg(DstLo);
2842 Amt->setIsKill(true);
2843 return true;
2844 }
2845
2846 bool Overlapped = MI->modifiesRegister(AmtReg, &TRI);
2847 Register NewReg;
2848 for (MCRegister Reg : Overlapped ? AMDGPU::VReg_64_Align2RegClass
2849 : AMDGPU::VGPR_32RegClass) {
2850 if (!MI->modifiesRegister(Reg, &TRI) && !MI->readsRegister(Reg, &TRI)) {
2851 NewReg = Reg;
2852 break;
2853 }
2854 }
2855
2856 Register NewAmt = Overlapped ? (Register)TRI.getSubReg(NewReg, AMDGPU::sub1)
2857 : NewReg;
2858 Register NewAmtLo;
2859
2860 if (Overlapped)
2861 NewAmtLo = TRI.getSubReg(NewReg, AMDGPU::sub0);
2862
2863 // Insert a full wait count because found register might be pending a wait.
2864 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::S_WAITCNT))
2865 .addImm(0);
2866
2867 // Insert V_SWAP_B32 instruction(s) and run hazard recognizer on them.
2868 if (Overlapped)
2869 runOnInstruction(
2870 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_SWAP_B32), NewAmtLo)
2871 .addDef(AmtReg - 1)
2872 .addReg(AmtReg - 1, RegState::Undef)
2873 .addReg(NewAmtLo, RegState::Undef));
2874 runOnInstruction(BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_SWAP_B32), NewAmt)
2875 .addDef(AmtReg)
2876 .addReg(AmtReg, RegState::Undef)
2877 .addReg(NewAmt, RegState::Undef));
2878
2879 // Instructions emitted after the current instruction will be processed by the
2880 // parent loop of the hazard recognizer in a natural way.
2881 BuildMI(*MBB, std::next(MI->getIterator()), DL, TII.get(AMDGPU::V_SWAP_B32),
2882 AmtReg)
2883 .addDef(NewAmt)
2884 .addReg(NewAmt)
2885 .addReg(AmtReg);
2886 if (Overlapped)
2887 BuildMI(*MBB, std::next(MI->getIterator()), DL, TII.get(AMDGPU::V_SWAP_B32),
2888 AmtReg - 1)
2889 .addDef(NewAmtLo)
2890 .addReg(NewAmtLo)
2891 .addReg(AmtReg - 1);
2892
2893 // Re-running hazard recognizer on the modified instruction is not necessary,
2894 // inserted V_SWAP_B32 has already both read and write new registers so
2895 // hazards related to these register has already been handled.
2896 Amt->setReg(NewAmt);
2897 Amt->setIsKill(false);
2898 // We do not update liveness, so verifier may see it as undef.
2899 Amt->setIsUndef();
2900 if (Overlapped) {
2901 MI->getOperand(0).setReg(NewReg);
2902 Src1->setReg(NewReg);
2903 Src1->setIsKill(false);
2904 Src1->setIsUndef();
2905 }
2906
2907 return true;
2908}
2909
2910int GCNHazardRecognizer::checkNSAtoVMEMHazard(MachineInstr *MI) const {
2911 int NSAtoVMEMWaitStates = 1;
2912
2913 if (!ST.hasNSAtoVMEMBug())
2914 return 0;
2915
2917 return 0;
2918
2919 const SIInstrInfo *TII = ST.getInstrInfo();
2920 const auto *Offset = TII->getNamedOperand(*MI, AMDGPU::OpName::offset);
2921 if (!Offset || (Offset->getImm() & 6) == 0)
2922 return 0;
2923
2924 auto IsHazardFn = [TII](const MachineInstr &I) {
2925 if (!SIInstrInfo::isMIMG(I))
2926 return false;
2927 const AMDGPU::MIMGInfo *Info = AMDGPU::getMIMGInfo(I.getOpcode());
2928 return Info->MIMGEncoding == AMDGPU::MIMGEncGfx10NSA &&
2929 TII->getInstSizeInBytes(I) >= 16;
2930 };
2931
2932 return NSAtoVMEMWaitStates - getWaitStatesSince(IsHazardFn, 1);
2933}
2934
2935int GCNHazardRecognizer::checkFPAtomicToDenormModeHazard(
2936 MachineInstr *MI) const {
2937 int FPAtomicToDenormModeWaitStates = 3;
2938
2939 if (!ST.hasFPAtomicToDenormModeHazard())
2940 return 0;
2941 assert(!ST.hasExtendedWaitCounts());
2942
2943 if (MI->getOpcode() != AMDGPU::S_DENORM_MODE)
2944 return 0;
2945
2946 auto IsHazardFn = [](const MachineInstr &I) {
2947 if (!SIInstrInfo::isVMEM(I))
2948 return false;
2949 return SIInstrInfo::isFPAtomic(I);
2950 };
2951
2952 auto IsExpiredFn = [](const MachineInstr &MI, int WaitStates) {
2953 if (WaitStates >= 3 || SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true))
2954 return true;
2955
2956 return SIInstrInfo::isWaitcnt(MI.getOpcode());
2957 };
2958
2959 return FPAtomicToDenormModeWaitStates -
2960 ::getWaitStatesSince(IsHazardFn, MI, IsExpiredFn);
2961}
2962
2963int GCNHazardRecognizer::checkMAIHazards(MachineInstr *MI) const {
2965
2966 return ST.hasGFX90AInsts() ? checkMAIHazards90A(MI) : checkMAIHazards908(MI);
2967}
2968
2969int GCNHazardRecognizer::checkMFMAPadding(MachineInstr *MI) const {
2970 // Early exit if no padding is requested.
2971 if (MFMAPaddingRatio == 0)
2972 return 0;
2973
2974 const SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
2975 if (!SIInstrInfo::isMFMA(*MI) || MFI->getOccupancy() < 2)
2976 return 0;
2977
2978 int NeighborMFMALatency = 0;
2979 auto IsNeighboringMFMA = [&NeighborMFMALatency,
2980 this](const MachineInstr &MI) {
2981 if (!SIInstrInfo::isMFMA(MI))
2982 return false;
2983
2984 NeighborMFMALatency = this->getMFMAPipelineWaitStates(MI);
2985 return true;
2986 };
2987
2988 const int MaxMFMAPipelineWaitStates = 16;
2989 int WaitStatesSinceNeighborMFMA =
2990 getWaitStatesSince(IsNeighboringMFMA, MaxMFMAPipelineWaitStates);
2991
2992 int NeighborMFMAPaddingNeeded =
2993 (NeighborMFMALatency * MFMAPaddingRatio / 100) -
2994 WaitStatesSinceNeighborMFMA;
2995
2996 return std::max(0, NeighborMFMAPaddingNeeded);
2997}
2998
2999int GCNHazardRecognizer::checkMAIHazards908(MachineInstr *MI) const {
3000 int WaitStatesNeeded = 0;
3001 unsigned Opc = MI->getOpcode();
3002
3003 auto IsVALUFn = [](const MachineInstr &MI) {
3004 return SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true) || MI.isInlineAsm();
3005 };
3006
3007 if (Opc != AMDGPU::V_ACCVGPR_READ_B32_e64) { // MFMA or v_accvgpr_write
3008 const int LegacyVALUWritesVGPRWaitStates = 2;
3009 const int VALUWritesExecWaitStates = 4;
3010 const int MaxWaitStates = 4;
3011
3012 int WaitStatesNeededForUse = VALUWritesExecWaitStates -
3013 getWaitStatesSinceDef(AMDGPU::EXEC, IsVALUFn, MaxWaitStates);
3014 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3015
3016 if (WaitStatesNeeded < MaxWaitStates) {
3017 for (const MachineOperand &Use : MI->explicit_uses()) {
3018 const int MaxWaitStates = 2;
3019
3020 if (!Use.isReg() || !TRI.isVGPR(MF.getRegInfo(), Use.getReg()))
3021 continue;
3022
3023 int WaitStatesNeededForUse = LegacyVALUWritesVGPRWaitStates -
3024 getWaitStatesSinceDef(Use.getReg(), IsVALUFn, MaxWaitStates);
3025 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3026
3027 if (WaitStatesNeeded == MaxWaitStates)
3028 break;
3029 }
3030 }
3031 }
3032
3033 for (const MachineOperand &Op : MI->explicit_operands()) {
3034 if (!Op.isReg() || !TRI.isAGPR(MF.getRegInfo(), Op.getReg()))
3035 continue;
3036
3037 if (Op.isDef() && Opc != AMDGPU::V_ACCVGPR_WRITE_B32_e64)
3038 continue;
3039
3040 const int MFMAWritesAGPROverlappedSrcABWaitStates = 4;
3041 const int MFMAWritesAGPROverlappedSrcCWaitStates = 2;
3042 const int MFMA4x4WritesAGPRAccVgprReadWaitStates = 4;
3043 const int MFMA16x16WritesAGPRAccVgprReadWaitStates = 10;
3044 const int MFMA32x32WritesAGPRAccVgprReadWaitStates = 18;
3045 const int MFMA4x4WritesAGPRAccVgprWriteWaitStates = 1;
3046 const int MFMA16x16WritesAGPRAccVgprWriteWaitStates = 7;
3047 const int MFMA32x32WritesAGPRAccVgprWriteWaitStates = 15;
3048 const int MaxWaitStates = 18;
3049 Register Reg = Op.getReg();
3050 unsigned HazardDefLatency = 0;
3051
3052 auto IsOverlappedMFMAFn = [Reg, &HazardDefLatency,
3053 this](const MachineInstr &MI) {
3054 if (!SIInstrInfo::isMFMA(MI))
3055 return false;
3056 Register DstReg = MI.getOperand(0).getReg();
3057 if (DstReg == Reg)
3058 return false;
3059 HazardDefLatency =
3060 std::max(HazardDefLatency, TSchedModel.computeInstrLatency(&MI));
3061 return TRI.regsOverlap(DstReg, Reg);
3062 };
3063
3064 int WaitStatesSinceDef = getWaitStatesSinceDef(Reg, IsOverlappedMFMAFn,
3065 MaxWaitStates);
3066 int NeedWaitStates = MFMAWritesAGPROverlappedSrcABWaitStates;
3067 int SrcCIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2);
3068 int OpNo = Op.getOperandNo();
3069 if (OpNo == SrcCIdx) {
3070 NeedWaitStates = MFMAWritesAGPROverlappedSrcCWaitStates;
3071 } else if (Opc == AMDGPU::V_ACCVGPR_READ_B32_e64) {
3072 switch (HazardDefLatency) {
3073 case 2: NeedWaitStates = MFMA4x4WritesAGPRAccVgprReadWaitStates;
3074 break;
3075 case 8: NeedWaitStates = MFMA16x16WritesAGPRAccVgprReadWaitStates;
3076 break;
3077 case 16: [[fallthrough]];
3078 default: NeedWaitStates = MFMA32x32WritesAGPRAccVgprReadWaitStates;
3079 break;
3080 }
3081 } else if (Opc == AMDGPU::V_ACCVGPR_WRITE_B32_e64) {
3082 switch (HazardDefLatency) {
3083 case 2: NeedWaitStates = MFMA4x4WritesAGPRAccVgprWriteWaitStates;
3084 break;
3085 case 8: NeedWaitStates = MFMA16x16WritesAGPRAccVgprWriteWaitStates;
3086 break;
3087 case 16: [[fallthrough]];
3088 default: NeedWaitStates = MFMA32x32WritesAGPRAccVgprWriteWaitStates;
3089 break;
3090 }
3091 }
3092
3093 int WaitStatesNeededForUse = NeedWaitStates - WaitStatesSinceDef;
3094 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3095
3096 if (WaitStatesNeeded == MaxWaitStates)
3097 return WaitStatesNeeded; // Early exit.
3098
3099 auto IsAccVgprWriteFn = [Reg, this](const MachineInstr &MI) {
3100 if (MI.getOpcode() != AMDGPU::V_ACCVGPR_WRITE_B32_e64)
3101 return false;
3102 Register DstReg = MI.getOperand(0).getReg();
3103 return TRI.regsOverlap(Reg, DstReg);
3104 };
3105
3106 const int AccVGPRWriteMFMAReadSrcCWaitStates = 1;
3107 const int AccVGPRWriteMFMAReadSrcABWaitStates = 3;
3108 const int AccVGPRWriteAccVgprReadWaitStates = 3;
3109 NeedWaitStates = AccVGPRWriteMFMAReadSrcABWaitStates;
3110 if (OpNo == SrcCIdx)
3111 NeedWaitStates = AccVGPRWriteMFMAReadSrcCWaitStates;
3112 else if (Opc == AMDGPU::V_ACCVGPR_READ_B32_e64)
3113 NeedWaitStates = AccVGPRWriteAccVgprReadWaitStates;
3114
3115 WaitStatesNeededForUse = NeedWaitStates -
3116 getWaitStatesSinceDef(Reg, IsAccVgprWriteFn, MaxWaitStates);
3117 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3118
3119 if (WaitStatesNeeded == MaxWaitStates)
3120 return WaitStatesNeeded; // Early exit.
3121 }
3122
3123 if (Opc == AMDGPU::V_ACCVGPR_WRITE_B32_e64) {
3124 const int MFMA4x4ReadSrcCAccVgprWriteWaitStates = 0;
3125 const int MFMA16x16ReadSrcCAccVgprWriteWaitStates = 5;
3126 const int MFMA32x32ReadSrcCAccVgprWriteWaitStates = 13;
3127 const int MaxWaitStates = 13;
3128 Register DstReg = MI->getOperand(0).getReg();
3129 unsigned HazardDefLatency = 0;
3130
3131 auto IsSrcCMFMAFn = [DstReg, &HazardDefLatency,
3132 this](const MachineInstr &MI) {
3133 if (!SIInstrInfo::isMFMA(MI))
3134 return false;
3135 Register Reg = TII.getNamedOperand(MI, AMDGPU::OpName::src2)->getReg();
3136 HazardDefLatency =
3137 std::max(HazardDefLatency, TSchedModel.computeInstrLatency(&MI));
3138 return TRI.regsOverlap(Reg, DstReg);
3139 };
3140
3141 int WaitStatesSince = getWaitStatesSince(IsSrcCMFMAFn, MaxWaitStates);
3142 int NeedWaitStates;
3143 switch (HazardDefLatency) {
3144 case 2: NeedWaitStates = MFMA4x4ReadSrcCAccVgprWriteWaitStates;
3145 break;
3146 case 8: NeedWaitStates = MFMA16x16ReadSrcCAccVgprWriteWaitStates;
3147 break;
3148 case 16: [[fallthrough]];
3149 default: NeedWaitStates = MFMA32x32ReadSrcCAccVgprWriteWaitStates;
3150 break;
3151 }
3152
3153 int WaitStatesNeededForUse = NeedWaitStates - WaitStatesSince;
3154 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3155 }
3156
3157 // Pad neighboring MFMA with noops for better inter-wave performance.
3158 WaitStatesNeeded = std::max(WaitStatesNeeded, checkMFMAPadding(MI));
3159
3160 return WaitStatesNeeded;
3161}
3162
3163/// One MFMA can be written with up to four opcodes that differ only in how vdst
3164/// and src2 are encoded: both are either AGPRs or VGPRs, and the mac form ties
3165/// vdst to src2 instead of taking them as separate operands. \returns the AGPR,
3166/// non-mac opcode, so that every form of the same MFMA maps to one value.
3167static unsigned getMFMANonMacAGPRFormOp(unsigned Opc) {
3168 if (int NonMacOp = AMDGPU::getMFMAEarlyClobberOp(Opc); NonMacOp != -1)
3169 Opc = NonMacOp;
3170 if (int AGPROp = AMDGPU::getAGPRFormOp(Opc); AGPROp != -1)
3171 Opc = AGPROp;
3172 return Opc;
3173}
3174
3175/// \returns true if \p Opc0 and \p Opc1 are the same MFMA, ignoring the mac
3176/// form and whether vdst/src2 are AGPRs or VGPRs.
3177static bool isSameMFMA(unsigned Opc0, unsigned Opc1) {
3179}
3180
3181static int
3183 bool IsGFX950) {
3184 // xdl def cycles | gfx940 | gfx950
3185 // 2 pass | 3 4
3186 // 4 pass | 5 6
3187 // 8 pass | 9 10
3188 // 16 pass | 17 18
3189 return NumPasses + 1 + IsGFX950;
3190}
3191
3192static int
3194 bool IsGFX950) {
3195 // xdl def cycles | gfx940 | gfx950
3196 // 2 pass | 3 3
3197 // 4 pass | 5 6
3198 // 8 pass | 9 10
3199 // 16 pass | 17 18
3200 return NumPasses + 1 + (NumPasses != 2 && IsGFX950);
3201}
3202
3203static int
3205 // 2 pass -> 2
3206 // 4 pass -> 4
3207 // 8 pass -> 8
3208 // 16 pass -> 16
3209 return NumPasses;
3210}
3211
3212static int
3214 // 2 pass -> 4
3215 // 4 pass -> 6
3216 // 8 pass -> 10
3217 // 16 pass -> 18
3218 return NumPasses + 2;
3219}
3220
3222 bool IsGFX950) {
3223 // xdl def cycles | gfx942 | gfx950
3224 // 2 pass | 5 5
3225 // 4 pass | 7 8
3226 // 8 pass | 11 12
3227 // 16 pass | 19 20
3228 return NumPasses + 3 + (NumPasses != 2 && IsGFX950);
3229}
3230
3231int GCNHazardRecognizer::getMFMAOverlappedSrcCWaitStates(
3232 const MachineInstr *Reader, const MachineInstr *Writer) const {
3233 constexpr int SMFMA4x4WritesVGPROverlappedSMFMASrcCWaitStates = 2;
3234 constexpr int SMFMA16x16WritesVGPROverlappedSMFMASrcCWaitStates = 8;
3235 constexpr int SMFMA32x32WritesVGPROverlappedSMFMASrcCWaitStates = 16;
3236 constexpr int SMFMA4x4WritesVGPROverlappedDMFMASrcCWaitStates = 3;
3237 constexpr int SMFMA16x16WritesVGPROverlappedDMFMASrcCWaitStates = 9;
3238 constexpr int SMFMA32x32WritesVGPROverlappedDMFMASrcCWaitStates = 17;
3239 constexpr int DMFMA16x16WritesVGPROverlappedSrcCWaitStates = 9;
3240 constexpr int GFX950_DMFMA16x16WritesVGPROverlappedSrcCWaitStates = 17;
3241 constexpr int DMFMA4x4WritesVGPROverlappedSrcCWaitStates = 4;
3242
3243 // An XDL read of a non-XDL result needs no wait states. DGEMM is never XDL,
3244 // so this also covers the f64 writers handled below.
3245 if (TII.isXDL(*Reader) && !TII.isXDL(*Writer))
3246 return 0;
3247
3248 switch (Writer->getOpcode()) {
3249 case AMDGPU::V_MFMA_F64_16X16X4F64_e64:
3250 case AMDGPU::V_MFMA_F64_16X16X4F64_vgprcd_e64:
3251 case AMDGPU::V_MFMA_F64_16X16X4F64_mac_e64:
3252 case AMDGPU::V_MFMA_F64_16X16X4F64_mac_vgprcd_e64:
3253 return ST.hasGFX950Insts()
3254 ? GFX950_DMFMA16x16WritesVGPROverlappedSrcCWaitStates
3255 : DMFMA16x16WritesVGPROverlappedSrcCWaitStates;
3256 case AMDGPU::V_MFMA_F64_4X4X4F64_e64:
3257 case AMDGPU::V_MFMA_F64_4X4X4F64_vgprcd_e64:
3258 return DMFMA4x4WritesVGPROverlappedSrcCWaitStates;
3259 default:
3260 break;
3261 }
3262
3263 int NumPasses = TSchedModel.computeInstrLatency(Writer);
3264 if (ST.hasGFX940Insts()) {
3265 if (!TII.isXDL(*Writer))
3267 NumPasses);
3268 return TII.isXDL(*Reader)
3270 NumPasses, ST.hasGFX950Insts())
3272 NumPasses, ST.hasGFX950Insts());
3273 }
3274
3275 bool IsDGEMM = SIInstrInfo::isDGEMM(Reader->getOpcode());
3276 switch (NumPasses) {
3277 case 2:
3278 return IsDGEMM ? SMFMA4x4WritesVGPROverlappedDMFMASrcCWaitStates
3279 : SMFMA4x4WritesVGPROverlappedSMFMASrcCWaitStates;
3280 case 8:
3281 return IsDGEMM ? SMFMA16x16WritesVGPROverlappedDMFMASrcCWaitStates
3282 : SMFMA16x16WritesVGPROverlappedSMFMASrcCWaitStates;
3283 case 16:
3284 return IsDGEMM ? SMFMA32x32WritesVGPROverlappedDMFMASrcCWaitStates
3285 : SMFMA32x32WritesVGPROverlappedSMFMASrcCWaitStates;
3286 default:
3287 llvm_unreachable("unexpected number of passes");
3288 }
3289}
3290
3291int GCNHazardRecognizer::checkMAIHazards90A(MachineInstr *MI) const {
3292 int WaitStatesNeeded = 0;
3293 unsigned Opc = MI->getOpcode();
3294
3295 auto IsLegacyVALUFn = [](const MachineInstr &MI) {
3296 return SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true) &&
3298 };
3299
3300 auto IsLegacyVALUNotDotFn = [](const MachineInstr &MI) {
3301 return SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true) &&
3303 };
3304
3305 if (!SIInstrInfo::isMFMA(*MI))
3306 return WaitStatesNeeded;
3307
3308 const int VALUWritesExecWaitStates = 4;
3309 int WaitStatesNeededForUse = VALUWritesExecWaitStates -
3310 getWaitStatesSinceDef(AMDGPU::EXEC, IsLegacyVALUFn,
3311 VALUWritesExecWaitStates);
3312 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3313
3314 int SrcCIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2);
3315
3316 // Loop for both DGEMM and S/HGEMM 2nd instruction.
3317 for (const MachineOperand &Use : MI->explicit_uses()) {
3318 const int LegacyVALUNotDotWritesVGPRWaitStates = 2;
3319 const int SMFMA4x4WritesVGPROverlappedSrcABWaitStates = 5;
3320 const int SMFMA16x16WritesVGPROverlappedSrcABWaitStates = 11;
3321 const int SMFMA32x32WritesVGPROverlappedSrcABWaitStates = 19;
3322 const int DMFMA4x4WritesVGPROverlappedMFMASrcABWaitStates = 6;
3323 const int DMFMA16x16WritesVGPROverlappedMFMASrcABWaitStates = 11;
3324 const int GFX950_DMFMA16x16WritesVGPROverlappedMFMASrcABWaitStates = 19;
3325 const int DMFMA4x4WritesVGPRFullSrcCWaitStates = 4;
3326 const int GFX940_SMFMA4x4WritesVGPRFullSrcCWaitStates = 2;
3327 const int MaxWaitStates =
3329 16, ST.hasGFX950Insts());
3330
3331 if (!Use.isReg())
3332 continue;
3333 Register Reg = Use.getReg();
3334 bool FullReg;
3335 const MachineInstr *MI1;
3336
3337 auto IsOverlappedMFMAFn = [Reg, &FullReg, &MI1,
3338 this](const MachineInstr &MI) {
3339 if (!SIInstrInfo::isMFMA(MI))
3340 return false;
3341 Register DstReg = MI.getOperand(0).getReg();
3342 FullReg = (DstReg == Reg);
3343 MI1 = &MI;
3344 return TRI.regsOverlap(DstReg, Reg);
3345 };
3346
3347 WaitStatesNeededForUse = LegacyVALUNotDotWritesVGPRWaitStates -
3348 getWaitStatesSinceDef(Reg, IsLegacyVALUNotDotFn, MaxWaitStates);
3349 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3350
3351 int NumWaitStates =
3352 getWaitStatesSinceDef(Reg, IsOverlappedMFMAFn, MaxWaitStates);
3353 if (NumWaitStates == std::numeric_limits<int>::max())
3354 continue;
3355
3356 int OpNo = Use.getOperandNo();
3357 unsigned Opc1 = MI1->getOpcode();
3358 int NeedWaitStates = 0;
3359 if (OpNo == SrcCIdx) {
3360 if (!SIInstrInfo::isDGEMM(Opc) &&
3361 (!ST.hasGFX940Insts() && SIInstrInfo::isDGEMM(Opc1))) {
3362 NeedWaitStates = 0;
3363 } else if (FullReg) {
3364 if ((Opc == AMDGPU::V_MFMA_F64_4X4X4F64_e64 ||
3365 Opc == AMDGPU::V_MFMA_F64_4X4X4F64_vgprcd_e64) &&
3366 (Opc1 == AMDGPU::V_MFMA_F64_4X4X4F64_e64 ||
3367 Opc1 == AMDGPU::V_MFMA_F64_4X4X4F64_vgprcd_e64))
3368 NeedWaitStates = DMFMA4x4WritesVGPRFullSrcCWaitStates;
3369 else if (ST.hasGFX940Insts() &&
3370 TSchedModel.computeInstrLatency(MI1) == 2)
3371 NeedWaitStates = GFX940_SMFMA4x4WritesVGPRFullSrcCWaitStates;
3372
3373 // The accumulator forwarding path that allows zero wait states is only
3374 // available while the chain stays on a single MFMA. Two different MFMAs
3375 // sharing an accumulator need the wait states of a partial overlap.
3376 if (ST.hasGFX940Insts() && !isSameMFMA(Opc, Opc1)) {
3377 NeedWaitStates = std::max(NeedWaitStates,
3378 getMFMAOverlappedSrcCWaitStates(MI, MI1));
3379 }
3380 } else {
3381 NeedWaitStates = getMFMAOverlappedSrcCWaitStates(MI, MI1);
3382 }
3383 } else {
3384 switch (Opc1) {
3385 case AMDGPU::V_MFMA_F64_16X16X4F64_e64:
3386 case AMDGPU::V_MFMA_F64_16X16X4F64_vgprcd_e64:
3387 case AMDGPU::V_MFMA_F64_16X16X4F64_mac_e64:
3388 case AMDGPU::V_MFMA_F64_16X16X4F64_mac_vgprcd_e64:
3389 NeedWaitStates =
3390 ST.hasGFX950Insts()
3391 ? GFX950_DMFMA16x16WritesVGPROverlappedMFMASrcABWaitStates
3392 : DMFMA16x16WritesVGPROverlappedMFMASrcABWaitStates;
3393 break;
3394 case AMDGPU::V_MFMA_F64_4X4X4F64_e64:
3395 case AMDGPU::V_MFMA_F64_4X4X4F64_vgprcd_e64:
3396 NeedWaitStates = DMFMA4x4WritesVGPROverlappedMFMASrcABWaitStates;
3397 break;
3398 default:
3399 int NumPasses = TSchedModel.computeInstrLatency(MI1);
3400
3401 if (ST.hasGFX940Insts()) {
3402 NeedWaitStates =
3403 TII.isXDL(*MI1)
3405 NumPasses, ST.hasGFX950Insts())
3407 NumPasses);
3408 break;
3409 }
3410
3411 switch (NumPasses) {
3412 case 2:
3413 NeedWaitStates = SMFMA4x4WritesVGPROverlappedSrcABWaitStates;
3414 break;
3415 case 4:
3416 llvm_unreachable("unexpected number of passes for mfma");
3417 case 8:
3418 NeedWaitStates = SMFMA16x16WritesVGPROverlappedSrcABWaitStates;
3419 break;
3420 case 16:
3421 default:
3422 NeedWaitStates = SMFMA32x32WritesVGPROverlappedSrcABWaitStates;
3423 }
3424 }
3425 }
3426 assert(NeedWaitStates <= MaxWaitStates &&
3427 "hazard requirement exceeds the scan window");
3428 if (WaitStatesNeeded >= NeedWaitStates)
3429 continue;
3430
3431 WaitStatesNeededForUse = NeedWaitStates - NumWaitStates;
3432 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3433
3434 if (WaitStatesNeeded == MaxWaitStates)
3435 break;
3436 }
3437
3438 // Pad neighboring MFMA with noops for better inter-wave performance.
3439 WaitStatesNeeded = std::max(WaitStatesNeeded, checkMFMAPadding(MI));
3440
3441 return WaitStatesNeeded;
3442}
3443
3444int GCNHazardRecognizer::checkMAILdStHazards(MachineInstr *MI) const {
3445 // On gfx90a+ relevant hazards are checked in checkMAIVALUHazards()
3446 if (!ST.hasMAIInsts() || ST.hasGFX90AInsts())
3447 return 0;
3448
3449 int WaitStatesNeeded = 0;
3450
3451 auto IsAccVgprReadFn = [](const MachineInstr &MI) {
3452 return MI.getOpcode() == AMDGPU::V_ACCVGPR_READ_B32_e64;
3453 };
3454
3455 for (const MachineOperand &Op : MI->explicit_uses()) {
3456 if (!Op.isReg() || !TRI.isVGPR(MF.getRegInfo(), Op.getReg()))
3457 continue;
3458
3459 Register Reg = Op.getReg();
3460
3461 const int AccVgprReadLdStWaitStates = 2;
3462 const int VALUWriteAccVgprRdWrLdStDepVALUWaitStates = 1;
3463 const int MaxWaitStates = 2;
3464
3465 int WaitStatesNeededForUse = AccVgprReadLdStWaitStates -
3466 getWaitStatesSinceDef(Reg, IsAccVgprReadFn, MaxWaitStates);
3467 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3468
3469 if (WaitStatesNeeded == MaxWaitStates)
3470 return WaitStatesNeeded; // Early exit.
3471
3472 auto IsVALUAccVgprRdWrCheckFn = [Reg, this](const MachineInstr &MI) {
3473 if (MI.getOpcode() != AMDGPU::V_ACCVGPR_READ_B32_e64 &&
3474 MI.getOpcode() != AMDGPU::V_ACCVGPR_WRITE_B32_e64)
3475 return false;
3476 auto IsVALUFn = [](const MachineInstr &MI) {
3477 return SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true) &&
3479 };
3480 return getWaitStatesSinceDef(Reg, IsVALUFn, 2 /*MaxWaitStates*/) <
3481 std::numeric_limits<int>::max();
3482 };
3483
3484 WaitStatesNeededForUse = VALUWriteAccVgprRdWrLdStDepVALUWaitStates -
3485 getWaitStatesSince(IsVALUAccVgprRdWrCheckFn, MaxWaitStates);
3486 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3487 }
3488
3489 return WaitStatesNeeded;
3490}
3491
3492int GCNHazardRecognizer::checkPermlaneHazards(MachineInstr *MI) const {
3493 assert(!ST.hasVcmpxPermlaneHazard() &&
3494 "this is a different vcmpx+permlane hazard");
3495 const SIRegisterInfo *TRI = ST.getRegisterInfo();
3496 const SIInstrInfo *TII = ST.getInstrInfo();
3497
3498 auto IsVCmpXWritesExecFn = [TII, TRI](const MachineInstr &MI) {
3499 return isVCmpXWritesExec(*TII, *TRI, MI);
3500 };
3501
3502 auto IsVALUFn = [](const MachineInstr &MI) {
3503 return SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true);
3504 };
3505
3506 const int VCmpXWritesExecWaitStates = 4;
3507 const int VALUWritesVDstWaitStates = 2;
3508 int WaitStatesNeeded = 0;
3509
3510 for (const MachineOperand &Op : MI->explicit_uses()) {
3511 if (!Op.isReg() || !TRI->isVGPR(MF.getRegInfo(), Op.getReg()))
3512 continue;
3513 Register Reg = Op.getReg();
3514
3515 int WaitStatesSinceDef =
3516 VALUWritesVDstWaitStates -
3517 getWaitStatesSinceDef(Reg, IsVALUFn,
3518 /*MaxWaitStates=*/VALUWritesVDstWaitStates);
3519 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesSinceDef);
3520 if (WaitStatesNeeded >= VALUWritesVDstWaitStates)
3521 break;
3522 }
3523
3524 int VCmpXHazardWaits =
3525 VCmpXWritesExecWaitStates -
3526 getWaitStatesSince(IsVCmpXWritesExecFn, VCmpXWritesExecWaitStates);
3527
3528 WaitStatesNeeded = std::max(WaitStatesNeeded, VCmpXHazardWaits);
3529 return WaitStatesNeeded;
3530}
3531
3533 // 2 pass -> 4
3534 // 4 pass -> 6
3535 // 8 pass -> 10
3536 // 16 pass -> 18
3537 return NumPasses + 2;
3538}
3539
3541 bool IsGFX950) {
3542 // xdl def cycles | gfx942 | gfx950
3543 // 2 pass | 5 5
3544 // 4 pass | 7 8
3545 // 8 pass | 11 12
3546 // 16 pass | 19 20
3547 return NumPasses + 3 + (NumPasses != 2 && IsGFX950);
3548}
3549
3551 bool IsGFX950) {
3552 // xdl def cycles | gfx942 | gfx950
3553 // 2 pass | 5 5
3554 // 4 pass | 7 8
3555 // 8 pass | 11 12
3556 // 16 pass | 19 20
3557 return NumPasses + 3 + (NumPasses != 2 && IsGFX950);
3558}
3559
3561 // 2 pass -> 4
3562 // 4 pass -> 6
3563 // 8 pass -> 10
3564 // 16 pass -> 18
3565 return NumPasses + 2;
3566}
3567
3568int GCNHazardRecognizer::checkMAIVALUHazards(MachineInstr *MI) const {
3569 if (!ST.hasGFX90AInsts())
3570 return 0;
3571
3572 auto IsDGEMMFn = [](const MachineInstr &MI) -> bool {
3573 return SIInstrInfo::isDGEMM(MI.getOpcode());
3574 };
3575
3576 // This is checked in checkMAIHazards90A()
3577 if (SIInstrInfo::isMFMA(*MI))
3578 return 0;
3579
3580 const MachineRegisterInfo &MRI = MF.getRegInfo();
3581
3582 int WaitStatesNeeded = 0;
3583
3584 bool IsMem = SIInstrInfo::isVMEM(*MI) || SIInstrInfo::isDS(*MI);
3585 bool IsMemOrExport = IsMem || SIInstrInfo::isEXP(*MI);
3586 bool IsVALU = SIInstrInfo::isVALU(*MI, /*AllowLDSDMA=*/true);
3587
3588 const MachineInstr *MFMA = nullptr;
3589 unsigned Reg;
3590 auto IsMFMAWriteFn = [&Reg, &MFMA, this](const MachineInstr &MI) {
3591 if (!SIInstrInfo::isMFMA(MI) ||
3592 !TRI.regsOverlap(MI.getOperand(0).getReg(), Reg))
3593 return false;
3594 MFMA = &MI;
3595 return true;
3596 };
3597
3598 const MachineInstr *DOT = nullptr;
3599 auto IsDotWriteFn = [&Reg, &DOT, this](const MachineInstr &MI) {
3600 if (!SIInstrInfo::isDOT(MI) ||
3601 !TRI.regsOverlap(MI.getOperand(0).getReg(), Reg))
3602 return false;
3603 DOT = &MI;
3604 return true;
3605 };
3606
3607 bool DGEMMAfterVALUWrite = false;
3608 auto IsDGEMMHazard = [&DGEMMAfterVALUWrite, this](const MachineInstr &MI) {
3609 // Found DGEMM on reverse traversal to def.
3610 if (SIInstrInfo::isDGEMM(MI.getOpcode()))
3611 DGEMMAfterVALUWrite = true;
3612
3613 // Only hazard if register is defined by a VALU and a DGEMM is found after
3614 // after the def.
3615 if (!TII.isVALU(MI, /*AllowLDSDMA=*/true) || !DGEMMAfterVALUWrite)
3616 return false;
3617
3618 return true;
3619 };
3620
3621 int SrcCIdx = AMDGPU::getNamedOperandIdx(MI->getOpcode(),
3622 AMDGPU::OpName::src2);
3623
3624 if (IsMemOrExport || IsVALU) {
3625 const int SMFMA4x4WriteVgprVALUMemExpReadWaitStates = 5;
3626 const int SMFMA16x16WriteVgprVALUMemExpReadWaitStates = 11;
3627 const int SMFMA32x32WriteVgprVALUMemExpReadWaitStates = 19;
3628 const int DMFMA4x4WriteVgprMemExpReadWaitStates = 9;
3629 const int DMFMA16x16WriteVgprMemExpReadWaitStates = 18;
3630 const int DMFMA4x4WriteVgprVALUReadWaitStates = 6;
3631 const int DMFMA16x16WriteVgprVALUReadWaitStates = 11;
3632 const int GFX950_DMFMA16x16WriteVgprVALUReadWaitStates = 19;
3633 const int DotWriteSameDotReadSrcAB = 3;
3634 const int DotWriteDifferentVALURead = 3;
3635 const int DMFMABetweenVALUWriteVMEMRead = 2;
3636 const int MaxWaitStates =
3638 ST.hasGFX950Insts());
3639
3640 for (const MachineOperand &Use : MI->explicit_uses()) {
3641 if (!Use.isReg())
3642 continue;
3643 Reg = Use.getReg();
3644
3645 DOT = nullptr;
3646 int WaitStatesSinceDef = getWaitStatesSinceDef(Reg, IsDotWriteFn,
3647 MaxWaitStates);
3648 if (DOT) {
3649 int NeedWaitStates = 0;
3650 if (DOT->getOpcode() == MI->getOpcode()) {
3651 if (&Use - &MI->getOperand(0) != SrcCIdx)
3652 NeedWaitStates = DotWriteSameDotReadSrcAB;
3653 } else {
3654 NeedWaitStates = DotWriteDifferentVALURead;
3655 }
3656
3657 int WaitStatesNeededForUse = NeedWaitStates - WaitStatesSinceDef;
3658 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3659 }
3660
3661 // Workaround for HW data hazard bug observed only in GFX90A. When there
3662 // is a DGEMM instruction in-between a VALU and a VMEM instruction it
3663 // causes the SQ to incorrectly not insert two wait states between the two
3664 // instructions needed to avoid data hazard.
3665 if (IsMem && ST.hasGFX90AInsts() && !ST.hasGFX940Insts()) {
3666 DGEMMAfterVALUWrite = false;
3667 if (TRI.isVectorRegister(MRI, Reg)) {
3668 int WaitStatesNeededForUse =
3669 DMFMABetweenVALUWriteVMEMRead -
3670 getWaitStatesSinceDef(Reg, IsDGEMMHazard,
3671 DMFMABetweenVALUWriteVMEMRead);
3672
3673 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3674 }
3675 }
3676
3677 MFMA = nullptr;
3678 WaitStatesSinceDef =
3679 getWaitStatesSinceDef(Reg, IsMFMAWriteFn, MaxWaitStates);
3680 if (!MFMA)
3681 continue;
3682
3683 unsigned HazardDefLatency = TSchedModel.computeInstrLatency(MFMA);
3684 int NumPasses = HazardDefLatency;
3685 int NeedWaitStates = MaxWaitStates;
3686
3687 if (SIInstrInfo::isDGEMM(MFMA->getOpcode())) {
3688 switch (HazardDefLatency) {
3689 case 4:
3690 NeedWaitStates = IsMemOrExport ? DMFMA4x4WriteVgprMemExpReadWaitStates
3691 : DMFMA4x4WriteVgprVALUReadWaitStates;
3692 break;
3693 case 8:
3694 case 16:
3695 NeedWaitStates =
3696 IsMemOrExport
3697 ? DMFMA16x16WriteVgprMemExpReadWaitStates
3698 : (ST.hasGFX950Insts()
3699 ? GFX950_DMFMA16x16WriteVgprVALUReadWaitStates
3700 : DMFMA16x16WriteVgprVALUReadWaitStates);
3701 break;
3702 default:
3703 llvm_unreachable("unexpected dgemm");
3704 }
3705 } else if (ST.hasGFX940Insts()) {
3706 NeedWaitStates =
3707 TII.isXDL(*MFMA)
3709 NumPasses, ST.hasGFX950Insts())
3711 NumPasses);
3712 } else {
3713 switch (HazardDefLatency) {
3714 case 2:
3715 NeedWaitStates = SMFMA4x4WriteVgprVALUMemExpReadWaitStates;
3716 break;
3717 case 8:
3718 NeedWaitStates = SMFMA16x16WriteVgprVALUMemExpReadWaitStates;
3719 break;
3720 case 16:
3721 NeedWaitStates = SMFMA32x32WriteVgprVALUMemExpReadWaitStates;
3722 break;
3723 default:
3724 llvm_unreachable("unexpected number of passes for mfma");
3725 }
3726 }
3727
3728 assert(NeedWaitStates <= MaxWaitStates &&
3729 "hazard requirement exceeds the scan window");
3730 int WaitStatesNeededForUse = NeedWaitStates - WaitStatesSinceDef;
3731 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3732
3733 if (WaitStatesNeeded == MaxWaitStates)
3734 break;
3735 }
3736 }
3737
3738 unsigned Opc = MI->getOpcode();
3739 const int DMFMAToFMA64WaitStates = 2;
3740 if ((Opc == AMDGPU::V_FMA_F64_e64 ||
3741 Opc == AMDGPU::V_FMAC_F64_e32 || Opc == AMDGPU::V_FMAC_F64_e64 ||
3742 Opc == AMDGPU::V_FMAC_F64_dpp) &&
3743 WaitStatesNeeded < DMFMAToFMA64WaitStates) {
3744 int WaitStatesNeededForUse = DMFMAToFMA64WaitStates -
3745 getWaitStatesSince(IsDGEMMFn, DMFMAToFMA64WaitStates);
3746 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3747 }
3748
3749 if (!IsVALU && !IsMemOrExport)
3750 return WaitStatesNeeded;
3751
3752 for (const MachineOperand &Def : MI->defs()) {
3753 const int SMFMA4x4WriteVgprVALUWawWaitStates = 5;
3754 const int SMFMA16x16WriteVgprVALUWawWaitStates = 11;
3755 const int SMFMA32x32WriteVgprVALUWawWaitStates = 19;
3756 const int SMFMA4x4ReadVgprVALUWarWaitStates = 1;
3757 const int GFX940_XDL4PassReadVgprVALUWarWaitStates = 3;
3758 const int SMFMA16x16ReadVgprVALUWarWaitStates = 7;
3759 const int SMFMA32x32ReadVgprVALUWarWaitStates = 15;
3760 const int DMFMA4x4WriteVgprVALUWriteWaitStates = 6;
3761 const int DMFMA16x16WriteVgprVALUWriteWaitStates = 11;
3762 const int DotWriteDifferentVALUWrite = 3;
3763 const int MaxWaitStates =
3764 GFX940_XDL_N_PassWriteVgprVALUWawWaitStates(16, ST.hasGFX950Insts());
3765 const int MaxWarWaitStates = 15;
3766
3767 Reg = Def.getReg();
3768
3769 DOT = nullptr;
3770 int WaitStatesSinceDef = getWaitStatesSinceDef(Reg, IsDotWriteFn,
3771 MaxWaitStates);
3772 if (DOT && DOT->getOpcode() != MI->getOpcode())
3773 WaitStatesNeeded = std::max(WaitStatesNeeded, DotWriteDifferentVALUWrite -
3774 WaitStatesSinceDef);
3775
3776 MFMA = nullptr;
3777 WaitStatesSinceDef =
3778 getWaitStatesSinceDef(Reg, IsMFMAWriteFn, MaxWaitStates);
3779 if (MFMA) {
3780 int NeedWaitStates = MaxWaitStates;
3781 int NumPasses = TSchedModel.computeInstrLatency(MFMA);
3782
3783 if (SIInstrInfo::isDGEMM(MFMA->getOpcode())) {
3784 switch (NumPasses) {
3785 case 4:
3786 NeedWaitStates = DMFMA4x4WriteVgprVALUWriteWaitStates;
3787 break;
3788 case 8:
3789 case 16:
3790 NeedWaitStates = DMFMA16x16WriteVgprVALUWriteWaitStates;
3791 break;
3792 default:
3793 llvm_unreachable("unexpected number of cycles for dgemm");
3794 }
3795 } else if (ST.hasGFX940Insts()) {
3796 NeedWaitStates =
3797 TII.isXDL(*MFMA)
3799 NumPasses, ST.hasGFX950Insts())
3801 } else {
3802 switch (NumPasses) {
3803 case 2:
3804 NeedWaitStates = SMFMA4x4WriteVgprVALUWawWaitStates;
3805 break;
3806 case 8:
3807 NeedWaitStates = SMFMA16x16WriteVgprVALUWawWaitStates;
3808 break;
3809 case 16:
3810 NeedWaitStates = SMFMA32x32WriteVgprVALUWawWaitStates;
3811 break;
3812 default:
3813 llvm_unreachable("Unexpected number of passes for mfma");
3814 }
3815 }
3816
3817 assert(NeedWaitStates <= MaxWaitStates &&
3818 "hazard requirement exceeds the scan window");
3819 int WaitStatesNeededForUse = NeedWaitStates - WaitStatesSinceDef;
3820 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3821
3822 if (WaitStatesNeeded == MaxWaitStates)
3823 break;
3824 }
3825
3826 auto IsSMFMAReadAsCFn = [&Reg, &MFMA, this](const MachineInstr &MI) {
3827 if (!SIInstrInfo::isMFMA(MI) || SIInstrInfo::isDGEMM(MI.getOpcode()) ||
3828 !MI.readsRegister(Reg, &TRI))
3829 return false;
3830
3831 if (ST.hasGFX940Insts() && !TII.isXDL(MI))
3832 return false;
3833
3834 const MachineOperand *SrcC =
3835 TII.getNamedOperand(MI, AMDGPU::OpName::src2);
3836 assert(SrcC);
3837 if (!SrcC->isReg() || !TRI.regsOverlap(SrcC->getReg(), Reg))
3838 return false;
3839
3840 MFMA = &MI;
3841 return true;
3842 };
3843
3844 MFMA = nullptr;
3845 int WaitStatesSinceUse = getWaitStatesSince(IsSMFMAReadAsCFn,
3846 MaxWarWaitStates);
3847 if (!MFMA)
3848 continue;
3849
3850 unsigned HazardDefLatency = TSchedModel.computeInstrLatency(MFMA);
3851 int NeedWaitStates = MaxWaitStates;
3852 switch (HazardDefLatency) {
3853 case 2: NeedWaitStates = SMFMA4x4ReadVgprVALUWarWaitStates;
3854 break;
3855 case 4: assert(ST.hasGFX940Insts());
3856 NeedWaitStates = GFX940_XDL4PassReadVgprVALUWarWaitStates;
3857 break;
3858 case 8: NeedWaitStates = SMFMA16x16ReadVgprVALUWarWaitStates;
3859 break;
3860 case 16: [[fallthrough]];
3861 default: NeedWaitStates = SMFMA32x32ReadVgprVALUWarWaitStates;
3862 break;
3863 }
3864
3865 int WaitStatesNeededForUse = NeedWaitStates - WaitStatesSinceUse;
3866 WaitStatesNeeded = std::max(WaitStatesNeeded, WaitStatesNeededForUse);
3867 }
3868
3869 return WaitStatesNeeded;
3870}
3871
3873 if (!SU->isInstr())
3874 return false;
3875
3876 const MachineInstr *MAI = nullptr;
3877
3878 auto IsMFMAFn = [&MAI](const MachineInstr &MI) {
3879 MAI = nullptr;
3881 MAI = &MI;
3882 return MAI != nullptr;
3883 };
3884
3885 MachineInstr *MI = SU->getInstr();
3886 if (IsMFMAFn(*MI)) {
3887 int W = getWaitStatesSince(IsMFMAFn, 16);
3888 if (MAI)
3889 return W < (int)TSchedModel.computeInstrLatency(MAI);
3890 }
3891
3892 return false;
3893}
3894
3895// Adjust global offsets for instructions bundled with S_GETPC_B64 after
3896// insertion of a new instruction.
3897static void updateGetPCBundle(MachineInstr *NewMI) {
3898 if (!NewMI->isBundled())
3899 return;
3900
3901 // Find start of bundle.
3902 auto I = NewMI->getIterator();
3903 while (I->isBundledWithPred())
3904 I--;
3905 if (I->isBundle())
3906 I++;
3907
3908 // Bail if this is not an S_GETPC bundle.
3909 if (I->getOpcode() != AMDGPU::S_GETPC_B64)
3910 return;
3911
3912 // Update offsets of any references in the bundle.
3913 const unsigned NewBytes = 4;
3914 assert(NewMI->getOpcode() == AMDGPU::S_WAITCNT_DEPCTR &&
3915 "Unexpected instruction insertion in bundle");
3916 auto NextMI = std::next(NewMI->getIterator());
3917 auto End = NewMI->getParent()->end();
3918 while (NextMI != End && NextMI->isBundledWithPred()) {
3919 for (auto &Operand : NextMI->operands()) {
3920 if (Operand.isGlobal())
3921 Operand.setOffset(Operand.getOffset() + NewBytes);
3922 }
3923 NextMI++;
3924 }
3925}
3926
3927bool GCNHazardRecognizer::fixVALUMaskWriteHazard(MachineInstr *MI) {
3928 if (!ST.hasVALUMaskWriteHazard())
3929 return false;
3930 assert(!ST.hasExtendedWaitCounts());
3931
3932 if (!ST.isWave64())
3933 return false;
3934
3935 const bool IsSALU = SIInstrInfo::isSALU(*MI);
3936 const bool IsVALU = SIInstrInfo::isVALU(*MI, /*AllowLDSDMA=*/true);
3937 if (!IsSALU && !IsVALU)
3938 return false;
3939
3940 // The hazard sequence is three instructions:
3941 // 1. VALU reads SGPR as mask
3942 // 2. VALU/SALU writes SGPR
3943 // 3. VALU/SALU reads SGPR
3944 // The hazard can expire if the distance between 2 and 3 is sufficient,
3945 // or (2) is VALU and (3) is SALU.
3946 // In practice this happens <10% of the time, hence always assume the hazard
3947 // exists if (1) and (2) are present to avoid searching all SGPR reads.
3948
3949 const SIRegisterInfo *TRI = ST.getRegisterInfo();
3950 const MachineRegisterInfo &MRI = MF.getRegInfo();
3951
3952 auto IgnoreableSGPR = [](const Register Reg) {
3953 switch (Reg) {
3954 case AMDGPU::EXEC:
3955 case AMDGPU::EXEC_LO:
3956 case AMDGPU::EXEC_HI:
3957 case AMDGPU::M0:
3958 case AMDGPU::SGPR_NULL:
3959 case AMDGPU::SGPR_NULL64:
3960 case AMDGPU::SCC:
3961 return true;
3962 default:
3963 return false;
3964 }
3965 };
3966 auto IsVCC = [](const Register Reg) {
3967 return Reg == AMDGPU::VCC || Reg == AMDGPU::VCC_LO || Reg == AMDGPU::VCC_HI;
3968 };
3969
3970 struct StateType {
3971 SmallSet<Register, 2> HazardSGPRs;
3972
3973 static unsigned getHashValue(const StateType &State) {
3974 return hash_combine_range(State.HazardSGPRs);
3975 }
3976 static bool isEqual(const StateType &LHS, const StateType &RHS) {
3977 return LHS.HazardSGPRs == RHS.HazardSGPRs;
3978 }
3979 };
3980
3981 SmallVector<const MachineInstr *> WaitInstrs;
3982 StateType InitialState;
3983
3984 // Look for SGPR write.
3985 MachineOperand *HazardDef = nullptr;
3986 for (MachineOperand &Op : MI->all_defs()) {
3987 Register Reg = Op.getReg();
3988 if (IgnoreableSGPR(Reg))
3989 continue;
3990 if (!IsVCC(Reg)) {
3991 if (Op.isImplicit())
3992 continue;
3993 if (!TRI->isSGPRReg(MRI, Reg))
3994 continue;
3995 }
3996
3997 HazardDef = &Op;
3998 break;
3999 }
4000
4001 if (!HazardDef)
4002 return false;
4003
4004 // Setup to track writes to individual SGPRs
4005 const Register HazardReg = HazardDef->getReg();
4006 if (AMDGPU::SReg_32RegClass.contains(HazardReg)) {
4007 InitialState.HazardSGPRs.insert(HazardReg);
4008 } else {
4009 assert(AMDGPU::SReg_64RegClass.contains(HazardReg));
4010 InitialState.HazardSGPRs.insert(TRI->getSubReg(HazardReg, AMDGPU::sub0));
4011 InitialState.HazardSGPRs.insert(TRI->getSubReg(HazardReg, AMDGPU::sub1));
4012 }
4013
4014 auto IsHazardFn = [&](StateType &State, const MachineInstr &I) {
4015 if (State.HazardSGPRs.empty())
4016 return HazardExpired;
4017
4018 switch (I.getOpcode()) {
4019 case AMDGPU::V_ADDC_U32_e32:
4020 case AMDGPU::V_ADDC_U32_dpp:
4021 case AMDGPU::V_CNDMASK_B16_t16_e32:
4022 case AMDGPU::V_CNDMASK_B16_fake16_e32:
4023 case AMDGPU::V_CNDMASK_B16_t16_dpp:
4024 case AMDGPU::V_CNDMASK_B16_fake16_dpp:
4025 case AMDGPU::V_CNDMASK_B32_e32:
4026 case AMDGPU::V_CNDMASK_B32_dpp:
4027 case AMDGPU::V_DIV_FMAS_F32_e64:
4028 case AMDGPU::V_DIV_FMAS_F64_e64:
4029 case AMDGPU::V_SUBB_U32_e32:
4030 case AMDGPU::V_SUBB_U32_dpp:
4031 case AMDGPU::V_SUBBREV_U32_e32:
4032 case AMDGPU::V_SUBBREV_U32_dpp: {
4033 // These implicitly read VCC as mask source.
4034 return IsVCC(HazardReg) ? HazardFound : NoHazardFound;
4035 }
4036 case AMDGPU::V_ADDC_U32_e64:
4037 case AMDGPU::V_ADDC_U32_e64_dpp:
4038 case AMDGPU::V_CNDMASK_B16_t16_e64:
4039 case AMDGPU::V_CNDMASK_B16_fake16_e64:
4040 case AMDGPU::V_CNDMASK_B16_t16_e64_dpp:
4041 case AMDGPU::V_CNDMASK_B16_fake16_e64_dpp:
4042 case AMDGPU::V_CNDMASK_B32_e64:
4043 case AMDGPU::V_CNDMASK_B32_e64_dpp:
4044 case AMDGPU::V_SUBB_U32_e64:
4045 case AMDGPU::V_SUBB_U32_e64_dpp:
4046 case AMDGPU::V_SUBBREV_U32_e64:
4047 case AMDGPU::V_SUBBREV_U32_e64_dpp: {
4048 // Only check mask register overlaps.
4049 const MachineOperand *SSRCOp = TII.getNamedOperand(I, AMDGPU::OpName::src2);
4050 assert(SSRCOp);
4051 bool Result = TRI->regsOverlap(SSRCOp->getReg(), HazardReg);
4052 return Result ? HazardFound : NoHazardFound;
4053 }
4054 default:
4055 return NoHazardFound;
4056 }
4057 };
4058
4059 auto UpdateStateFn = [&](StateType &State, const MachineInstr &I) {
4060 // Update tracking of SGPR writes.
4061 for (auto &Op : I.all_defs()) {
4062 Register Reg = Op.getReg();
4063 if (IgnoreableSGPR(Reg))
4064 continue;
4065 if (!IsVCC(Reg)) {
4066 if (Op.isImplicit())
4067 continue;
4068 if (!TRI->isSGPRReg(MRI, Reg))
4069 continue;
4070 }
4071
4072 // Stop tracking any SGPRs with writes on the basis that they will
4073 // already have an appropriate wait inserted afterwards.
4075 for (Register SGPR : State.HazardSGPRs) {
4076 if (Reg == SGPR || TRI->regsOverlap(Reg, SGPR))
4077 Found.push_back(SGPR);
4078 }
4079 for (Register SGPR : Found)
4080 State.HazardSGPRs.erase(SGPR);
4081 }
4082 };
4083
4084 // Check for hazard
4085 if (!hasHazard<StateType>(InitialState, IsHazardFn, UpdateStateFn,
4086 MI->getParent(),
4087 std::next(MI->getReverseIterator())))
4088 return false;
4089
4090 // Compute counter mask
4091 unsigned DepCtr =
4092 IsVALU ? (IsVCC(HazardReg) ? AMDGPU::DepCtr::encodeFieldVaVcc(0, ST)
4093 : AMDGPU::DepCtr::encodeFieldVaSdst(0, ST))
4094 : AMDGPU::DepCtr::encodeFieldSaSdst(0, ST);
4095
4096 // Add s_waitcnt_depctr after SGPR write.
4097 auto NextMI = std::next(MI->getIterator());
4098 auto NewMI = BuildMI(*MI->getParent(), NextMI, MI->getDebugLoc(),
4099 TII.get(AMDGPU::S_WAITCNT_DEPCTR))
4100 .addImm(DepCtr);
4101
4102 // SALU write may be s_getpc in a bundle.
4103 updateGetPCBundle(NewMI);
4104
4105 return true;
4106}
4107
4108static bool ensureEntrySetPrio(MachineFunction *MF, int Priority,
4109 const SIInstrInfo &TII) {
4110 MachineBasicBlock &EntryMBB = MF->front();
4111 if (EntryMBB.begin() != EntryMBB.end()) {
4112 auto &EntryMI = *EntryMBB.begin();
4113 if (EntryMI.getOpcode() == AMDGPU::S_SETPRIO &&
4114 EntryMI.getOperand(0).getImm() >= Priority)
4115 return false;
4116 }
4117
4118 BuildMI(EntryMBB, EntryMBB.begin(), DebugLoc(), TII.get(AMDGPU::S_SETPRIO))
4119 .addImm(Priority);
4120 return true;
4121}
4122
4123bool GCNHazardRecognizer::fixRequiredExportPriority(MachineInstr *MI) {
4124 if (!ST.hasRequiredExportPriority())
4125 return false;
4126
4127 // Assume the following shader types will never have exports,
4128 // and avoid adding or adjusting S_SETPRIO.
4129 MachineBasicBlock *MBB = MI->getParent();
4130 MachineFunction *MF = MBB->getParent();
4131 auto CC = MF->getFunction().getCallingConv();
4132 switch (CC) {
4137 return false;
4138 default:
4139 break;
4140 }
4141
4142 const int MaxPriority = 3;
4143 const int NormalPriority = 2;
4144 const int PostExportPriority = 0;
4145
4146 auto It = MI->getIterator();
4147 switch (MI->getOpcode()) {
4148 case AMDGPU::S_ENDPGM:
4149 case AMDGPU::S_ENDPGM_SAVED:
4150 case AMDGPU::S_ENDPGM_ORDERED_PS_DONE:
4151 case AMDGPU::SI_RETURN_TO_EPILOG:
4152 // Ensure shader with calls raises priority at entry.
4153 // This ensures correct priority if exports exist in callee.
4154 if (MF->getFrameInfo().hasCalls())
4155 return ensureEntrySetPrio(MF, NormalPriority, TII);
4156 return false;
4157 case AMDGPU::S_SETPRIO: {
4158 // Raise minimum priority unless in workaround.
4159 auto &PrioOp = MI->getOperand(0);
4160 int Prio = PrioOp.getImm();
4161 bool InWA = (Prio == PostExportPriority) &&
4162 (It != MBB->begin() && TII.isEXP(*std::prev(It)));
4163 if (InWA || Prio >= NormalPriority)
4164 return false;
4165 PrioOp.setImm(std::min(Prio + NormalPriority, MaxPriority));
4166 return true;
4167 }
4168 default:
4169 if (!TII.isEXP(*MI))
4170 return false;
4171 break;
4172 }
4173
4174 // Check entry priority at each export (as there will only be a few).
4175 // Note: amdgpu_gfx can only be a callee, so defer to caller setprio.
4176 bool Changed = false;
4178 Changed = ensureEntrySetPrio(MF, NormalPriority, TII);
4179
4180 auto NextMI = std::next(It);
4181 bool EndOfShader = false;
4182 if (NextMI != MBB->end()) {
4183 // Only need WA at end of sequence of exports.
4184 if (TII.isEXP(*NextMI))
4185 return Changed;
4186 // Assume appropriate S_SETPRIO after export means WA already applied.
4187 if (NextMI->getOpcode() == AMDGPU::S_SETPRIO &&
4188 NextMI->getOperand(0).getImm() == PostExportPriority)
4189 return Changed;
4190 EndOfShader = NextMI->getOpcode() == AMDGPU::S_ENDPGM;
4191 }
4192
4193 const DebugLoc &DL = MI->getDebugLoc();
4194
4195 // Lower priority.
4196 BuildMI(*MBB, NextMI, DL, TII.get(AMDGPU::S_SETPRIO))
4197 .addImm(PostExportPriority);
4198
4199 if (!EndOfShader) {
4200 // Wait for exports to complete.
4201 BuildMI(*MBB, NextMI, DL, TII.get(AMDGPU::S_WAITCNT_EXPCNT))
4202 .addReg(AMDGPU::SGPR_NULL)
4203 .addImm(0);
4204 }
4205
4206 BuildMI(*MBB, NextMI, DL, TII.get(AMDGPU::S_NOP)).addImm(0);
4207 BuildMI(*MBB, NextMI, DL, TII.get(AMDGPU::S_NOP)).addImm(0);
4208
4209 if (!EndOfShader) {
4210 // Return to normal (higher) priority.
4211 BuildMI(*MBB, NextMI, DL, TII.get(AMDGPU::S_SETPRIO))
4212 .addImm(NormalPriority);
4213 }
4214
4215 return true;
4216}
4217
4218// Advance past meta instructions (debug values, labels, CFI, KILL, etc.) to the
4219// next instruction that actually issues. Unlike skipDebugInstructionsForward /
4220// next_nodbg, this skips the full isMetaInstruction() set.
4224 while (I != End && I->isMetaInstruction())
4225 ++I;
4226 return I;
4227}
4228
4229bool GCNHazardRecognizer::fixVPermPk16Hazard(MachineInstr *MI) {
4230 // Requirement #1 of 2:
4231 // The cross-wave entry-block mitigation is delegated to the mandatory
4232 // unclaused-VMEM entry prologue (GLOBAL_PREFETCH_B8 + V_NOP).
4233 assert(ST.hasRequiresInitialUnclausedVmem() &&
4234 "V_PERM_PK16-hazard subtarget must provide the unclaused-VMEM entry "
4235 "prologue to satisfy the cross-wave entry mitigation");
4236
4237 if (!SIInstrInfo::isVPermPk16(MI->getOpcode()))
4238 return false;
4239
4240 MachineBasicBlock *MBB = MI->getParent();
4241
4242 // Requirement #2 of 2:
4243 // V_PERM_PK16 must be immediately followed by a safe instruction.
4245 skipMetaInstructionsForward(std::next(MI->getIterator()), MBB->end());
4246 if (NextI != MBB->end() && TII.isVPermPk16SafeInstr(*NextI))
4247 return false;
4248
4249 // EXEC is guaranteed non-zero here: V_PERM_PK16 reports unwanted effects
4250 // when EXEC is empty, so s_cbranch_execz over this region is retained.
4251 // A plain V_NOP is therefore a real VALU nop and clears the hazard.
4252 BuildMI(*MBB, NextI, MI->getDebugLoc(), TII.get(AMDGPU::V_NOP_e32));
4253 return true;
4254}
4255
4256bool GCNHazardRecognizer::fixGetRegWaitIdle(MachineInstr *MI) {
4257 if (!isSGetReg(MI->getOpcode()))
4258 return false;
4259
4260 const SIInstrInfo *TII = ST.getInstrInfo();
4261 switch (getHWReg(TII, *MI)) {
4262 default:
4263 return false;
4268 break;
4269 }
4270
4271 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(),
4272 TII->get(AMDGPU::S_WAITCNT_DEPCTR))
4273 .addImm(0);
4274 return true;
4275}
4276
4277bool GCNHazardRecognizer::fixDsAtomicAsyncBarrierArriveB64(MachineInstr *MI) {
4278 if (MI->getOpcode() != AMDGPU::DS_ATOMIC_ASYNC_BARRIER_ARRIVE_B64)
4279 return false;
4280
4281 const SIInstrInfo *TII = ST.getInstrInfo();
4282 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(),
4283 TII->get(AMDGPU::S_WAITCNT_DEPCTR))
4285 BuildMI(*MI->getParent(), std::next(MI->getIterator()), MI->getDebugLoc(),
4286 TII->get(AMDGPU::S_WAITCNT_DEPCTR))
4288
4289 return true;
4290}
4291
4292bool GCNHazardRecognizer::fixScratchBaseForwardingHazard(MachineInstr *MI) {
4293 // No reason to check this in pre-RA scheduling, SGPRs have to be allocated
4294 // for hazard to trigger.
4296 return false;
4297
4298 const SIRegisterInfo *TRI = ST.getRegisterInfo();
4299 const SIInstrInfo *TII = ST.getInstrInfo();
4300 // Hazard expires after 10 SGPR writes by SALU or 8 SGPR writes by VALU.
4301 const int FlatScrBaseWaitStates = 10;
4302
4303 bool ReadsFlatScrLo =
4304 MI->readsRegister(AMDGPU::SRC_FLAT_SCRATCH_BASE_LO, TRI);
4305 bool ReadsFlatScrHi =
4306 MI->readsRegister(AMDGPU::SRC_FLAT_SCRATCH_BASE_HI, TRI);
4307 if (isSGetReg(MI->getOpcode())) {
4308 switch (getHWReg(TII, *MI)) {
4309 default:
4310 break;
4312 ReadsFlatScrLo = true;
4313 break;
4315 ReadsFlatScrHi = true;
4316 break;
4317 }
4318 }
4319
4320 const MachineRegisterInfo &MRI = MF.getRegInfo();
4321
4322 auto IsRegDefHazard = [&](Register Reg) -> bool {
4323 DenseSet<const MachineBasicBlock *> Visited;
4324 auto IsHazardFn = [TRI, Reg](const MachineInstr &MI) {
4325 return MI.modifiesRegister(Reg, TRI);
4326 };
4327
4328 // This literally abuses the idea of waitstates. Instead of waitstates it
4329 // returns 1 for SGPR written and 0 otherwise.
4330 auto IsSGPRDef = [TII, TRI, &MRI](const MachineInstr &MI) -> unsigned {
4331 if (!TII->isSALU(MI) && !TII->isVALU(MI, /*AllowLDSDMA=*/true))
4332 return 0;
4333 for (const MachineOperand &MO : MI.all_defs()) {
4334 if (TRI->isSGPRReg(MRI, MO.getReg()))
4335 return 1;
4336 }
4337 return 0;
4338 };
4339
4340 auto IsExpiredFn = [=](const MachineInstr &MI, int SgprWrites) {
4341 if (MI.getOpcode() == AMDGPU::S_WAITCNT_DEPCTR) {
4342 unsigned Wait = MI.getOperand(0).getImm();
4345 return true;
4346 }
4347 return SgprWrites >= FlatScrBaseWaitStates;
4348 };
4349
4350 return ::getWaitStatesSince(
4351 IsHazardFn, MI->getParent(), std::next(MI->getReverseIterator()),
4352 0, IsExpiredFn, Visited, IsSGPRDef) < FlatScrBaseWaitStates;
4353 };
4354
4355 if ((!ReadsFlatScrLo || MRI.isConstantPhysReg(AMDGPU::SGPR102) ||
4356 !IsRegDefHazard(AMDGPU::SGPR102)) &&
4357 (!ReadsFlatScrHi || MRI.isConstantPhysReg(AMDGPU::SGPR103) ||
4358 !IsRegDefHazard(AMDGPU::SGPR103)))
4359 return false;
4360
4361 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(),
4362 TII->get(AMDGPU::S_WAITCNT_DEPCTR))
4365 return true;
4366}
4367
4368bool GCNHazardRecognizer::fixSetRegMode(MachineInstr *MI) {
4369 if (!isSSetReg(MI->getOpcode()) ||
4370 MI->getOperand(1).getImm() != AMDGPU::Hwreg::ID_MODE)
4371 return false;
4372
4373 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(), TII.get(AMDGPU::V_NOP_e32));
4374 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(), TII.get(AMDGPU::V_NOP_e32));
4375 return true;
4376}
4377
4378bool GCNHazardRecognizer::fixTDM(MachineInstr *MI) {
4379 auto IsTDM = [&](const MachineInstr &MI) -> bool {
4381 MI.getOpcode() != AMDGPU::S_WAIT_TENSORCNT;
4382 };
4383
4384 if (!IsTDM(*MI))
4385 return false;
4386
4387 auto IsExpiredFn = [](const MachineInstr &MI, int) {
4388 if (MI.getOpcode() != AMDGPU::S_WAIT_TENSORCNT)
4389 return false;
4390 return MI.getOperand(0).getImm() <= 10;
4391 };
4392
4393 if (::getWaitStatesSince(IsTDM, MI, IsExpiredFn) ==
4394 std::numeric_limits<int>::max())
4395 return false;
4396
4397 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(),
4398 TII.get(AMDGPU::S_WAIT_TENSORCNT))
4399 .addImm(10);
4400 return true;
4401}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
AMDGPU Rewrite AGPR Copy MFMA
The AMDGPU TargetMachine interface definition for hw codegen targets.
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static bool isEqual(const Function &Caller, const Function &Callee)
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static cl::opt< unsigned, false, MFMAPaddingRatioParser > MFMAPaddingRatio("amdgpu-mfma-padding-ratio", cl::init(0), cl::Hidden, cl::desc("Fill a percentage of the latency between " "neighboring MFMA with s_nops."))
static bool shouldRunLdsBranchVmemWARHazardFixup(const MachineFunction &MF, const GCNSubtarget &ST)
static cl::opt< bool > EnableWMMAVnopHoisting("amdgpu-wmma-vnop-hoisting", cl::init(true), cl::Hidden, cl::desc("Hoist WMMA hazard V_NOPs from loops to preheaders"))
static bool consumesDstSelForwardingOperand(const MachineInstr *VALU, const MachineOperand *Dst, const SIRegisterInfo *TRI)
Checks whether the provided MI "consumes" the operand with a Dest sel fowarding issue Dst .
static bool isSGetReg(unsigned Opcode)
static bool breaksSMEMSoftClause(MachineInstr *MI)
static bool isLdsDma(const MachineInstr &MI)
static int GFX940_XDL_N_PassWritesVGPROverlappedSrcABWaitStates(int NumPasses, bool IsGFX950)
static unsigned getWMMAHazardInstInCategory(const MachineInstr &MI, const SIInstrInfo *TII, const TargetSchedModel &SchedModel, const GCNSubtarget &ST)
static bool isRFE(unsigned Opcode)
static bool isRWLane(unsigned Opcode)
static bool isSMovRel(unsigned Opcode)
static unsigned getMFMANonMacAGPRFormOp(unsigned Opc)
One MFMA can be written with up to four opcodes that differ only in how vdst and src2 are encoded: bo...
#define DEBUG_TYPE_VERBOSE
static const MachineOperand * getDstSelForwardingOperand(const MachineInstr &MI, const GCNSubtarget &ST)
Dest sel forwarding issue occurs if additional logic is needed to swizzle / pack the computed value i...
static int GFX940_XDL_N_PassWritesVGPROverlappedSGEMMDGEMMSrcCWaitStates(int NumPasses, bool IsGFX950)
static void updateGetPCBundle(MachineInstr *NewMI)
static int GFX940_XDL_N_PassWriteVgprVALUMemExpReadWaitStates(int NumPasses, bool IsGFX950)
static bool isStoreCountWaitZero(const MachineInstr &I)
static bool breaksVMEMSoftClause(MachineInstr *MI)
static bool isVCmpXWritesExec(const SIInstrInfo &TII, const SIRegisterInfo &TRI, const MachineInstr &MI)
static bool isSSetReg(unsigned Opcode)
static void addRegUnits(const SIRegisterInfo &TRI, BitVector &BV, MCRegister Reg)
static unsigned getHWReg(const SIInstrInfo *TII, const MachineInstr &RegInstr)
static bool isSameMFMA(unsigned Opc0, unsigned Opc1)
static bool isDivFMas(unsigned Opcode)
static bool hasHazard(StateT InitialState, function_ref< HazardFnResult(StateT &, const MachineInstr &)> IsHazard, function_ref< void(StateT &, const MachineInstr &)> UpdateState, const MachineBasicBlock *InitialMBB, MachineBasicBlock::const_reverse_instr_iterator InitialI)
static int getWaitStatesSince(GCNHazardRecognizer::IsHazardFn IsHazard, const MachineBasicBlock *MBB, MachineBasicBlock::const_reverse_instr_iterator I, int WaitStates, GCNHazardRecognizer::IsExpiredFn IsExpired, DenseSet< const MachineBasicBlock * > &Visited, GCNHazardRecognizer::GetNumWaitStatesFn GetNumWaitStates=SIInstrInfo::getNumWaitStates)
static int GFX940_SMFMA_N_PassWritesVGPROverlappedSrcABWaitStates(int NumPasses)
static int GFX940_XDL_N_PassWriteVgprVALUWawWaitStates(int NumPasses, bool IsGFX950)
static int GFX940_SMFMA_N_PassWriteVgprVALUMemExpReadWaitStates(int NumPasses)
static MachineBasicBlock::iterator skipMetaInstructionsForward(MachineBasicBlock::iterator I, MachineBasicBlock::iterator End)
static int GFX940_SMFMA_N_PassWritesVGPROverlappedSMFMASrcCWaitStates(int NumPasses)
static bool isCoexecutableVALUInst(const MachineInstr &MI)
static bool ensureEntrySetPrio(MachineFunction *MF, int Priority, const SIInstrInfo &TII)
static void addRegsToSet(const SIRegisterInfo &TRI, iterator_range< MachineInstr::const_mop_iterator > Ops, BitVector &DefSet, BitVector &UseSet)
static void insertNoopsInBundle(MachineInstr *MI, const SIInstrInfo &TII, unsigned Quantity)
static bool isSendMsgTraceDataOrGDS(const SIInstrInfo &TII, const MachineInstr &MI)
static cl::opt< unsigned > NopPadding("amdgpu-snop-padding", cl::init(0), cl::Hidden, cl::desc("Insert a s_nop x before every instruction"))
static bool isPermlane(const MachineInstr &MI)
static int GFX940_SMFMA_N_PassWriteVgprVALUWawWaitStates(int NumPasses)
static int GFX940_XDL_N_PassWritesVGPROverlappedXDLOrSMFMASrcCWaitStates(int NumPasses, bool IsGFX950)
AMD GCN specific subclass of TargetSubtarget.
static Register UseReg(const MachineOperand &MO)
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static llvm::Error parse(GsymDataExtractor &Data, uint64_t BaseAddr, LineEntryCallback const &Callback)
Definition LineTable.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
#define H(x, y, z)
Definition MD5.cpp:56
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
Func MI getDebugLoc()))
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
#define LLVM_DEBUG(...)
Definition Debug.h:119
#define DEBUG_WITH_TYPE(TYPE,...)
DEBUG_WITH_TYPE macro - This macro should be used by passes to emit debug information.
Definition Debug.h:72
Value * RHS
Value * LHS
static const uint32_t IV[8]
Definition blake3_impl.h:83
unsigned get(InstCounterType T) const
BitVector & set()
Set all bits in the bitvector.
Definition BitVector.h:366
A debug info location.
Definition DebugLoc.h:126
std::pair< iterator, bool > insert_as(std::pair< KeyT, ValueT > &&KV, const LookupKeyT &Val)
Alternate version of insert() which allows a different, and possibly less expensive,...
Definition DenseMap.h:890
Implements a dense probed hash-table based set.
Definition DenseSet.h:281
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
unsigned getHazardWaitStates(MachineInstr *MI) const
Returns the number of wait states until all hazards for MI are resolved.
unsigned PreEmitNoopsCommon(MachineInstr *) const
OperatingMode
Operating mode for the hazard recognizer.
void EmitNoop() override
EmitNoop - This callback is invoked when a noop was added to the instruction stream.
void Reset() override
Reset - This callback is invoked when a new block of instructions is about to be schedule.
unsigned PreEmitNoops(MachineInstr *) override
This overload will be used when the hazard recognizer is being used by a non-scheduling pass,...
void EmitInstruction(SUnit *SU) override
EmitInstruction - This callback is invoked when an instruction is emitted, to advance the hazard stat...
function_ref< bool(const MachineInstr &)> IsHazardFn
void AdvanceCycle() override
AdvanceCycle - This callback is invoked whenever the next top-down instruction to be scheduled cannot...
function_ref< unsigned int(const MachineInstr &)> GetNumWaitStatesFn
bool ShouldPreferAnother(SUnit *SU) const override
ShouldPreferAnother - This callback may be invoked if getHazardType returns NoHazard.
bool hasPhysRegs() const
Returns true if instruction operands are physical registers, so that hazards defined by register depe...
function_ref< bool(const MachineInstr &, int WaitStates)> IsExpiredFn
bool isSchedulerMode() const
Returns true if running as a scheduler (pre-RA or post-RA).
GCNHazardRecognizer(const MachineFunction &MF, OperatingMode Mode, MachineLoopInfo *MLI=nullptr)
Construct with explicit operating mode.
static AMDGPU::CoExecMaskT getCoExecMaskForMI(const MachineInstr &MI, const SIInstrInfo &TII)
Get the CoExecMask for a given instruction.
HazardType getHazardType(SUnit *SU, int Stalls) override
getHazardType - Return the hazard type of emitting this node.
void RecedeCycle() override
RecedeCycle - This callback is invoked whenever the next bottom-up instruction to be scheduled cannot...
bool isHazardRecognizerMode() const
Returns true if running as the standalone hazard recognizer pass.
bool isPreRA() const
Returns true if running in pre-RA scheduling mode.
BlockT * getLoopPreheader() const
If there is a preheader for this loop, return it.
LoopT * getParentLoop() const
Return the parent loop if it exists or nullptr for top level loops.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
Instructions::const_reverse_iterator const_reverse_instr_iterator
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
Instructions::iterator instr_iterator
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
MachineInstrBundleIterator< MachineInstr > iterator
Function & getFunction()
Return the LLVM function that this machine code represents.
const MachineBasicBlock & front() const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
bool readsRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr reads the specified register.
bool isBundled() const
Return true if this instruction part of a bundle.
MachineOperand class - Representation of each machine instruction operand.
void setImm(int64_t immVal)
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
void setIsKill(bool Val=true)
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool isConstantPhysReg(MCRegister PhysReg) const
Returns true if PhysReg is unallocatable and constant throughout the function.
LLVM_ABI bool isPhysRegUsed(MCRegister PhysReg, bool SkipRegMaskTest=false) const
Return true if the specified register is modified or read in this function.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
static bool isDS(const MachineInstr &MI)
static bool isVMEM(const MachineInstr &MI)
static bool isSMRD(const MachineInstr &MI)
static bool isMTBUF(const MachineInstr &MI)
static bool isDGEMM(unsigned Opcode)
static bool isEXP(const MachineInstr &MI)
static bool isSALU(const MachineInstr &MI)
static bool isSDWA(const MachineInstr &MI)
static bool isDOT(const MachineInstr &MI)
static bool usesTENSOR_CNT(const MachineInstr &MI)
static bool isSWMMAC(const MachineInstr &MI)
static bool isLDSDIR(const MachineInstr &MI)
static bool isVPermPk16(unsigned Opcode)
static bool isVALU(const MachineInstr &MI, bool AllowLDSDMA)
static bool isTRANS(const MachineInstr &MI)
static bool isMUBUF(const MachineInstr &MI)
static bool isWaitcnt(unsigned Opcode)
static bool isDPP(const MachineInstr &MI)
static bool isMFMA(const MachineInstr &MI)
static bool isMAI(const MCInstrDesc &Desc)
static bool isFPAtomic(const MachineInstr &MI)
static bool isMIMG(const MachineInstr &MI)
static unsigned getNumWaitStates(const MachineInstr &MI)
Return the number of wait states that result from executing this instruction.
static bool isWMMA(const MachineInstr &MI)
static bool isLDSDMA(const MachineInstr &MI)
Scheduling unit. This is a node in the scheduling DAG.
bool isInstr() const
Returns true if this SUnit refers to a machine instruction as opposed to an SDNode.
MachineInstr * getInstr() const
Returns the representative MachineInstr for this SUnit.
unsigned MaxLookAhead
MaxLookAhead - Indicate the number of cycles in the scoreboard state.
virtual void EmitNoops(unsigned Quantity)
EmitNoops - This callback is invoked when noops were added to the instruction stream.
size_type size() const
Determine the number of elements in the SetVector.
Definition SetVector.h:103
bool insert(const value_type &X)
Insert a new element into the SetVector.
Definition SetVector.h:157
A SetVector that performs no allocations if smaller than a certain size.
Definition SetVector.h:345
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
Definition SmallSet.h:184
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Definition StringRef.h:490
Provide an instruction scheduling machine model to CodeGen passes.
std::pair< iterator, bool > insert(const ValueT &V)
Definition DenseSet.h:209
An efficient, type-erasing, non-owning reference to a callable.
self_iterator getIterator()
Definition ilist_node.h:123
A range adaptor for a pair of iterators.
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
unsigned encodeFieldVaVcc(unsigned Encoded, unsigned VaVcc)
unsigned encodeFieldVaVdst(unsigned Encoded, unsigned VaVdst)
unsigned decodeFieldSaSdst(unsigned Encoded)
unsigned decodeFieldVaSdst(unsigned Encoded)
unsigned encodeFieldVmVsrc(unsigned Encoded, unsigned VmVsrc)
unsigned encodeFieldSaSdst(unsigned Encoded, unsigned SaSdst)
unsigned decodeFieldVaVdst(unsigned Encoded)
unsigned decodeFieldVmVsrc(unsigned Encoded)
unsigned encodeFieldVaSdst(unsigned Encoded, unsigned VaSdst)
const char * getCoExecMaskName(CoExecMaskT Mask)
Return a human-readable name for a mask holding a single instruction class, as produced by getCoExecM...
LLVM_READONLY const MIMGInfo * getMIMGInfo(unsigned Opc)
StringRef getSchedStrategy(const Function &F)
constexpr unsigned MaxCoExecStages
Max stages: INT8 16x16x64 = 17 cycles, round up for safety.
FPType getFPDstSelType(unsigned Opc)
bool isGFX12Plus(const MCSubtargetInfo &STI)
const char * getStageTypeName(CoExecStageType T)
LLVM_ABI IsaVersion getIsaVersion(StringRef GPU)
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
LLVM_READONLY int32_t getMFMAEarlyClobberOp(uint32_t Opcode)
CoExecMask CoExecMaskT
Waitcnt decodeWaitcnt(const IsaVersion &Version, unsigned Encoded)
CoExecInfo getCoExecInfo(const MachineInstr &MI, const SIInstrInfo &TII)
Get co-execution info for a WMMA instruction, selecting the per-cycle slot pattern from the opcode (a...
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
LLVM_READONLY int32_t getAGPRFormOp(uint32_t Opcode)
InstructionFlavor classifyFlavor(const MachineInstr &MI, const SIInstrInfo &SII)
Classify MI into the execution flavor that drives both the scheduler's slot preferences and the hazar...
constexpr CoExecMaskT getCoExecMask(InstructionFlavor F)
Map a flavor to the co-execution class it occupies in a window slot.
bool isGFX1250(const MCSubtargetInfo &STI)
@ Entry
Definition COFF.h:862
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ AMDGPU_Gfx
Used for AMD graphics targets.
@ AMDGPU_CS_ChainPreserve
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_CS_Chain
Used on AMDGPUs to give the middle-end more control over argument placement.
This namespace contains all of the command line option processing machinery.
Definition MCSchedule.h:35
initializer< Ty > init(const Ty &Val)
constexpr double e
NodeAddr< DefNode * > Def
Definition RDFGraph.h:384
NodeAddr< UseNode * > Use
Definition RDFGraph.h:385
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
@ Offset
Definition DWP.cpp:577
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
@ Kill
The last use of a register.
@ Undef
Value of the register doesn't matter.
@ Define
Register definition.
@ Wait
Definition Threading.h:60
constexpr RegState getDeadRegState(bool B)
Op::Description Desc
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
DWARFExpression::Operation Op
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
Definition InstrProf.h:147
hash_code hash_combine(const Ts &...args)
Combine values into a single hash_code.
Definition Hashing.h:307
LLVM_ABI Printable printMBBReference(const MachineBasicBlock &MBB)
Prints a machine basic block reference.
hash_code hash_combine_range(InputIteratorT first, InputIteratorT last)
Compute a hash_code for a sequence of values.
Definition Hashing.h:287
Co-execution characteristics for a multi-cycle instruction.
static std::tuple< typename Fields::ValueType... > decode(uint64_t Encoded)
An information struct used to provide DenseMap with the various necessary components for a given valu...