LLVM 24.0.0git
SIPreEmitPeephole.cpp
Go to the documentation of this file.
1//===-- SIPreEmitPeephole.cpp ------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This pass performs the peephole optimizations before code emission.
11///
12/// Additionally, this pass also unpacks packed instructions (V_PK_MUL_F32/F16,
13/// V_PK_ADD_F32/F16, V_PK_FMA_F32) adjacent to MFMAs such that they can be
14/// co-issued. This helps with overlapping MFMA and certain vector instructions
15/// in machine schedules and is expected to improve performance. Only those
16/// packed instructions are unpacked that are overlapped by the MFMA latency.
17/// Rest should remain untouched.
18/// TODO: Add support for F16 packed instructions
19//===----------------------------------------------------------------------===//
20
21#include "AMDGPU.h"
22#include "GCNSubtarget.h"
23#include "llvm/ADT/Statistic.h"
29using namespace llvm;
30
31#define DEBUG_TYPE "si-pre-emit-peephole"
32
33STATISTIC(NumModeWritesRemoved,
34 "Number of redundant mode register writes removed");
35
36namespace {
37
38/// The state of one independent field of the MODE register, as tracked by
39/// removeRedundantModeWrites.
40struct ModeFieldState {
41 std::optional<int64_t> Value;
42 std::optional<int64_t> ValueBeforePendingWrite;
43 MachineInstr *PendingWrite = nullptr;
44
45 bool isTracked() const { return PendingWrite || Value; }
46};
47
48class SIPreEmitPeephole {
49private:
50 const SIInstrInfo *TII = nullptr;
51 const SIRegisterInfo *TRI = nullptr;
52 MachineLoopInfo *MLI = nullptr;
53
54 bool optimizeVccBranch(MachineInstr &MI) const;
55 void updateMLIBeforeRemovingEdge(MachineBasicBlock *From,
56 MachineBasicBlock *To) const;
57 bool optimizeSetGPR(MachineInstr &First, MachineInstr &MI) const;
58 bool getBlockDestinations(MachineBasicBlock &SrcMBB,
59 MachineBasicBlock *&TrueMBB,
60 MachineBasicBlock *&FalseMBB,
61 SmallVectorImpl<MachineOperand> &Cond);
62 bool mustRetainExeczBranch(const MachineInstr &Branch,
63 const MachineBasicBlock &From,
64 const MachineBasicBlock &To) const;
65 bool removeExeczBranch(MachineInstr &MI, MachineBasicBlock &SrcMBB);
66 bool removeRedundantModeWrites(MachineBasicBlock &SrcMBB) const;
67 // Creates a list of packed instructions following an MFMA that are suitable
68 // for unpacking.
69 void collectUnpackingCandidates(MachineInstr &BeginMI,
70 SetVector<MachineInstr *> &InstrsToUnpack,
71 uint16_t NumMFMACycles);
72 // v_pk_fma_f32 v[0:1], v[0:1], v[2:3], v[2:3] op_sel:[1,1,1]
73 // op_sel_hi:[0,0,0]
74 // ==>
75 // v_fma_f32 v0, v1, v3, v3
76 // v_fma_f32 v1, v0, v2, v2
77 // Here, we have overwritten v0 before we use it. This function checks if
78 // unpacking can lead to such a situation.
79 bool canUnpackingClobberRegister(const MachineInstr &MI);
80 // Unpack and insert F32 packed instructions, such as V_PK_MUL, V_PK_ADD, and
81 // V_PK_FMA. Currently, only V_PK_MUL, V_PK_ADD, V_PK_FMA are supported for
82 // this transformation.
83 void performF32Unpacking(MachineInstr &I);
84 // Select corresponding unpacked instruction
85 uint32_t mapToUnpackedOpcode(MachineInstr &I);
86 // Creates the unpacked instruction to be inserted. Adds source modifiers to
87 // the unpacked instructions based on the source modifiers in the packed
88 // instruction.
89 MachineInstrBuilder createUnpackedMI(MachineInstr &I, uint32_t UnpackedOpcode,
90 bool IsHiBits);
91 // Process operands/source modifiers from packed instructions and insert the
92 // appropriate source modifers and operands into the unpacked instructions.
93 void addOperandAndMods(MachineInstrBuilder &NewMI, unsigned SrcMods,
94 bool IsHiBits, const MachineOperand &SrcMO);
95
96public:
97 bool run(MachineFunction &MF, MachineLoopInfo *MLI);
98};
99
100class SIPreEmitPeepholeLegacy : public MachineFunctionPass {
101public:
102 static char ID;
103
104 SIPreEmitPeepholeLegacy() : MachineFunctionPass(ID) {}
105
106 void getAnalysisUsage(AnalysisUsage &AU) const override {
107 AU.addUsedIfAvailable<MachineLoopInfoWrapperPass>();
108 AU.addPreserved<MachineLoopInfoWrapperPass>();
110 }
111
112 bool runOnMachineFunction(MachineFunction &MF) override {
113 auto *MLIWrapper = getAnalysisIfAvailable<MachineLoopInfoWrapperPass>();
114 MachineLoopInfo *MLI = MLIWrapper ? &MLIWrapper->getLI() : nullptr;
115 return SIPreEmitPeephole().run(MF, MLI);
116 }
117};
118
119} // End anonymous namespace.
120
121INITIALIZE_PASS(SIPreEmitPeepholeLegacy, DEBUG_TYPE,
122 "SI peephole optimizations", false, false)
123
124char SIPreEmitPeepholeLegacy::ID = 0;
125
126char &llvm::SIPreEmitPeepholeID = SIPreEmitPeepholeLegacy::ID;
127
128void SIPreEmitPeephole::updateMLIBeforeRemovingEdge(
129 MachineBasicBlock *From, MachineBasicBlock *To) const {
130 if (!MLI)
131 return;
132
133 // Only handle back-edges: To must be a loop header with From inside the loop.
134 MachineLoop *Loop = MLI->getLoopFor(To);
135 if (!Loop || Loop->getHeader() != To || !Loop->contains(From))
136 return;
137
138 // Count back-edges
139 unsigned BackEdgeCount = 0;
140 for (MachineBasicBlock *Pred : To->predecessors()) {
141 if (Loop->contains(Pred))
142 BackEdgeCount++;
143 }
144
145 if (BackEdgeCount > 1)
146 return;
147
148 MachineLoop *ParentLoop = Loop->getParentLoop();
149
150 // Re-map blocks directly owned by this loop to the parent.
151 for (MachineBasicBlock *BB : Loop->blocks()) {
152 if (MLI->getLoopFor(BB) == Loop)
153 MLI->changeLoopFor(BB, ParentLoop);
154 }
155
156 // Reparent all child loops.
157 while (!Loop->isInnermost()) {
158 MachineLoop *Child = Loop->removeChildLoop(std::prev(Loop->end()));
159 if (ParentLoop)
160 ParentLoop->addChildLoop(Child);
161 else
162 MLI->addTopLevelLoop(Child);
163 }
164
165 if (ParentLoop)
166 ParentLoop->removeChildLoop(Loop);
167 else
168 MLI->removeLoop(llvm::find(*MLI, Loop));
169
170 MLI->destroy(Loop);
171}
172
173bool SIPreEmitPeephole::optimizeVccBranch(MachineInstr &MI) const {
174 // Match:
175 // sreg = -1 or 0
176 // vcc = S_AND_B64 exec, sreg or S_ANDN2_B64 exec, sreg
177 // S_CBRANCH_VCC[N]Z
178 // =>
179 // S_CBRANCH_EXEC[N]Z
180 // We end up with this pattern sometimes after basic block placement.
181 // It happens while combining a block which assigns -1 or 0 to a saved mask
182 // and another block which consumes that saved mask and then a branch.
183
184 bool Changed = false;
185 MachineBasicBlock &MBB = *MI.getParent();
186 const GCNSubtarget &ST = MBB.getParent()->getSubtarget<GCNSubtarget>();
187 const bool IsWave32 = ST.isWave32();
188 const unsigned CondReg = TRI->getVCC();
189 const unsigned ExecReg = IsWave32 ? AMDGPU::EXEC_LO : AMDGPU::EXEC;
190 const unsigned And = IsWave32 ? AMDGPU::S_AND_B32 : AMDGPU::S_AND_B64;
191 const unsigned AndN2 = IsWave32 ? AMDGPU::S_ANDN2_B32 : AMDGPU::S_ANDN2_B64;
192 const unsigned Mov = IsWave32 ? AMDGPU::S_MOV_B32 : AMDGPU::S_MOV_B64;
193
194 MachineBasicBlock::reverse_iterator A = MI.getReverseIterator(),
195 E = MBB.rend();
196 bool ReadsCond = false;
197 unsigned Threshold = 5;
198 for (++A; A != E; ++A) {
199 if (!--Threshold)
200 return false;
201 if (A->modifiesRegister(ExecReg, TRI))
202 return false;
203 if (A->modifiesRegister(CondReg, TRI)) {
204 if (!A->definesRegister(CondReg, TRI) ||
205 (A->getOpcode() != And && A->getOpcode() != AndN2))
206 return false;
207 break;
208 }
209 ReadsCond |= A->readsRegister(CondReg, TRI);
210 }
211 if (A == E)
212 return false;
213
214 MachineOperand &Op1 = A->getOperand(1);
215 MachineOperand &Op2 = A->getOperand(2);
216 if ((!Op1.isReg() || Op1.getReg() != ExecReg) && Op2.isReg() &&
217 Op2.getReg() == ExecReg) {
218 TII->commuteInstruction(*A);
219 Changed = true;
220 }
221 if (!Op1.isReg() || Op1.getReg() != ExecReg)
222 return Changed;
223 if (Op2.isImm() && !(Op2.getImm() == -1 || Op2.getImm() == 0))
224 return Changed;
225
226 int64_t MaskValue = 0;
228 if (Op2.isReg()) {
229 SReg = Op2.getReg();
230 auto M = std::next(A);
231 bool ReadsSreg = false;
232 for (; M != E; ++M) {
233 if (M->definesRegister(SReg, TRI))
234 break;
235 if (M->modifiesRegister(SReg, TRI))
236 return Changed;
237 ReadsSreg |= M->readsRegister(SReg, TRI);
238 }
239 if (M == E)
240 return Changed;
241
242 if (!M->isMoveImmediate() || !M->getOperand(1).isImm() ||
243 (M->getOperand(1).getImm() != -1 && M->getOperand(1).getImm() != 0))
244 return Changed;
245 MaskValue = M->getOperand(1).getImm();
246 // First if sreg is only used in the AND instruction fold the immediate
247 // into the AND.
248 if (!ReadsSreg && Op2.isKill()) {
249 A->getOperand(2).ChangeToImmediate(MaskValue);
250 M->eraseFromParent();
251 }
252 } else if (Op2.isImm()) {
253 MaskValue = Op2.getImm();
254 } else {
255 llvm_unreachable("Op2 must be register or immediate");
256 }
257
258 // Invert mask for s_andn2
259 assert(MaskValue == 0 || MaskValue == -1);
260 if (A->getOpcode() == AndN2)
261 MaskValue = ~MaskValue;
262
263 if (!ReadsCond && A->registerDefIsDead(AMDGPU::SCC, /*TRI=*/nullptr)) {
264 if (!MI.killsRegister(CondReg, TRI)) {
265 // Replace AND with MOV
266 if (MaskValue == 0) {
267 BuildMI(*A->getParent(), *A, A->getDebugLoc(), TII->get(Mov), CondReg)
268 .addImm(0);
269 } else {
270 BuildMI(*A->getParent(), *A, A->getDebugLoc(), TII->get(Mov), CondReg)
271 .addReg(ExecReg);
272 }
273 }
274 // Remove AND instruction
275 A->eraseFromParent();
276 }
277
278 bool IsVCCZ = MI.getOpcode() == AMDGPU::S_CBRANCH_VCCZ;
279 if (SReg == ExecReg) {
280 // EXEC is updated directly
281 if (IsVCCZ) {
282 MI.eraseFromParent();
283 return true;
284 }
285 MI.setDesc(TII->get(AMDGPU::S_BRANCH));
286 } else if (IsVCCZ && MaskValue == 0) {
287 // Will always branch
288 // Remove all successors shadowed by new unconditional branch
289 MachineBasicBlock *Parent = MI.getParent();
290 SmallVector<MachineInstr *, 4> ToRemove;
291 bool Found = false;
292 for (MachineInstr &Term : Parent->terminators()) {
293 if (Found) {
294 if (Term.isBranch())
295 ToRemove.push_back(&Term);
296 } else {
297 Found = Term.isIdenticalTo(MI);
298 }
299 }
300 assert(Found && "conditional branch is not terminator");
301 for (auto *BranchMI : ToRemove) {
302 MachineOperand &Dst = BranchMI->getOperand(0);
303 assert(Dst.isMBB() && "destination is not basic block");
304 updateMLIBeforeRemovingEdge(Parent, Dst.getMBB());
305 Parent->removeSuccessor(Dst.getMBB());
306 BranchMI->eraseFromParent();
307 }
308
309 if (MachineBasicBlock *Succ = Parent->getFallThrough()) {
310 updateMLIBeforeRemovingEdge(Parent, Succ);
311 Parent->removeSuccessor(Succ);
312 }
313
314 // Rewrite to unconditional branch
315 MI.setDesc(TII->get(AMDGPU::S_BRANCH));
316 } else if (!IsVCCZ && MaskValue == 0) {
317 // Will never branch
318 MachineOperand &Dst = MI.getOperand(0);
319 assert(Dst.isMBB() && "destination is not basic block");
320 MachineBasicBlock *Parent = MI.getParent();
321 updateMLIBeforeRemovingEdge(Parent, Dst.getMBB());
322 Parent->removeSuccessor(Dst.getMBB());
323 MI.eraseFromParent();
324 return true;
325 } else if (MaskValue == -1) {
326 // Depends only on EXEC
327 MI.setDesc(
328 TII->get(IsVCCZ ? AMDGPU::S_CBRANCH_EXECZ : AMDGPU::S_CBRANCH_EXECNZ));
329 }
330
331 MI.removeOperand(MI.findRegisterUseOperandIdx(CondReg, TRI, false /*Kill*/));
332 MI.addImplicitDefUseOperands(*MBB.getParent());
333
334 return true;
335}
336
337bool SIPreEmitPeephole::optimizeSetGPR(MachineInstr &First,
338 MachineInstr &MI) const {
339 MachineBasicBlock &MBB = *MI.getParent();
340 const MachineFunction &MF = *MBB.getParent();
341 const MachineRegisterInfo &MRI = MF.getRegInfo();
342 MachineOperand *Idx = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
343 Register IdxReg = Idx->isReg() ? Idx->getReg() : Register();
344 SmallVector<MachineInstr *, 4> ToRemove;
345 bool IdxOn = true;
346
347 if (!MI.isIdenticalTo(First))
348 return false;
349
350 // Scan back to find an identical S_SET_GPR_IDX_ON
351 for (MachineBasicBlock::instr_iterator I = std::next(First.getIterator()),
352 E = MI.getIterator();
353 I != E; ++I) {
354 if (I->isBundle() || I->isDebugInstr())
355 continue;
356 switch (I->getOpcode()) {
357 case AMDGPU::S_SET_GPR_IDX_MODE:
358 return false;
359 case AMDGPU::S_SET_GPR_IDX_OFF:
360 IdxOn = false;
361 ToRemove.push_back(&*I);
362 break;
363 default:
364 if (I->modifiesRegister(AMDGPU::M0, TRI))
365 return false;
366 if (IdxReg && I->modifiesRegister(IdxReg, TRI))
367 return false;
368 if (llvm::any_of(I->operands(), [&MRI, this](const MachineOperand &MO) {
369 return MO.isReg() && TRI->isVectorRegister(MRI, MO.getReg());
370 })) {
371 // The only exception allowed here is another indirect vector move
372 // with the same mode.
373 if (!IdxOn || !(I->getOpcode() == AMDGPU::V_MOV_B32_indirect_write ||
374 I->getOpcode() == AMDGPU::V_MOV_B32_indirect_read))
375 return false;
376 }
377 }
378 }
379
380 MI.eraseFromBundle();
381 for (MachineInstr *RI : ToRemove)
382 RI->eraseFromBundle();
383 return true;
384}
385
386bool SIPreEmitPeephole::getBlockDestinations(
387 MachineBasicBlock &SrcMBB, MachineBasicBlock *&TrueMBB,
388 MachineBasicBlock *&FalseMBB, SmallVectorImpl<MachineOperand> &Cond) {
389 if (TII->analyzeBranch(SrcMBB, TrueMBB, FalseMBB, Cond))
390 return false;
391
392 if (!FalseMBB)
393 FalseMBB = SrcMBB.getNextNode();
394
395 return true;
396}
397
398namespace {
399class BranchWeightCostModel {
400 const SIInstrInfo &TII;
401 const TargetSchedModel &SchedModel;
402 BranchProbability BranchProb;
403 static constexpr uint64_t BranchNotTakenCost = 1;
404 uint64_t BranchTakenCost;
405 uint64_t ThenCyclesCost = 0;
406
407public:
408 BranchWeightCostModel(const SIInstrInfo &TII, const MachineInstr &Branch,
409 const MachineBasicBlock &Succ)
410 : TII(TII), SchedModel(TII.getSchedModel()) {
411 const MachineBasicBlock &Head = *Branch.getParent();
412 const auto *FromIt = find(Head.successors(), &Succ);
413 assert(FromIt != Head.succ_end());
414
415 BranchProb = Head.getSuccProbability(FromIt);
416 if (BranchProb.isUnknown())
417 BranchProb = BranchProbability::getZero();
418 BranchTakenCost = SchedModel.computeInstrLatency(&Branch);
419 }
420
421 bool isProfitable(const MachineInstr &MI) {
422 if (TII.isWaitcnt(MI.getOpcode()))
423 return false;
424
425 ThenCyclesCost += SchedModel.computeInstrLatency(&MI);
426
427 // Consider `P = N/D` to be the probability of execz being false (skipping
428 // the then-block) The transformation is profitable if always executing the
429 // 'then' block is cheaper than executing sometimes 'then' and always
430 // executing s_cbranch_execz:
431 // * ThenCost <= P*ThenCost + (1-P)*BranchTakenCost + P*BranchNotTakenCost
432 // * (1-P) * ThenCost <= (1-P)*BranchTakenCost + P*BranchNotTakenCost
433 // * (D-N)/D * ThenCost <= (D-N)/D * BranchTakenCost + N/D *
434 // BranchNotTakenCost
435 uint64_t Numerator = BranchProb.getNumerator();
436 uint64_t Denominator = BranchProb.getDenominator();
437 return (Denominator - Numerator) * ThenCyclesCost <=
438 ((Denominator - Numerator) * BranchTakenCost +
439 Numerator * BranchNotTakenCost);
440 }
441};
442
443bool SIPreEmitPeephole::mustRetainExeczBranch(
444 const MachineInstr &Branch, const MachineBasicBlock &From,
445 const MachineBasicBlock &To) const {
446 assert(is_contained(Branch.getParent()->successors(), &From));
447 BranchWeightCostModel CostModel{*TII, Branch, From};
448
449 const MachineFunction *MF = From.getParent();
450 for (MachineFunction::const_iterator MBBI(&From), ToI(&To), End = MF->end();
451 MBBI != End && MBBI != ToI; ++MBBI) {
452 const MachineBasicBlock &MBB = *MBBI;
453
454 for (const MachineInstr &MI : MBB) {
455 // When a uniform loop is inside non-uniform control flow, the branch
456 // leaving the loop might never be taken when EXEC = 0.
457 // Hence we should retain cbranch out of the loop lest it become infinite.
458 if (MI.isConditionalBranch())
459 return true;
460
461 if (MI.isUnconditionalBranch() &&
462 TII->getBranchDestBlock(MI) != MBB.getNextNode())
463 return true;
464
465 if (MI.isMetaInstruction())
466 continue;
467
468 if (TII->hasUnwantedEffectsWhenEXECEmpty(MI))
469 return true;
470
471 if (!CostModel.isProfitable(MI))
472 return true;
473 }
474 }
475
476 return false;
477}
478} // namespace
479
480// Returns true if the skip branch instruction is removed.
481bool SIPreEmitPeephole::removeExeczBranch(MachineInstr &MI,
482 MachineBasicBlock &SrcMBB) {
483
484 if (!TII->getSchedModel().hasInstrSchedModel())
485 return false;
486
487 MachineBasicBlock *TrueMBB = nullptr;
488 MachineBasicBlock *FalseMBB = nullptr;
490
491 if (!getBlockDestinations(SrcMBB, TrueMBB, FalseMBB, Cond))
492 return false;
493
494 // Consider only the forward branches.
495 if (SrcMBB.getNumber() >= TrueMBB->getNumber())
496 return false;
497
498 // Consider only when it is legal and profitable
499 if (mustRetainExeczBranch(MI, *FalseMBB, *TrueMBB))
500 return false;
501
502 LLVM_DEBUG(dbgs() << "Removing the execz branch: " << MI);
503 MI.eraseFromParent();
504 SrcMBB.removeSuccessor(TrueMBB);
505
506 return true;
507}
508
509/// Remove writes to the FP round mode and FP denorm mode that can never be
510/// observed: either the value written is already live in MODE, or a mode write
511/// replaces the whole mode field before anything reads it.
512///
513/// s_round_mode and s_denorm_mode each assign one field of MODE and preserve
514/// the rest of the register, so the two fields are tracked independently and a
515/// write to one is transparent to the other.
516///
517/// This is a purely intra-block analysis: the mode on entry to \p SrcMBB is
518/// unknown, and a write that is still live at the end of the block is kept for
519/// the benefit of the successors.
520bool SIPreEmitPeephole::removeRedundantModeWrites(
521 MachineBasicBlock &SrcMBB) const {
522 bool Changed = false;
523 ModeFieldState DenormMode;
524 ModeFieldState RoundMode;
525
526 for (MachineInstr &MI : make_early_inc_range(SrcMBB)) {
527 if (MI.isDebugInstr())
528 continue;
529
530 unsigned Opc = MI.getOpcode();
531 if (Opc == AMDGPU::S_DENORM_MODE || Opc == AMDGPU::S_ROUND_MODE) {
532 ModeFieldState &Field =
533 Opc == AMDGPU::S_DENORM_MODE ? DenormMode : RoundMode;
534 int64_t NewValue = MI.getOperand(0).getImm();
535
536 if (Field.PendingWrite) {
537 LLVM_DEBUG(dbgs() << "Removing dead mode write: "
538 << *Field.PendingWrite);
539 Field.PendingWrite->eraseFromParent();
540 ++NumModeWritesRemoved;
541 Changed = true;
542 Field.PendingWrite = nullptr;
543 Field.Value = Field.ValueBeforePendingWrite;
544 }
545
546 if (Field.Value == NewValue) {
547 LLVM_DEBUG(dbgs() << "Removing redundant mode write: " << MI);
548 MI.eraseFromParent();
549 ++NumModeWritesRemoved;
550 Changed = true;
551 continue;
552 }
553
554 Field.ValueBeforePendingWrite = Field.Value;
555 Field.PendingWrite = &MI;
556 Field.Value = NewValue;
557 continue;
558 }
559
560 // Nothing tracked yet; skip register checks below.
561 if (!DenormMode.isTracked() && !RoundMode.isTracked())
562 continue;
563
564 // Inline asm cannot declare a MODE clobber, so assume it writes both.
565 if (MI.isInlineAsm() || MI.modifiesRegister(AMDGPU::MODE, TRI)) {
566 DenormMode = ModeFieldState();
567 RoundMode = ModeFieldState();
568 continue;
569 }
570
571 if (MI.readsRegister(AMDGPU::MODE, TRI) || MI.hasUnmodeledSideEffects()) {
572 DenormMode.PendingWrite = nullptr;
573 RoundMode.PendingWrite = nullptr;
574 }
575 }
576 return Changed;
577}
578
579bool SIPreEmitPeephole::canUnpackingClobberRegister(const MachineInstr &MI) {
580 unsigned OpCode = MI.getOpcode();
581 Register DstReg = MI.getOperand(0).getReg();
582 // Only the first register in the register pair needs to be checked due to the
583 // unpacking order. Packed instructions are unpacked such that the lower 32
584 // bits (i.e., the first register in the pair) are written first. This can
585 // introduce dependencies if the first register is written in one instruction
586 // and then read as part of the higher 32 bits in the subsequent instruction.
587 // Such scenarios can arise due to specific combinations of op_sel and
588 // op_sel_hi modifiers.
589 Register UnpackedDstReg = TRI->getSubReg(DstReg, AMDGPU::sub0);
590
591 const MachineOperand *Src0MO = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
592 if (Src0MO && Src0MO->isReg()) {
593 Register SrcReg0 = Src0MO->getReg();
594 unsigned Src0Mods =
595 TII->getNamedOperand(MI, AMDGPU::OpName::src0_modifiers)->getImm();
596 Register HiSrc0Reg = (Src0Mods & SISrcMods::OP_SEL_1)
597 ? TRI->getSubReg(SrcReg0, AMDGPU::sub1)
598 : TRI->getSubReg(SrcReg0, AMDGPU::sub0);
599 // Check if the register selected by op_sel_hi is the same as the first
600 // register in the destination register pair.
601 if (TRI->regsOverlap(UnpackedDstReg, HiSrc0Reg))
602 return true;
603 }
604
605 const MachineOperand *Src1MO = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
606 if (Src1MO && Src1MO->isReg()) {
607 Register SrcReg1 = Src1MO->getReg();
608 unsigned Src1Mods =
609 TII->getNamedOperand(MI, AMDGPU::OpName::src1_modifiers)->getImm();
610 Register HiSrc1Reg = (Src1Mods & SISrcMods::OP_SEL_1)
611 ? TRI->getSubReg(SrcReg1, AMDGPU::sub1)
612 : TRI->getSubReg(SrcReg1, AMDGPU::sub0);
613 if (TRI->regsOverlap(UnpackedDstReg, HiSrc1Reg))
614 return true;
615 }
616
617 // Applicable for packed instructions with 3 source operands, such as
618 // V_PK_FMA.
619 if (AMDGPU::hasNamedOperand(OpCode, AMDGPU::OpName::src2)) {
620 const MachineOperand *Src2MO =
621 TII->getNamedOperand(MI, AMDGPU::OpName::src2);
622 if (Src2MO && Src2MO->isReg()) {
623 Register SrcReg2 = Src2MO->getReg();
624 unsigned Src2Mods =
625 TII->getNamedOperand(MI, AMDGPU::OpName::src2_modifiers)->getImm();
626 Register HiSrc2Reg = (Src2Mods & SISrcMods::OP_SEL_1)
627 ? TRI->getSubReg(SrcReg2, AMDGPU::sub1)
628 : TRI->getSubReg(SrcReg2, AMDGPU::sub0);
629 if (TRI->regsOverlap(UnpackedDstReg, HiSrc2Reg))
630 return true;
631 }
632 }
633 return false;
634}
635
636uint32_t SIPreEmitPeephole::mapToUnpackedOpcode(MachineInstr &I) {
637 unsigned Opcode = I.getOpcode();
638 // Use 64 bit encoding to allow use of VOP3 instructions.
639 // VOP3 e64 instructions allow source modifiers
640 // e32 instructions don't allow source modifiers.
641 switch (Opcode) {
642 case AMDGPU::V_PK_ADD_F32:
643 case AMDGPU::V_PK_ADD_F32_gfx1250:
644 return AMDGPU::V_ADD_F32_e64;
645 case AMDGPU::V_PK_MUL_F32:
646 case AMDGPU::V_PK_MUL_F32_gfx1250:
647 return AMDGPU::V_MUL_F32_e64;
648 case AMDGPU::V_PK_FMA_F32:
649 case AMDGPU::V_PK_FMA_F32_gfx1250:
650 return AMDGPU::V_FMA_F32_e64;
651 default:
652 return std::numeric_limits<uint32_t>::max();
653 }
654 llvm_unreachable("Fully covered switch");
655}
656
657void SIPreEmitPeephole::addOperandAndMods(MachineInstrBuilder &NewMI,
658 unsigned SrcMods, bool IsHiBits,
659 const MachineOperand &SrcMO) {
660 unsigned NewSrcMods = 0;
661 unsigned NegModifier = IsHiBits ? SISrcMods::NEG_HI : SISrcMods::NEG;
662 unsigned OpSelModifier = IsHiBits ? SISrcMods::OP_SEL_1 : SISrcMods::OP_SEL_0;
663 // Packed instructions (VOP3P) do not support ABS. Hence, no checks are done
664 // for ABS modifiers.
665 // If NEG or NEG_HI is true, we need to negate the corresponding 32 bit
666 // lane.
667 // NEG_HI shares the same bit position with ABS. But packed instructions do
668 // not support ABS. Therefore, NEG_HI must be translated to NEG source
669 // modifier for the higher 32 bits. Unpacked VOP3 instructions support
670 // ABS, but do not support NEG_HI. Therefore we need to explicitly add the
671 // NEG modifier if present in the packed instruction.
672 if (SrcMods & NegModifier)
673 NewSrcMods |= SISrcMods::NEG;
674 // Src modifiers. Only negative modifiers are added if needed. Unpacked
675 // operations do not have op_sel, therefore it must be handled explicitly as
676 // done below.
677 NewMI.addImm(NewSrcMods);
678 if (SrcMO.isImm()) {
679 NewMI.addImm(SrcMO.getImm());
680 return;
681 }
682 // If op_sel == 0, select register 0 of reg:sub0_sub1.
683 Register UnpackedSrcReg = (SrcMods & OpSelModifier)
684 ? TRI->getSubReg(SrcMO.getReg(), AMDGPU::sub1)
685 : TRI->getSubReg(SrcMO.getReg(), AMDGPU::sub0);
686
687 MachineOperand UnpackedSrcMO =
688 MachineOperand::CreateReg(UnpackedSrcReg, /*isDef=*/false);
689 if (SrcMO.isKill()) {
690 // For each unpacked instruction, mark its source registers as killed if the
691 // corresponding source register in the original packed instruction was
692 // marked as killed.
693 //
694 // Exception:
695 // If the op_sel and op_sel_hi modifiers require both unpacked instructions
696 // to use the same register (e.g., due to overlapping access to low/high
697 // bits of the same packed register), then only the *second* (latter)
698 // instruction should mark the register as killed. This is because the
699 // second instruction handles the higher bits and is effectively the last
700 // user of the full register pair.
701
702 bool OpSel = SrcMods & SISrcMods::OP_SEL_0;
703 bool OpSelHi = SrcMods & SISrcMods::OP_SEL_1;
704 bool KillState = true;
705 if ((OpSel == OpSelHi) && !IsHiBits)
706 KillState = false;
707 UnpackedSrcMO.setIsKill(KillState);
708 }
709 NewMI.add(UnpackedSrcMO);
710}
711
712void SIPreEmitPeephole::collectUnpackingCandidates(
713 MachineInstr &BeginMI, SetVector<MachineInstr *> &InstrsToUnpack,
714 uint16_t NumMFMACycles) {
715 auto *BB = BeginMI.getParent();
716 auto E = BB->end();
717 int TotalCyclesBetweenCandidates = 0;
718 auto SchedModel = TII->getSchedModel();
719 Register MFMADef = BeginMI.getOperand(0).getReg();
720
721 for (auto I = std::next(BeginMI.getIterator()); I != E; ++I) {
722 MachineInstr &Instr = *I;
723 uint32_t UnpackedOpCode = mapToUnpackedOpcode(Instr);
724 bool IsUnpackable =
725 !(UnpackedOpCode == std::numeric_limits<uint32_t>::max());
726 if (Instr.isMetaInstruction())
727 continue;
728 if ((Instr.isTerminator()) ||
729 (TII->isNeverCoissue(Instr) && !IsUnpackable) ||
731 Instr.modifiesRegister(AMDGPU::EXEC, TRI)))
732 return;
733
734 const MCSchedClassDesc *InstrSchedClassDesc =
735 SchedModel.resolveSchedClass(&Instr);
736 uint16_t Latency =
737 SchedModel.getWriteProcResBegin(InstrSchedClassDesc)->ReleaseAtCycle;
738 TotalCyclesBetweenCandidates += Latency;
739
740 if (TotalCyclesBetweenCandidates >= NumMFMACycles - 1)
741 return;
742 // Identify register dependencies between those used by the MFMA
743 // instruction and the following packed instructions. Also checks for
744 // transitive dependencies between the MFMA def and candidate instruction
745 // def and uses. Conservatively ensures that we do not incorrectly
746 // read/write registers.
747 for (const MachineOperand &InstrMO : Instr.operands()) {
748 if (!InstrMO.isReg() || !InstrMO.getReg().isValid())
749 continue;
750 if (TRI->regsOverlap(MFMADef, InstrMO.getReg()))
751 return;
752 }
753 if (!IsUnpackable)
754 continue;
755
756 if (canUnpackingClobberRegister(Instr))
757 return;
758 // If it's a packed instruction, adjust latency: remove the packed
759 // latency, add latency of two unpacked instructions (currently estimated
760 // as 2 cycles).
761 TotalCyclesBetweenCandidates -= Latency;
762 // TODO: improve latency handling based on instruction modeling.
763 TotalCyclesBetweenCandidates += 2;
764 // Subtract 1 to account for MFMA issue latency.
765 if (TotalCyclesBetweenCandidates < NumMFMACycles - 1)
766 InstrsToUnpack.insert(&Instr);
767 }
768}
769
770void SIPreEmitPeephole::performF32Unpacking(MachineInstr &I) {
771 const MachineOperand &DstOp = I.getOperand(0);
772
773 uint32_t UnpackedOpcode = mapToUnpackedOpcode(I);
774 assert(UnpackedOpcode != std::numeric_limits<uint32_t>::max() &&
775 "Unsupported Opcode");
776
777 MachineInstrBuilder Op0LOp1L =
778 createUnpackedMI(I, UnpackedOpcode, /*IsHiBits=*/false);
779 MachineOperand LoDstOp = Op0LOp1L->getOperand(0);
780
781 LoDstOp.setIsUndef(DstOp.isUndef());
782
783 MachineInstrBuilder Op0HOp1H =
784 createUnpackedMI(I, UnpackedOpcode, /*IsHiBits=*/true);
785 MachineOperand HiDstOp = Op0HOp1H->getOperand(0);
786
787 uint32_t IFlags = I.getFlags();
788 Op0LOp1L->setFlags(IFlags);
789 Op0HOp1H->setFlags(IFlags);
790 LoDstOp.setIsRenamable(DstOp.isRenamable());
791 HiDstOp.setIsRenamable(DstOp.isRenamable());
792
793 I.eraseFromParent();
794}
795
796MachineInstrBuilder SIPreEmitPeephole::createUnpackedMI(MachineInstr &I,
797 uint32_t UnpackedOpcode,
798 bool IsHiBits) {
799 MachineBasicBlock &MBB = *I.getParent();
800 const DebugLoc &DL = I.getDebugLoc();
801 const MachineOperand *SrcMO0 = TII->getNamedOperand(I, AMDGPU::OpName::src0);
802 const MachineOperand *SrcMO1 = TII->getNamedOperand(I, AMDGPU::OpName::src1);
803 Register DstReg = I.getOperand(0).getReg();
804 unsigned OpCode = I.getOpcode();
805 Register UnpackedDstReg = IsHiBits ? TRI->getSubReg(DstReg, AMDGPU::sub1)
806 : TRI->getSubReg(DstReg, AMDGPU::sub0);
807
808 int64_t ClampVal = TII->getNamedOperand(I, AMDGPU::OpName::clamp)->getImm();
809 unsigned Src0Mods =
810 TII->getNamedOperand(I, AMDGPU::OpName::src0_modifiers)->getImm();
811 unsigned Src1Mods =
812 TII->getNamedOperand(I, AMDGPU::OpName::src1_modifiers)->getImm();
813
814 MachineInstrBuilder NewMI = BuildMI(MBB, I, DL, TII->get(UnpackedOpcode));
815 NewMI.addDef(UnpackedDstReg); // vdst
816 addOperandAndMods(NewMI, Src0Mods, IsHiBits, *SrcMO0);
817 addOperandAndMods(NewMI, Src1Mods, IsHiBits, *SrcMO1);
818
819 if (AMDGPU::hasNamedOperand(OpCode, AMDGPU::OpName::src2)) {
820 const MachineOperand *SrcMO2 =
821 TII->getNamedOperand(I, AMDGPU::OpName::src2);
822 unsigned Src2Mods =
823 TII->getNamedOperand(I, AMDGPU::OpName::src2_modifiers)->getImm();
824 addOperandAndMods(NewMI, Src2Mods, IsHiBits, *SrcMO2);
825 }
826 NewMI.addImm(ClampVal); // clamp
827 // Packed instructions do not support output modifiers. safe to assign them 0
828 // for this use case
829 NewMI.addImm(0); // omod
830 return NewMI;
831}
832
833PreservedAnalyses
836 auto *MLI = MFAM.getCachedResult<MachineLoopAnalysis>(MF);
837 SIPreEmitPeephole Impl;
838
839 if (Impl.run(MF, MLI)) {
841 PA.preserve<MachineLoopAnalysis>();
842 return PA;
843 }
844
845 return PreservedAnalyses::all();
846}
847
848bool SIPreEmitPeephole::run(MachineFunction &MF, MachineLoopInfo *LoopInfo) {
849 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
850 TII = ST.getInstrInfo();
851 TRI = &TII->getRegisterInfo();
852 MLI = LoopInfo;
853 bool Changed = false;
854
855 MF.RenumberBlocks();
856
857 for (MachineBasicBlock &MBB : MF) {
858 Changed |= removeRedundantModeWrites(MBB);
859
860 MachineBasicBlock::iterator TermI = MBB.getFirstTerminator();
861 // Check first terminator for branches to optimize
862 if (TermI != MBB.end()) {
863 MachineInstr &MI = *TermI;
864 switch (MI.getOpcode()) {
865 case AMDGPU::S_CBRANCH_VCCZ:
866 case AMDGPU::S_CBRANCH_VCCNZ:
867 Changed |= optimizeVccBranch(MI);
868 break;
869 case AMDGPU::S_CBRANCH_EXECZ:
870 Changed |= removeExeczBranch(MI, MBB);
871 break;
872 }
873 }
874
875 if (!ST.hasVGPRIndexMode())
876 continue;
877
878 MachineInstr *SetGPRMI = nullptr;
879 const unsigned Threshold = 20;
880 unsigned Count = 0;
881 // Scan the block for two S_SET_GPR_IDX_ON instructions to see if a
882 // second is not needed. Do expensive checks in the optimizeSetGPR()
883 // and limit the distance to 20 instructions for compile time purposes.
884 // Note: this needs to work on bundles as S_SET_GPR_IDX* instructions
885 // may be bundled with the instructions they modify.
886 for (auto &MI : make_early_inc_range(MBB.instrs())) {
887 if (Count == Threshold)
888 SetGPRMI = nullptr;
889 else
890 ++Count;
891
892 if (MI.getOpcode() != AMDGPU::S_SET_GPR_IDX_ON)
893 continue;
894
895 Count = 0;
896 if (!SetGPRMI) {
897 SetGPRMI = &MI;
898 continue;
899 }
900
901 if (optimizeSetGPR(*SetGPRMI, MI))
902 Changed = true;
903 else
904 SetGPRMI = &MI;
905 }
906 }
907
908 // TODO: Fold this into previous block, if possible. Evaluate and handle any
909 // side effects.
910
911 // Perform the extra MF scans only for supported archs
912 if (!ST.hasGFX940Insts())
913 return Changed;
914 for (MachineBasicBlock &MBB : MF) {
915 // Unpack packed instructions overlapped by MFMAs. This allows the
916 // compiler to co-issue unpacked instructions with MFMA
917 auto SchedModel = TII->getSchedModel();
918 SetVector<MachineInstr *> InstrsToUnpack;
919 for (auto &MI : make_early_inc_range(MBB.instrs())) {
921 continue;
922 const MCSchedClassDesc *SchedClassDesc =
923 SchedModel.resolveSchedClass(&MI);
924 uint16_t NumMFMACycles =
925 SchedModel.getWriteProcResBegin(SchedClassDesc)->ReleaseAtCycle;
926 collectUnpackingCandidates(MI, InstrsToUnpack, NumMFMACycles);
927 }
928 for (MachineInstr *MI : InstrsToUnpack) {
929 performF32Unpacking(*MI);
930 }
931 }
932
933 return Changed;
934}
for(const MachineOperand &MO :llvm::drop_begin(OldMI.operands(), Desc.getNumOperands()))
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
ReachingDefInfo InstSet & ToRemove
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
AMD GCN specific subclass of TargetSubtarget.
#define DEBUG_TYPE
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
#define I(x, y, z)
Definition MD5.cpp:57
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
OptimizedStructLayoutField Field
if(PassOpts->AAPipeline)
#define INITIALIZE_PASS(passName, arg, name, cfg, analysis)
Definition PassSupport.h:56
const SmallVectorImpl< MachineOperand > & Cond
static bool isProfitable(const StableFunctionMap::StableFunctionEntries &SFS)
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
#define LLVM_DEBUG(...)
Definition Debug.h:119
PassT::Result * getCachedResult(IRUnitT &IR) const
Get the cached result of an analysis pass for a given IR unit.
AnalysisUsage & addUsedIfAvailable()
Add the specified Pass class to the set of analyses used by this pass.
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
static uint32_t getDenominator()
static constexpr BranchProbability getZero()
uint32_t getNumerator() const
bool analyzeBranch(MachineBasicBlock &MBB, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, SmallVectorImpl< MachineOperand > &Cond, bool AllowModify) const override
Analyze the branching code at the end of MBB, returning true if it cannot be understood (e....
bool contains(const LoopT *L) const
Return true if the specified loop is contained within this loop.
bool isInnermost() const
Return true if the loop does not contain any (natural) loops.
BlockT * getHeader() const
iterator_range< block_iterator > blocks() const
LoopT * getParentLoop() const
Return the parent loop if it exists or nullptr for top level loops.
LoopT * removeChildLoop(iterator I)
This removes the specified child from being a subloop of this loop.
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
LLVM_ABI MachineBasicBlock * getFallThrough(bool JumpToFallThrough=true)
Return the fallthrough block if the block can implicitly transfer control to the block after it by fa...
LLVM_ABI BranchProbability getSuccProbability(const_succ_iterator Succ) const
Return probability of the edge from this block to MBB.
int getNumber() const
MachineBasicBlocks are uniquely numbered at the function level, unless they're not in a MachineFuncti...
LLVM_ABI void removeSuccessor(MachineBasicBlock *Succ, bool NormalizeSuccProbs=false)
Remove successor from the successors list of this MachineBasicBlock.
Instructions::iterator instr_iterator
MachineInstrBundleIterator< MachineInstr, true > reverse_iterator
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
iterator_range< iterator > terminators()
iterator_range< succ_iterator > successors()
iterator_range< pred_iterator > predecessors()
MachineInstrBundleIterator< MachineInstr > iterator
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
void RenumberBlocks(MachineBasicBlock *MBBFrom=nullptr)
RenumberBlocks - This discards all of the MachineBasicBlock numbers and recomputes them.
BasicBlockListType::const_iterator const_iterator
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
Representation of each machine instruction.
const MachineBasicBlock * getParent() const
void setFlags(unsigned flags)
const MachineOperand & getOperand(unsigned i) const
Analysis pass that exposes the MachineLoopInfo for a machine function.
int64_t getImm() const
LLVM_ABI void setIsRenamable(bool Val=true)
bool isReg() const
isReg - Tests if this is a MO_Register operand.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
void setIsKill(bool Val=true)
LLVM_ABI bool isRenamable() const
isRenamable - Returns true if this register may be renamed, i.e.
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
static bool isMFMA(const MachineInstr &MI)
static bool modifiesModeRegister(const MachineInstr &MI)
Return true if the instruction modifies the mode register.q.
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
bool insert(const value_type &X)
Insert a new element into the SetVector.
Definition SetVector.h:157
LLVM_ABI const MCSchedClassDesc * resolveSchedClass(const MachineInstr *MI) const
Return the MCSchedClassDesc for this instruction.
ProcResIter getWriteProcResBegin(const MCSchedClassDesc *SC) const
LLVM Value Representation.
Definition Value.h:75
self_iterator getIterator()
Definition ilist_node.h:123
NodeTy * getNextNode()
Get the next node, or nullptr for the list tail.
Definition ilist_node.h:348
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
PointerTypeMap run(const Module &M)
Compute the PointerTypeMap for the module M.
NodeAddr< InstrNode * > Instr
Definition RDFGraph.h:389
This is an optimization pass for GlobalISel generic memory operations.
auto find(R &&Range, const T &Val)
Provide wrappers to std::find which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1781
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:649
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
char & SIPreEmitPeepholeID
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
Definition ModRef.h:74
@ And
Bitwise or logical AND of integers.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
uint16_t ReleaseAtCycle
Cycle at which the resource will be released by an instruction, relatively to the cycle in which the ...
Definition MCSchedule.h:79