LLVM 24.0.0git
SIInstrInfo.cpp
Go to the documentation of this file.
1//===- SIInstrInfo.cpp - SI Instruction Information ----------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// SI Implementation of TargetInstrInfo.
11//
12//===----------------------------------------------------------------------===//
13
14#include "SIInstrInfo.h"
15#include "AMDGPU.h"
16#include "AMDGPUInstrInfo.h"
17#include "AMDGPULaneMaskUtils.h"
18#include "GCNHazardRecognizer.h"
19#include "GCNSubtarget.h"
22#include "llvm/ADT/STLExtras.h"
33#include "llvm/IR/IntrinsicsAMDGPU.h"
34#include "llvm/MC/MCContext.h"
37#include <tuple>
38
39using namespace llvm;
40
41#define DEBUG_TYPE "si-instr-info"
42
43#define GET_INSTRINFO_CTOR_DTOR
44#include "AMDGPUGenInstrInfo.inc"
45
46namespace llvm::AMDGPU {
47#define GET_ImageDimIntrinsicTable_IMPL
48#define GET_RsrcIntrinsics_IMPL
49#define GET_GFX1250BlockingCyclesTable_DECL
50#define GET_GFX1250BlockingCyclesTable_IMPL
51
56
57#include "AMDGPUGenSearchableTables.inc"
58} // namespace llvm::AMDGPU
59
60// Must be at least 4 to be able to branch over minimum unconditional branch
61// code. This is only for making it possible to write reasonably small tests for
62// long branches.
64BranchOffsetBits("amdgpu-s-branch-bits", cl::ReallyHidden, cl::init(16),
65 cl::desc("Restrict range of branch instructions (DEBUG)"));
66
68 "amdgpu-fix-16-bit-physreg-copies",
69 cl::desc("Fix copies between 32 and 16 bit registers by extending to 32 bit"),
70 cl::init(true),
72
74 : AMDGPUGenInstrInfo(ST, RI, AMDGPU::ADJCALLSTACKUP,
75 AMDGPU::ADJCALLSTACKDOWN),
76 RI(ST), ST(ST) {
77 SchedModel.init(&ST);
78}
79
80//===----------------------------------------------------------------------===//
81// TargetInstrInfo callbacks
82//===----------------------------------------------------------------------===//
83
84static unsigned getNumOperandsNoGlue(SDNode *Node) {
85 unsigned N = Node->getNumOperands();
86 while (N && Node->getOperand(N - 1).getValueType() == MVT::Glue)
87 --N;
88 return N;
89}
90
91/// Returns true if both nodes have the same value for the given
92/// operand \p Op, or if both nodes do not have this operand.
94 AMDGPU::OpName OpName) {
95 unsigned Opc0 = N0->getMachineOpcode();
96 unsigned Opc1 = N1->getMachineOpcode();
97
98 int Op0Idx = AMDGPU::getNamedOperandIdx(Opc0, OpName);
99 int Op1Idx = AMDGPU::getNamedOperandIdx(Opc1, OpName);
100
101 if (Op0Idx == -1 && Op1Idx == -1)
102 return true;
103
104
105 if ((Op0Idx == -1 && Op1Idx != -1) ||
106 (Op1Idx == -1 && Op0Idx != -1))
107 return false;
108
109 // getNamedOperandIdx returns the index for the MachineInstr's operands,
110 // which includes the result as the first operand. We are indexing into the
111 // MachineSDNode's operands, so we need to skip the result operand to get
112 // the real index.
113 --Op0Idx;
114 --Op1Idx;
115
116 return N0->getOperand(Op0Idx) == N1->getOperand(Op1Idx);
117}
118
119static bool canRemat(const MachineInstr &MI) {
120
124 return true;
125
126 if (SIInstrInfo::isSMRD(MI)) {
127 return !MI.memoperands_empty() &&
128 llvm::all_of(MI.memoperands(), [](const MachineMemOperand *MMO) {
129 return MMO->isLoad() && MMO->isInvariant();
130 });
131 }
132
133 return false;
134}
135
136// Split relocation flags for 64-bit global-address materialization into a
137// common base and the hi/lo relocation variants.
138static std::tuple<unsigned, unsigned, unsigned>
140 const MachineOperand &SrcOp) {
141 const unsigned BaseFlags = SrcOp.getTargetFlags() & ~SIInstrInfo::MO_MASK;
142 const unsigned Reloc = SrcOp.getTargetFlags() & SIInstrInfo::MO_MASK;
143
144 // Infer the relocation type from the existing flags on the global operand.
145 // The relocation type should have been determined earlier in the pipeline.
146 unsigned LoReloc, HiReloc;
147 switch (Reloc) {
151 LoReloc = SIInstrInfo::MO_REL32_LO;
152 HiReloc = SIInstrInfo::MO_REL32_HI;
153 break;
158 break;
161 // For 64-bit GOT-relative, use the 64-bit relocation.
164 break;
168 LoReloc = SIInstrInfo::MO_ABS32_LO;
169 HiReloc = SIInstrInfo::MO_ABS32_HI;
170 break;
171 default:
172 llvm_unreachable("unknown relocation type for global address");
173 break;
174 }
175
176 return {BaseFlags, LoReloc, HiReloc};
177}
178
180 const MachineInstr &MI) const {
181
182 if (canRemat(MI)) {
183 // Normally VALU use of exec would block the rematerialization, but that
184 // is OK in this case to have an implicit exec read as all VALU do.
185 // We really want all of the generic logic for this except for this.
186
187 // Another potential implicit use is mode register. The core logic of
188 // the RA will not attempt rematerialization if mode is set anywhere
189 // in the function, otherwise it is safe since mode is not changed.
190
191 // There is difference to generic method which does not allow
192 // rematerialization if there are virtual register uses. We allow this,
193 // therefore this method includes SOP instructions as well.
194 if (!MI.hasImplicitDef() &&
195 MI.getNumImplicitOperands() == MI.getDesc().implicit_uses().size() &&
196 !MI.mayRaiseFPException())
197 return true;
198 }
199
200 // Everything below copied from TargetInstrInfo::isReMaterializableImpl. The
201 // only difference is that we allow operations that perform read-modify-write
202 // on sub-registers.
203
204 // Remat clients assume operand 0 is the defined register.
205 if (!MI.getNumOperands() || !MI.getOperand(0).isReg())
206 return false;
207 Register DefReg = MI.getOperand(0).getReg();
208
209 const MachineFunction &MF = *MI.getMF();
210
211 // A load from a fixed stack slot can be rematerialized. This may be
212 // redundant with subsequent checks, but it's target-independent,
213 // simple, and a common case.
214 int FrameIdx = 0;
215 if (isLoadFromStackSlot(MI, FrameIdx) &&
217 return true;
218
219 // Avoid instructions obviously unsafe for remat.
220 if (MI.isNotDuplicable() || MI.mayStore() || MI.mayRaiseFPException() ||
221 MI.hasUnmodeledSideEffects())
222 return false;
223
224 // Don't remat inline asm. We have no idea how expensive it is
225 // even if it's side effect free.
226 if (MI.isInlineAsm())
227 return false;
228
229 // Avoid instructions which load from potentially varying memory.
230 if (MI.mayLoad() && !MI.isDereferenceableInvariantLoad())
231 return false;
232
233 const MachineRegisterInfo &MRI = MF.getRegInfo();
234
235 // If any of the registers accessed are non-constant, conservatively assume
236 // the instruction is not rematerializable.
237 for (const MachineOperand &MO : MI.operands()) {
238 if (!MO.isReg())
239 continue;
240 Register Reg = MO.getReg();
241 if (Reg == 0)
242 continue;
243
244 // Check for a well-behaved physical register.
245 if (Reg.isPhysical()) {
246 if (MO.isUse()) {
247 // If the physreg has no defs anywhere, it's just an ambient register
248 // and we can freely move its uses. Alternatively, if it's allocatable,
249 // it could get allocated to something with a def during allocation.
250 if (!MRI.isConstantPhysReg(Reg))
251 return false;
252 } else {
253 // A physreg def. We can't remat it.
254 return false;
255 }
256 continue;
257 }
258
259 // Only allow one virtual-register def. There may be multiple defs of the
260 // same virtual register, though.
261 if (MO.isDef() && Reg != DefReg)
262 return false;
263 }
264
265 return true;
266}
267
269 switch (Opcode) {
270 // v_subrev_u16 (gfx9)
271 case AMDGPU::V_SUBREV_U16_e32:
272 case AMDGPU::V_SUBREV_U16_e64:
273 // v_subrev_u32 (gfx9) / v_subrev_nc_u32 (gfx10+)
274 case AMDGPU::V_SUBREV_U32_e32:
275 case AMDGPU::V_SUBREV_U32_e64:
276 // v_subrev_co_u32
277 case AMDGPU::V_SUBREV_CO_U32_e32:
278 case AMDGPU::V_SUBREV_CO_U32_e64:
279 // v_subbrev_u32 (gfx9) / v_subrev_co_ci_u32 (gfx10+)
280 case AMDGPU::V_SUBBREV_U32_e32:
281 case AMDGPU::V_SUBBREV_U32_e64:
282 return true;
283 // REV shift opcodes worked this way before GFX11, verified on hardware
284 case AMDGPU::V_ASHRREV_I16_e32:
285 case AMDGPU::V_ASHRREV_I16_e64:
286 case AMDGPU::V_ASHRREV_I32_e32:
287 case AMDGPU::V_ASHRREV_I32_e64:
288 case AMDGPU::V_ASHRREV_I64_e64:
289 case AMDGPU::V_LSHLREV_B16_e32:
290 case AMDGPU::V_LSHLREV_B16_e64:
291 case AMDGPU::V_LSHLREV_B32_e32:
292 case AMDGPU::V_LSHLREV_B32_e64:
293 case AMDGPU::V_LSHLREV_B64_e64:
294 case AMDGPU::V_LSHRREV_B16_e32:
295 case AMDGPU::V_LSHRREV_B16_e64:
296 case AMDGPU::V_LSHRREV_B32_e32:
297 case AMDGPU::V_LSHRREV_B32_e64:
298 case AMDGPU::V_LSHRREV_B64_e64:
299 return !ST.hasGFX11Insts();
300 default:
301 return false;
302 }
303}
304
305// Returns true if the result of a VALU instruction depends on exec.
306bool SIInstrInfo::resultDependsOnExec(const MachineInstr &MI) const {
307 assert(isVALU(MI, /*AllowLDSDMA=*/true));
308
309 // If it is convergent it depends on EXEC.
310 if (MI.isConvergent())
311 return true;
312
313 // If it defines an SGPR it depends on EXEC, unless it's dead.
314 const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
315 for (const MachineOperand &Def : MI.defs()) {
316 if (Def.isDead())
317 continue;
318
319 Register Reg = Def.getReg();
320 if (Reg && RI.isSGPRReg(MRI, Reg))
321 return true;
322 }
323
324 return false;
325}
326
327bool SIInstrInfo::isIgnorableUse(const MachineInstr &MI, unsigned OpIdx) const {
328 const MachineOperand &MO = MI.getOperand(OpIdx);
329 // Any implicit use of exec by VALU is not a real register read.
330 return MO.getReg() == AMDGPU::EXEC && MO.isImplicit() &&
331 isVALU(MI, /*AllowLDSDMA=*/true) && !resultDependsOnExec(MI);
332}
333
335 MachineBasicBlock *SuccToSinkTo,
336 MachineCycleInfo *CI) const {
337 // Allow sinking if MI edits lane mask (divergent i1 in sgpr).
338 if (MI.getOpcode() == AMDGPU::SI_IF_BREAK)
339 return true;
340
341 MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
342 // Check if sinking of MI would create temporal divergent use.
343 for (auto Op : MI.uses()) {
344 if (Op.isReg() && Op.getReg().isVirtual() &&
345 RI.isSGPRClass(MRI.getRegClass(Op.getReg()))) {
346 MachineInstr *SgprDef = MRI.getVRegDef(Op.getReg());
347 if (!SgprDef)
348 continue;
349
350 // SgprDef defined inside cycle
351 CycleRef FromCycle = CI->getCycle(SgprDef->getParent());
352 if (!FromCycle)
353 continue;
354
355 CycleRef ToCycle = CI->getCycle(SuccToSinkTo);
356 // Check if there is a FromCycle that contains SgprDef's basic block but
357 // does not contain SuccToSinkTo and also has divergent exit condition.
358 while (FromCycle && !(ToCycle && CI->contains(FromCycle, ToCycle))) {
360 CI->getExitingBlocks(FromCycle, ExitingBlocks);
361
362 // FromCycle has divergent exit condition.
363 for (MachineBasicBlock *ExitingBlock : ExitingBlocks) {
364 if (hasDivergentBranch(ExitingBlock))
365 return false;
366 }
367
368 FromCycle = CI->getParentCycle(FromCycle);
369 }
370 }
371 }
372
373 return true;
374}
375
377 int64_t &Offset0,
378 int64_t &Offset1) const {
379 if (!Load0->isMachineOpcode() || !Load1->isMachineOpcode())
380 return false;
381
382 unsigned Opc0 = Load0->getMachineOpcode();
383 unsigned Opc1 = Load1->getMachineOpcode();
384
385 // Make sure both are actually loads.
386 if (!get(Opc0).mayLoad() || !get(Opc1).mayLoad())
387 return false;
388
389 // A mayLoad instruction without a def is not a load. Likely a prefetch.
390 if (!get(Opc0).getNumDefs() || !get(Opc1).getNumDefs())
391 return false;
392
393 if (isDS(Opc0) && isDS(Opc1)) {
394
395 // FIXME: Handle this case:
396 if (getNumOperandsNoGlue(Load0) != getNumOperandsNoGlue(Load1))
397 return false;
398
399 // Check base reg.
400 if (Load0->getOperand(0) != Load1->getOperand(0))
401 return false;
402
403 // Skip read2 / write2 variants for simplicity.
404 // TODO: We should report true if the used offsets are adjacent (excluded
405 // st64 versions).
406 int Offset0Idx = AMDGPU::getNamedOperandIdx(Opc0, AMDGPU::OpName::offset);
407 int Offset1Idx = AMDGPU::getNamedOperandIdx(Opc1, AMDGPU::OpName::offset);
408 if (Offset0Idx == -1 || Offset1Idx == -1)
409 return false;
410
411 // XXX - be careful of dataless loads
412 // getNamedOperandIdx returns the index for MachineInstrs. Since they
413 // include the output in the operand list, but SDNodes don't, we need to
414 // subtract the index by one.
415 Offset0Idx -= get(Opc0).NumDefs;
416 Offset1Idx -= get(Opc1).NumDefs;
417 Offset0 = Load0->getConstantOperandVal(Offset0Idx);
418 Offset1 = Load1->getConstantOperandVal(Offset1Idx);
419 return true;
420 }
421
422 if (isSMRD(Opc0) && isSMRD(Opc1)) {
423 // Skip time and cache invalidation instructions.
424 if (!AMDGPU::hasNamedOperand(Opc0, AMDGPU::OpName::sbase) ||
425 !AMDGPU::hasNamedOperand(Opc1, AMDGPU::OpName::sbase))
426 return false;
427
428 unsigned NumOps = getNumOperandsNoGlue(Load0);
429 if (NumOps != getNumOperandsNoGlue(Load1))
430 return false;
431
432 // Check base reg.
433 if (Load0->getOperand(0) != Load1->getOperand(0))
434 return false;
435
436 // Match register offsets, if both register and immediate offsets present.
437 assert(NumOps == 4 || NumOps == 5);
438 if (NumOps == 5 && Load0->getOperand(1) != Load1->getOperand(1))
439 return false;
440
441 const ConstantSDNode *Load0Offset =
443 const ConstantSDNode *Load1Offset =
445
446 if (!Load0Offset || !Load1Offset)
447 return false;
448
449 Offset0 = Load0Offset->getZExtValue();
450 Offset1 = Load1Offset->getZExtValue();
451 return true;
452 }
453
454 // MUBUF and MTBUF can access the same addresses.
455 if ((isMUBUF(Opc0) || isMTBUF(Opc0)) && (isMUBUF(Opc1) || isMTBUF(Opc1))) {
456
457 // MUBUF and MTBUF have vaddr at different indices.
458 if (!nodesHaveSameOperandValue(Load0, Load1, AMDGPU::OpName::soffset) ||
459 !nodesHaveSameOperandValue(Load0, Load1, AMDGPU::OpName::vaddr) ||
460 !nodesHaveSameOperandValue(Load0, Load1, AMDGPU::OpName::srsrc))
461 return false;
462
463 int OffIdx0 = AMDGPU::getNamedOperandIdx(Opc0, AMDGPU::OpName::offset);
464 int OffIdx1 = AMDGPU::getNamedOperandIdx(Opc1, AMDGPU::OpName::offset);
465
466 if (OffIdx0 == -1 || OffIdx1 == -1)
467 return false;
468
469 // getNamedOperandIdx returns the index for MachineInstrs. Since they
470 // include the output in the operand list, but SDNodes don't, we need to
471 // subtract the index by one.
472 OffIdx0 -= get(Opc0).NumDefs;
473 OffIdx1 -= get(Opc1).NumDefs;
474
475 SDValue Off0 = Load0->getOperand(OffIdx0);
476 SDValue Off1 = Load1->getOperand(OffIdx1);
477
478 // The offset might be a FrameIndexSDNode.
479 if (!isa<ConstantSDNode>(Off0) || !isa<ConstantSDNode>(Off1))
480 return false;
481
482 Offset0 = Off0->getAsZExtVal();
483 Offset1 = Off1->getAsZExtVal();
484 return true;
485 }
486
487 return false;
488}
489
490static bool isStride64(unsigned Opc) {
491 switch (Opc) {
492 case AMDGPU::DS_READ2ST64_B32:
493 case AMDGPU::DS_READ2ST64_B64:
494 case AMDGPU::DS_WRITE2ST64_B32:
495 case AMDGPU::DS_WRITE2ST64_B64:
496 return true;
497 default:
498 return false;
499 }
500}
501
504 int64_t &Offset, bool &OffsetIsScalable, LocationSize &Width,
505 const TargetRegisterInfo *TRI) const {
506 if (!LdSt.mayLoadOrStore())
507 return false;
508
509 unsigned Opc = LdSt.getOpcode();
510 OffsetIsScalable = false;
511 const MachineOperand *BaseOp, *OffsetOp;
512 int DataOpIdx;
513
514 if (isDS(LdSt)) {
515 BaseOp = getNamedOperand(LdSt, AMDGPU::OpName::addr);
516 OffsetOp = getNamedOperand(LdSt, AMDGPU::OpName::offset);
517 if (OffsetOp) {
518 // Normal, single offset LDS instruction.
519 if (!BaseOp) {
520 // DS_CONSUME/DS_APPEND use M0 for the base address.
521 // TODO: find the implicit use operand for M0 and use that as BaseOp?
522 return false;
523 }
524 BaseOps.push_back(BaseOp);
525 Offset = OffsetOp->getImm();
526 // Get appropriate operand, and compute width accordingly.
527 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst);
528 if (DataOpIdx == -1)
529 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::data0);
530 if (Opc == AMDGPU::DS_ATOMIC_ASYNC_BARRIER_ARRIVE_B64)
531 Width = LocationSize::precise(64);
532 else
533 Width = LocationSize::precise(getOpSize(LdSt, DataOpIdx));
534 } else {
535 // The 2 offset instructions use offset0 and offset1 instead. We can treat
536 // these as a load with a single offset if the 2 offsets are consecutive.
537 // We will use this for some partially aligned loads.
538 const MachineOperand *Offset0Op =
539 getNamedOperand(LdSt, AMDGPU::OpName::offset0);
540 const MachineOperand *Offset1Op =
541 getNamedOperand(LdSt, AMDGPU::OpName::offset1);
542
543 unsigned Offset0 = Offset0Op->getImm() & 0xff;
544 unsigned Offset1 = Offset1Op->getImm() & 0xff;
545 if (Offset0 + 1 != Offset1)
546 return false;
547
548 // Each of these offsets is in element sized units, so we need to convert
549 // to bytes of the individual reads.
550
551 unsigned EltSize;
552 if (LdSt.mayLoad())
553 EltSize = TRI->getRegSizeInBits(*getOpRegClass(LdSt, 0)) / 16;
554 else {
555 assert(LdSt.mayStore());
556 int Data0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::data0);
557 EltSize = TRI->getRegSizeInBits(*getOpRegClass(LdSt, Data0Idx)) / 8;
558 }
559
560 if (isStride64(Opc))
561 EltSize *= 64;
562
563 BaseOps.push_back(BaseOp);
564 Offset = EltSize * Offset0;
565 // Get appropriate operand(s), and compute width accordingly.
566 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst);
567 if (DataOpIdx == -1) {
568 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::data0);
569 Width = LocationSize::precise(getOpSize(LdSt, DataOpIdx));
570 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::data1);
571 Width = LocationSize::precise(
572 Width.getValue() + TypeSize::getFixed(getOpSize(LdSt, DataOpIdx)));
573 } else {
574 Width = LocationSize::precise(getOpSize(LdSt, DataOpIdx));
575 }
576 }
577 return true;
578 }
579
580 if (isMUBUF(LdSt) || isMTBUF(LdSt)) {
581 const MachineOperand *RSrc = getNamedOperand(LdSt, AMDGPU::OpName::srsrc);
582 if (!RSrc) // e.g. BUFFER_WBINVL1_VOL
583 return false;
584 BaseOps.push_back(RSrc);
585 BaseOp = getNamedOperand(LdSt, AMDGPU::OpName::vaddr);
586 if (BaseOp && !BaseOp->isFI())
587 BaseOps.push_back(BaseOp);
588 const MachineOperand *OffsetImm =
589 getNamedOperand(LdSt, AMDGPU::OpName::offset);
590 Offset = OffsetImm->getImm();
591 const MachineOperand *SOffset =
592 getNamedOperand(LdSt, AMDGPU::OpName::soffset);
593 if (SOffset) {
594 if (SOffset->isReg())
595 BaseOps.push_back(SOffset);
596 else
597 Offset += SOffset->getImm();
598 }
599 // Get appropriate operand, and compute width accordingly.
600 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst);
601 if (DataOpIdx == -1)
602 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdata);
603 if (DataOpIdx == -1) // LDS DMA
604 return false;
605 Width = LocationSize::precise(getOpSize(LdSt, DataOpIdx));
606 return true;
607 }
608
609 if (isImage(LdSt)) {
610 auto RsrcOpName =
611 isMIMG(LdSt) ? AMDGPU::OpName::srsrc : AMDGPU::OpName::rsrc;
612 int SRsrcIdx = AMDGPU::getNamedOperandIdx(Opc, RsrcOpName);
613 BaseOps.push_back(&LdSt.getOperand(SRsrcIdx));
614 int VAddr0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vaddr0);
615 if (VAddr0Idx >= 0) {
616 // GFX10 possible NSA encoding.
617 for (int I = VAddr0Idx; I < SRsrcIdx; ++I)
618 BaseOps.push_back(&LdSt.getOperand(I));
619 } else {
620 BaseOps.push_back(getNamedOperand(LdSt, AMDGPU::OpName::vaddr));
621 }
622 Offset = 0;
623 // Get appropriate operand, and compute width accordingly.
624 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdata);
625 if (DataOpIdx == -1)
626 return false; // no return sampler
627 Width = LocationSize::precise(getOpSize(LdSt, DataOpIdx));
628 return true;
629 }
630
631 if (isSMRD(LdSt)) {
632 BaseOp = getNamedOperand(LdSt, AMDGPU::OpName::sbase);
633 if (!BaseOp) // e.g. S_MEMTIME
634 return false;
635 BaseOps.push_back(BaseOp);
636 OffsetOp = getNamedOperand(LdSt, AMDGPU::OpName::offset);
637 Offset = OffsetOp ? OffsetOp->getImm() : 0;
638 // Get appropriate operand, and compute width accordingly.
639 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::sdst);
640 if (DataOpIdx == -1)
641 return false;
642 Width = LocationSize::precise(getOpSize(LdSt, DataOpIdx));
643 return true;
644 }
645
646 if (isFLAT(LdSt)) {
647 // Instructions have either vaddr or saddr or both or none.
648 BaseOp = getNamedOperand(LdSt, AMDGPU::OpName::vaddr);
649 if (BaseOp)
650 BaseOps.push_back(BaseOp);
651 BaseOp = getNamedOperand(LdSt, AMDGPU::OpName::saddr);
652 if (BaseOp)
653 BaseOps.push_back(BaseOp);
654 Offset = getNamedOperand(LdSt, AMDGPU::OpName::offset)->getImm();
655 // Get appropriate operand, and compute width accordingly.
656 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst);
657 if (DataOpIdx == -1)
658 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdata);
659 if (DataOpIdx == -1) // LDS DMA
660 return false;
661 Width = LocationSize::precise(getOpSize(LdSt, DataOpIdx));
662 return true;
663 }
664
665 return false;
666}
667
668static bool memOpsHaveSameBasePtr(const MachineInstr &MI1,
670 const MachineInstr &MI2,
672 // Only examine the first "base" operand of each instruction, on the
673 // assumption that it represents the real base address of the memory access.
674 // Other operands are typically offsets or indices from this base address.
675 if (BaseOps1.front()->isIdenticalTo(*BaseOps2.front()))
676 return true;
677
678 if (!MI1.hasOneMemOperand() || !MI2.hasOneMemOperand())
679 return false;
680
681 auto *MO1 = *MI1.memoperands_begin();
682 auto *MO2 = *MI2.memoperands_begin();
683 if (MO1->getAddrSpace() != MO2->getAddrSpace())
684 return false;
685
686 const auto *Base1 = MO1->getValue();
687 const auto *Base2 = MO2->getValue();
688 if (!Base1 || !Base2)
689 return false;
690 Base1 = getUnderlyingObject(Base1);
691 Base2 = getUnderlyingObject(Base2);
692
693 if (isa<UndefValue>(Base1) || isa<UndefValue>(Base2))
694 return false;
695
696 return Base1 == Base2;
697}
698
700 int64_t Offset1, bool OffsetIsScalable1,
702 int64_t Offset2, bool OffsetIsScalable2,
703 unsigned ClusterSize,
704 unsigned NumBytes) const {
705 // If the mem ops (to be clustered) do not have the same base ptr, then they
706 // should not be clustered
707 unsigned MaxMemoryClusterDWords = DefaultMemoryClusterDWordsLimit;
708 if (!BaseOps1.empty() && !BaseOps2.empty()) {
709 const MachineInstr &FirstLdSt = *BaseOps1.front()->getParent();
710 const MachineInstr &SecondLdSt = *BaseOps2.front()->getParent();
711 if (!memOpsHaveSameBasePtr(FirstLdSt, BaseOps1, SecondLdSt, BaseOps2))
712 return false;
713
714 const SIMachineFunctionInfo *MFI =
715 FirstLdSt.getMF()->getInfo<SIMachineFunctionInfo>();
716 MaxMemoryClusterDWords = MFI->getMaxMemoryClusterDWords();
717 } else if (!BaseOps1.empty() || !BaseOps2.empty()) {
718 // If only one base op is empty, they do not have the same base ptr
719 return false;
720 }
721
722 // In order to avoid register pressure, on an average, the number of DWORDS
723 // loaded together by all clustered mem ops should not exceed
724 // MaxMemoryClusterDWords. This is an empirical value based on certain
725 // observations and performance related experiments.
726 // The good thing about this heuristic is - it avoids clustering of too many
727 // sub-word loads, and also avoids clustering of wide loads. Below is the
728 // brief summary of how the heuristic behaves for various `LoadSize` when
729 // MaxMemoryClusterDWords is 8.
730 //
731 // (1) 1 <= LoadSize <= 4: cluster at max 8 mem ops
732 // (2) 5 <= LoadSize <= 8: cluster at max 4 mem ops
733 // (3) 9 <= LoadSize <= 12: cluster at max 2 mem ops
734 // (4) 13 <= LoadSize <= 16: cluster at max 2 mem ops
735 // (5) LoadSize >= 17: do not cluster
736 const unsigned LoadSize = NumBytes / ClusterSize;
737 const unsigned NumDWords = ((LoadSize + 3) / 4) * ClusterSize;
738 return NumDWords <= MaxMemoryClusterDWords;
739}
740
741// FIXME: This behaves strangely. If, for example, you have 32 load + stores,
742// the first 16 loads will be interleaved with the stores, and the next 16 will
743// be clustered as expected. It should really split into 2 16 store batches.
744//
745// Loads are clustered until this returns false, rather than trying to schedule
746// groups of stores. This also means we have to deal with saying different
747// address space loads should be clustered, and ones which might cause bank
748// conflicts.
749//
750// This might be deprecated so it might not be worth that much effort to fix.
752 int64_t Offset0, int64_t Offset1,
753 unsigned NumLoads) const {
754 assert(Offset1 > Offset0 &&
755 "Second offset should be larger than first offset!");
756 // If we have less than 16 loads in a row, and the offsets are within 64
757 // bytes, then schedule together.
758
759 // A cacheline is 64 bytes (for global memory).
760 return (NumLoads <= 16 && (Offset1 - Offset0) < 64);
761}
762
765 const DebugLoc &DL, MCRegister DestReg,
766 MCRegister SrcReg, bool KillSrc,
767 const char *Msg = "illegal VGPR to SGPR copy") {
768 MachineFunction *MF = MBB.getParent();
769
772
773 BuildMI(MBB, MI, DL, TII->get(AMDGPU::SI_ILLEGAL_COPY), DestReg)
774 .addReg(SrcReg, getKillRegState(KillSrc));
775}
776
777/// Handle copying from SGPR to AGPR, or from AGPR to AGPR on GFX908. It is not
778/// possible to have a direct copy in these cases on GFX908, so an intermediate
779/// VGPR copy is required.
782 const DebugLoc &DL, MCRegister DestReg,
783 MCRegister SrcReg, bool KillSrc,
784 RegScavenger &RS, bool RegsOverlap,
785 Register ImpUseSuperReg = Register()) {
786 assert((TII.getSubtarget().hasMAIInsts() &&
787 !TII.getSubtarget().hasGFX90AInsts()) &&
788 "Expected GFX908 subtarget.");
789
790 assert((AMDGPU::SReg_32RegClass.contains(SrcReg) ||
791 AMDGPU::AGPR_32RegClass.contains(SrcReg)) &&
792 "Source register of the copy should be either an SGPR or an AGPR.");
793
794 assert(AMDGPU::AGPR_32RegClass.contains(DestReg) &&
795 "Destination register of the copy should be an AGPR.");
796
797 const SIRegisterInfo &RI = TII.getRegisterInfo();
798
799 // First try to find defining accvgpr_write to avoid temporary registers.
800 // In the case of copies of overlapping AGPRs, we conservatively do not
801 // reuse previous accvgpr_writes. Otherwise, we may incorrectly pick up
802 // an accvgpr_write used for this same copy due to implicit-defs
803 if (!RegsOverlap) {
804 for (auto Def = MI, E = MBB.begin(); Def != E; ) {
805 --Def;
806
807 if (!Def->modifiesRegister(SrcReg, &RI))
808 continue;
809
810 if (Def->getOpcode() != AMDGPU::V_ACCVGPR_WRITE_B32_e64 ||
811 Def->getOperand(0).getReg() != SrcReg)
812 break;
813
814 MachineOperand &DefOp = Def->getOperand(1);
815 assert(DefOp.isReg() || DefOp.isImm());
816
817 if (DefOp.isReg()) {
818 bool SafeToPropagate = true;
819 // Check that register source operand is not clobbered before MI.
820 // Immediate operands are always safe to propagate.
821 for (auto I = Def; I != MI && SafeToPropagate; ++I)
822 if (I->modifiesRegister(DefOp.getReg(), &RI))
823 SafeToPropagate = false;
824
825 if (!SafeToPropagate)
826 break;
827
828 for (auto I = Def; I != MI; ++I)
829 I->clearRegisterKills(DefOp.getReg(), &RI);
830 }
831
832 MachineInstrBuilder Builder =
833 BuildMI(MBB, MI, DL, TII.get(AMDGPU::V_ACCVGPR_WRITE_B32_e64),
834 DestReg)
835 .add(DefOp);
836
837 if (ImpUseSuperReg) {
838 Builder.addReg(ImpUseSuperReg,
840 }
841
842 return;
843 }
844 }
845
846 RS.enterBasicBlockEnd(MBB);
847 RS.backward(std::next(MI));
848
849 // Ideally we want to have three registers for a long reg_sequence copy
850 // to hide 2 waitstates between v_mov_b32 and accvgpr_write.
851 unsigned MaxVGPRs = RI.getRegPressureLimit(&AMDGPU::VGPR_32RegClass,
852 *MBB.getParent());
853
854 // Registers in the sequence are allocated contiguously so we can just
855 // use register number to pick one of three round-robin temps.
856 unsigned RegNo = (DestReg - AMDGPU::AGPR0) % 3;
857 Register Tmp =
858 MBB.getParent()->getInfo<SIMachineFunctionInfo>()->getVGPRForAGPRCopy();
859 assert(MBB.getParent()->getRegInfo().isReserved(Tmp) &&
860 "VGPR used for an intermediate copy should have been reserved.");
861
862 // Only loop through if there are any free registers left. We don't want to
863 // spill.
864 while (RegNo--) {
865 Register Tmp2 = RS.scavengeRegisterBackwards(AMDGPU::VGPR_32RegClass, MI,
866 /* RestoreAfter */ false, 0,
867 /* AllowSpill */ false);
868 if (!Tmp2 || RI.getHWRegIndex(Tmp2) >= MaxVGPRs)
869 break;
870 Tmp = Tmp2;
871 RS.setRegUsed(Tmp);
872 }
873
874 // Insert copy to temporary VGPR.
875 unsigned TmpCopyOp = AMDGPU::V_MOV_B32_e32;
876 if (AMDGPU::AGPR_32RegClass.contains(SrcReg)) {
877 TmpCopyOp = AMDGPU::V_ACCVGPR_READ_B32_e64;
878 } else {
879 assert(AMDGPU::SReg_32RegClass.contains(SrcReg));
880 }
881
882 MachineInstrBuilder UseBuilder = BuildMI(MBB, MI, DL, TII.get(TmpCopyOp), Tmp)
883 .addReg(SrcReg, getKillRegState(KillSrc));
884 if (ImpUseSuperReg) {
885 UseBuilder.addReg(ImpUseSuperReg,
887 }
888
889 BuildMI(MBB, MI, DL, TII.get(AMDGPU::V_ACCVGPR_WRITE_B32_e64), DestReg)
890 .addReg(Tmp, RegState::Kill);
891}
892
895 MCRegister DestReg, MCRegister SrcReg, bool KillSrc,
896 const TargetRegisterClass *RC, bool Forward) {
897 const SIRegisterInfo &RI = TII.getRegisterInfo();
898 ArrayRef<int16_t> BaseIndices = RI.getRegSplitParts(RC, 4);
900 MachineInstr *FirstMI = nullptr, *LastMI = nullptr;
901
902 for (unsigned Idx = 0; Idx < BaseIndices.size(); ++Idx) {
903 int16_t SubIdx = BaseIndices[Idx];
904 Register DestSubReg = RI.getSubReg(DestReg, SubIdx);
905 Register SrcSubReg = RI.getSubReg(SrcReg, SubIdx);
906 assert(DestSubReg && SrcSubReg && "Failed to find subregs!");
907 unsigned Opcode = AMDGPU::S_MOV_B32;
908
909 // Is SGPR aligned? If so try to combine with next.
910 bool AlignedDest = ((DestSubReg - AMDGPU::SGPR0) % 2) == 0;
911 bool AlignedSrc = ((SrcSubReg - AMDGPU::SGPR0) % 2) == 0;
912 if (AlignedDest && AlignedSrc && (Idx + 1 < BaseIndices.size())) {
913 // Can use SGPR64 copy
914 unsigned Channel = RI.getChannelFromSubReg(SubIdx);
915 SubIdx = RI.getSubRegFromChannel(Channel, 2);
916 DestSubReg = RI.getSubReg(DestReg, SubIdx);
917 SrcSubReg = RI.getSubReg(SrcReg, SubIdx);
918 assert(DestSubReg && SrcSubReg && "Failed to find subregs!");
919 Opcode = AMDGPU::S_MOV_B64;
920 Idx++;
921 }
922
923 LastMI = BuildMI(MBB, I, DL, TII.get(Opcode), DestSubReg)
924 .addReg(SrcSubReg)
925 .addReg(SrcReg, RegState::Implicit);
926
927 if (!FirstMI)
928 FirstMI = LastMI;
929
930 if (!Forward)
931 I--;
932 }
933
934 assert(FirstMI && LastMI);
935 if (!Forward)
936 std::swap(FirstMI, LastMI);
937
938 if (KillSrc)
939 LastMI->addRegisterKilled(SrcReg, &RI);
940}
941
944 const DebugLoc &DL, Register DestReg,
945 Register SrcReg, bool KillSrc, bool RenamableDest,
946 bool RenamableSrc) const {
947 const TargetRegisterClass *RC = RI.getPhysRegBaseClass(DestReg);
948 unsigned Size = RI.getRegSizeInBits(*RC);
949 const TargetRegisterClass *SrcRC = RI.getPhysRegBaseClass(SrcReg);
950 unsigned SrcSize = RI.getRegSizeInBits(*SrcRC);
951
952 // The rest of copyPhysReg assumes Src and Dst size are the same size.
953 // TODO-GFX11_16BIT If all true 16 bit instruction patterns are completed can
954 // we remove Fix16BitCopies and this code block?
955 if (Fix16BitCopies) {
956 if (((Size == 16) != (SrcSize == 16))) {
957 // Non-VGPR Src and Dst will later be expanded back to 32 bits.
958 assert(ST.useRealTrue16Insts());
959 Register &RegToFix = (Size == 32) ? DestReg : SrcReg;
960 MCRegister SubReg = RI.getSubReg(RegToFix, AMDGPU::lo16);
961 RegToFix = SubReg;
962
963 if (DestReg == SrcReg) {
964 // Identity copy. Insert empty bundle since ExpandPostRA expects an
965 // instruction here.
966 BuildMI(MBB, MI, DL, get(AMDGPU::BUNDLE));
967 return;
968 }
969 RC = RI.getPhysRegBaseClass(DestReg);
970 Size = RI.getRegSizeInBits(*RC);
971 SrcRC = RI.getPhysRegBaseClass(SrcReg);
972 SrcSize = RI.getRegSizeInBits(*SrcRC);
973 }
974 }
975
976 if (RC == &AMDGPU::VGPR_32RegClass) {
977 assert(AMDGPU::VGPR_32RegClass.contains(SrcReg) ||
978 AMDGPU::SReg_32RegClass.contains(SrcReg) ||
979 AMDGPU::AGPR_32RegClass.contains(SrcReg));
980 unsigned Opc = AMDGPU::AGPR_32RegClass.contains(SrcReg) ?
981 AMDGPU::V_ACCVGPR_READ_B32_e64 : AMDGPU::V_MOV_B32_e32;
982 BuildMI(MBB, MI, DL, get(Opc), DestReg)
983 .addReg(SrcReg, getKillRegState(KillSrc));
984 return;
985 }
986
987 if (RC == &AMDGPU::SReg_32_XM0RegClass ||
988 RC == &AMDGPU::SReg_32RegClass) {
989 if (SrcReg == AMDGPU::SCC) {
990 BuildMI(MBB, MI, DL, get(AMDGPU::S_CSELECT_B32), DestReg)
991 .addImm(1)
992 .addImm(0);
993 return;
994 }
995
996 if (!AMDGPU::SReg_32RegClass.contains(SrcReg)) {
997 if (DestReg == AMDGPU::VCC_LO) {
998 // FIXME: Hack until VReg_1 removed.
999 assert(AMDGPU::VGPR_32RegClass.contains(SrcReg));
1000 BuildMI(MBB, MI, DL, get(AMDGPU::V_CMP_NE_U32_e32))
1001 .addImm(0)
1002 .addReg(SrcReg, getKillRegState(KillSrc));
1003 return;
1004 }
1005
1006 reportIllegalCopy(this, MBB, MI, DL, DestReg, SrcReg, KillSrc);
1007 return;
1008 }
1009
1010 BuildMI(MBB, MI, DL, get(AMDGPU::S_MOV_B32), DestReg)
1011 .addReg(SrcReg, getKillRegState(KillSrc));
1012 return;
1013 }
1014
1015 if (RC == &AMDGPU::SReg_64RegClass) {
1016 if (SrcReg == AMDGPU::SCC) {
1017 BuildMI(MBB, MI, DL, get(AMDGPU::S_CSELECT_B64), DestReg)
1018 .addImm(1)
1019 .addImm(0);
1020 return;
1021 }
1022
1023 if (!AMDGPU::SReg_64_EncodableRegClass.contains(SrcReg)) {
1024 if (DestReg == AMDGPU::VCC) {
1025 // FIXME: Hack until VReg_1 removed.
1026 assert(AMDGPU::VGPR_32RegClass.contains(SrcReg));
1027 BuildMI(MBB, MI, DL, get(AMDGPU::V_CMP_NE_U32_e32))
1028 .addImm(0)
1029 .addReg(SrcReg, getKillRegState(KillSrc));
1030 return;
1031 }
1032
1033 reportIllegalCopy(this, MBB, MI, DL, DestReg, SrcReg, KillSrc);
1034 return;
1035 }
1036
1037 BuildMI(MBB, MI, DL, get(AMDGPU::S_MOV_B64), DestReg)
1038 .addReg(SrcReg, getKillRegState(KillSrc));
1039 return;
1040 }
1041
1042 if (DestReg == AMDGPU::SCC) {
1043 // Copying 64-bit or 32-bit sources to SCC barely makes sense,
1044 // but SelectionDAG emits such copies for i1 sources.
1045 if (AMDGPU::SReg_64RegClass.contains(SrcReg)) {
1046 // This copy can only be produced by patterns
1047 // with explicit SCC, which are known to be enabled
1048 // only for subtargets with S_CMP_LG_U64 present.
1049 assert(ST.hasScalarCompareEq64());
1050 BuildMI(MBB, MI, DL, get(AMDGPU::S_CMP_LG_U64))
1051 .addReg(SrcReg, getKillRegState(KillSrc))
1052 .addImm(0);
1053 } else {
1054 assert(AMDGPU::SReg_32RegClass.contains(SrcReg));
1055 BuildMI(MBB, MI, DL, get(AMDGPU::S_CMP_LG_U32))
1056 .addReg(SrcReg, getKillRegState(KillSrc))
1057 .addImm(0);
1058 }
1059
1060 return;
1061 }
1062
1063 if (RC == &AMDGPU::AGPR_32RegClass) {
1064 if (AMDGPU::VGPR_32RegClass.contains(SrcReg) ||
1065 (ST.hasGFX90AInsts() && AMDGPU::SReg_32RegClass.contains(SrcReg))) {
1066 BuildMI(MBB, MI, DL, get(AMDGPU::V_ACCVGPR_WRITE_B32_e64), DestReg)
1067 .addReg(SrcReg, getKillRegState(KillSrc));
1068 return;
1069 }
1070
1071 if (AMDGPU::AGPR_32RegClass.contains(SrcReg) && ST.hasGFX90AInsts()) {
1072 BuildMI(MBB, MI, DL, get(AMDGPU::V_ACCVGPR_MOV_B32), DestReg)
1073 .addReg(SrcReg, getKillRegState(KillSrc));
1074 return;
1075 }
1076
1077 // FIXME: Pass should maintain scavenger to avoid scan through the block on
1078 // every AGPR spill.
1079 RegScavenger RS;
1080 const bool Overlap = RI.regsOverlap(SrcReg, DestReg);
1081 indirectCopyToAGPR(*this, MBB, MI, DL, DestReg, SrcReg, KillSrc, RS, Overlap);
1082 return;
1083 }
1084
1085 if (Size == 16) {
1086 assert(AMDGPU::VGPR_16RegClass.contains(SrcReg) ||
1087 AMDGPU::SReg_LO16RegClass.contains(SrcReg) ||
1088 AMDGPU::AGPR_LO16RegClass.contains(SrcReg));
1089
1090 bool IsSGPRDst = AMDGPU::SReg_LO16RegClass.contains(DestReg);
1091 bool IsSGPRSrc = AMDGPU::SReg_LO16RegClass.contains(SrcReg);
1092 bool IsAGPRDst = AMDGPU::AGPR_LO16RegClass.contains(DestReg);
1093 bool IsAGPRSrc = AMDGPU::AGPR_LO16RegClass.contains(SrcReg);
1094 bool DstLow = !AMDGPU::isHi16Reg(DestReg, RI);
1095 bool SrcLow = !AMDGPU::isHi16Reg(SrcReg, RI);
1096 MCRegister NewDestReg = RI.get32BitRegister(DestReg);
1097 MCRegister NewSrcReg = RI.get32BitRegister(SrcReg);
1098
1099 if (IsSGPRDst) {
1100 if (!IsSGPRSrc) {
1101 reportIllegalCopy(this, MBB, MI, DL, DestReg, SrcReg, KillSrc);
1102 return;
1103 }
1104
1105 BuildMI(MBB, MI, DL, get(AMDGPU::S_MOV_B32), NewDestReg)
1106 .addReg(NewSrcReg, getKillRegState(KillSrc));
1107 return;
1108 }
1109
1110 if (IsAGPRDst || IsAGPRSrc) {
1111 if (!DstLow || !SrcLow) {
1112 reportIllegalCopy(this, MBB, MI, DL, DestReg, SrcReg, KillSrc,
1113 "Cannot use hi16 subreg with an AGPR!");
1114 }
1115
1116 copyPhysReg(MBB, MI, DL, NewDestReg, NewSrcReg, KillSrc);
1117 return;
1118 }
1119
1120 if (ST.useRealTrue16Insts()) {
1121 if (IsSGPRSrc) {
1122 assert(SrcLow);
1123 SrcReg = NewSrcReg;
1124 }
1125 // Use the smaller instruction encoding if possible.
1126 if (AMDGPU::VGPR_16_Lo128RegClass.contains(DestReg) &&
1127 (IsSGPRSrc || AMDGPU::VGPR_16_Lo128RegClass.contains(SrcReg))) {
1128 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B16_t16_e32), DestReg)
1129 .addReg(SrcReg);
1130 } else {
1131 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B16_t16_e64), DestReg)
1132 .addImm(0) // src0_modifiers
1133 .addReg(SrcReg)
1134 .addImm(0); // op_sel
1135 }
1136 return;
1137 }
1138
1139 if (IsSGPRSrc && !ST.hasSDWAScalar()) {
1140 if (!DstLow || !SrcLow) {
1141 reportIllegalCopy(this, MBB, MI, DL, DestReg, SrcReg, KillSrc,
1142 "Cannot use hi16 subreg on VI!");
1143 }
1144
1145 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_e32), NewDestReg)
1146 .addReg(NewSrcReg, getKillRegState(KillSrc));
1147 return;
1148 }
1149
1150 auto MIB = BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_sdwa), NewDestReg)
1151 .addImm(0) // src0_modifiers
1152 .addReg(NewSrcReg)
1153 .addImm(0) // clamp
1160 // First implicit operand is $exec.
1161 MIB->tieOperands(0, MIB->getNumOperands() - 1);
1162 return;
1163 }
1164
1165 // Returns true if Dst and Src are in Opc's HwMode-resolved destination and
1166 // source operand classes.
1167 auto CanCopyWith = [&](unsigned Opc, MCRegister Dst, MCRegister Src,
1168 unsigned SrcOp = 1) {
1169 const MCInstrDesc &Desc = get(Opc);
1170 const TargetRegisterClass *DstOpRC = getRegClass(Desc, 0);
1171 const TargetRegisterClass *SrcOpRC = getRegClass(Desc, SrcOp);
1172 return DstOpRC && SrcOpRC && DstOpRC->contains(Dst) &&
1173 SrcOpRC->contains(Src);
1174 };
1175
1176 if (RC == RI.getVGPR64Class() && (SrcRC == RC || RI.isSGPRClass(SrcRC))) {
1177 if (ST.hasVMovB64Inst() &&
1178 CanCopyWith(AMDGPU::V_MOV_B64_e32, DestReg, SrcReg)) {
1179 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B64_e32), DestReg)
1180 .addReg(SrcReg, getKillRegState(KillSrc));
1181 return;
1182 }
1183 if (ST.hasPkMovB32() &&
1184 CanCopyWith(AMDGPU::V_PK_MOV_B32, DestReg, SrcReg, /*SrcOp=*/2)) {
1185 BuildMI(MBB, MI, DL, get(AMDGPU::V_PK_MOV_B32), DestReg)
1187 .addReg(SrcReg)
1189 .addReg(SrcReg)
1190 .addImm(0) // op_sel_lo
1191 .addImm(0) // op_sel_hi
1192 .addImm(0) // neg_lo
1193 .addImm(0) // neg_hi
1194 .addImm(0) // clamp
1195 .addReg(SrcReg, getKillRegState(KillSrc) | RegState::Implicit);
1196 return;
1197 }
1198 }
1199
1200 const bool Forward = RI.getHWRegIndex(DestReg) <= RI.getHWRegIndex(SrcReg);
1201 if (RI.isSGPRClass(RC)) {
1202 if (!RI.isSGPRClass(SrcRC)) {
1203 reportIllegalCopy(this, MBB, MI, DL, DestReg, SrcReg, KillSrc);
1204 return;
1205 }
1206 const bool CanKillSuperReg = KillSrc && !RI.regsOverlap(SrcReg, DestReg);
1207 expandSGPRCopy(*this, MBB, MI, DL, DestReg, SrcReg, CanKillSuperReg, RC,
1208 Forward);
1209 return;
1210 }
1211
1212 unsigned Opcode = AMDGPU::V_MOV_B32_e32;
1213 unsigned WideOpcode = AMDGPU::INSTRUCTION_LIST_END;
1214 if (RI.isAGPRClass(RC)) {
1215 if (ST.hasGFX90AInsts() && RI.isAGPRClass(SrcRC))
1216 Opcode = AMDGPU::V_ACCVGPR_MOV_B32;
1217 else if (RI.hasVGPRs(SrcRC) ||
1218 (ST.hasGFX90AInsts() && RI.isSGPRClass(SrcRC)))
1219 Opcode = AMDGPU::V_ACCVGPR_WRITE_B32_e64;
1220 else
1221 Opcode = AMDGPU::INSTRUCTION_LIST_END;
1222 } else if (RI.hasVGPRs(RC) && RI.isAGPRClass(SrcRC)) {
1223 Opcode = AMDGPU::V_ACCVGPR_READ_B32_e64;
1224 } else if (RI.isVGPRClass(RC)) {
1225 if (ST.hasVMovB64Inst())
1226 WideOpcode = AMDGPU::V_MOV_B64_e32;
1227 else if (ST.hasPkMovB32())
1228 WideOpcode = AMDGPU::V_PK_MOV_B32;
1229 }
1230
1231 const TargetRegisterClass *WideDstRC{}, *WideSrcRC{};
1232 if (WideOpcode != AMDGPU::INSTRUCTION_LIST_END) {
1233 const MCInstrDesc &Desc = get(WideOpcode);
1234 unsigned SrcOp = WideOpcode == AMDGPU::V_PK_MOV_B32 ? 2 : 1;
1235 WideDstRC = getRegClass(Desc, 0);
1236 WideSrcRC = getRegClass(Desc, SrcOp);
1237 }
1238
1239 // If there is an overlap, we can't kill the super-register on the last
1240 // instruction, since it will also kill the components made live by this def.
1241 const bool Overlap = RI.regsOverlap(SrcReg, DestReg);
1242 const bool CanKillSuperReg = KillSrc && !Overlap;
1243
1244 // For the cases where we need an intermediate instruction/temporary register
1245 // (destination is an AGPR), we need a scavenger.
1246 //
1247 // FIXME: The pass should maintain this for us so we don't have to re-scan the
1248 // whole block for every handled copy.
1249 std::unique_ptr<RegScavenger> RS;
1250 if (Opcode == AMDGPU::INSTRUCTION_LIST_END)
1251 RS = std::make_unique<RegScavenger>();
1252
1253 ArrayRef<int16_t> SubIndices = RI.getRegSplitParts(RC, 4);
1254
1255 for (unsigned Idx{}; Idx < SubIndices.size();) {
1256 unsigned NumRegs = 1;
1257 unsigned ThisOpcode = Opcode;
1258 unsigned SubIdx =
1259 Forward ? SubIndices[Idx] : SubIndices[SubIndices.size() - Idx - 1];
1260
1261 if (WideDstRC && WideSrcRC && Idx + 1 < SubIndices.size()) {
1262 unsigned Channel = RI.getChannelFromSubReg(SubIdx);
1263 if (!Forward)
1264 --Channel;
1265
1266 unsigned WideSubIdx = RI.getSubRegFromChannel(Channel, 2);
1267 Register WideDst = RI.getSubReg(DestReg, WideSubIdx);
1268 Register WideSrc = RI.getSubReg(SrcReg, WideSubIdx);
1269
1270 if (WideDst && WideSrc && WideDstRC->contains(WideDst) &&
1271 WideSrcRC->contains(WideSrc)) {
1272 SubIdx = WideSubIdx;
1273 NumRegs = 2;
1274 ThisOpcode = WideOpcode;
1275 }
1276 }
1277
1278 Register DestSubReg = RI.getSubReg(DestReg, SubIdx);
1279 Register SrcSubReg = RI.getSubReg(SrcReg, SubIdx);
1280 assert(DestSubReg && SrcSubReg && "Failed to find subregs!");
1281
1282 Idx += NumRegs;
1283 bool UseKill = CanKillSuperReg && Idx == SubIndices.size();
1284
1285 if (ThisOpcode == AMDGPU::INSTRUCTION_LIST_END) {
1286 Register ImpUseSuper = SrcReg;
1287 indirectCopyToAGPR(*this, MBB, MI, DL, DestSubReg, SrcSubReg, UseKill,
1288 *RS, Overlap, ImpUseSuper);
1289 } else if (ThisOpcode == AMDGPU::V_PK_MOV_B32) {
1290 BuildMI(MBB, MI, DL, get(AMDGPU::V_PK_MOV_B32), DestSubReg)
1292 .addReg(SrcSubReg)
1294 .addReg(SrcSubReg)
1295 .addImm(0) // op_sel_lo
1296 .addImm(0) // op_sel_hi
1297 .addImm(0) // neg_lo
1298 .addImm(0) // neg_hi
1299 .addImm(0) // clamp
1300 .addReg(SrcReg, getKillRegState(UseKill) | RegState::Implicit);
1301 } else {
1302 MachineInstrBuilder Builder =
1303 BuildMI(MBB, MI, DL, get(ThisOpcode), DestSubReg).addReg(SrcSubReg);
1304
1305 Builder.addReg(SrcReg, getKillRegState(UseKill) | RegState::Implicit);
1306 }
1307 }
1308}
1309
1310int SIInstrInfo::commuteOpcode(unsigned Opcode) const {
1311 int32_t NewOpc;
1312
1313 // Try to map original to commuted opcode
1314 NewOpc = AMDGPU::getCommuteRev(Opcode);
1315 if (NewOpc != -1)
1316 // Check if the commuted (REV) opcode exists on the target.
1317 return pseudoToMCOpcode(NewOpc) != -1 ? NewOpc : -1;
1318
1319 // Try to map commuted to original opcode
1320 NewOpc = AMDGPU::getCommuteOrig(Opcode);
1321 if (NewOpc != -1)
1322 // Check if the original (non-REV) opcode exists on the target.
1323 return pseudoToMCOpcode(NewOpc) != -1 ? NewOpc : -1;
1324
1325 return Opcode;
1326}
1327
1329 const Register Reg,
1330 int64_t &ImmVal) const {
1331 switch (MI.getOpcode()) {
1332 case AMDGPU::V_MOV_B32_e32:
1333 case AMDGPU::S_MOV_B32:
1334 case AMDGPU::S_MOVK_I32:
1335 case AMDGPU::S_MOV_B64:
1336 case AMDGPU::V_MOV_B64_e32:
1337 case AMDGPU::V_ACCVGPR_WRITE_B32_e64:
1338 case AMDGPU::AV_MOV_B32_IMM_PSEUDO:
1339 case AMDGPU::AV_MOV_B64_IMM_PSEUDO:
1340 case AMDGPU::S_MOV_B64_IMM_PSEUDO:
1341 case AMDGPU::V_MOV_B64_PSEUDO:
1342 case AMDGPU::V_MOV_B16_t16_e32: {
1343 const MachineOperand &Src0 = MI.getOperand(1);
1344 if (Src0.isImm()) {
1345 ImmVal = Src0.getImm();
1346 return MI.getOperand(0).getReg() == Reg;
1347 }
1348
1349 return false;
1350 }
1351 case AMDGPU::V_MOV_B16_t16_e64: {
1352 const MachineOperand &Src0 = MI.getOperand(2);
1353 if (Src0.isImm() && !MI.getOperand(1).getImm()) {
1354 ImmVal = Src0.getImm();
1355 return MI.getOperand(0).getReg() == Reg;
1356 }
1357
1358 return false;
1359 }
1360 case AMDGPU::S_BREV_B32:
1361 case AMDGPU::V_BFREV_B32_e32:
1362 case AMDGPU::V_BFREV_B32_e64: {
1363 const MachineOperand &Src0 = MI.getOperand(1);
1364 if (Src0.isImm()) {
1365 ImmVal = static_cast<int64_t>(reverseBits<int32_t>(Src0.getImm()));
1366 return MI.getOperand(0).getReg() == Reg;
1367 }
1368
1369 return false;
1370 }
1371 case AMDGPU::S_NOT_B32:
1372 case AMDGPU::V_NOT_B32_e32:
1373 case AMDGPU::V_NOT_B32_e64: {
1374 const MachineOperand &Src0 = MI.getOperand(1);
1375 if (Src0.isImm()) {
1376 ImmVal = static_cast<int64_t>(~static_cast<int32_t>(Src0.getImm()));
1377 return MI.getOperand(0).getReg() == Reg;
1378 }
1379
1380 return false;
1381 }
1382 default:
1383 return false;
1384 }
1385}
1386
1387std::optional<int64_t>
1389 const MachineOperand &Op,
1390 MachineInstr **DefMI) const {
1391 if (DefMI)
1392 *DefMI = nullptr;
1393
1394 if (Op.isImm())
1395 return Op.getImm();
1396
1397 if (!Op.isReg() || !Op.getReg().isVirtual())
1398 return std::nullopt;
1399 MachineInstr *Def = MRI.getUniqueVRegDef(Op.getReg());
1400 if (Def && Def->isMoveImmediate()) {
1401 const MachineOperand &ImmSrc = Def->getOperand(1);
1402 if (ImmSrc.isImm()) {
1403 if (DefMI)
1404 *DefMI = Def;
1405 return extractSubregFromImm(ImmSrc.getImm(), Op.getSubReg());
1406 }
1407 }
1408
1409 return std::nullopt;
1410}
1411
1412std::optional<int64_t>
1418
1420
1421 if (RI.isAGPRClass(DstRC))
1422 return AMDGPU::COPY;
1423 if (RI.getRegSizeInBits(*DstRC) == 16) {
1424 // Assume hi bits are unneeded. Only _e64 true16 instructions are legal
1425 // before RA.
1426 return RI.isSGPRClass(DstRC) ? AMDGPU::COPY : AMDGPU::V_MOV_B16_t16_e64;
1427 }
1428 if (RI.getRegSizeInBits(*DstRC) == 32)
1429 return RI.isSGPRClass(DstRC) ? AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32;
1430 if (RI.getRegSizeInBits(*DstRC) == 64 && RI.isSGPRClass(DstRC))
1431 return AMDGPU::S_MOV_B64;
1432 if (RI.getRegSizeInBits(*DstRC) == 64 && !RI.isSGPRClass(DstRC))
1433 return AMDGPU::V_MOV_B64_PSEUDO;
1434 return AMDGPU::COPY;
1435}
1436
1437const MCInstrDesc &
1439 bool IsIndirectSrc) const {
1440 if (IsIndirectSrc) {
1441 if (VecSize <= 32) // 4 bytes
1442 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V1);
1443 if (VecSize <= 64) // 8 bytes
1444 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V2);
1445 if (VecSize <= 96) // 12 bytes
1446 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V3);
1447 if (VecSize <= 128) // 16 bytes
1448 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V4);
1449 if (VecSize <= 160) // 20 bytes
1450 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V5);
1451 if (VecSize <= 192) // 24 bytes
1452 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V6);
1453 if (VecSize <= 224) // 28 bytes
1454 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V7);
1455 if (VecSize <= 256) // 32 bytes
1456 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V8);
1457 if (VecSize <= 288) // 36 bytes
1458 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V9);
1459 if (VecSize <= 320) // 40 bytes
1460 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V10);
1461 if (VecSize <= 352) // 44 bytes
1462 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V11);
1463 if (VecSize <= 384) // 48 bytes
1464 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V12);
1465 if (VecSize <= 512) // 64 bytes
1466 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V16);
1467 if (VecSize <= 1024) // 128 bytes
1468 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V32);
1469
1470 llvm_unreachable("unsupported size for IndirectRegReadGPRIDX pseudos");
1471 }
1472
1473 if (VecSize <= 32) // 4 bytes
1474 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V1);
1475 if (VecSize <= 64) // 8 bytes
1476 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V2);
1477 if (VecSize <= 96) // 12 bytes
1478 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V3);
1479 if (VecSize <= 128) // 16 bytes
1480 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V4);
1481 if (VecSize <= 160) // 20 bytes
1482 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V5);
1483 if (VecSize <= 192) // 24 bytes
1484 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V6);
1485 if (VecSize <= 224) // 28 bytes
1486 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V7);
1487 if (VecSize <= 256) // 32 bytes
1488 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V8);
1489 if (VecSize <= 288) // 36 bytes
1490 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V9);
1491 if (VecSize <= 320) // 40 bytes
1492 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V10);
1493 if (VecSize <= 352) // 44 bytes
1494 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V11);
1495 if (VecSize <= 384) // 48 bytes
1496 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V12);
1497 if (VecSize <= 512) // 64 bytes
1498 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V16);
1499 if (VecSize <= 1024) // 128 bytes
1500 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V32);
1501
1502 llvm_unreachable("unsupported size for IndirectRegWriteGPRIDX pseudos");
1503}
1504
1505static unsigned getIndirectVGPRWriteMovRelPseudoOpc(unsigned VecSize) {
1506 if (VecSize <= 32) // 4 bytes
1507 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V1;
1508 if (VecSize <= 64) // 8 bytes
1509 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V2;
1510 if (VecSize <= 96) // 12 bytes
1511 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V3;
1512 if (VecSize <= 128) // 16 bytes
1513 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V4;
1514 if (VecSize <= 160) // 20 bytes
1515 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V5;
1516 if (VecSize <= 192) // 24 bytes
1517 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V6;
1518 if (VecSize <= 224) // 28 bytes
1519 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V7;
1520 if (VecSize <= 256) // 32 bytes
1521 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V8;
1522 if (VecSize <= 288) // 36 bytes
1523 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V9;
1524 if (VecSize <= 320) // 40 bytes
1525 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V10;
1526 if (VecSize <= 352) // 44 bytes
1527 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V11;
1528 if (VecSize <= 384) // 48 bytes
1529 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V12;
1530 if (VecSize <= 512) // 64 bytes
1531 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V16;
1532 if (VecSize <= 1024) // 128 bytes
1533 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V32;
1534
1535 llvm_unreachable("unsupported size for IndirectRegWrite pseudos");
1536}
1537
1538static unsigned getIndirectSGPRWriteMovRelPseudo32(unsigned VecSize) {
1539 if (VecSize <= 32) // 4 bytes
1540 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V1;
1541 if (VecSize <= 64) // 8 bytes
1542 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V2;
1543 if (VecSize <= 96) // 12 bytes
1544 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V3;
1545 if (VecSize <= 128) // 16 bytes
1546 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V4;
1547 if (VecSize <= 160) // 20 bytes
1548 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V5;
1549 if (VecSize <= 192) // 24 bytes
1550 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V6;
1551 if (VecSize <= 224) // 28 bytes
1552 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V7;
1553 if (VecSize <= 256) // 32 bytes
1554 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V8;
1555 if (VecSize <= 288) // 36 bytes
1556 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V9;
1557 if (VecSize <= 320) // 40 bytes
1558 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V10;
1559 if (VecSize <= 352) // 44 bytes
1560 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V11;
1561 if (VecSize <= 384) // 48 bytes
1562 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V12;
1563 if (VecSize <= 512) // 64 bytes
1564 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V16;
1565 if (VecSize <= 1024) // 128 bytes
1566 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V32;
1567
1568 llvm_unreachable("unsupported size for IndirectRegWrite pseudos");
1569}
1570
1571static unsigned getIndirectSGPRWriteMovRelPseudo64(unsigned VecSize) {
1572 if (VecSize <= 64) // 8 bytes
1573 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V1;
1574 if (VecSize <= 128) // 16 bytes
1575 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V2;
1576 if (VecSize <= 256) // 32 bytes
1577 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V4;
1578 if (VecSize <= 512) // 64 bytes
1579 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V8;
1580 if (VecSize <= 1024) // 128 bytes
1581 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V16;
1582
1583 llvm_unreachable("unsupported size for IndirectRegWrite pseudos");
1584}
1585
1586const MCInstrDesc &
1587SIInstrInfo::getIndirectRegWriteMovRelPseudo(unsigned VecSize, unsigned EltSize,
1588 bool IsSGPR) const {
1589 if (IsSGPR) {
1590 switch (EltSize) {
1591 case 32:
1592 return get(getIndirectSGPRWriteMovRelPseudo32(VecSize));
1593 case 64:
1594 return get(getIndirectSGPRWriteMovRelPseudo64(VecSize));
1595 default:
1596 llvm_unreachable("invalid reg indexing elt size");
1597 }
1598 }
1599
1600 assert(EltSize == 32 && "invalid reg indexing elt size");
1602}
1603
1604static unsigned getSGPRSpillSaveOpcode(unsigned Size, bool NeedsCFI) {
1605 switch (Size) {
1606 case 4:
1607 return NeedsCFI ? AMDGPU::SI_SPILL_S32_CFI_SAVE : AMDGPU::SI_SPILL_S32_SAVE;
1608 case 8:
1609 return NeedsCFI ? AMDGPU::SI_SPILL_S64_CFI_SAVE : AMDGPU::SI_SPILL_S64_SAVE;
1610 case 12:
1611 return NeedsCFI ? AMDGPU::SI_SPILL_S96_CFI_SAVE : AMDGPU::SI_SPILL_S96_SAVE;
1612 case 16:
1613 return NeedsCFI ? AMDGPU::SI_SPILL_S128_CFI_SAVE
1614 : AMDGPU::SI_SPILL_S128_SAVE;
1615 case 20:
1616 return NeedsCFI ? AMDGPU::SI_SPILL_S160_CFI_SAVE
1617 : AMDGPU::SI_SPILL_S160_SAVE;
1618 case 24:
1619 return NeedsCFI ? AMDGPU::SI_SPILL_S192_CFI_SAVE
1620 : AMDGPU::SI_SPILL_S192_SAVE;
1621 case 28:
1622 return NeedsCFI ? AMDGPU::SI_SPILL_S224_CFI_SAVE
1623 : AMDGPU::SI_SPILL_S224_SAVE;
1624 case 32:
1625 return AMDGPU::SI_SPILL_S256_SAVE;
1626 case 36:
1627 return AMDGPU::SI_SPILL_S288_SAVE;
1628 case 40:
1629 return AMDGPU::SI_SPILL_S320_SAVE;
1630 case 44:
1631 return AMDGPU::SI_SPILL_S352_SAVE;
1632 case 48:
1633 return AMDGPU::SI_SPILL_S384_SAVE;
1634 case 64:
1635 return NeedsCFI ? AMDGPU::SI_SPILL_S512_CFI_SAVE
1636 : AMDGPU::SI_SPILL_S512_SAVE;
1637 case 128:
1638 return NeedsCFI ? AMDGPU::SI_SPILL_S1024_CFI_SAVE
1639 : AMDGPU::SI_SPILL_S1024_SAVE;
1640 default:
1641 llvm_unreachable("unknown register size");
1642 }
1643}
1644
1645static unsigned getVGPRSpillSaveOpcode(unsigned Size, bool NeedsCFI) {
1646 switch (Size) {
1647 case 2:
1648 return AMDGPU::SI_SPILL_V16_SAVE;
1649 case 4:
1650 return NeedsCFI ? AMDGPU::SI_SPILL_V32_CFI_SAVE : AMDGPU::SI_SPILL_V32_SAVE;
1651 case 8:
1652 return NeedsCFI ? AMDGPU::SI_SPILL_V64_CFI_SAVE : AMDGPU::SI_SPILL_V64_SAVE;
1653 case 12:
1654 return NeedsCFI ? AMDGPU::SI_SPILL_V96_CFI_SAVE : AMDGPU::SI_SPILL_V96_SAVE;
1655 case 16:
1656 return NeedsCFI ? AMDGPU::SI_SPILL_V128_CFI_SAVE
1657 : AMDGPU::SI_SPILL_V128_SAVE;
1658 case 20:
1659 return NeedsCFI ? AMDGPU::SI_SPILL_V160_CFI_SAVE
1660 : AMDGPU::SI_SPILL_V160_SAVE;
1661 case 24:
1662 return NeedsCFI ? AMDGPU::SI_SPILL_V192_CFI_SAVE
1663 : AMDGPU::SI_SPILL_V192_SAVE;
1664 case 28:
1665 return NeedsCFI ? AMDGPU::SI_SPILL_V224_CFI_SAVE
1666 : AMDGPU::SI_SPILL_V224_SAVE;
1667 case 32:
1668 return NeedsCFI ? AMDGPU::SI_SPILL_V256_CFI_SAVE
1669 : AMDGPU::SI_SPILL_V256_SAVE;
1670 case 36:
1671 return NeedsCFI ? AMDGPU::SI_SPILL_V288_CFI_SAVE
1672 : AMDGPU::SI_SPILL_V288_SAVE;
1673 case 40:
1674 return NeedsCFI ? AMDGPU::SI_SPILL_V320_CFI_SAVE
1675 : AMDGPU::SI_SPILL_V320_SAVE;
1676 case 44:
1677 return NeedsCFI ? AMDGPU::SI_SPILL_V352_CFI_SAVE
1678 : AMDGPU::SI_SPILL_V352_SAVE;
1679 case 48:
1680 return NeedsCFI ? AMDGPU::SI_SPILL_V384_CFI_SAVE
1681 : AMDGPU::SI_SPILL_V384_SAVE;
1682 case 64:
1683 return NeedsCFI ? AMDGPU::SI_SPILL_V512_CFI_SAVE
1684 : AMDGPU::SI_SPILL_V512_SAVE;
1685 case 128:
1686 return NeedsCFI ? AMDGPU::SI_SPILL_V1024_CFI_SAVE
1687 : AMDGPU::SI_SPILL_V1024_SAVE;
1688 default:
1689 llvm_unreachable("unknown register size");
1690 }
1691}
1692
1693static unsigned getAVSpillSaveOpcode(unsigned Size, bool NeedsCFI) {
1694 switch (Size) {
1695 case 4:
1696 return NeedsCFI ? AMDGPU::SI_SPILL_AV32_CFI_SAVE
1697 : AMDGPU::SI_SPILL_AV32_SAVE;
1698 case 8:
1699 return NeedsCFI ? AMDGPU::SI_SPILL_AV64_CFI_SAVE
1700 : AMDGPU::SI_SPILL_AV64_SAVE;
1701 case 12:
1702 return NeedsCFI ? AMDGPU::SI_SPILL_AV96_CFI_SAVE
1703 : AMDGPU::SI_SPILL_AV96_SAVE;
1704 case 16:
1705 return NeedsCFI ? AMDGPU::SI_SPILL_AV128_CFI_SAVE
1706 : AMDGPU::SI_SPILL_AV128_SAVE;
1707 case 20:
1708 return NeedsCFI ? AMDGPU::SI_SPILL_AV160_CFI_SAVE
1709 : AMDGPU::SI_SPILL_AV160_SAVE;
1710 case 24:
1711 return NeedsCFI ? AMDGPU::SI_SPILL_AV192_CFI_SAVE
1712 : AMDGPU::SI_SPILL_AV192_SAVE;
1713 case 28:
1714 return NeedsCFI ? AMDGPU::SI_SPILL_AV224_CFI_SAVE
1715 : AMDGPU::SI_SPILL_AV224_SAVE;
1716 case 32:
1717 return NeedsCFI ? AMDGPU::SI_SPILL_AV256_CFI_SAVE
1718 : AMDGPU::SI_SPILL_AV256_SAVE;
1719 case 36:
1720 return AMDGPU::SI_SPILL_AV288_SAVE;
1721 case 40:
1722 return AMDGPU::SI_SPILL_AV320_SAVE;
1723 case 44:
1724 return AMDGPU::SI_SPILL_AV352_SAVE;
1725 case 48:
1726 return AMDGPU::SI_SPILL_AV384_SAVE;
1727 case 64:
1728 return NeedsCFI ? AMDGPU::SI_SPILL_AV512_CFI_SAVE
1729 : AMDGPU::SI_SPILL_AV512_SAVE;
1730 case 128:
1731 return NeedsCFI ? AMDGPU::SI_SPILL_AV1024_CFI_SAVE
1732 : AMDGPU::SI_SPILL_AV1024_SAVE;
1733 default:
1734 llvm_unreachable("unknown register size");
1735 }
1736}
1737
1738static unsigned getWWMRegSpillSaveOpcode(unsigned Size,
1739 bool IsVectorSuperClass) {
1740 // Currently, there is only 32-bit WWM register spills needed.
1741 if (Size != 4)
1742 llvm_unreachable("unknown wwm register spill size");
1743
1744 if (IsVectorSuperClass)
1745 return AMDGPU::SI_SPILL_WWM_AV32_SAVE;
1746
1747 return AMDGPU::SI_SPILL_WWM_V32_SAVE;
1748}
1749
1751 Register Reg, const TargetRegisterClass *RC, unsigned Size,
1752 const SIMachineFunctionInfo &MFI, bool NeedsCFI) const {
1753 bool IsVectorSuperClass = RI.isVectorSuperClass(RC);
1754
1755 // Choose the right opcode if spilling a WWM register.
1757 return getWWMRegSpillSaveOpcode(Size, IsVectorSuperClass);
1758
1759 // TODO: Check if AGPRs are available
1760 if (ST.hasMAIInsts())
1761 return getAVSpillSaveOpcode(Size, NeedsCFI);
1762
1763 return getVGPRSpillSaveOpcode(Size, NeedsCFI);
1764}
1765
1766void SIInstrInfo::storeRegToStackSlotImpl(
1768 bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg,
1769 MachineInstr::MIFlag Flags, bool NeedsCFI) const {
1770 MachineFunction *MF = MBB.getParent();
1772 MachineFrameInfo &FrameInfo = MF->getFrameInfo();
1773 const DebugLoc &DL = MBB.findDebugLoc(MI);
1774
1775 MachinePointerInfo PtrInfo
1776 = MachinePointerInfo::getFixedStack(*MF, FrameIndex);
1778 PtrInfo, MachineMemOperand::MOStore, FrameInfo.getObjectSize(FrameIndex),
1779 FrameInfo.getObjectAlign(FrameIndex));
1780 unsigned SpillSize = RI.getSpillSize(*RC);
1781
1782 MachineRegisterInfo &MRI = MF->getRegInfo();
1783 if (RI.isSGPRClass(RC)) {
1784 if (FrameInfo.getStackID(FrameIndex) == TargetStackID::SGPRSpill)
1785 MFI->setHasSpilledSGPRs();
1786 assert(SrcReg != AMDGPU::M0 && "m0 should not be spilled");
1787 assert(SrcReg != AMDGPU::EXEC_LO && SrcReg != AMDGPU::EXEC_HI &&
1788 SrcReg != AMDGPU::EXEC && "exec should not be spilled");
1789
1790 // We are only allowed to create one new instruction when spilling
1791 // registers, so we need to use pseudo instruction for spilling SGPRs.
1792 const MCInstrDesc &OpDesc =
1793 get(getSGPRSpillSaveOpcode(SpillSize, NeedsCFI));
1794
1795 // The SGPR spill/restore instructions only work on number sgprs, so we need
1796 // to make sure we are using the correct register class.
1797 if (SrcReg.isVirtual() && SpillSize == 4) {
1798 MRI.constrainRegClass(SrcReg, &AMDGPU::SReg_32_XM0_XEXECRegClass);
1799 }
1800
1801 BuildMI(MBB, MI, DL, OpDesc)
1802 .addReg(SrcReg, getKillRegState(isKill)) // data
1803 .addFrameIndex(FrameIndex) // addr
1804 .addMemOperand(MMO)
1806
1807 return;
1808 }
1809
1810 unsigned Opcode = getVectorRegSpillSaveOpcode(VReg ? VReg : SrcReg, RC,
1811 SpillSize, *MFI, NeedsCFI);
1812 MFI->setHasSpilledVGPRs();
1813
1814 BuildMI(MBB, MI, DL, get(Opcode))
1815 .addReg(SrcReg, getKillRegState(isKill)) // data
1816 .addFrameIndex(FrameIndex) // addr
1817 .addReg(MFI->getStackPtrOffsetReg()) // scratch_offset
1818 .addImm(0) // offset
1819 .addMemOperand(MMO);
1820}
1821
1824 bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg,
1825 MachineInstr::MIFlag Flags) const {
1826 storeRegToStackSlotImpl(MBB, MI, SrcReg, isKill, FrameIndex, RC, VReg, Flags,
1827 false);
1828}
1829
1832 Register SrcReg, bool isKill,
1833 int FrameIndex,
1834 const TargetRegisterClass *RC) const {
1835 storeRegToStackSlotImpl(MBB, MI, SrcReg, isKill, FrameIndex, RC, Register(),
1836 MachineInstr::NoFlags, true);
1837}
1838
1839static unsigned getSGPRSpillRestoreOpcode(unsigned Size) {
1840 switch (Size) {
1841 case 4:
1842 return AMDGPU::SI_SPILL_S32_RESTORE;
1843 case 8:
1844 return AMDGPU::SI_SPILL_S64_RESTORE;
1845 case 12:
1846 return AMDGPU::SI_SPILL_S96_RESTORE;
1847 case 16:
1848 return AMDGPU::SI_SPILL_S128_RESTORE;
1849 case 20:
1850 return AMDGPU::SI_SPILL_S160_RESTORE;
1851 case 24:
1852 return AMDGPU::SI_SPILL_S192_RESTORE;
1853 case 28:
1854 return AMDGPU::SI_SPILL_S224_RESTORE;
1855 case 32:
1856 return AMDGPU::SI_SPILL_S256_RESTORE;
1857 case 36:
1858 return AMDGPU::SI_SPILL_S288_RESTORE;
1859 case 40:
1860 return AMDGPU::SI_SPILL_S320_RESTORE;
1861 case 44:
1862 return AMDGPU::SI_SPILL_S352_RESTORE;
1863 case 48:
1864 return AMDGPU::SI_SPILL_S384_RESTORE;
1865 case 64:
1866 return AMDGPU::SI_SPILL_S512_RESTORE;
1867 case 128:
1868 return AMDGPU::SI_SPILL_S1024_RESTORE;
1869 default:
1870 llvm_unreachable("unknown register size");
1871 }
1872}
1873
1874static unsigned getVGPRSpillRestoreOpcode(unsigned Size) {
1875 switch (Size) {
1876 case 2:
1877 return AMDGPU::SI_SPILL_V16_RESTORE;
1878 case 4:
1879 return AMDGPU::SI_SPILL_V32_RESTORE;
1880 case 8:
1881 return AMDGPU::SI_SPILL_V64_RESTORE;
1882 case 12:
1883 return AMDGPU::SI_SPILL_V96_RESTORE;
1884 case 16:
1885 return AMDGPU::SI_SPILL_V128_RESTORE;
1886 case 20:
1887 return AMDGPU::SI_SPILL_V160_RESTORE;
1888 case 24:
1889 return AMDGPU::SI_SPILL_V192_RESTORE;
1890 case 28:
1891 return AMDGPU::SI_SPILL_V224_RESTORE;
1892 case 32:
1893 return AMDGPU::SI_SPILL_V256_RESTORE;
1894 case 36:
1895 return AMDGPU::SI_SPILL_V288_RESTORE;
1896 case 40:
1897 return AMDGPU::SI_SPILL_V320_RESTORE;
1898 case 44:
1899 return AMDGPU::SI_SPILL_V352_RESTORE;
1900 case 48:
1901 return AMDGPU::SI_SPILL_V384_RESTORE;
1902 case 64:
1903 return AMDGPU::SI_SPILL_V512_RESTORE;
1904 case 128:
1905 return AMDGPU::SI_SPILL_V1024_RESTORE;
1906 default:
1907 llvm_unreachable("unknown register size");
1908 }
1909}
1910
1911static unsigned getAVSpillRestoreOpcode(unsigned Size) {
1912 switch (Size) {
1913 case 4:
1914 return AMDGPU::SI_SPILL_AV32_RESTORE;
1915 case 8:
1916 return AMDGPU::SI_SPILL_AV64_RESTORE;
1917 case 12:
1918 return AMDGPU::SI_SPILL_AV96_RESTORE;
1919 case 16:
1920 return AMDGPU::SI_SPILL_AV128_RESTORE;
1921 case 20:
1922 return AMDGPU::SI_SPILL_AV160_RESTORE;
1923 case 24:
1924 return AMDGPU::SI_SPILL_AV192_RESTORE;
1925 case 28:
1926 return AMDGPU::SI_SPILL_AV224_RESTORE;
1927 case 32:
1928 return AMDGPU::SI_SPILL_AV256_RESTORE;
1929 case 36:
1930 return AMDGPU::SI_SPILL_AV288_RESTORE;
1931 case 40:
1932 return AMDGPU::SI_SPILL_AV320_RESTORE;
1933 case 44:
1934 return AMDGPU::SI_SPILL_AV352_RESTORE;
1935 case 48:
1936 return AMDGPU::SI_SPILL_AV384_RESTORE;
1937 case 64:
1938 return AMDGPU::SI_SPILL_AV512_RESTORE;
1939 case 128:
1940 return AMDGPU::SI_SPILL_AV1024_RESTORE;
1941 default:
1942 llvm_unreachable("unknown register size");
1943 }
1944}
1945
1946static unsigned getWWMRegSpillRestoreOpcode(unsigned Size,
1947 bool IsVectorSuperClass) {
1948 // Currently, there is only 32-bit WWM register spills needed.
1949 if (Size != 4)
1950 llvm_unreachable("unknown wwm register spill size");
1951
1952 if (IsVectorSuperClass) // TODO: Always use this if there are AGPRs
1953 return AMDGPU::SI_SPILL_WWM_AV32_RESTORE;
1954
1955 return AMDGPU::SI_SPILL_WWM_V32_RESTORE;
1956}
1957
1959 Register Reg, const TargetRegisterClass *RC, unsigned Size,
1960 const SIMachineFunctionInfo &MFI) const {
1961 bool IsVectorSuperClass = RI.isVectorSuperClass(RC);
1962
1963 // Choose the right opcode if restoring a WWM register.
1965 return getWWMRegSpillRestoreOpcode(Size, IsVectorSuperClass);
1966
1967 // TODO: Check if AGPRs are available
1968 if (ST.hasMAIInsts())
1970
1971 assert(!RI.isAGPRClass(RC));
1973}
1974
1977 Register DestReg, int FrameIndex,
1978 const TargetRegisterClass *RC,
1979 Register VReg, unsigned SubReg,
1980 MachineInstr::MIFlag Flags) const {
1981 MachineFunction *MF = MBB.getParent();
1983 MachineFrameInfo &FrameInfo = MF->getFrameInfo();
1984 const DebugLoc &DL = MBB.findDebugLoc(MI);
1985 unsigned SpillSize = RI.getSpillSize(*RC);
1986
1987 MachinePointerInfo PtrInfo
1988 = MachinePointerInfo::getFixedStack(*MF, FrameIndex);
1989
1991 PtrInfo, MachineMemOperand::MOLoad, FrameInfo.getObjectSize(FrameIndex),
1992 FrameInfo.getObjectAlign(FrameIndex));
1993
1994 if (RI.isSGPRClass(RC)) {
1995 if (FrameInfo.getStackID(FrameIndex) == TargetStackID::SGPRSpill)
1996 MFI->setHasSpilledSGPRs();
1997 assert(DestReg != AMDGPU::M0 && "m0 should not be reloaded into");
1998 assert(DestReg != AMDGPU::EXEC_LO && DestReg != AMDGPU::EXEC_HI &&
1999 DestReg != AMDGPU::EXEC && "exec should not be spilled");
2000
2001 // FIXME: Maybe this should not include a memoperand because it will be
2002 // lowered to non-memory instructions.
2003 const MCInstrDesc &OpDesc = get(getSGPRSpillRestoreOpcode(SpillSize));
2004 if (DestReg.isVirtual() && SpillSize == 4) {
2005 MachineRegisterInfo &MRI = MF->getRegInfo();
2006 MRI.constrainRegClass(DestReg, &AMDGPU::SReg_32_XM0_XEXECRegClass);
2007 }
2008
2009 BuildMI(MBB, MI, DL, OpDesc, DestReg)
2010 .addFrameIndex(FrameIndex) // addr
2011 .addMemOperand(MMO)
2013
2014 return;
2015 }
2016
2017 unsigned Opcode = getVectorRegSpillRestoreOpcode(VReg ? VReg : DestReg, RC,
2018 SpillSize, *MFI);
2019 BuildMI(MBB, MI, DL, get(Opcode), DestReg)
2020 .addFrameIndex(FrameIndex) // vaddr
2021 .addReg(MFI->getStackPtrOffsetReg()) // scratch_offset
2022 .addImm(0) // offset
2023 .addMemOperand(MMO);
2024}
2025
2030
2033 unsigned Quantity) const {
2034 DebugLoc DL = MBB.findDebugLoc(MI);
2035 unsigned MaxSNopCount = 1u << ST.getSNopBits();
2036 while (Quantity > 0) {
2037 unsigned Arg = std::min(Quantity, MaxSNopCount);
2038 Quantity -= Arg;
2039 BuildMI(MBB, MI, DL, get(AMDGPU::S_NOP)).addImm(Arg - 1);
2040 }
2041}
2042
2046 const DebugLoc &DL) const {
2047 MachineFunction *MF = MBB.getParent();
2048 constexpr unsigned DoorbellIDMask = 0x3ff;
2049 constexpr unsigned ECQueueWaveAbort = 0x400;
2050
2051 MachineBasicBlock *TrapBB = &MBB;
2052 MachineBasicBlock *HaltLoopBB = MF->CreateMachineBasicBlock();
2053
2054 if (!MBB.succ_empty() || std::next(MI.getIterator()) != MBB.end()) {
2055 MBB.splitAt(MI, /*UpdateLiveIns=*/false);
2056 TrapBB = MF->CreateMachineBasicBlock();
2057 BuildMI(MBB, MI, DL, get(AMDGPU::S_CBRANCH_EXECNZ)).addMBB(TrapBB);
2058 MF->push_back(TrapBB);
2059 MBB.addSuccessor(TrapBB);
2060 }
2061 // Start with a `s_trap 2`, if we're in PRIV=1 and we need the workaround this
2062 // will be a nop.
2063 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_TRAP))
2064 .addImm(static_cast<unsigned>(GCNSubtarget::TrapID::LLVMAMDHSATrap));
2065 Register DoorbellReg = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
2066 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_SENDMSG_RTN_B32),
2067 DoorbellReg)
2069 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_MOV_B32), AMDGPU::TTMP2)
2070 .addUse(AMDGPU::M0);
2071 Register DoorbellRegMasked =
2072 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
2073 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_AND_B32), DoorbellRegMasked)
2074 .addUse(DoorbellReg)
2075 .addImm(DoorbellIDMask)
2076 .setOperandDead(3); // implicit-def $scc
2077 Register SetWaveAbortBit =
2078 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
2079 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_OR_B32), SetWaveAbortBit)
2080 .addUse(DoorbellRegMasked)
2081 .addImm(ECQueueWaveAbort)
2082 .setOperandDead(3); // implicit-def $scc
2083 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_MOV_B32), AMDGPU::M0)
2084 .addUse(SetWaveAbortBit);
2085 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_SENDMSG))
2087 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_MOV_B32), AMDGPU::M0)
2088 .addUse(AMDGPU::TTMP2);
2089 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_BRANCH)).addMBB(HaltLoopBB);
2090 TrapBB->addSuccessor(HaltLoopBB);
2091
2092 BuildMI(*HaltLoopBB, HaltLoopBB->end(), DL, get(AMDGPU::S_SETHALT)).addImm(5);
2093 BuildMI(*HaltLoopBB, HaltLoopBB->end(), DL, get(AMDGPU::S_BRANCH))
2094 .addMBB(HaltLoopBB);
2095 MF->push_back(HaltLoopBB);
2096 HaltLoopBB->addSuccessor(HaltLoopBB);
2097
2098 return MBB.getNextNode();
2099}
2100
2102 switch (MI.getOpcode()) {
2103 default:
2104 if (MI.isMetaInstruction())
2105 return 0;
2106 return 1; // FIXME: Do wait states equal cycles?
2107
2108 case AMDGPU::S_NOP:
2109 return MI.getOperand(0).getImm() + 1;
2110 // SI_RETURN_TO_EPILOG is a fallthrough to code outside of the function. The
2111 // hazard, even if one exist, won't really be visible. Should we handle it?
2112 }
2113}
2114
2116 MachineBasicBlock &MBB = *MI.getParent();
2117 DebugLoc DL = MBB.findDebugLoc(MI);
2119
2120 switch (MI.getOpcode()) {
2121 default: return TargetInstrInfo::expandPostRAPseudo(MI);
2122 case AMDGPU::S_MOV_B64_term:
2123 // This is only a terminator to get the correct spill code placement during
2124 // register allocation.
2125 MI.setDesc(get(AMDGPU::S_MOV_B64));
2126 break;
2127
2128 case AMDGPU::S_MOV_B32_term:
2129 // This is only a terminator to get the correct spill code placement during
2130 // register allocation.
2131 MI.setDesc(get(AMDGPU::S_MOV_B32));
2132 break;
2133
2134 case AMDGPU::S_XOR_B64_term:
2135 // This is only a terminator to get the correct spill code placement during
2136 // register allocation.
2137 MI.setDesc(get(AMDGPU::S_XOR_B64));
2138 break;
2139
2140 case AMDGPU::S_XOR_B32_term:
2141 // This is only a terminator to get the correct spill code placement during
2142 // register allocation.
2143 MI.setDesc(get(AMDGPU::S_XOR_B32));
2144 break;
2145 case AMDGPU::S_OR_B64_term:
2146 // This is only a terminator to get the correct spill code placement during
2147 // register allocation.
2148 MI.setDesc(get(AMDGPU::S_OR_B64));
2149 break;
2150 case AMDGPU::S_OR_B32_term:
2151 // This is only a terminator to get the correct spill code placement during
2152 // register allocation.
2153 MI.setDesc(get(AMDGPU::S_OR_B32));
2154 break;
2155
2156 case AMDGPU::S_ANDN2_B64_term:
2157 // This is only a terminator to get the correct spill code placement during
2158 // register allocation.
2159 MI.setDesc(get(AMDGPU::S_ANDN2_B64));
2160 break;
2161
2162 case AMDGPU::S_ANDN2_B32_term:
2163 // This is only a terminator to get the correct spill code placement during
2164 // register allocation.
2165 MI.setDesc(get(AMDGPU::S_ANDN2_B32));
2166 break;
2167
2168 case AMDGPU::S_AND_B64_term:
2169 // This is only a terminator to get the correct spill code placement during
2170 // register allocation.
2171 MI.setDesc(get(AMDGPU::S_AND_B64));
2172 break;
2173
2174 case AMDGPU::S_AND_B32_term:
2175 // This is only a terminator to get the correct spill code placement during
2176 // register allocation.
2177 MI.setDesc(get(AMDGPU::S_AND_B32));
2178 break;
2179
2180 case AMDGPU::S_AND_SAVEEXEC_B64_term:
2181 // This is only a terminator to get the correct spill code placement during
2182 // register allocation.
2183 MI.setDesc(get(AMDGPU::S_AND_SAVEEXEC_B64));
2184 break;
2185
2186 case AMDGPU::S_AND_SAVEEXEC_B32_term:
2187 // This is only a terminator to get the correct spill code placement during
2188 // register allocation.
2189 MI.setDesc(get(AMDGPU::S_AND_SAVEEXEC_B32));
2190 break;
2191
2192 case AMDGPU::V_CMPX_EQ_U32_nosdst_e32_term:
2193 MI.setDesc(get(AMDGPU::V_CMPX_EQ_U32_nosdst_e32));
2194 break;
2195 case AMDGPU::V_CMPX_EQ_U64_nosdst_e32_term:
2196 MI.setDesc(get(AMDGPU::V_CMPX_EQ_U64_nosdst_e32));
2197 break;
2198
2199 case AMDGPU::SI_SPILL_S32_TO_VGPR:
2200 MI.setDesc(get(AMDGPU::V_WRITELANE_B32));
2201 break;
2202
2203 case AMDGPU::SI_RESTORE_S32_FROM_VGPR:
2204 MI.setDesc(get(AMDGPU::V_READLANE_B32));
2205 break;
2206 case AMDGPU::AV_MOV_B32_IMM_PSEUDO: {
2207 Register Dst = MI.getOperand(0).getReg();
2208 bool IsAGPR = SIRegisterInfo::isAGPRClass(RI.getPhysRegBaseClass(Dst));
2209 MI.setDesc(
2210 get(IsAGPR ? AMDGPU::V_ACCVGPR_WRITE_B32_e64 : AMDGPU::V_MOV_B32_e32));
2211 break;
2212 }
2213 case AMDGPU::AV_MOV_B64_IMM_PSEUDO: {
2214 Register Dst = MI.getOperand(0).getReg();
2215 if (SIRegisterInfo::isAGPRClass(RI.getPhysRegBaseClass(Dst))) {
2216 int64_t Imm = MI.getOperand(1).getImm();
2217
2218 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2219 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2220 BuildMI(MBB, MI, DL, get(AMDGPU::V_ACCVGPR_WRITE_B32_e64), DstLo)
2222 BuildMI(MBB, MI, DL, get(AMDGPU::V_ACCVGPR_WRITE_B32_e64), DstHi)
2223 .addImm(SignExtend64<32>(Imm >> 32));
2224 MI.eraseFromParent();
2225 break;
2226 }
2227
2228 [[fallthrough]];
2229 }
2230 case AMDGPU::V_MOV_B64_PSEUDO: {
2231 Register Dst = MI.getOperand(0).getReg();
2232 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2233 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2234
2235 const MCInstrDesc &Mov64Desc = get(AMDGPU::V_MOV_B64_e32);
2236 const TargetRegisterClass *Mov64RC = getRegClass(Mov64Desc, /*OpNum=*/0);
2237
2238 const MachineOperand &SrcOp = MI.getOperand(1);
2239 // FIXME: Will this work for 64-bit floating point immediates?
2240 assert(!SrcOp.isFPImm());
2241 if (ST.hasVMovB64Inst() && Mov64RC->contains(Dst)) {
2242 MI.setDesc(Mov64Desc);
2243 if (SrcOp.isReg() || isInlineConstant(MI, 1) ||
2244 (SrcOp.isImm() &&
2245 (isUInt<32>(SrcOp.getImm()) || ST.has64BitLiterals())) ||
2246 (SrcOp.isGlobal() && ST.has64BitLiterals()))
2247 break;
2248 }
2249 if (SrcOp.isGlobal()) {
2250 // The address is unknown until link time, so the PK_MOV inline-constant
2251 // shortcut cannot apply.
2252 const GlobalValue *GV = SrcOp.getGlobal();
2253 int64_t Offset = SrcOp.getOffset();
2254 unsigned BaseFlags, LoReloc, HiReloc;
2255 std::tie(BaseFlags, LoReloc, HiReloc) =
2257
2258 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_e32), DstLo)
2259 .addGlobalAddress(GV, Offset, BaseFlags | LoReloc);
2260 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_e32), DstHi)
2261 .addGlobalAddress(GV, Offset, BaseFlags | HiReloc);
2262 } else if (SrcOp.isImm()) {
2263 APInt Imm(64, SrcOp.getImm());
2264 APInt Lo(32, Imm.getLoBits(32).getZExtValue());
2265 APInt Hi(32, Imm.getHiBits(32).getZExtValue());
2266 const MCInstrDesc &PkMovDesc = get(AMDGPU::V_PK_MOV_B32);
2267 const TargetRegisterClass *PkMovRC = getRegClass(PkMovDesc, /*OpNum=*/0);
2268
2269 if (ST.hasPkMovB32() && Lo == Hi && isInlineConstant(Lo) &&
2270 PkMovRC->contains(Dst)) {
2271 BuildMI(MBB, MI, DL, PkMovDesc, Dst)
2273 .addImm(Lo.getSExtValue())
2275 .addImm(Lo.getSExtValue())
2276 .addImm(0) // op_sel_lo
2277 .addImm(0) // op_sel_hi
2278 .addImm(0) // neg_lo
2279 .addImm(0) // neg_hi
2280 .addImm(0); // clamp
2281 } else {
2282 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_e32), DstLo)
2283 .addImm(Lo.getSExtValue());
2284 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_e32), DstHi)
2285 .addImm(Hi.getSExtValue());
2286 }
2287 } else {
2288 assert(SrcOp.isReg());
2289 if (ST.hasPkMovB32() &&
2290 !RI.isAGPR(MBB.getParent()->getRegInfo(), SrcOp.getReg())) {
2291 BuildMI(MBB, MI, DL, get(AMDGPU::V_PK_MOV_B32), Dst)
2292 .addImm(SISrcMods::OP_SEL_1) // src0_mod
2293 .addReg(SrcOp.getReg())
2295 .addReg(SrcOp.getReg())
2296 .addImm(0) // op_sel_lo
2297 .addImm(0) // op_sel_hi
2298 .addImm(0) // neg_lo
2299 .addImm(0) // neg_hi
2300 .addImm(0); // clamp
2301 } else {
2302 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_e32), DstLo)
2303 .addReg(RI.getSubReg(SrcOp.getReg(), AMDGPU::sub0));
2304 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_e32), DstHi)
2305 .addReg(RI.getSubReg(SrcOp.getReg(), AMDGPU::sub1));
2306 }
2307 }
2308 MI.eraseFromParent();
2309 break;
2310 }
2311 case AMDGPU::V_MOV_B64_DPP_PSEUDO: {
2313 break;
2314 }
2315 case AMDGPU::S_MOV_B64_IMM_PSEUDO: {
2316 const MachineOperand &SrcOp = MI.getOperand(1);
2317 assert(!SrcOp.isFPImm());
2318
2319 if (ST.has64BitLiterals()) {
2320 MI.setDesc(get(AMDGPU::S_MOV_B64));
2321 break;
2322 }
2323
2324 if (SrcOp.isGlobal()) {
2325 Register Dst = MI.getOperand(0).getReg();
2326 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2327 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2328 const GlobalValue *GV = SrcOp.getGlobal();
2329 int64_t Offset = SrcOp.getOffset();
2330 unsigned BaseFlags, LoReloc, HiReloc;
2331 std::tie(BaseFlags, LoReloc, HiReloc) =
2333
2334 BuildMI(MBB, MI, DL, get(AMDGPU::S_MOV_B32), DstLo)
2335 .addGlobalAddress(GV, Offset, BaseFlags | LoReloc);
2336 BuildMI(MBB, MI, DL, get(AMDGPU::S_MOV_B32), DstHi)
2337 .addGlobalAddress(GV, Offset, BaseFlags | HiReloc);
2338 MI.eraseFromParent();
2339 break;
2340 }
2341
2342 // SrcOp is immediate
2343 APInt Imm(64, SrcOp.getImm());
2344 if (Imm.isIntN(32) || isInlineConstant(Imm)) {
2345 MI.setDesc(get(AMDGPU::S_MOV_B64));
2346 break;
2347 }
2348
2349 Register Dst = MI.getOperand(0).getReg();
2350 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2351 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2352
2353 APInt Lo(32, Imm.getLoBits(32).getZExtValue());
2354 APInt Hi(32, Imm.getHiBits(32).getZExtValue());
2355 BuildMI(MBB, MI, DL, get(AMDGPU::S_MOV_B32), DstLo)
2356 .addImm(Lo.getSExtValue());
2357 BuildMI(MBB, MI, DL, get(AMDGPU::S_MOV_B32), DstHi)
2358 .addImm(Hi.getSExtValue());
2359 MI.eraseFromParent();
2360 break;
2361 }
2362 case AMDGPU::V_SET_INACTIVE_B32: {
2363 // Lower V_SET_INACTIVE_B32 to V_CNDMASK_B32.
2364 Register DstReg = MI.getOperand(0).getReg();
2365 BuildMI(MBB, MI, DL, get(AMDGPU::V_CNDMASK_B32_e64), DstReg)
2366 .add(MI.getOperand(3))
2367 .add(MI.getOperand(4))
2368 .add(MI.getOperand(1))
2369 .add(MI.getOperand(2))
2370 .add(MI.getOperand(5));
2371 MI.eraseFromParent();
2372 break;
2373 }
2374 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V1:
2375 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V2:
2376 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V3:
2377 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V4:
2378 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V5:
2379 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V6:
2380 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V7:
2381 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V8:
2382 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V9:
2383 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V10:
2384 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V11:
2385 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V12:
2386 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V16:
2387 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V32:
2388 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V1:
2389 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V2:
2390 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V3:
2391 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V4:
2392 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V5:
2393 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V6:
2394 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V7:
2395 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V8:
2396 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V9:
2397 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V10:
2398 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V11:
2399 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V12:
2400 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V16:
2401 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V32:
2402 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V1:
2403 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V2:
2404 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V4:
2405 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V8:
2406 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V16: {
2407 const TargetRegisterClass *EltRC = getOpRegClass(MI, 2);
2408
2409 unsigned Opc;
2410 if (RI.hasVGPRs(EltRC)) {
2411 Opc = AMDGPU::V_MOVRELD_B32_e32;
2412 } else {
2413 Opc = RI.getRegSizeInBits(*EltRC) == 64 ? AMDGPU::S_MOVRELD_B64
2414 : AMDGPU::S_MOVRELD_B32;
2415 }
2416
2417 const MCInstrDesc &OpDesc = get(Opc);
2418 Register VecReg = MI.getOperand(0).getReg();
2419 bool IsUndef = MI.getOperand(1).isUndef();
2420 unsigned SubReg = MI.getOperand(3).getImm();
2421 assert(VecReg == MI.getOperand(1).getReg());
2422
2424 BuildMI(MBB, MI, DL, OpDesc)
2425 .addReg(RI.getSubReg(VecReg, SubReg), RegState::Undef)
2426 .add(MI.getOperand(2))
2428 .addReg(VecReg, RegState::Implicit | getUndefRegState(IsUndef));
2429
2430 const int ImpDefIdx =
2431 OpDesc.getNumOperands() + OpDesc.implicit_uses().size();
2432 const int ImpUseIdx = ImpDefIdx + 1;
2433 MIB->tieOperands(ImpDefIdx, ImpUseIdx);
2434 MI.eraseFromParent();
2435 break;
2436 }
2437 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V1:
2438 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V2:
2439 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V3:
2440 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V4:
2441 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V5:
2442 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V6:
2443 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V7:
2444 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V8:
2445 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V9:
2446 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V10:
2447 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V11:
2448 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V12:
2449 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V16:
2450 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V32: {
2451 assert(ST.useVGPRIndexMode());
2452 Register VecReg = MI.getOperand(0).getReg();
2453 bool IsUndef = MI.getOperand(1).isUndef();
2454 MachineOperand &Idx = MI.getOperand(3);
2455 Register SubReg = MI.getOperand(4).getImm();
2456
2457 MachineInstr *SetOn = BuildMI(MBB, MI, DL, get(AMDGPU::S_SET_GPR_IDX_ON))
2458 .add(Idx)
2460 SetOn->getOperand(3).setIsUndef();
2461
2462 const MCInstrDesc &OpDesc = get(AMDGPU::V_MOV_B32_indirect_write);
2464 BuildMI(MBB, MI, DL, OpDesc)
2465 .addReg(RI.getSubReg(VecReg, SubReg), RegState::Undef)
2466 .add(MI.getOperand(2))
2468 .addReg(VecReg, RegState::Implicit | getUndefRegState(IsUndef));
2469
2470 const int ImpDefIdx =
2471 OpDesc.getNumOperands() + OpDesc.implicit_uses().size();
2472 const int ImpUseIdx = ImpDefIdx + 1;
2473 MIB->tieOperands(ImpDefIdx, ImpUseIdx);
2474
2475 MachineInstr *SetOff = BuildMI(MBB, MI, DL, get(AMDGPU::S_SET_GPR_IDX_OFF));
2476
2477 finalizeBundle(MBB, SetOn->getIterator(), std::next(SetOff->getIterator()));
2478
2479 MI.eraseFromParent();
2480 break;
2481 }
2482 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V1:
2483 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V2:
2484 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V3:
2485 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V4:
2486 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V5:
2487 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V6:
2488 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V7:
2489 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V8:
2490 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V9:
2491 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V10:
2492 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V11:
2493 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V12:
2494 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V16:
2495 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V32: {
2496 assert(ST.useVGPRIndexMode());
2497 Register Dst = MI.getOperand(0).getReg();
2498 Register VecReg = MI.getOperand(1).getReg();
2499 bool IsUndef = MI.getOperand(1).isUndef();
2500 Register SubReg = MI.getOperand(3).getImm();
2501
2502 MachineInstr *SetOn = BuildMI(MBB, MI, DL, get(AMDGPU::S_SET_GPR_IDX_ON))
2503 .add(MI.getOperand(2))
2505 SetOn->getOperand(3).setIsUndef();
2506
2507 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_indirect_read))
2508 .addDef(Dst)
2509 .addReg(RI.getSubReg(VecReg, SubReg), RegState::Undef)
2510 .addReg(VecReg, RegState::Implicit | getUndefRegState(IsUndef));
2511
2512 MachineInstr *SetOff = BuildMI(MBB, MI, DL, get(AMDGPU::S_SET_GPR_IDX_OFF));
2513
2514 finalizeBundle(MBB, SetOn->getIterator(), std::next(SetOff->getIterator()));
2515
2516 MI.eraseFromParent();
2517 break;
2518 }
2519 case AMDGPU::SI_PC_ADD_REL_OFFSET: {
2520 MachineFunction &MF = *MBB.getParent();
2521 Register Reg = MI.getOperand(0).getReg();
2522 Register RegLo = RI.getSubReg(Reg, AMDGPU::sub0);
2523 Register RegHi = RI.getSubReg(Reg, AMDGPU::sub1);
2524 MachineOperand OpLo = MI.getOperand(1);
2525 MachineOperand OpHi = MI.getOperand(2);
2526
2527 // Create a bundle so these instructions won't be re-ordered by the
2528 // post-RA scheduler.
2529 MIBundleBuilder Bundler(MBB, MI);
2530 Bundler.append(BuildMI(MF, DL, get(AMDGPU::S_GETPC_B64), Reg));
2531
2532 // What we want here is an offset from the value returned by s_getpc (which
2533 // is the address of the s_add_u32 instruction) to the global variable, but
2534 // since the encoding of $symbol starts 4 bytes after the start of the
2535 // s_add_u32 instruction, we end up with an offset that is 4 bytes too
2536 // small. This requires us to add 4 to the global variable offset in order
2537 // to compute the correct address. Similarly for the s_addc_u32 instruction,
2538 // the encoding of $symbol starts 12 bytes after the start of the s_add_u32
2539 // instruction.
2540
2541 int64_t Adjust = 0;
2542 if (ST.hasGetPCZeroExtension()) {
2543 // Fix up hardware that does not sign-extend the 48-bit PC value by
2544 // inserting: s_sext_i32_i16 reghi, reghi
2545 Bundler.append(
2546 BuildMI(MF, DL, get(AMDGPU::S_SEXT_I32_I16), RegHi).addReg(RegHi));
2547 Adjust += 4;
2548 }
2549
2550 if (OpLo.isGlobal())
2551 OpLo.setOffset(OpLo.getOffset() + Adjust + 4);
2552 Bundler.append(
2553 BuildMI(MF, DL, get(AMDGPU::S_ADD_U32), RegLo).addReg(RegLo).add(OpLo));
2554
2555 if (OpHi.isGlobal())
2556 OpHi.setOffset(OpHi.getOffset() + Adjust + 12);
2557 Bundler.append(BuildMI(MF, DL, get(AMDGPU::S_ADDC_U32), RegHi)
2558 .addReg(RegHi)
2559 .add(OpHi));
2560
2561 finalizeBundle(MBB, Bundler.begin());
2562
2563 MI.eraseFromParent();
2564 break;
2565 }
2566 case AMDGPU::SI_PC_ADD_REL_OFFSET64: {
2567 MachineFunction &MF = *MBB.getParent();
2568 Register Reg = MI.getOperand(0).getReg();
2569 MachineOperand Op = MI.getOperand(1);
2570
2571 // Create a bundle so these instructions won't be re-ordered by the
2572 // post-RA scheduler.
2573 MIBundleBuilder Bundler(MBB, MI);
2574 Bundler.append(BuildMI(MF, DL, get(AMDGPU::S_GETPC_B64), Reg));
2575 if (Op.isGlobal())
2576 Op.setOffset(Op.getOffset() + 4);
2577 Bundler.append(
2578 BuildMI(MF, DL, get(AMDGPU::S_ADD_U64), Reg).addReg(Reg).add(Op));
2579
2580 finalizeBundle(MBB, Bundler.begin());
2581
2582 MI.eraseFromParent();
2583 break;
2584 }
2585 case AMDGPU::ENTER_STRICT_WWM: {
2586 // This only gets its own opcode so that SIPreAllocateWWMRegs can tell when
2587 // Whole Wave Mode is entered.
2588 MI.setDesc(get(LMC.OrSaveExecOpc));
2589 break;
2590 }
2591 case AMDGPU::ENTER_STRICT_WQM: {
2592 // This only gets its own opcode so that SIPreAllocateWWMRegs can tell when
2593 // STRICT_WQM is entered.
2594 BuildMI(MBB, MI, DL, get(LMC.MovOpc), MI.getOperand(0).getReg())
2595 .addReg(LMC.ExecReg);
2596 BuildMI(MBB, MI, DL, get(LMC.WQMOpc), LMC.ExecReg).addReg(LMC.ExecReg);
2597
2598 MI.eraseFromParent();
2599 break;
2600 }
2601 case AMDGPU::EXIT_STRICT_WWM:
2602 case AMDGPU::EXIT_STRICT_WQM: {
2603 // This only gets its own opcode so that SIPreAllocateWWMRegs can tell when
2604 // WWM/STICT_WQM is exited.
2605 MI.setDesc(get(LMC.MovOpc));
2606 break;
2607 }
2608 case AMDGPU::SI_RETURN: {
2609 const MachineFunction *MF = MBB.getParent();
2610 const GCNSubtarget &ST = MF->getSubtarget<GCNSubtarget>();
2611 const SIRegisterInfo *TRI = ST.getRegisterInfo();
2612 // Hiding the return address use with SI_RETURN may lead to extra kills in
2613 // the function and missing live-ins. We are fine in practice because callee
2614 // saved register handling ensures the register value is restored before
2615 // RET, but we need the undef flag here to appease the MachineVerifier
2616 // liveness checks.
2618 BuildMI(MBB, MI, DL, get(AMDGPU::S_SETPC_B64_return))
2619 .addReg(TRI->getReturnAddressReg(*MF), RegState::Undef);
2620
2621 MIB.copyImplicitOps(MI);
2622 MI.eraseFromParent();
2623 break;
2624 }
2625
2626 case AMDGPU::S_MUL_U64_U32_PSEUDO:
2627 case AMDGPU::S_MUL_I64_I32_PSEUDO:
2628 MI.setDesc(get(AMDGPU::S_MUL_U64));
2629 break;
2630
2631 case AMDGPU::S_GETPC_B64_pseudo:
2632 MI.setDesc(get(AMDGPU::S_GETPC_B64));
2633 if (ST.hasGetPCZeroExtension()) {
2634 Register Dst = MI.getOperand(0).getReg();
2635 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2636 // Fix up hardware that does not sign-extend the 48-bit PC value by
2637 // inserting: s_sext_i32_i16 dsthi, dsthi
2638 BuildMI(MBB, std::next(MI.getIterator()), DL, get(AMDGPU::S_SEXT_I32_I16),
2639 DstHi)
2640 .addReg(DstHi);
2641 }
2642 break;
2643
2644 case AMDGPU::V_MAX_BF16_PSEUDO_e64: {
2645 assert(ST.hasBF16PackedInsts());
2646 MI.setDesc(get(AMDGPU::V_PK_MAX_NUM_BF16));
2647 MI.addOperand(MachineOperand::CreateImm(0)); // op_sel
2648 MI.addOperand(MachineOperand::CreateImm(0)); // neg_lo
2649 MI.addOperand(MachineOperand::CreateImm(0)); // neg_hi
2650 auto Op0 = getNamedOperand(MI, AMDGPU::OpName::src0_modifiers);
2651 Op0->setImm(Op0->getImm() | SISrcMods::OP_SEL_1);
2652 auto Op1 = getNamedOperand(MI, AMDGPU::OpName::src1_modifiers);
2653 Op1->setImm(Op1->getImm() | SISrcMods::OP_SEL_1);
2654 break;
2655 }
2656
2657 case AMDGPU::GET_STACK_BASE:
2658 // The stack starts at offset 0 unless we need to reserve some space at the
2659 // bottom.
2660 if (ST.getFrameLowering()->mayReserveScratchForCWSR(*MBB.getParent())) {
2661 // When CWSR is used in dynamic VGPR mode, the trap handler needs to save
2662 // some of the VGPRs. The size of the required scratch space has already
2663 // been computed by prolog epilog insertion.
2664 const SIMachineFunctionInfo *MFI =
2665 MBB.getParent()->getInfo<SIMachineFunctionInfo>();
2666 unsigned VGPRSize = MFI->getScratchReservedForDynamicVGPRs();
2667 Register DestReg = MI.getOperand(0).getReg();
2668 BuildMI(MBB, MI, DL, get(AMDGPU::S_GETREG_B32), DestReg)
2671 // The MicroEngine ID is 0 for the graphics queue, and 1 or 2 for compute
2672 // (3 is unused, so we ignore it). Unfortunately, S_GETREG doesn't set
2673 // SCC, so we need to check for 0 manually.
2674 BuildMI(MBB, MI, DL, get(AMDGPU::S_CMP_LG_U32)).addImm(0).addReg(DestReg);
2675 // Change the implicif-def of SCC to an explicit use (but first remove
2676 // the dead flag if present).
2677 MI.getOperand(MI.getNumExplicitOperands()).setIsDead(false);
2678 MI.getOperand(MI.getNumExplicitOperands()).setIsUse();
2679 MI.setDesc(get(AMDGPU::S_CMOVK_I32));
2680 MI.addOperand(MachineOperand::CreateImm(VGPRSize));
2681 } else {
2682 MI.setDesc(get(AMDGPU::S_MOV_B32));
2683 MI.addOperand(MachineOperand::CreateImm(0));
2684 MI.removeOperand(
2685 MI.getNumExplicitOperands()); // Drop implicit def of SCC.
2686 }
2687 break;
2688 }
2689
2690 return true;
2691}
2692
2695 unsigned SubIdx, const MachineInstr &Orig,
2696 LaneBitmask UsedLanes) const {
2697
2698 // Try shrinking the instruction to remat only the part needed for current
2699 // context.
2700 // TODO: Handle more cases.
2701 unsigned Opcode = Orig.getOpcode();
2702 switch (Opcode) {
2703 case AMDGPU::S_MOV_B64:
2704 case AMDGPU::S_MOV_B64_IMM_PSEUDO: {
2705 if (SubIdx != 0)
2706 break;
2707
2708 if (!Orig.getOperand(1).isImm())
2709 break;
2710
2711 // Shrink S_MOV_B64 to S_MOV_B32 when UsedLanes indicates only a single
2712 // 32-bit lane of the 64-bit value is live at the rematerialization point.
2713 if (UsedLanes.all())
2714 break;
2715
2716 // Determine which half of the 64-bit immediate corresponds to the use.
2717 unsigned OrigSubReg = Orig.getOperand(0).getSubReg();
2718 unsigned LoSubReg = RI.composeSubRegIndices(OrigSubReg, AMDGPU::sub0);
2719 unsigned HiSubReg = RI.composeSubRegIndices(OrigSubReg, AMDGPU::sub1);
2720
2721 bool NeedLo = (UsedLanes & RI.getSubRegIndexLaneMask(LoSubReg)).any();
2722 bool NeedHi = (UsedLanes & RI.getSubRegIndexLaneMask(HiSubReg)).any();
2723
2724 if (NeedLo && NeedHi)
2725 break;
2726
2727 int64_t Imm64 = Orig.getOperand(1).getImm();
2728 int32_t Imm32 = NeedLo ? Lo_32(Imm64) : Hi_32(Imm64);
2729
2730 unsigned UseSubReg = NeedLo ? LoSubReg : HiSubReg;
2731
2732 // Emit S_MOV_B32 defining just the needed 32-bit subreg of DestReg.
2733 BuildMI(MBB, I, Orig.getDebugLoc(), get(AMDGPU::S_MOV_B32))
2734 .addReg(DestReg, RegState::Define | RegState::Undef, UseSubReg)
2735 .addImm(Imm32);
2736 return;
2737 }
2738
2739 case AMDGPU::S_LOAD_DWORDX16_IMM:
2740 case AMDGPU::S_LOAD_DWORDX8_IMM: {
2741 if (SubIdx != 0)
2742 break;
2743
2744 if (I == MBB.end())
2745 break;
2746
2747 if (I->isBundled())
2748 break;
2749
2750 // Look for a single use of the register that is also a subreg.
2751 Register RegToFind = Orig.getOperand(0).getReg();
2752 MachineOperand *UseMO = nullptr;
2753 for (auto &CandMO : I->operands()) {
2754 if (!CandMO.isReg() || CandMO.getReg() != RegToFind || CandMO.isDef())
2755 continue;
2756 if (UseMO) {
2757 UseMO = nullptr;
2758 break;
2759 }
2760 UseMO = &CandMO;
2761 }
2762 if (!UseMO || UseMO->getSubReg() == AMDGPU::NoSubRegister)
2763 break;
2764
2765 unsigned Offset = RI.getSubRegIdxOffset(UseMO->getSubReg());
2766 unsigned SubregSize = RI.getSubRegIdxSize(UseMO->getSubReg());
2767
2768 MachineFunction *MF = MBB.getParent();
2769 MachineRegisterInfo &MRI = MF->getRegInfo();
2770 assert(MRI.use_nodbg_empty(DestReg) && "DestReg should have no users yet.");
2771
2772 unsigned NewOpcode = -1;
2773 if (SubregSize == 256)
2774 NewOpcode = AMDGPU::S_LOAD_DWORDX8_IMM;
2775 else if (SubregSize == 128)
2776 NewOpcode = AMDGPU::S_LOAD_DWORDX4_IMM;
2777 else
2778 break;
2779
2780 const MCInstrDesc &TID = get(NewOpcode);
2781 const TargetRegisterClass *NewRC =
2782 RI.getAllocatableClass(getRegClass(TID, 0));
2783 MRI.setRegClass(DestReg, NewRC);
2784
2785 UseMO->setReg(DestReg);
2786 UseMO->setSubReg(AMDGPU::NoSubRegister);
2787
2788 // Use a smaller load with the desired size, possibly with updated offset.
2789 MachineInstr *MI = MF->CloneMachineInstr(&Orig);
2790 MI->setDesc(TID);
2791 MI->getOperand(0).setReg(DestReg);
2792 MI->getOperand(0).setSubReg(AMDGPU::NoSubRegister);
2793 if (Offset) {
2794 MachineOperand *OffsetMO = getNamedOperand(*MI, AMDGPU::OpName::offset);
2795 int64_t FinalOffset = OffsetMO->getImm() + Offset / 8;
2796 OffsetMO->setImm(FinalOffset);
2797 }
2799 for (const MachineMemOperand *MemOp : Orig.memoperands())
2800 NewMMOs.push_back(MF->getMachineMemOperand(MemOp, MemOp->getPointerInfo(),
2801 SubregSize / 8));
2802 MI->setMemRefs(*MF, NewMMOs);
2803
2804 MBB.insert(I, MI);
2805 return;
2806 }
2807
2808 default:
2809 break;
2810 }
2811
2812 TargetInstrInfo::reMaterialize(MBB, I, DestReg, SubIdx, Orig, UsedLanes);
2813}
2814
2815std::pair<MachineInstr*, MachineInstr*>
2817 assert (MI.getOpcode() == AMDGPU::V_MOV_B64_DPP_PSEUDO);
2818
2819 if (ST.hasVMovB64Inst() && ST.hasFeature(AMDGPU::FeatureDPALU_DPP) &&
2821 ST, getNamedOperand(MI, AMDGPU::OpName::dpp_ctrl)->getImm())) {
2822 MI.setDesc(get(AMDGPU::V_MOV_B64_dpp));
2823 return std::pair(&MI, nullptr);
2824 }
2825
2826 MachineBasicBlock &MBB = *MI.getParent();
2827 DebugLoc DL = MBB.findDebugLoc(MI);
2828 MachineFunction *MF = MBB.getParent();
2829 MachineRegisterInfo &MRI = MF->getRegInfo();
2830 Register Dst = MI.getOperand(0).getReg();
2831 unsigned Part = 0;
2832 MachineInstr *Split[2];
2833
2834 for (auto Sub : { AMDGPU::sub0, AMDGPU::sub1 }) {
2835 auto MovDPP = BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_dpp));
2836 if (Dst.isPhysical()) {
2837 MovDPP.addDef(RI.getSubReg(Dst, Sub));
2838 } else {
2839 assert(MRI.isSSA());
2840 auto Tmp = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
2841 MovDPP.addDef(Tmp);
2842 }
2843
2844 for (unsigned I = 1; I <= 2; ++I) { // old and src operands.
2845 const MachineOperand &SrcOp = MI.getOperand(I);
2846 assert(!SrcOp.isFPImm());
2847 if (SrcOp.isImm()) {
2848 APInt Imm(64, SrcOp.getImm());
2849 Imm.ashrInPlace(Part * 32);
2850 MovDPP.addImm(Imm.getLoBits(32).getZExtValue());
2851 } else {
2852 assert(SrcOp.isReg());
2853 Register Src = SrcOp.getReg();
2854 if (Src.isPhysical())
2855 MovDPP.addReg(RI.getSubReg(Src, Sub));
2856 else
2857 MovDPP.addReg(Src, getUndefRegState(SrcOp.isUndef()), Sub);
2858 }
2859 }
2860
2861 for (const MachineOperand &MO : llvm::drop_begin(MI.explicit_operands(), 3))
2862 MovDPP.addImm(MO.getImm());
2863
2864 Split[Part] = MovDPP;
2865 ++Part;
2866 }
2867
2868 if (Dst.isVirtual())
2869 BuildMI(MBB, MI, DL, get(AMDGPU::REG_SEQUENCE), Dst)
2870 .addReg(Split[0]->getOperand(0).getReg())
2871 .addImm(AMDGPU::sub0)
2872 .addReg(Split[1]->getOperand(0).getReg())
2873 .addImm(AMDGPU::sub1);
2874
2875 MI.eraseFromParent();
2876 return std::pair(Split[0], Split[1]);
2877}
2878
2879std::optional<DestSourcePair>
2881 if (MI.getOpcode() == AMDGPU::WWM_COPY)
2882 return DestSourcePair{MI.getOperand(0), MI.getOperand(1)};
2883
2884 return std::nullopt;
2885}
2886
2888 AMDGPU::OpName Src0OpName,
2889 MachineOperand &Src1,
2890 AMDGPU::OpName Src1OpName) const {
2891 MachineOperand *Src0Mods = getNamedOperand(MI, Src0OpName);
2892 if (!Src0Mods)
2893 return false;
2894
2895 MachineOperand *Src1Mods = getNamedOperand(MI, Src1OpName);
2896 assert(Src1Mods &&
2897 "All commutable instructions have both src0 and src1 modifiers");
2898
2899 int Src0ModsVal = Src0Mods->getImm();
2900 int Src1ModsVal = Src1Mods->getImm();
2901
2902 Src1Mods->setImm(Src0ModsVal);
2903 Src0Mods->setImm(Src1ModsVal);
2904 return true;
2905}
2906
2908 MachineOperand &RegOp,
2909 MachineOperand &NonRegOp) {
2910 Register Reg = RegOp.getReg();
2911 unsigned SubReg = RegOp.getSubReg();
2912 bool IsKill = RegOp.isKill();
2913 bool IsDead = RegOp.isDead();
2914 bool IsUndef = RegOp.isUndef();
2915 bool IsDebug = RegOp.isDebug();
2916
2917 if (NonRegOp.isImm())
2918 RegOp.ChangeToImmediate(NonRegOp.getImm());
2919 else if (NonRegOp.isFI())
2920 RegOp.ChangeToFrameIndex(NonRegOp.getIndex());
2921 else if (NonRegOp.isGlobal()) {
2922 RegOp.ChangeToGA(NonRegOp.getGlobal(), NonRegOp.getOffset(),
2923 NonRegOp.getTargetFlags());
2924 } else
2925 return nullptr;
2926
2927 // Make sure we don't reinterpret a subreg index in the target flags.
2928 RegOp.setTargetFlags(NonRegOp.getTargetFlags());
2929
2930 NonRegOp.ChangeToRegister(Reg, false, false, IsKill, IsDead, IsUndef, IsDebug);
2931 NonRegOp.setSubReg(SubReg);
2932
2933 return &MI;
2934}
2935
2937 MachineOperand &NonRegOp1,
2938 MachineOperand &NonRegOp2) {
2939 unsigned TargetFlags = NonRegOp1.getTargetFlags();
2940 int64_t NonRegVal = NonRegOp1.getImm();
2941
2942 NonRegOp1.setImm(NonRegOp2.getImm());
2943 NonRegOp2.setImm(NonRegVal);
2944 NonRegOp1.setTargetFlags(NonRegOp2.getTargetFlags());
2945 NonRegOp2.setTargetFlags(TargetFlags);
2946 return &MI;
2947}
2948
2949bool SIInstrInfo::isLegalToSwap(const MachineInstr &MI, unsigned OpIdx0,
2950 unsigned OpIdx1) const {
2951 const MCInstrDesc &InstDesc = MI.getDesc();
2952 const MCOperandInfo &OpInfo0 = InstDesc.operands()[OpIdx0];
2953 const MCOperandInfo &OpInfo1 = InstDesc.operands()[OpIdx1];
2954
2955 unsigned Opc = MI.getOpcode();
2956 int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
2957
2958 const MachineOperand &MO0 = MI.getOperand(OpIdx0);
2959 const MachineOperand &MO1 = MI.getOperand(OpIdx1);
2960
2961 // Swap doesn't breach constant bus or literal limits
2962 // It may move literal to position other than src0, this is not allowed
2963 // pre-gfx10 However, most test cases need literals in Src0 for VOP
2964 // FIXME: After gfx9, literal can be in place other than Src0
2965 if (isVALU(MI, /*AllowLDSDMA=*/false)) {
2966 if ((int)OpIdx0 == Src0Idx && !MO0.isReg() &&
2967 !isInlineConstant(MO0, OpInfo1))
2968 return false;
2969 if ((int)OpIdx1 == Src0Idx && !MO1.isReg() &&
2970 !isInlineConstant(MO1, OpInfo0))
2971 return false;
2972 }
2973
2974 if ((int)OpIdx1 != Src0Idx && MO0.isReg()) {
2975 if (OpInfo1.RegClass == -1)
2976 return OpInfo1.OperandType == MCOI::OPERAND_UNKNOWN;
2977 return isLegalRegOperand(MI, OpIdx1, MO0) &&
2978 (!MO1.isReg() || isLegalRegOperand(MI, OpIdx0, MO1));
2979 }
2980 if ((int)OpIdx0 != Src0Idx && MO1.isReg()) {
2981 if (OpInfo0.RegClass == -1)
2982 return OpInfo0.OperandType == MCOI::OPERAND_UNKNOWN;
2983 return (!MO0.isReg() || isLegalRegOperand(MI, OpIdx1, MO0)) &&
2984 isLegalRegOperand(MI, OpIdx0, MO1);
2985 }
2986
2987 // No need to check 64-bit literals since swapping does not bring new
2988 // 64-bit literals into current instruction to fold to 32-bit
2989
2990 return isImmOperandLegal(MI, OpIdx1, MO0);
2991}
2992
2994 if (!isDPP(MI))
2995 return false;
2996 const MachineOperand *DppCtrl = getNamedOperand(MI, AMDGPU::OpName::dpp_ctrl);
2997 return !DppCtrl || DppCtrl->getImm() != AMDGPU::DPP::QUAD_PERM_ID;
2998}
2999
3001 unsigned Src0Idx,
3002 unsigned Src1Idx) const {
3003 assert(!NewMI && "this should never be used");
3004
3006 return nullptr;
3007
3008 unsigned Opc = MI.getOpcode();
3009 int CommutedOpcode = commuteOpcode(Opc);
3010 if (CommutedOpcode == -1)
3011 return nullptr;
3012
3013 if (Src0Idx > Src1Idx)
3014 std::swap(Src0Idx, Src1Idx);
3015
3016 assert(AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0) ==
3017 static_cast<int>(Src0Idx) &&
3018 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1) ==
3019 static_cast<int>(Src1Idx) &&
3020 "inconsistency with findCommutedOpIndices");
3021
3022 if (!isLegalToSwap(MI, Src0Idx, Src1Idx))
3023 return nullptr;
3024
3025 MachineInstr *CommutedMI = nullptr;
3026 MachineOperand &Src0 = MI.getOperand(Src0Idx);
3027 MachineOperand &Src1 = MI.getOperand(Src1Idx);
3028 if (Src0.isReg() && Src1.isReg()) {
3029 // Be sure to copy the source modifiers to the right place.
3030 CommutedMI =
3031 TargetInstrInfo::commuteInstructionImpl(MI, NewMI, Src0Idx, Src1Idx);
3032 } else if (Src0.isReg() && !Src1.isReg()) {
3033 CommutedMI = swapRegAndNonRegOperand(MI, Src0, Src1);
3034 } else if (!Src0.isReg() && Src1.isReg()) {
3035 CommutedMI = swapRegAndNonRegOperand(MI, Src1, Src0);
3036 } else if (Src0.isImm() && Src1.isImm()) {
3037 CommutedMI = swapImmOperands(MI, Src0, Src1);
3038 } else {
3039 // FIXME: Found two non registers to commute. This does happen.
3040 return nullptr;
3041 }
3042
3043 if (CommutedMI) {
3044 swapSourceModifiers(MI, Src0, AMDGPU::OpName::src0_modifiers,
3045 Src1, AMDGPU::OpName::src1_modifiers);
3046
3047 swapSourceModifiers(MI, Src0, AMDGPU::OpName::src0_sel, Src1,
3048 AMDGPU::OpName::src1_sel);
3049
3050 CommutedMI->setDesc(get(CommutedOpcode));
3051 }
3052
3053 return CommutedMI;
3054}
3055
3056// This needs to be implemented because the source modifiers may be inserted
3057// between the true commutable operands, and the base
3058// TargetInstrInfo::commuteInstruction uses it.
3060 unsigned &SrcOpIdx0,
3061 unsigned &SrcOpIdx1) const {
3063 return false;
3064
3065 return findCommutedOpIndices(MI.getDesc(), SrcOpIdx0, SrcOpIdx1);
3066}
3067
3069 unsigned &SrcOpIdx0,
3070 unsigned &SrcOpIdx1) const {
3071 if (!Desc.isCommutable())
3072 return false;
3073
3074 unsigned Opc = Desc.getOpcode();
3075 int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
3076 if (Src0Idx == -1)
3077 return false;
3078
3079 int Src1Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1);
3080 if (Src1Idx == -1)
3081 return false;
3082
3083 return fixCommutedOpIndices(SrcOpIdx0, SrcOpIdx1, Src0Idx, Src1Idx);
3084}
3085
3087 int64_t BrOffset) const {
3088 // BranchRelaxation should never have to check s_setpc_b64 or s_add_pc_i64
3089 // because its dest block is unanalyzable.
3090 assert(isSOPP(BranchOp) || isSOPK(BranchOp));
3091
3092 // Convert to dwords.
3093 BrOffset /= 4;
3094
3095 // The branch instructions do PC += signext(SIMM16 * 4) + 4, so the offset is
3096 // from the next instruction.
3097 BrOffset -= 1;
3098
3099 return isIntN(BranchOffsetBits, BrOffset);
3100}
3101
3104 return MI.getOperand(0).getMBB();
3105}
3106
3108 for (const MachineInstr &MI : MBB->terminators()) {
3109 if (MI.getOpcode() == AMDGPU::SI_IF || MI.getOpcode() == AMDGPU::SI_ELSE ||
3110 MI.getOpcode() == AMDGPU::SI_LOOP ||
3111 MI.getOpcode() == AMDGPU::SI_WATERFALL_LOOP)
3112 return true;
3113 }
3114 return false;
3115}
3116
3118 MachineBasicBlock &DestBB,
3119 MachineBasicBlock &RestoreBB,
3120 const DebugLoc &DL, int64_t BrOffset,
3121 RegScavenger *RS) const {
3122 assert(MBB.empty() &&
3123 "new block should be inserted for expanding unconditional branch");
3124 assert(MBB.pred_size() == 1);
3125 assert(RestoreBB.empty() &&
3126 "restore block should be inserted for restoring clobbered registers");
3127
3128 MachineFunction *MF = MBB.getParent();
3129 MachineRegisterInfo &MRI = MF->getRegInfo();
3131 auto I = MBB.end();
3132 auto &MCCtx = MF->getContext();
3133
3134 if (ST.useAddPC64Inst()) {
3135 MCSymbol *Offset =
3136 MCCtx.createTempSymbol("offset", /*AlwaysAddSuffix=*/true);
3137 auto AddPC = BuildMI(MBB, I, DL, get(AMDGPU::S_ADD_PC_I64))
3139 MCSymbol *PostAddPCLabel =
3140 MCCtx.createTempSymbol("post_addpc", /*AlwaysAddSuffix=*/true);
3141 AddPC->setPostInstrSymbol(*MF, PostAddPCLabel);
3142 auto *OffsetExpr = MCBinaryExpr::createSub(
3143 MCSymbolRefExpr::create(DestBB.getSymbol(), MCCtx),
3144 MCSymbolRefExpr::create(PostAddPCLabel, MCCtx), MCCtx);
3145 Offset->setVariableValue(OffsetExpr);
3146 return;
3147 }
3148
3149 assert(RS && "RegScavenger required for long branching");
3150
3151 // FIXME: Virtual register workaround for RegScavenger not working with empty
3152 // blocks.
3153 Register PCReg = MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
3154
3155 // Note: as this is used after hazard recognizer we need to apply some hazard
3156 // workarounds directly.
3157 const bool FlushSGPRWrites = (ST.isWave64() && ST.hasVALUMaskWriteHazard()) ||
3158 ST.hasVALUReadSGPRHazard();
3159 auto ApplyHazardWorkarounds = [this, &MBB, &I, &DL, FlushSGPRWrites]() {
3160 if (FlushSGPRWrites)
3161 BuildMI(MBB, I, DL, get(AMDGPU::S_WAITCNT_DEPCTR))
3163 };
3164
3165 // We need to compute the offset relative to the instruction immediately after
3166 // s_getpc_b64. Insert pc arithmetic code before last terminator.
3167 MachineInstr *GetPC = BuildMI(MBB, I, DL, get(AMDGPU::S_GETPC_B64), PCReg);
3168 ApplyHazardWorkarounds();
3169
3170 MCSymbol *PostGetPCLabel =
3171 MCCtx.createTempSymbol("post_getpc", /*AlwaysAddSuffix=*/true);
3172 GetPC->setPostInstrSymbol(*MF, PostGetPCLabel);
3173
3174 MCSymbol *OffsetLo =
3175 MCCtx.createTempSymbol("offset_lo", /*AlwaysAddSuffix=*/true);
3176 MCSymbol *OffsetHi =
3177 MCCtx.createTempSymbol("offset_hi", /*AlwaysAddSuffix=*/true);
3178 BuildMI(MBB, I, DL, get(AMDGPU::S_ADD_U32))
3179 .addReg(PCReg, RegState::Define, AMDGPU::sub0)
3180 .addReg(PCReg, {}, AMDGPU::sub0)
3181 .addSym(OffsetLo, MO_FAR_BRANCH_OFFSET);
3182 BuildMI(MBB, I, DL, get(AMDGPU::S_ADDC_U32))
3183 .addReg(PCReg, RegState::Define, AMDGPU::sub1)
3184 .addReg(PCReg, {}, AMDGPU::sub1)
3185 .addSym(OffsetHi, MO_FAR_BRANCH_OFFSET);
3186 ApplyHazardWorkarounds();
3187
3188 // Insert the indirect branch after the other terminator.
3189 BuildMI(&MBB, DL, get(AMDGPU::S_SETPC_B64))
3190 .addReg(PCReg);
3191
3192 // If a spill is needed for the pc register pair, we need to insert a spill
3193 // restore block right before the destination block, and insert a short branch
3194 // into the old destination block's fallthrough predecessor.
3195 // e.g.:
3196 //
3197 // s_cbranch_scc0 skip_long_branch:
3198 //
3199 // long_branch_bb:
3200 // spill s[8:9]
3201 // s_getpc_b64 s[8:9]
3202 // s_add_u32 s8, s8, restore_bb
3203 // s_addc_u32 s9, s9, 0
3204 // s_setpc_b64 s[8:9]
3205 //
3206 // skip_long_branch:
3207 // foo;
3208 //
3209 // .....
3210 //
3211 // dest_bb_fallthrough_predecessor:
3212 // bar;
3213 // s_branch dest_bb
3214 //
3215 // restore_bb:
3216 // restore s[8:9]
3217 // fallthrough dest_bb
3218 ///
3219 // dest_bb:
3220 // buzz;
3221
3222 Register LongBranchReservedReg = MFI->getLongBranchReservedReg();
3223 Register Scav;
3224
3225 // If we've previously reserved a register for long branches
3226 // avoid running the scavenger and just use those registers
3227 if (LongBranchReservedReg) {
3228 RS->enterBasicBlock(MBB);
3229 Scav = LongBranchReservedReg;
3230 } else {
3231 RS->enterBasicBlockEnd(MBB);
3232 Scav = RS->scavengeRegisterBackwards(
3233 AMDGPU::SReg_64RegClass, MachineBasicBlock::iterator(GetPC),
3234 /* RestoreAfter */ false, 0, /* AllowSpill */ false);
3235 }
3236 if (Scav) {
3237 RS->setRegUsed(Scav);
3238 MRI.replaceRegWith(PCReg, Scav);
3239 MRI.clearVirtRegs();
3240 } else {
3241 // As SGPR needs VGPR to be spilled, we reuse the slot of temporary VGPR for
3242 // SGPR spill.
3243 const GCNSubtarget &ST = MF->getSubtarget<GCNSubtarget>();
3244 const SIRegisterInfo *TRI = ST.getRegisterInfo();
3245 TRI->spillEmergencySGPR(GetPC, RestoreBB, AMDGPU::SGPR0_SGPR1, RS);
3246 MRI.replaceRegWith(PCReg, AMDGPU::SGPR0_SGPR1);
3247 MRI.clearVirtRegs();
3248 }
3249
3250 MCSymbol *DestLabel = Scav ? DestBB.getSymbol() : RestoreBB.getSymbol();
3251 // Now, the distance could be defined.
3253 MCSymbolRefExpr::create(DestLabel, MCCtx),
3254 MCSymbolRefExpr::create(PostGetPCLabel, MCCtx), MCCtx);
3255 // Add offset assignments.
3256 auto *Mask = MCConstantExpr::create(0xFFFFFFFFULL, MCCtx);
3257 OffsetLo->setVariableValue(MCBinaryExpr::createAnd(Offset, Mask, MCCtx));
3258 auto *ShAmt = MCConstantExpr::create(32, MCCtx);
3259 OffsetHi->setVariableValue(MCBinaryExpr::createAShr(Offset, ShAmt, MCCtx));
3260}
3261
3262unsigned SIInstrInfo::getBranchOpcode(SIInstrInfo::BranchPredicate Cond) {
3263 switch (Cond) {
3264 case SIInstrInfo::SCC_TRUE:
3265 return AMDGPU::S_CBRANCH_SCC1;
3266 case SIInstrInfo::SCC_FALSE:
3267 return AMDGPU::S_CBRANCH_SCC0;
3268 case SIInstrInfo::VCCNZ:
3269 return AMDGPU::S_CBRANCH_VCCNZ;
3270 case SIInstrInfo::VCCZ:
3271 return AMDGPU::S_CBRANCH_VCCZ;
3272 case SIInstrInfo::EXECNZ:
3273 return AMDGPU::S_CBRANCH_EXECNZ;
3274 case SIInstrInfo::EXECZ:
3275 return AMDGPU::S_CBRANCH_EXECZ;
3276 default:
3277 llvm_unreachable("invalid branch predicate");
3278 }
3279}
3280
3281SIInstrInfo::BranchPredicate SIInstrInfo::getBranchPredicate(unsigned Opcode) {
3282 switch (Opcode) {
3283 case AMDGPU::S_CBRANCH_SCC0:
3284 return SCC_FALSE;
3285 case AMDGPU::S_CBRANCH_SCC1:
3286 return SCC_TRUE;
3287 case AMDGPU::S_CBRANCH_VCCNZ:
3288 return VCCNZ;
3289 case AMDGPU::S_CBRANCH_VCCZ:
3290 return VCCZ;
3291 case AMDGPU::S_CBRANCH_EXECNZ:
3292 return EXECNZ;
3293 case AMDGPU::S_CBRANCH_EXECZ:
3294 return EXECZ;
3295 default:
3296 return INVALID_BR;
3297 }
3298}
3299
3303 MachineBasicBlock *&FBB,
3305 bool AllowModify) const {
3306 if (I->getOpcode() == AMDGPU::S_BRANCH) {
3307 // Unconditional Branch
3308 TBB = I->getOperand(0).getMBB();
3309 return false;
3310 }
3311
3312 BranchPredicate Pred = getBranchPredicate(I->getOpcode());
3313 if (Pred == INVALID_BR)
3314 return true;
3315
3316 MachineBasicBlock *CondBB = I->getOperand(0).getMBB();
3317 Cond.push_back(MachineOperand::CreateImm(Pred));
3318 Cond.push_back(I->getOperand(1)); // Save the branch register.
3319
3320 ++I;
3321
3322 if (I == MBB.end()) {
3323 // Conditional branch followed by fall-through.
3324 TBB = CondBB;
3325 return false;
3326 }
3327
3328 if (I->getOpcode() == AMDGPU::S_BRANCH) {
3329 TBB = CondBB;
3330 FBB = I->getOperand(0).getMBB();
3331 return false;
3332 }
3333
3334 return true;
3335}
3336
3338 MachineBasicBlock *&FBB,
3340 bool AllowModify) const {
3341 MachineBasicBlock::iterator I = MBB.getFirstTerminator();
3342 auto E = MBB.end();
3343 if (I == E)
3344 return false;
3345
3346 // Skip over the instructions that are artificially terminators for special
3347 // exec management.
3348 while (I != E && !I->isBranch() && !I->isReturn()) {
3349 switch (I->getOpcode()) {
3350 case AMDGPU::S_MOV_B64_term:
3351 case AMDGPU::S_XOR_B64_term:
3352 case AMDGPU::S_OR_B64_term:
3353 case AMDGPU::S_ANDN2_B64_term:
3354 case AMDGPU::S_AND_B64_term:
3355 case AMDGPU::S_AND_SAVEEXEC_B64_term:
3356 case AMDGPU::S_MOV_B32_term:
3357 case AMDGPU::S_XOR_B32_term:
3358 case AMDGPU::S_OR_B32_term:
3359 case AMDGPU::S_ANDN2_B32_term:
3360 case AMDGPU::S_AND_B32_term:
3361 case AMDGPU::S_AND_SAVEEXEC_B32_term:
3362 case AMDGPU::V_CMPX_EQ_U32_nosdst_e32_term:
3363 case AMDGPU::V_CMPX_EQ_U64_nosdst_e32_term:
3364 break;
3365 case AMDGPU::SI_IF:
3366 case AMDGPU::SI_ELSE:
3367 case AMDGPU::SI_KILL_I1_TERMINATOR:
3368 case AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR:
3369 // FIXME: It's messy that these need to be considered here at all.
3370 return true;
3371 default:
3372 llvm_unreachable("unexpected non-branch terminator inst");
3373 }
3374
3375 ++I;
3376 }
3377
3378 if (I == E)
3379 return false;
3380
3381 return analyzeBranchImpl(MBB, I, TBB, FBB, Cond, AllowModify);
3382}
3383
3385 int *BytesRemoved) const {
3386 unsigned Count = 0;
3387 unsigned RemovedSize = 0;
3388 for (MachineInstr &MI : llvm::make_early_inc_range(MBB.terminators())) {
3389 // Skip over artificial terminators when removing instructions.
3390 if (MI.isBranch() || MI.isReturn()) {
3391 RemovedSize += getInstSizeInBytes(MI);
3392 MI.eraseFromParent();
3393 ++Count;
3394 }
3395 }
3396
3397 if (BytesRemoved)
3398 *BytesRemoved = RemovedSize;
3399
3400 return Count;
3401}
3402
3403// Copy the flags onto the implicit condition register operand.
3405 const MachineOperand &OrigCond) {
3406 CondReg.setIsUndef(OrigCond.isUndef());
3407 CondReg.setIsKill(OrigCond.isKill());
3408}
3409
3412 MachineBasicBlock *FBB,
3414 const DebugLoc &DL,
3415 int *BytesAdded) const {
3416 if (!FBB && Cond.empty()) {
3417 BuildMI(&MBB, DL, get(AMDGPU::S_BRANCH))
3418 .addMBB(TBB);
3419 if (BytesAdded)
3420 *BytesAdded = ST.hasOffset3fBug() ? 8 : 4;
3421 return 1;
3422 }
3423
3424 assert(TBB && Cond[0].isImm());
3425
3426 unsigned Opcode
3427 = getBranchOpcode(static_cast<BranchPredicate>(Cond[0].getImm()));
3428
3429 if (!FBB) {
3430 MachineInstr *CondBr =
3431 BuildMI(&MBB, DL, get(Opcode))
3432 .addMBB(TBB);
3433
3434 // Copy the flags onto the implicit condition register operand.
3435 preserveCondRegFlags(CondBr->getOperand(1), Cond[1]);
3436 fixImplicitOperands(*CondBr);
3437
3438 if (BytesAdded)
3439 *BytesAdded = ST.hasOffset3fBug() ? 8 : 4;
3440 return 1;
3441 }
3442
3443 assert(TBB && FBB);
3444
3445 MachineInstr *CondBr =
3446 BuildMI(&MBB, DL, get(Opcode))
3447 .addMBB(TBB);
3448 fixImplicitOperands(*CondBr);
3449 BuildMI(&MBB, DL, get(AMDGPU::S_BRANCH))
3450 .addMBB(FBB);
3451
3452 MachineOperand &CondReg = CondBr->getOperand(1);
3453 CondReg.setIsUndef(Cond[1].isUndef());
3454 CondReg.setIsKill(Cond[1].isKill());
3455
3456 if (BytesAdded)
3457 *BytesAdded = ST.hasOffset3fBug() ? 16 : 8;
3458
3459 return 2;
3460}
3461
3464 if (Cond.size() != 2) {
3465 return true;
3466 }
3467
3468 if (Cond[0].isImm()) {
3469 Cond[0].setImm(-Cond[0].getImm());
3470 return false;
3471 }
3472
3473 return true;
3474}
3475
3476namespace {
3477class AMDGPUPipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
3478private:
3479 /// The compare instruction for loop control
3480 const MachineInstr *CmpInst = nullptr;
3481 /// The normalized condition used by createTripCountGreaterCondition()
3483
3484public:
3485 AMDGPUPipelinerLoopInfo(MachineInstr *CmpInst,
3487 : CmpInst(CmpInst), Cond(Cond.begin(), Cond.end()) {}
3488
3489 bool shouldIgnoreForPipelining(const MachineInstr *MI) const override {
3490 return CmpInst && MI == CmpInst;
3491 }
3492
3493 std::optional<bool> createTripCountGreaterCondition(
3494 int TC, MachineBasicBlock &MBB,
3495 SmallVectorImpl<MachineOperand> &CondParam) override {
3496 CondParam = this->Cond;
3497 return {};
3498 }
3499
3500 void adjustTripCount(int TripCountAdjust) override {}
3501
3502 void setPreheader(MachineBasicBlock *NewPreheader) override {}
3503};
3504} // namespace
3505
3506std::unique_ptr<TargetInstrInfo::PipelinerLoopInfo>
3508 MachineBasicBlock *TBB = nullptr, *FBB = nullptr;
3510 // Unanalyzable terminator.
3511 if (analyzeBranch(*LoopBB, TBB, FBB, Cond, /*AllowModify=*/false))
3512 return nullptr;
3513
3514 // Infinite loops are not supported.
3515 if (TBB == LoopBB && FBB == LoopBB)
3516 return nullptr;
3517
3518 // Must be conditional branch.
3519 if (FBB == nullptr)
3520 return nullptr;
3521
3522 assert((TBB == LoopBB || FBB == LoopBB) &&
3523 "The Loop must be a single-basic-block loop");
3524
3525 // Divergent (VCC/EXEC) back-edge is not supported.
3526 BranchPredicate Pred = static_cast<BranchPredicate>(Cond[0].getImm());
3527 if (Pred != SCC_TRUE && Pred != SCC_FALSE)
3528 return nullptr;
3529
3530 // Calls and inline assembly are not supported.
3531 for (const MachineInstr &MI : *LoopBB)
3532 if (MI.isCall() || MI.isInlineAsm())
3533 return nullptr;
3534
3535 // Normalization for createTripCountGreaterCondition(): make Cond mean
3536 // "exit the loop" so the expander emits correct prolog guard branches.
3537 if (TBB == LoopBB)
3539
3540 auto Instructions = make_range(
3542 LoopBB->rend());
3543 auto CmpI = llvm::find_if(Instructions, [&](const MachineInstr &MI) {
3544 return MI.modifiesRegister(Cond[1].getReg(), &RI);
3545 });
3546
3547 if (CmpI == Instructions.end() || CmpI->isPHI())
3548 return nullptr;
3549 MachineInstr *CmpInst = &*CmpI;
3550
3551 return std::make_unique<AMDGPUPipelinerLoopInfo>(CmpInst, Cond);
3552}
3553
3556 Register DstReg, Register TrueReg,
3557 Register FalseReg, int &CondCycles,
3558 int &TrueCycles, int &FalseCycles) const {
3559 switch (Cond[0].getImm()) {
3560 case VCCNZ:
3561 case VCCZ: {
3562 const MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
3563 const TargetRegisterClass *RC = MRI.getRegClass(TrueReg);
3564 if (MRI.getRegClass(FalseReg) != RC)
3565 return false;
3566
3567 int NumInsts = AMDGPU::getRegBitWidth(*RC) / 32;
3568 CondCycles = TrueCycles = FalseCycles = NumInsts; // ???
3569
3570 // Limit to equal cost for branch vs. N v_cndmask_b32s.
3571 return RI.hasVGPRs(RC) && NumInsts <= 6;
3572 }
3573 case SCC_TRUE:
3574 case SCC_FALSE: {
3575 // FIXME: We could insert for VGPRs if we could replace the original compare
3576 // with a vector one.
3577 const MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
3578 const TargetRegisterClass *RC = MRI.getRegClass(TrueReg);
3579 if (MRI.getRegClass(FalseReg) != RC)
3580 return false;
3581
3582 int NumInsts = AMDGPU::getRegBitWidth(*RC) / 32;
3583
3584 // Multiples of 8 can do s_cselect_b64
3585 if (NumInsts % 2 == 0)
3586 NumInsts /= 2;
3587
3588 CondCycles = TrueCycles = FalseCycles = NumInsts; // ???
3589 return RI.isSGPRClass(RC);
3590 }
3591 default:
3592 return false;
3593 }
3594}
3595
3599 Register TrueReg, Register FalseReg) const {
3600 BranchPredicate Pred = static_cast<BranchPredicate>(Cond[0].getImm());
3601 if (Pred == VCCZ || Pred == SCC_FALSE) {
3602 Pred = static_cast<BranchPredicate>(-Pred);
3603 std::swap(TrueReg, FalseReg);
3604 }
3605
3606 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
3607 const TargetRegisterClass *DstRC = MRI.getRegClass(DstReg);
3608 unsigned DstSize = RI.getRegSizeInBits(*DstRC);
3609
3610 if (DstSize == 32) {
3612 if (Pred == SCC_TRUE) {
3613 Select = BuildMI(MBB, I, DL, get(AMDGPU::S_CSELECT_B32), DstReg)
3614 .addReg(TrueReg)
3615 .addReg(FalseReg);
3616 } else {
3617 // Instruction's operands are backwards from what is expected.
3618 Select = BuildMI(MBB, I, DL, get(AMDGPU::V_CNDMASK_B32_e32), DstReg)
3619 .addReg(FalseReg)
3620 .addReg(TrueReg);
3621 }
3622
3623 preserveCondRegFlags(Select->getOperand(3), Cond[1]);
3624 return;
3625 }
3626
3627 if (DstSize == 64 && Pred == SCC_TRUE) {
3629 BuildMI(MBB, I, DL, get(AMDGPU::S_CSELECT_B64), DstReg)
3630 .addReg(TrueReg)
3631 .addReg(FalseReg);
3632
3633 preserveCondRegFlags(Select->getOperand(3), Cond[1]);
3634 return;
3635 }
3636
3637 static const int16_t Sub0_15[] = {
3638 AMDGPU::sub0, AMDGPU::sub1, AMDGPU::sub2, AMDGPU::sub3,
3639 AMDGPU::sub4, AMDGPU::sub5, AMDGPU::sub6, AMDGPU::sub7,
3640 AMDGPU::sub8, AMDGPU::sub9, AMDGPU::sub10, AMDGPU::sub11,
3641 AMDGPU::sub12, AMDGPU::sub13, AMDGPU::sub14, AMDGPU::sub15,
3642 };
3643
3644 static const int16_t Sub0_15_64[] = {
3645 AMDGPU::sub0_sub1, AMDGPU::sub2_sub3,
3646 AMDGPU::sub4_sub5, AMDGPU::sub6_sub7,
3647 AMDGPU::sub8_sub9, AMDGPU::sub10_sub11,
3648 AMDGPU::sub12_sub13, AMDGPU::sub14_sub15,
3649 };
3650
3651 unsigned SelOp = AMDGPU::V_CNDMASK_B32_e32;
3652 const TargetRegisterClass *EltRC = &AMDGPU::VGPR_32RegClass;
3653 const int16_t *SubIndices = Sub0_15;
3654 int NElts = DstSize / 32;
3655
3656 // 64-bit select is only available for SALU.
3657 // TODO: Split 96-bit into 64-bit and 32-bit, not 3x 32-bit.
3658 if (Pred == SCC_TRUE) {
3659 if (NElts % 2) {
3660 SelOp = AMDGPU::S_CSELECT_B32;
3661 EltRC = &AMDGPU::SGPR_32RegClass;
3662 } else {
3663 SelOp = AMDGPU::S_CSELECT_B64;
3664 EltRC = &AMDGPU::SGPR_64RegClass;
3665 SubIndices = Sub0_15_64;
3666 NElts /= 2;
3667 }
3668 }
3669
3671 MBB, I, DL, get(AMDGPU::REG_SEQUENCE), DstReg);
3672
3673 I = MIB->getIterator();
3674
3676 for (int Idx = 0; Idx != NElts; ++Idx) {
3677 Register DstElt = MRI.createVirtualRegister(EltRC);
3678 Regs.push_back(DstElt);
3679
3680 unsigned SubIdx = SubIndices[Idx];
3681
3683 if (SelOp == AMDGPU::V_CNDMASK_B32_e32) {
3684 Select = BuildMI(MBB, I, DL, get(SelOp), DstElt)
3685 .addReg(FalseReg, {}, SubIdx)
3686 .addReg(TrueReg, {}, SubIdx);
3687 } else {
3688 Select = BuildMI(MBB, I, DL, get(SelOp), DstElt)
3689 .addReg(TrueReg, {}, SubIdx)
3690 .addReg(FalseReg, {}, SubIdx);
3691 }
3692
3693 preserveCondRegFlags(Select->getOperand(3), Cond[1]);
3695
3696 MIB.addReg(DstElt)
3697 .addImm(SubIdx);
3698 }
3699}
3700
3702
3703 if (MI.isBranch() || MI.isCall() || MI.isReturn() || MI.isIndirectBranch())
3704 return true;
3705
3706 switch (MI.getOpcode()) {
3707 case AMDGPU::S_ENDPGM:
3708 case AMDGPU::S_ENDPGM_SAVED:
3709 case AMDGPU::S_TRAP:
3710 case AMDGPU::S_GETREG_B32:
3711 case AMDGPU::S_SETREG_B32:
3712 case AMDGPU::S_SETREG_B32_mode:
3713 case AMDGPU::S_SETREG_IMM32_B32:
3714 case AMDGPU::S_SETREG_IMM32_B32_mode:
3715 case AMDGPU::S_SENDMSG:
3716 case AMDGPU::S_SENDMSGHALT:
3717 case AMDGPU::S_SENDMSG_RTN_B32:
3718 case AMDGPU::S_SENDMSG_RTN_B64:
3719 case AMDGPU::S_BARRIER_WAIT:
3720 case AMDGPU::S_BARRIER_SIGNAL_M0:
3721 case AMDGPU::S_BARRIER_SIGNAL_IMM:
3722 case AMDGPU::S_BARRIER_SIGNAL_ISFIRST_M0:
3723 case AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM:
3724 return true;
3725 default:
3726 return false;
3727 }
3728}
3729
3731 switch (MI.getOpcode()) {
3732 case AMDGPU::V_MOV_B16_t16_e32:
3733 case AMDGPU::V_MOV_B16_t16_e64:
3734 case AMDGPU::V_MOV_B32_e32:
3735 case AMDGPU::V_MOV_B32_e64:
3736 case AMDGPU::V_MOV_B64_PSEUDO:
3737 case AMDGPU::V_MOV_B64_e32:
3738 case AMDGPU::V_MOV_B64_e64:
3739 case AMDGPU::S_MOV_B32:
3740 case AMDGPU::S_MOV_B64:
3741 case AMDGPU::S_MOV_B64_IMM_PSEUDO:
3742 case AMDGPU::COPY:
3743 case AMDGPU::WWM_COPY:
3744 case AMDGPU::V_ACCVGPR_WRITE_B32_e64:
3745 case AMDGPU::V_ACCVGPR_READ_B32_e64:
3746 case AMDGPU::V_ACCVGPR_MOV_B32:
3747 case AMDGPU::AV_MOV_B32_IMM_PSEUDO:
3748 case AMDGPU::AV_MOV_B64_IMM_PSEUDO:
3749 return true;
3750 default:
3751 return false;
3752 }
3753}
3754
3756 switch (MI.getOpcode()) {
3757 case AMDGPU::V_MOV_B16_t16_e32:
3758 case AMDGPU::V_MOV_B16_t16_e64:
3759 return 2;
3760 case AMDGPU::V_MOV_B32_e32:
3761 case AMDGPU::V_MOV_B32_e64:
3762 case AMDGPU::V_MOV_B64_PSEUDO:
3763 case AMDGPU::V_MOV_B64_e32:
3764 case AMDGPU::V_MOV_B64_e64:
3765 case AMDGPU::S_MOV_B32:
3766 case AMDGPU::S_MOV_B64:
3767 case AMDGPU::S_MOV_B64_IMM_PSEUDO:
3768 case AMDGPU::COPY:
3769 case AMDGPU::WWM_COPY:
3770 case AMDGPU::V_ACCVGPR_WRITE_B32_e64:
3771 case AMDGPU::V_ACCVGPR_READ_B32_e64:
3772 case AMDGPU::V_ACCVGPR_MOV_B32:
3773 case AMDGPU::AV_MOV_B32_IMM_PSEUDO:
3774 case AMDGPU::AV_MOV_B64_IMM_PSEUDO:
3775 return 1;
3776 default:
3777 llvm_unreachable("MI is not a foldable copy");
3778 }
3779}
3780
3781static constexpr AMDGPU::OpName ModifierOpNames[] = {
3782 AMDGPU::OpName::src0_modifiers, AMDGPU::OpName::src1_modifiers,
3783 AMDGPU::OpName::src2_modifiers, AMDGPU::OpName::clamp,
3784 AMDGPU::OpName::omod, AMDGPU::OpName::op_sel};
3785
3787 unsigned Opc = MI.getOpcode();
3788 for (AMDGPU::OpName Name : reverse(ModifierOpNames)) {
3789 int Idx = AMDGPU::getNamedOperandIdx(Opc, Name);
3790 if (Idx >= 0)
3791 MI.removeOperand(Idx);
3792 }
3793}
3794
3796 const MCInstrDesc &NewDesc) const {
3797 MI.setDesc(NewDesc);
3798
3799 // Remove any leftover implicit operands from mutating the instruction. e.g.
3800 // if we replace an s_and_b32 with a copy, we don't need the implicit scc def
3801 // anymore.
3802 const MCInstrDesc &Desc = MI.getDesc();
3803 unsigned NumOps = Desc.getNumOperands() + Desc.implicit_uses().size() +
3804 Desc.implicit_defs().size();
3805
3806 for (unsigned I = MI.getNumOperands() - 1; I >= NumOps; --I)
3807 MI.removeOperand(I);
3808}
3809
3810std::optional<int64_t> SIInstrInfo::extractSubregFromImm(int64_t Imm,
3811 unsigned SubRegIndex) {
3812 switch (SubRegIndex) {
3813 case AMDGPU::NoSubRegister:
3814 return Imm;
3815 case AMDGPU::sub0:
3816 return SignExtend64<32>(Imm);
3817 case AMDGPU::sub1:
3818 return SignExtend64<32>(Imm >> 32);
3819 case AMDGPU::lo16:
3820 return SignExtend64<16>(Imm);
3821 case AMDGPU::hi16:
3822 return SignExtend64<16>(Imm >> 16);
3823 case AMDGPU::sub1_lo16:
3824 return SignExtend64<16>(Imm >> 32);
3825 case AMDGPU::sub1_hi16:
3826 return SignExtend64<16>(Imm >> 48);
3827 default:
3828 return std::nullopt;
3829 }
3830
3831 llvm_unreachable("covered subregister switch");
3832}
3833
3834static unsigned getNewFMAAKInst(const GCNSubtarget &ST, unsigned Opc) {
3835 switch (Opc) {
3836 case AMDGPU::V_MAC_F16_e32:
3837 case AMDGPU::V_MAC_F16_e64:
3838 case AMDGPU::V_MAD_F16_e64:
3839 return AMDGPU::V_MADAK_F16;
3840 case AMDGPU::V_MAC_F32_e32:
3841 case AMDGPU::V_MAC_F32_e64:
3842 case AMDGPU::V_MAD_F32_e64:
3843 return AMDGPU::V_MADAK_F32;
3844 case AMDGPU::V_FMAC_F32_e32:
3845 case AMDGPU::V_FMAC_F32_e64:
3846 case AMDGPU::V_FMA_F32_e64:
3847 return AMDGPU::V_FMAAK_F32;
3848 case AMDGPU::V_FMAC_F16_e32:
3849 case AMDGPU::V_FMAC_F16_e64:
3850 case AMDGPU::V_FMAC_F16_t16_e64:
3851 case AMDGPU::V_FMAC_F16_fake16_e64:
3852 case AMDGPU::V_FMAC_F16_t16_e32:
3853 case AMDGPU::V_FMAC_F16_fake16_e32:
3854 case AMDGPU::V_FMA_F16_e64:
3855 return ST.hasTrue16BitInsts() ? ST.useRealTrue16Insts()
3856 ? AMDGPU::V_FMAAK_F16_t16
3857 : AMDGPU::V_FMAAK_F16_fake16
3858 : AMDGPU::V_FMAAK_F16;
3859 case AMDGPU::V_FMAC_F64_e32:
3860 case AMDGPU::V_FMAC_F64_e64:
3861 case AMDGPU::V_FMA_F64_e64:
3862 return AMDGPU::V_FMAAK_F64;
3863 default:
3864 llvm_unreachable("invalid instruction");
3865 }
3866}
3867
3868static unsigned getNewFMAMKInst(const GCNSubtarget &ST, unsigned Opc) {
3869 switch (Opc) {
3870 case AMDGPU::V_MAC_F16_e32:
3871 case AMDGPU::V_MAC_F16_e64:
3872 case AMDGPU::V_MAD_F16_e64:
3873 return AMDGPU::V_MADMK_F16;
3874 case AMDGPU::V_MAC_F32_e32:
3875 case AMDGPU::V_MAC_F32_e64:
3876 case AMDGPU::V_MAD_F32_e64:
3877 return AMDGPU::V_MADMK_F32;
3878 case AMDGPU::V_FMAC_F32_e32:
3879 case AMDGPU::V_FMAC_F32_e64:
3880 case AMDGPU::V_FMA_F32_e64:
3881 return AMDGPU::V_FMAMK_F32;
3882 case AMDGPU::V_FMAC_F16_e32:
3883 case AMDGPU::V_FMAC_F16_e64:
3884 case AMDGPU::V_FMAC_F16_t16_e64:
3885 case AMDGPU::V_FMAC_F16_fake16_e64:
3886 case AMDGPU::V_FMAC_F16_t16_e32:
3887 case AMDGPU::V_FMAC_F16_fake16_e32:
3888 case AMDGPU::V_FMA_F16_e64:
3889 return ST.hasTrue16BitInsts() ? ST.useRealTrue16Insts()
3890 ? AMDGPU::V_FMAMK_F16_t16
3891 : AMDGPU::V_FMAMK_F16_fake16
3892 : AMDGPU::V_FMAMK_F16;
3893 case AMDGPU::V_FMAC_F64_e32:
3894 case AMDGPU::V_FMAC_F64_e64:
3895 case AMDGPU::V_FMA_F64_e64:
3896 return AMDGPU::V_FMAMK_F64;
3897 default:
3898 llvm_unreachable("invalid instruction");
3899 }
3900}
3901
3903 Register Reg, MachineRegisterInfo *MRI) const {
3904 int64_t Imm;
3905 if (!getConstValDefinedInReg(DefMI, Reg, Imm))
3906 return false;
3907
3908 const bool HasMultipleUses = !MRI->hasOneNonDBGUse(Reg);
3909
3910 assert(!DefMI.getOperand(0).getSubReg() && "Expected SSA form");
3911
3912 unsigned Opc = UseMI.getOpcode();
3913 if (Opc == AMDGPU::COPY) {
3914 assert(!UseMI.getOperand(0).getSubReg() && "Expected SSA form");
3915
3916 Register DstReg = UseMI.getOperand(0).getReg();
3917 Register UseSubReg = UseMI.getOperand(1).getSubReg();
3918
3919 const TargetRegisterClass *DstRC = RI.getRegClassForReg(*MRI, DstReg);
3920
3921 if (HasMultipleUses) {
3922 // TODO: This should fold in more cases with multiple use, but we need to
3923 // more carefully consider what those uses are.
3924 unsigned ImmDefSize = RI.getRegSizeInBits(*MRI->getRegClass(Reg));
3925
3926 // Avoid breaking up a 64-bit inline immediate into a subregister extract.
3927 if (UseSubReg != AMDGPU::NoSubRegister && ImmDefSize == 64)
3928 return false;
3929
3930 // Most of the time folding a 32-bit inline constant is free (though this
3931 // might not be true if we can't later fold it into a real user).
3932 //
3933 // FIXME: This isInlineConstant check is imprecise if
3934 // getConstValDefinedInReg handled the tricky non-mov cases.
3935 if (ImmDefSize == 32 &&
3937 return false;
3938 }
3939
3940 bool Is16Bit = UseSubReg != AMDGPU::NoSubRegister &&
3941 RI.getSubRegIdxSize(UseSubReg) == 16;
3942
3943 if (Is16Bit) {
3944 if (RI.hasVGPRs(DstRC))
3945 return false; // Do not clobber vgpr_hi16
3946
3947 if (DstReg.isVirtual() && UseSubReg != AMDGPU::lo16)
3948 return false;
3949 }
3950
3951 MachineFunction *MF = UseMI.getMF();
3952
3953 unsigned NewOpc = AMDGPU::INSTRUCTION_LIST_END;
3954 MCRegister MovDstPhysReg =
3955 DstReg.isPhysical() ? DstReg.asMCReg() : MCRegister();
3956
3957 std::optional<int64_t> SubRegImm = extractSubregFromImm(Imm, UseSubReg);
3958
3959 // TODO: Try to fold with AMDGPU::V_MOV_B16_t16_e64
3960 for (unsigned MovOp :
3961 {AMDGPU::S_MOV_B32, AMDGPU::V_MOV_B32_e32, AMDGPU::S_MOV_B64,
3962 AMDGPU::V_MOV_B64_PSEUDO, AMDGPU::V_ACCVGPR_WRITE_B32_e64}) {
3963 const MCInstrDesc &MovDesc = get(MovOp);
3964
3965 const TargetRegisterClass *MovDstRC = getRegClass(MovDesc, 0);
3966 if (Is16Bit) {
3967 // We just need to find a correctly sized register class, so the
3968 // subregister index compatibility doesn't matter since we're statically
3969 // extracting the immediate value.
3970 MovDstRC = RI.getMatchingSuperRegClass(MovDstRC, DstRC, AMDGPU::lo16);
3971 if (!MovDstRC)
3972 continue;
3973
3974 if (MovDstPhysReg) {
3975 // FIXME: We probably should not do this. If there is a live value in
3976 // the high half of the register, it will be corrupted.
3977 MovDstPhysReg =
3978 RI.getMatchingSuperReg(MovDstPhysReg, AMDGPU::lo16, MovDstRC);
3979 if (!MovDstPhysReg)
3980 continue;
3981 }
3982 }
3983
3984 // Result class isn't the right size, try the next instruction.
3985 if (MovDstPhysReg) {
3986 if (!MovDstRC->contains(MovDstPhysReg))
3987 return false;
3988 } else if (!MRI->constrainRegClass(DstReg, MovDstRC)) {
3989 // TODO: This will be overly conservative in the case of 16-bit virtual
3990 // SGPRs. We could hack up the virtual register uses to use a compatible
3991 // 32-bit class.
3992 continue;
3993 }
3994
3995 const MCOperandInfo &OpInfo = MovDesc.operands()[1];
3996
3997 // Ensure the interpreted immediate value is a valid operand in the new
3998 // mov.
3999 //
4000 // FIXME: isImmOperandLegal should have form that doesn't require existing
4001 // MachineInstr or MachineOperand
4002 if (!RI.opCanUseLiteralConstant(OpInfo.OperandType) &&
4003 !isInlineConstant(*SubRegImm, OpInfo.OperandType))
4004 break;
4005
4006 NewOpc = MovOp;
4007 break;
4008 }
4009
4010 if (NewOpc == AMDGPU::INSTRUCTION_LIST_END)
4011 return false;
4012
4013 if (Is16Bit) {
4014 UseMI.getOperand(0).setSubReg(AMDGPU::NoSubRegister);
4015 if (MovDstPhysReg)
4016 UseMI.getOperand(0).setReg(MovDstPhysReg);
4017 assert(UseMI.getOperand(1).getReg().isVirtual());
4018 }
4019
4020 const MCInstrDesc &NewMCID = get(NewOpc);
4021 UseMI.setDesc(NewMCID);
4022 UseMI.getOperand(1).ChangeToImmediate(*SubRegImm);
4023 UseMI.addImplicitDefUseOperands(*MF);
4024 return true;
4025 }
4026
4027 if (HasMultipleUses)
4028 return false;
4029
4030 if (Opc == AMDGPU::V_MAD_F32_e64 || Opc == AMDGPU::V_MAC_F32_e64 ||
4031 Opc == AMDGPU::V_MAD_F16_e64 || Opc == AMDGPU::V_MAC_F16_e64 ||
4032 Opc == AMDGPU::V_FMA_F32_e64 || Opc == AMDGPU::V_FMAC_F32_e64 ||
4033 Opc == AMDGPU::V_FMA_F16_e64 || Opc == AMDGPU::V_FMAC_F16_e64 ||
4034 Opc == AMDGPU::V_FMAC_F16_t16_e64 ||
4035 Opc == AMDGPU::V_FMAC_F16_fake16_e64 || Opc == AMDGPU::V_FMA_F64_e64 ||
4036 Opc == AMDGPU::V_FMAC_F64_e64) {
4037 // Don't fold if we are using source or output modifiers. The new VOP2
4038 // instructions don't have them.
4040 return false;
4041
4042 // If this is a free constant, there's no reason to do this.
4043 // TODO: We could fold this here instead of letting SIFoldOperands do it
4044 // later.
4045 int Src0Idx = getNamedOperandIdx(UseMI.getOpcode(), AMDGPU::OpName::src0);
4046
4047 // Any src operand can be used for the legality check.
4048 if (isInlineConstant(UseMI, Src0Idx, Imm))
4049 return false;
4050
4051 MachineOperand *Src0 = &UseMI.getOperand(Src0Idx);
4052
4053 MachineOperand *Src1 = getNamedOperand(UseMI, AMDGPU::OpName::src1);
4054 MachineOperand *Src2 = getNamedOperand(UseMI, AMDGPU::OpName::src2);
4055
4056 auto CopyRegOperandToNarrowerRC =
4057 [MRI, this](MachineInstr &MI, unsigned OpNo,
4058 const TargetRegisterClass *NewRC) -> void {
4059 if (!MI.getOperand(OpNo).isReg())
4060 return;
4061 Register Reg = MI.getOperand(OpNo).getReg();
4062 const TargetRegisterClass *RC = RI.getRegClassForReg(*MRI, Reg);
4063 if (RI.getCommonSubClass(RC, NewRC) != NewRC)
4064 return;
4065 Register Tmp = MRI->createVirtualRegister(NewRC);
4066 BuildMI(*MI.getParent(), MI.getIterator(), MI.getDebugLoc(),
4067 get(AMDGPU::COPY), Tmp)
4068 .addReg(Reg);
4069 MI.getOperand(OpNo).setReg(Tmp);
4070 MI.getOperand(OpNo).setIsKill();
4071 };
4072
4073 // Multiplied part is the constant: Use v_madmk_{f16, f32}.
4074 if ((Src0->isReg() && Src0->getReg() == Reg) ||
4075 (Src1->isReg() && Src1->getReg() == Reg)) {
4076 MachineOperand *RegSrc =
4077 Src1->isReg() && Src1->getReg() == Reg ? Src0 : Src1;
4078 if (!RegSrc->isReg())
4079 return false;
4080 if (RI.isSGPRClass(MRI->getRegClass(RegSrc->getReg())) &&
4081 ST.getConstantBusLimit(Opc) < 2)
4082 return false;
4083
4084 if (!Src2->isReg() || RI.isSGPRClass(MRI->getRegClass(Src2->getReg())))
4085 return false;
4086
4087 // If src2 is also a literal constant then we have to choose which one to
4088 // fold. In general it is better to choose madak so that the other literal
4089 // can be materialized in an sgpr instead of a vgpr:
4090 // s_mov_b32 s0, literal
4091 // v_madak_f32 v0, s0, v0, literal
4092 // Instead of:
4093 // v_mov_b32 v1, literal
4094 // v_madmk_f32 v0, v0, literal, v1
4095 MachineInstr *Def = MRI->getUniqueVRegDef(Src2->getReg());
4096 if (Def && Def->isMoveImmediate() &&
4097 !isInlineConstant(Def->getOperand(1)))
4098 return false;
4099
4100 unsigned NewOpc = getNewFMAMKInst(ST, Opc);
4101 if (pseudoToMCOpcode(NewOpc) == -1)
4102 return false;
4103
4104 const std::optional<int64_t> SubRegImm = extractSubregFromImm(
4105 Imm, RegSrc == Src1 ? Src0->getSubReg() : Src1->getSubReg());
4106
4107 // FIXME: This would be a lot easier if we could return a new instruction
4108 // instead of having to modify in place.
4109
4110 Register SrcReg = RegSrc->getReg();
4111 unsigned SrcSubReg = RegSrc->getSubReg();
4112 Src0->setReg(SrcReg);
4113 Src0->setSubReg(SrcSubReg);
4114 Src0->setIsKill(RegSrc->isKill());
4115
4116 if (Opc == AMDGPU::V_MAC_F32_e64 || Opc == AMDGPU::V_MAC_F16_e64 ||
4117 Opc == AMDGPU::V_FMAC_F32_e64 || Opc == AMDGPU::V_FMAC_F16_t16_e64 ||
4118 Opc == AMDGPU::V_FMAC_F16_fake16_e64 ||
4119 Opc == AMDGPU::V_FMAC_F16_e64 || Opc == AMDGPU::V_FMAC_F64_e64)
4120 UseMI.untieRegOperand(
4121 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2));
4122
4123 Src1->ChangeToImmediate(*SubRegImm);
4124
4126 UseMI.setDesc(get(NewOpc));
4127
4128 if (NewOpc == AMDGPU::V_FMAMK_F16_t16 ||
4129 NewOpc == AMDGPU::V_FMAMK_F16_fake16) {
4130 const TargetRegisterClass *NewRC = getRegClass(get(NewOpc), 0);
4131 Register Tmp = MRI->createVirtualRegister(NewRC);
4132 BuildMI(*UseMI.getParent(), std::next(UseMI.getIterator()),
4133 UseMI.getDebugLoc(), get(AMDGPU::COPY),
4134 UseMI.getOperand(0).getReg())
4135 .addReg(Tmp, RegState::Kill);
4136 UseMI.getOperand(0).setReg(Tmp);
4137 CopyRegOperandToNarrowerRC(UseMI, 1, NewRC);
4138 CopyRegOperandToNarrowerRC(UseMI, 3, NewRC);
4139 }
4140
4141 bool DeleteDef = MRI->use_nodbg_empty(Reg);
4142 if (DeleteDef)
4143 DefMI.eraseFromParent();
4144
4145 return true;
4146 }
4147
4148 // Added part is the constant: Use v_madak_{f16, f32}.
4149 if (Src2->isReg() && Src2->getReg() == Reg) {
4150 if (ST.getConstantBusLimit(Opc) < 2) {
4151 // Not allowed to use constant bus for another operand.
4152 // We can however allow an inline immediate as src0.
4153 bool Src0Inlined = false;
4154 if (Src0->isReg()) {
4155 // Try to inline constant if possible.
4156 // If the Def moves immediate and the use is single
4157 // We are saving VGPR here.
4158 MachineInstr *Def = MRI->getUniqueVRegDef(Src0->getReg());
4159 if (Def && Def->isMoveImmediate() &&
4160 isInlineConstant(Def->getOperand(1)) &&
4161 MRI->hasOneNonDBGUse(Src0->getReg())) {
4162 Src0->ChangeToImmediate(Def->getOperand(1).getImm());
4163 Src0Inlined = true;
4164 } else if (ST.getConstantBusLimit(Opc) <= 1 &&
4165 RI.isSGPRReg(*MRI, Src0->getReg())) {
4166 return false;
4167 }
4168 // VGPR is okay as Src0 - fallthrough
4169 }
4170
4171 if (Src1->isReg() && !Src0Inlined) {
4172 // We have one slot for inlinable constant so far - try to fill it
4173 MachineInstr *Def = MRI->getUniqueVRegDef(Src1->getReg());
4174 if (Def && Def->isMoveImmediate() &&
4175 isInlineConstant(Def->getOperand(1)) &&
4176 MRI->hasOneNonDBGUse(Src1->getReg()) && commuteInstruction(UseMI))
4177 Src0->ChangeToImmediate(Def->getOperand(1).getImm());
4178 else if (RI.isSGPRReg(*MRI, Src1->getReg()))
4179 return false;
4180 // VGPR is okay as Src1 - fallthrough
4181 }
4182 }
4183
4184 unsigned NewOpc = getNewFMAAKInst(ST, Opc);
4185 if (pseudoToMCOpcode(NewOpc) == -1)
4186 return false;
4187
4188 // FIXME: This would be a lot easier if we could return a new instruction
4189 // instead of having to modify in place.
4190
4191 if (Opc == AMDGPU::V_MAC_F32_e64 || Opc == AMDGPU::V_MAC_F16_e64 ||
4192 Opc == AMDGPU::V_FMAC_F32_e64 || Opc == AMDGPU::V_FMAC_F16_t16_e64 ||
4193 Opc == AMDGPU::V_FMAC_F16_fake16_e64 ||
4194 Opc == AMDGPU::V_FMAC_F16_e64 || Opc == AMDGPU::V_FMAC_F64_e64)
4195 UseMI.untieRegOperand(
4196 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2));
4197
4198 const std::optional<int64_t> SubRegImm =
4200
4201 // ChangingToImmediate adds Src2 back to the instruction.
4202 Src2->ChangeToImmediate(*SubRegImm);
4203
4204 // These come before src2.
4206 UseMI.setDesc(get(NewOpc));
4207
4208 if (NewOpc == AMDGPU::V_FMAAK_F16_t16 ||
4209 NewOpc == AMDGPU::V_FMAAK_F16_fake16) {
4210 const TargetRegisterClass *NewRC = getRegClass(get(NewOpc), 0);
4211 Register Tmp = MRI->createVirtualRegister(NewRC);
4212 BuildMI(*UseMI.getParent(), std::next(UseMI.getIterator()),
4213 UseMI.getDebugLoc(), get(AMDGPU::COPY),
4214 UseMI.getOperand(0).getReg())
4215 .addReg(Tmp, RegState::Kill);
4216 UseMI.getOperand(0).setReg(Tmp);
4217 CopyRegOperandToNarrowerRC(UseMI, 1, NewRC);
4218 CopyRegOperandToNarrowerRC(UseMI, 2, NewRC);
4219 }
4220
4221 // It might happen that UseMI was commuted
4222 // and we now have SGPR as SRC1. If so 2 inlined
4223 // constant and SGPR are illegal.
4225
4226 int NewSrc0Idx =
4227 AMDGPU::getNamedOperandIdx(UseMI.getOpcode(), AMDGPU::OpName::src0);
4228 if (!isOperandLegal(UseMI, NewSrc0Idx))
4229 legalizeOpWithMove(UseMI, NewSrc0Idx);
4230
4231 bool DeleteDef = MRI->use_nodbg_empty(Reg);
4232 if (DeleteDef)
4233 DefMI.eraseFromParent();
4234
4235 return true;
4236 }
4237 }
4238
4239 return false;
4240}
4241
4242static bool
4245 if (BaseOps1.size() != BaseOps2.size())
4246 return false;
4247 for (size_t I = 0, E = BaseOps1.size(); I < E; ++I) {
4248 if (!BaseOps1[I]->isIdenticalTo(*BaseOps2[I]))
4249 return false;
4250 }
4251 return true;
4252}
4253
4254static bool offsetsDoNotOverlap(LocationSize WidthA, int OffsetA,
4255 LocationSize WidthB, int OffsetB) {
4256 int LowOffset = OffsetA < OffsetB ? OffsetA : OffsetB;
4257 int HighOffset = OffsetA < OffsetB ? OffsetB : OffsetA;
4258 LocationSize LowWidth = (LowOffset == OffsetA) ? WidthA : WidthB;
4259 return LowWidth.hasValue() &&
4260 LowOffset + (int)LowWidth.getValue() <= HighOffset;
4261}
4262
4263bool SIInstrInfo::checkInstOffsetsDoNotOverlap(const MachineInstr &MIa,
4264 const MachineInstr &MIb) const {
4265 SmallVector<const MachineOperand *, 4> BaseOps0, BaseOps1;
4266 int64_t Offset0, Offset1;
4267 LocationSize Dummy0 = LocationSize::precise(0);
4268 LocationSize Dummy1 = LocationSize::precise(0);
4269 bool Offset0IsScalable, Offset1IsScalable;
4270 if (!getMemOperandsWithOffsetWidth(MIa, BaseOps0, Offset0, Offset0IsScalable,
4271 Dummy0, &RI) ||
4272 !getMemOperandsWithOffsetWidth(MIb, BaseOps1, Offset1, Offset1IsScalable,
4273 Dummy1, &RI))
4274 return false;
4275
4276 if (!memOpsHaveSameBaseOperands(BaseOps0, BaseOps1))
4277 return false;
4278
4279 if (!MIa.hasOneMemOperand() || !MIb.hasOneMemOperand()) {
4280 // FIXME: Handle ds_read2 / ds_write2.
4281 return false;
4282 }
4283 LocationSize Width0 = MIa.memoperands().front()->getSize();
4284 LocationSize Width1 = MIb.memoperands().front()->getSize();
4285 return offsetsDoNotOverlap(Width0, Offset0, Width1, Offset1);
4286}
4287
4289 const MachineInstr &MIb) const {
4290 assert(MIa.mayLoadOrStore() &&
4291 "MIa must load from or modify a memory location");
4292 assert(MIb.mayLoadOrStore() &&
4293 "MIb must load from or modify a memory location");
4294
4296 return false;
4297
4298 // XXX - Can we relax this between address spaces?
4299 if (MIa.hasOrderedMemoryRef() || MIb.hasOrderedMemoryRef())
4300 return false;
4301
4302 if (isLDSDMA(MIa) || isLDSDMA(MIb))
4303 return false;
4304
4305 if (MIa.isBundle() || MIb.isBundle())
4306 return false;
4307
4308 // TODO: Should we check the address space from the MachineMemOperand? That
4309 // would allow us to distinguish objects we know don't alias based on the
4310 // underlying address space, even if it was lowered to a different one,
4311 // e.g. private accesses lowered to use MUBUF instructions on a scratch
4312 // buffer.
4313 if (isDS(MIa)) {
4314 if (isDS(MIb))
4315 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4316
4317 return !isFLAT(MIb) || isSegmentSpecificFLAT(MIb);
4318 }
4319
4320 if (isMUBUF(MIa) || isMTBUF(MIa)) {
4321 if (isMUBUF(MIb) || isMTBUF(MIb))
4322 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4323
4324 if (isFLAT(MIb))
4325 return isFLATScratch(MIb);
4326
4327 return !isSMRD(MIb);
4328 }
4329
4330 if (isSMRD(MIa)) {
4331 if (isSMRD(MIb))
4332 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4333
4334 if (isFLAT(MIb))
4335 return isFLATScratch(MIb);
4336
4337 return !isMUBUF(MIb) && !isMTBUF(MIb);
4338 }
4339
4340 if (isFLAT(MIa)) {
4341 if (isFLAT(MIb)) {
4342 if ((isFLATScratch(MIa) && isFLATGlobal(MIb)) ||
4343 (isFLATGlobal(MIa) && isFLATScratch(MIb)))
4344 return true;
4345
4346 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4347 }
4348
4349 return false;
4350 }
4351
4352 return false;
4353}
4354
4355static unsigned getNewFMAInst(const GCNSubtarget &ST, unsigned Opc) {
4356 switch (Opc) {
4357 case AMDGPU::V_MAC_F16_e32:
4358 case AMDGPU::V_MAC_F16_e64:
4359 return AMDGPU::V_MAD_F16_e64;
4360 case AMDGPU::V_MAC_F32_e32:
4361 case AMDGPU::V_MAC_F32_e64:
4362 return AMDGPU::V_MAD_F32_e64;
4363 case AMDGPU::V_MAC_LEGACY_F32_e32:
4364 case AMDGPU::V_MAC_LEGACY_F32_e64:
4365 return AMDGPU::V_MAD_LEGACY_F32_e64;
4366 case AMDGPU::V_FMAC_LEGACY_F32_e32:
4367 case AMDGPU::V_FMAC_LEGACY_F32_e64:
4368 return AMDGPU::V_FMA_LEGACY_F32_e64;
4369 case AMDGPU::V_FMAC_F16_e32:
4370 case AMDGPU::V_FMAC_F16_e64:
4371 case AMDGPU::V_FMAC_F16_t16_e64:
4372 case AMDGPU::V_FMAC_F16_fake16_e64:
4373 return ST.hasTrue16BitInsts() ? ST.useRealTrue16Insts()
4374 ? AMDGPU::V_FMA_F16_gfx9_t16_e64
4375 : AMDGPU::V_FMA_F16_gfx9_fake16_e64
4376 : AMDGPU::V_FMA_F16_gfx9_e64;
4377 case AMDGPU::V_FMAC_F32_e32:
4378 case AMDGPU::V_FMAC_F32_e64:
4379 return AMDGPU::V_FMA_F32_e64;
4380 case AMDGPU::V_FMAC_F64_e32:
4381 case AMDGPU::V_FMAC_F64_e64:
4382 return AMDGPU::V_FMA_F64_e64;
4383 default:
4384 llvm_unreachable("invalid instruction");
4385 }
4386}
4387
4388/// Helper struct for the implementation of 3-address conversion to communicate
4389/// updates made to instruction operands.
4391 /// Other instruction whose def is no longer used by the converted
4392 /// instruction.
4394};
4395
4397 LiveIntervals *LIS) const {
4398 MachineBasicBlock &MBB = *MI.getParent();
4399 MachineInstr *CandidateMI = &MI;
4400
4401 if (MI.isBundle()) {
4402 // This is a temporary placeholder for bundle handling that enables us to
4403 // exercise the relevant code paths in the two-address instruction pass.
4404 if (MI.getBundleSize() != 1)
4405 return nullptr;
4406 CandidateMI = MI.getNextNode();
4407 }
4408
4410 MachineInstr *NewMI = convertToThreeAddressImpl(*CandidateMI, U);
4411 if (!NewMI)
4412 return nullptr;
4413
4414 if (MI.isBundle()) {
4415 CandidateMI->eraseFromBundle();
4416
4417 for (MachineOperand &MO : MI.all_defs()) {
4418 if (MO.isTied())
4419 MI.untieRegOperand(MO.getOperandNo());
4420 }
4421 } else {
4422 if (LIS) {
4423 LIS->ReplaceMachineInstrInMaps(MI, *NewMI);
4424 // SlotIndex of defs needs to be updated when converting to early-clobber
4425 MachineOperand &Def = NewMI->getOperand(0);
4426 if (Def.isEarlyClobber() && Def.isReg() &&
4427 LIS->hasInterval(Def.getReg())) {
4428 SlotIndex OldIndex = LIS->getInstructionIndex(*NewMI).getRegSlot(false);
4429 SlotIndex NewIndex = LIS->getInstructionIndex(*NewMI).getRegSlot(true);
4430 auto &LI = LIS->getInterval(Def.getReg());
4431 auto UpdateDefIndex = [&](LiveRange &LR) {
4432 auto *S = LR.find(OldIndex);
4433 if (S != LR.end() && S->start == OldIndex) {
4434 assert(S->valno && S->valno->def == OldIndex);
4435 S->start = NewIndex;
4436 S->valno->def = NewIndex;
4437 }
4438 };
4439 UpdateDefIndex(LI);
4440 for (auto &SR : LI.subranges())
4441 UpdateDefIndex(SR);
4442 }
4443 }
4444 }
4445
4446 if (U.RemoveMIUse) {
4447 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
4448 // The only user is the instruction which will be killed.
4449 Register DefReg = U.RemoveMIUse->getOperand(0).getReg();
4450
4451 if (MRI.hasOneNonDBGUse(DefReg)) {
4452 // We cannot just remove the DefMI here, calling pass will crash.
4453 U.RemoveMIUse->setDesc(get(AMDGPU::IMPLICIT_DEF));
4454 U.RemoveMIUse->getOperand(0).setIsDead(true);
4455 for (unsigned I = U.RemoveMIUse->getNumOperands() - 1; I != 0; --I)
4456 U.RemoveMIUse->removeOperand(I);
4457 }
4458
4459 if (MI.isBundle()) {
4460 VirtRegInfo VRI = AnalyzeVirtRegInBundle(MI, DefReg);
4461 if (!VRI.Reads && !VRI.Writes) {
4462 for (MachineOperand &MO : MI.all_uses()) {
4463 if (MO.isReg() && MO.getReg() == DefReg) {
4464 assert(MO.getSubReg() == 0 &&
4465 "tied sub-registers in bundles currently not supported");
4466 MI.removeOperand(MO.getOperandNo());
4467 break;
4468 }
4469 }
4470
4471 if (LIS)
4472 LIS->shrinkToUses(&LIS->getInterval(DefReg));
4473 }
4474 } else if (LIS) {
4475 LiveInterval &DefLI = LIS->getInterval(DefReg);
4476
4477 // We cannot delete the original instruction here, so hack out the use
4478 // in the original instruction with a dummy register so we can use
4479 // shrinkToUses to deal with any multi-use edge cases. Other targets do
4480 // not have the complexity of deleting a use to consider here.
4481 Register DummyReg = MRI.cloneVirtualRegister(DefReg);
4482 for (MachineOperand &MIOp : MI.uses()) {
4483 if (MIOp.isReg() && MIOp.getReg() == DefReg) {
4484 MIOp.setIsUndef(true);
4485 MIOp.setReg(DummyReg);
4486 }
4487 }
4488
4489 if (MI.isBundle()) {
4490 VirtRegInfo VRI = AnalyzeVirtRegInBundle(MI, DefReg);
4491 if (!VRI.Reads && !VRI.Writes) {
4492 for (MachineOperand &MIOp : MI.uses()) {
4493 if (MIOp.isReg() && MIOp.getReg() == DefReg) {
4494 MIOp.setIsUndef(true);
4495 MIOp.setReg(DummyReg);
4496 }
4497 }
4498 }
4499
4500 MI.addOperand(MachineOperand::CreateReg(DummyReg, false, false, false,
4501 false, /*isUndef=*/true));
4502 }
4503
4504 LIS->shrinkToUses(&DefLI);
4505 }
4506 }
4507
4508 return MI.isBundle() ? &MI : NewMI;
4509}
4510
4512SIInstrInfo::convertToThreeAddressImpl(MachineInstr &MI,
4513 ThreeAddressUpdates &U) const {
4514 MachineBasicBlock &MBB = *MI.getParent();
4515 unsigned Opc = MI.getOpcode();
4516
4517 // Handle MFMA.
4518 int NewMFMAOpc = AMDGPU::getMFMAEarlyClobberOp(Opc);
4519 if (NewMFMAOpc != -1) {
4521 BuildMI(MBB, MI, MI.getDebugLoc(), get(NewMFMAOpc));
4522 for (unsigned I = 0, E = MI.getNumExplicitOperands(); I != E; ++I)
4523 MIB.add(MI.getOperand(I));
4524 return MIB;
4525 }
4526
4527 if (SIInstrInfo::isWMMA(MI)) {
4528 unsigned NewOpc = AMDGPU::mapWMMA2AddrTo3AddrOpcode(MI.getOpcode());
4529 MachineInstrBuilder MIB = BuildMI(MBB, MI, MI.getDebugLoc(), get(NewOpc))
4530 .setMIFlags(MI.getFlags());
4531 for (unsigned I = 0, E = MI.getNumExplicitOperands(); I != E; ++I)
4532 MIB->addOperand(MI.getOperand(I));
4533 return MIB;
4534 }
4535
4536 assert(Opc != AMDGPU::V_FMAC_F16_t16_e32 &&
4537 Opc != AMDGPU::V_FMAC_F16_fake16_e32 &&
4538 "V_FMAC_F16_t16/fake16_e32 is not supported and not expected to be "
4539 "present pre-RA");
4540
4541 // Handle MAC/FMAC.
4542 bool IsF64 = Opc == AMDGPU::V_FMAC_F64_e32 || Opc == AMDGPU::V_FMAC_F64_e64;
4543 bool IsLegacy = Opc == AMDGPU::V_MAC_LEGACY_F32_e32 ||
4544 Opc == AMDGPU::V_MAC_LEGACY_F32_e64 ||
4545 Opc == AMDGPU::V_FMAC_LEGACY_F32_e32 ||
4546 Opc == AMDGPU::V_FMAC_LEGACY_F32_e64;
4547 bool Src0Literal = false;
4548
4549 switch (Opc) {
4550 default:
4551 return nullptr;
4552 case AMDGPU::V_MAC_F16_e64:
4553 case AMDGPU::V_FMAC_F16_e64:
4554 case AMDGPU::V_FMAC_F16_t16_e64:
4555 case AMDGPU::V_FMAC_F16_fake16_e64:
4556 case AMDGPU::V_MAC_F32_e64:
4557 case AMDGPU::V_MAC_LEGACY_F32_e64:
4558 case AMDGPU::V_FMAC_F32_e64:
4559 case AMDGPU::V_FMAC_LEGACY_F32_e64:
4560 case AMDGPU::V_FMAC_F64_e64:
4561 break;
4562 case AMDGPU::V_MAC_F16_e32:
4563 case AMDGPU::V_FMAC_F16_e32:
4564 case AMDGPU::V_MAC_F32_e32:
4565 case AMDGPU::V_MAC_LEGACY_F32_e32:
4566 case AMDGPU::V_FMAC_F32_e32:
4567 case AMDGPU::V_FMAC_LEGACY_F32_e32:
4568 case AMDGPU::V_FMAC_F64_e32: {
4569 int Src0Idx = AMDGPU::getNamedOperandIdx(MI.getOpcode(),
4570 AMDGPU::OpName::src0);
4571 const MachineOperand *Src0 = &MI.getOperand(Src0Idx);
4572 if (!Src0->isReg() && !Src0->isImm())
4573 return nullptr;
4574
4575 if (Src0->isImm() && !isInlineConstant(MI, Src0Idx, *Src0))
4576 Src0Literal = true;
4577
4578 break;
4579 }
4580 }
4581
4582 MachineInstrBuilder MIB;
4583 const MachineOperand *Dst = getNamedOperand(MI, AMDGPU::OpName::vdst);
4584 const MachineOperand *Src0 = getNamedOperand(MI, AMDGPU::OpName::src0);
4585 const MachineOperand *Src0Mods =
4586 getNamedOperand(MI, AMDGPU::OpName::src0_modifiers);
4587 const MachineOperand *Src1 = getNamedOperand(MI, AMDGPU::OpName::src1);
4588 const MachineOperand *Src1Mods =
4589 getNamedOperand(MI, AMDGPU::OpName::src1_modifiers);
4590 const MachineOperand *Src2 = getNamedOperand(MI, AMDGPU::OpName::src2);
4591 const MachineOperand *Src2Mods =
4592 getNamedOperand(MI, AMDGPU::OpName::src2_modifiers);
4593 const MachineOperand *Clamp = getNamedOperand(MI, AMDGPU::OpName::clamp);
4594 const MachineOperand *Omod = getNamedOperand(MI, AMDGPU::OpName::omod);
4595 const MachineOperand *OpSel = getNamedOperand(MI, AMDGPU::OpName::op_sel);
4596
4597 if (!Src0Mods && !Src1Mods && !Src2Mods && !Clamp && !Omod && !IsLegacy &&
4598 (!IsF64 || ST.hasFmaakFmamkF64Insts()) &&
4599 // If we have an SGPR input, we will violate the constant bus restriction.
4600 (ST.getConstantBusLimit(Opc) > 1 || !Src0->isReg() ||
4601 !RI.isSGPRReg(MBB.getParent()->getRegInfo(), Src0->getReg()))) {
4602 MachineInstr *DefMI = nullptr;
4603 const MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
4604 std::optional<int64_t> ImmOpt;
4605 int64_t Imm;
4606
4607 if (!Src0Literal &&
4608 (ImmOpt = getImmOrMaterializedImm(MRI, *Src2, &DefMI))) {
4609 unsigned NewOpc = getNewFMAAKInst(ST, Opc);
4610 if (pseudoToMCOpcode(NewOpc) != -1) {
4611 MIB = BuildMI(MBB, MI, MI.getDebugLoc(), get(NewOpc))
4612 .add(*Dst)
4613 .add(*Src0)
4614 .add(*Src1)
4615 .addImm(*ImmOpt)
4616 .setMIFlags(MI.getFlags());
4617 U.RemoveMIUse = DefMI;
4618 return MIB;
4619 }
4620 }
4621 unsigned NewOpc = getNewFMAMKInst(ST, Opc);
4622 if (!Src0Literal &&
4623 (ImmOpt = getImmOrMaterializedImm(MRI, *Src1, &DefMI))) {
4624 if (pseudoToMCOpcode(NewOpc) != -1) {
4625 MIB = BuildMI(MBB, MI, MI.getDebugLoc(), get(NewOpc))
4626 .add(*Dst)
4627 .add(*Src0)
4628 .addImm(*ImmOpt)
4629 .add(*Src2)
4630 .setMIFlags(MI.getFlags());
4631 U.RemoveMIUse = DefMI;
4632 return MIB;
4633 }
4634 }
4635 if ((ImmOpt = getImmOrMaterializedImm(MRI, *Src0, &DefMI))) {
4636 Imm = *ImmOpt;
4637 if (pseudoToMCOpcode(NewOpc) != -1 &&
4639 MI, AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::src0),
4640 Src1)) {
4641 MIB = BuildMI(MBB, MI, MI.getDebugLoc(), get(NewOpc))
4642 .add(*Dst)
4643 .add(*Src1)
4644 .addImm(Imm)
4645 .add(*Src2)
4646 .setMIFlags(MI.getFlags());
4647 U.RemoveMIUse = DefMI;
4648 return MIB;
4649 }
4650 }
4651 }
4652
4653 // VOP2 mac/fmac with a literal operand cannot be converted to VOP3 mad/fma
4654 // if VOP3 does not allow a literal operand.
4655 if (Src0Literal && !ST.hasVOP3Literal())
4656 return nullptr;
4657
4658 unsigned NewOpc = getNewFMAInst(ST, Opc);
4659
4660 if (pseudoToMCOpcode(NewOpc) == -1)
4661 return nullptr;
4662
4663 MIB = BuildMI(MBB, MI, MI.getDebugLoc(), get(NewOpc))
4664 .add(*Dst)
4665 .addImm(Src0Mods ? Src0Mods->getImm() : 0)
4666 .add(*Src0)
4667 .addImm(Src1Mods ? Src1Mods->getImm() : 0)
4668 .add(*Src1)
4669 .addImm(Src2Mods ? Src2Mods->getImm() : 0)
4670 .add(*Src2)
4671 .addImm(Clamp ? Clamp->getImm() : 0)
4672 .addImm(Omod ? Omod->getImm() : 0)
4673 .setMIFlags(MI.getFlags());
4674 if (AMDGPU::hasNamedOperand(NewOpc, AMDGPU::OpName::op_sel))
4675 MIB.addImm(OpSel ? OpSel->getImm() : 0);
4676 return MIB;
4677}
4678
4679// It's not generally safe to move VALU instructions across these since it will
4680// start using the register as a base index rather than directly.
4681// XXX - Why isn't hasSideEffects sufficient for these?
4683 switch (MI.getOpcode()) {
4684 case AMDGPU::S_SET_GPR_IDX_ON:
4685 case AMDGPU::S_SET_GPR_IDX_MODE:
4686 case AMDGPU::S_SET_GPR_IDX_OFF:
4687 return true;
4688 default:
4689 return false;
4690 }
4691}
4692
4694 const MachineBasicBlock *MBB,
4695 const MachineFunction &MF) const {
4696 // Skipping the check for SP writes in the base implementation. The reason it
4697 // was added was apparently due to compile time concerns.
4698 //
4699 // TODO: Do we really want this barrier? It triggers unnecessary hazard nops
4700 // but is probably avoidable.
4701
4702 // Copied from base implementation.
4703 // Terminators and labels can't be scheduled around.
4704 if (MI.isTerminator() || MI.isPosition())
4705 return true;
4706
4707 // INLINEASM_BR can jump to another block
4708 if (MI.getOpcode() == TargetOpcode::INLINEASM_BR)
4709 return true;
4710
4711 if (MI.getOpcode() == AMDGPU::SCHED_BARRIER && MI.getOperand(0).getImm() == 0)
4712 return true;
4713
4714 // Target-independent instructions do not have an implicit-use of EXEC, even
4715 // when they operate on VGPRs. Treating EXEC modifications as scheduling
4716 // boundaries prevents incorrect movements of such instructions.
4717 return MI.modifiesRegister(AMDGPU::EXEC, &RI) ||
4718 MI.getOpcode() == AMDGPU::S_SETREG_IMM32_B32 ||
4719 MI.getOpcode() == AMDGPU::S_SETREG_B32 ||
4720 MI.getOpcode() == AMDGPU::S_SETPRIO ||
4721 MI.getOpcode() == AMDGPU::S_SETPRIO_INC_WG ||
4723}
4724
4726 return Opcode == AMDGPU::DS_ORDERED_COUNT ||
4727 Opcode == AMDGPU::DS_ADD_GS_REG_RTN ||
4728 Opcode == AMDGPU::DS_SUB_GS_REG_RTN || isGWS(Opcode);
4729}
4730
4732 // Instructions that access scratch use FLAT encoding or BUF encodings.
4733 if ((!isFLAT(MI) || isFLATGlobal(MI)) && !isBUF(MI))
4734 return false;
4735
4736 // SCRATCH instructions always access scratch.
4737 if (isFLATScratch(MI))
4738 return true;
4739
4740 // If FLAT_SCRATCH registers are not initialized, we can never access scratch
4741 // via the aperture.
4742 if (MI.getMF()->getFunction().hasFnAttribute("amdgpu-no-flat-scratch-init"))
4743 return false;
4744
4745 // If there are no memory operands then conservatively assume the flat
4746 // operation may access scratch.
4747 if (MI.memoperands_empty())
4748 return true;
4749
4750 // See if any memory operand specifies an address space that involves scratch.
4751 return any_of(MI.memoperands(), [](const MachineMemOperand *Memop) {
4752 unsigned AS = Memop->getAddrSpace();
4753 if (AS == AMDGPUAS::FLAT_ADDRESS) {
4754 const MDNode *MD = Memop->getAAInfo().NoAliasAddrSpace;
4755 return !MD || !AMDGPU::hasValueInRangeLikeMetadata(
4756 *MD, AMDGPUAS::PRIVATE_ADDRESS);
4757 }
4758 return AS == AMDGPUAS::PRIVATE_ADDRESS;
4759 });
4760}
4761
4763 assert(isFLAT(MI));
4764
4765 // All flat instructions use the VMEM counter except prefetch.
4766 if (!usesVM_CNT(MI))
4767 return false;
4768
4769 // If there are no memory operands then conservatively assume the flat
4770 // operation may access VMEM.
4771 if (MI.memoperands_empty())
4772 return true;
4773
4774 // See if any memory operand specifies an address space that involves VMEM.
4775 // Flat operations only supported FLAT, LOCAL (LDS), or address spaces
4776 // involving VMEM such as GLOBAL, CONSTANT, PRIVATE (SCRATCH), etc. The REGION
4777 // (GDS) address space is not supported by flat operations. Therefore, simply
4778 // return true unless only the LDS address space is found.
4779 for (const MachineMemOperand *Memop : MI.memoperands()) {
4780 unsigned AS = Memop->getAddrSpace();
4782 if (AS != AMDGPUAS::LOCAL_ADDRESS)
4783 return true;
4784 }
4785
4786 return false;
4787}
4788
4790 bool TgSplit) const {
4791 assert(isFLAT(MI));
4792
4793 // Flat instruction such as SCRATCH and GLOBAL do not use the lgkm counter.
4794 if (!usesLGKM_CNT(MI))
4795 return false;
4796
4797 // If in tgsplit mode then there can be no use of LDS.
4798 if (TgSplit)
4799 return false;
4800
4801 // If there are no memory operands then conservatively assume the flat
4802 // operation may access LDS.
4803 if (MI.memoperands_empty())
4804 return true;
4805
4806 // See if any memory operand specifies an address space that involves LDS.
4807 for (const MachineMemOperand *Memop : MI.memoperands()) {
4808 unsigned AS = Memop->getAddrSpace();
4810 return true;
4811 }
4812
4813 return false;
4814}
4815
4817 // Skip the full operand and register alias search modifiesRegister
4818 // does. There's only a handful of instructions that touch this, it's only an
4819 // implicit def, and doesn't alias any other registers.
4820 return is_contained(MI.getDesc().implicit_defs(), AMDGPU::MODE);
4821}
4822
4824 unsigned Opcode = MI.getOpcode();
4825
4826 if (MI.mayStore() && isSMRD(MI))
4827 return true; // scalar store or atomic
4828
4829 // This will terminate the function when other lanes may need to continue.
4830 if (MI.isReturn())
4831 return true;
4832
4833 // These instructions cause shader I/O that may cause hardware lockups
4834 // when executed with an empty EXEC mask.
4835 //
4836 // Note: exp with VM = DONE = 0 is automatically skipped by hardware when
4837 // EXEC = 0, but checking for that case here seems not worth it
4838 // given the typical code patterns.
4839 if (Opcode == AMDGPU::S_SENDMSG || Opcode == AMDGPU::S_SENDMSGHALT ||
4840 isEXP(Opcode) || Opcode == AMDGPU::DS_ORDERED_COUNT ||
4841 Opcode == AMDGPU::S_TRAP || Opcode == AMDGPU::S_WAIT_EVENT ||
4842 Opcode == AMDGPU::S_SETHALT)
4843 return true;
4844
4845 if (MI.isCall() || MI.isInlineAsm())
4846 return true; // conservative assumption
4847
4848 // V_PERM_PK16 must issue with EXEC != 0 so its follower (or an inserted
4849 // V_NOP) actually runs on the VALU pipe. Returning true here keeps the
4850 // s_cbranch_execz that skips this region when EXEC is empty.
4851 if (ST.hasVPermPk16Hazard() && isVPermPk16(Opcode))
4852 return true;
4853
4854 // Assume that barrier interactions are only intended with active lanes.
4855 if (isBarrier(Opcode))
4856 return true;
4857
4858 // A mode change is a scalar operation that influences vector instructions.
4860 return true;
4861
4862 // These are like SALU instructions in terms of effects, so it's questionable
4863 // whether we should return true for those.
4864 //
4865 // However, executing them with EXEC = 0 causes them to operate on undefined
4866 // data, which we avoid by returning true here.
4867 if (Opcode == AMDGPU::V_READFIRSTLANE_B32 ||
4868 Opcode == AMDGPU::V_READLANE_B32 || Opcode == AMDGPU::V_WRITELANE_B32 ||
4869 Opcode == AMDGPU::SI_RESTORE_S32_FROM_VGPR ||
4870 Opcode == AMDGPU::SI_SPILL_S32_TO_VGPR)
4871 return true;
4872
4873 return false;
4874}
4875
4877 const MachineInstr &MI) const {
4878 if (MI.isMetaInstruction())
4879 return false;
4880
4881 // This won't read exec if this is an SGPR->SGPR copy.
4882 if (MI.isCopyLike()) {
4883 if (!RI.isSGPRReg(MRI, MI.getOperand(0).getReg()))
4884 return true;
4885
4886 // Make sure this isn't copying exec as a normal operand
4887 return MI.readsRegister(AMDGPU::EXEC, &RI);
4888 }
4889
4890 // Make a conservative assumption about the callee.
4891 if (MI.isCall())
4892 return true;
4893
4894 // Be conservative with any unhandled generic opcodes.
4895 if (!isTargetSpecificOpcode(MI.getOpcode()))
4896 return true;
4897
4898 return !isSALU(MI) || MI.readsRegister(AMDGPU::EXEC, &RI);
4899}
4900
4902 switch (Imm.getBitWidth()) {
4903 case 1: // This likely will be a condition code mask.
4904 return true;
4905
4906 case 32:
4907 return AMDGPU::isInlinableLiteral32(Imm.getSExtValue(),
4908 ST.hasInv2PiInlineImm());
4909 case 64:
4910 return AMDGPU::isInlinableLiteral64(Imm.getSExtValue(),
4911 ST.hasInv2PiInlineImm());
4912 case 16:
4913 return ST.has16BitInsts() &&
4914 AMDGPU::isInlinableLiteralI16(Imm.getSExtValue(),
4915 ST.hasInv2PiInlineImm());
4916 default:
4917 llvm_unreachable("invalid bitwidth");
4918 }
4919}
4920
4922 APInt IntImm = Imm.bitcastToAPInt();
4923 int64_t IntImmVal = IntImm.getSExtValue();
4924 bool HasInv2Pi = ST.hasInv2PiInlineImm();
4925 switch (APFloat::SemanticsToEnum(Imm.getSemantics())) {
4926 default:
4927 llvm_unreachable("invalid fltSemantics");
4930 return isInlineConstant(IntImm);
4932 return ST.has16BitInsts() &&
4933 AMDGPU::isInlinableLiteralBF16(IntImmVal, HasInv2Pi);
4935 return ST.has16BitInsts() &&
4936 AMDGPU::isInlinableLiteralFP16(IntImmVal, HasInv2Pi);
4937 }
4938}
4939
4940bool SIInstrInfo::isInlineConstant(int64_t Imm, uint8_t OperandType) const {
4941 // MachineOperand provides no way to tell the true operand size, since it only
4942 // records a 64-bit value. We need to know the size to determine if a 32-bit
4943 // floating point immediate bit pattern is legal for an integer immediate. It
4944 // would be for any 32-bit integer operand, but would not be for a 64-bit one.
4945 switch (OperandType) {
4955 int32_t Trunc = static_cast<int32_t>(Imm);
4956 return AMDGPU::isInlinableLiteral32(Trunc, ST.hasInv2PiInlineImm());
4957 }
4965 return AMDGPU::isInlinableLiteral64(Imm, ST.hasInv2PiInlineImm());
4968 // We would expect inline immediates to not be concerned with an integer/fp
4969 // distinction. However, in the case of 16-bit integer operations, the
4970 // "floating point" values appear to not work. It seems read the low 16-bits
4971 // of 32-bit immediates, which happens to always work for the integer
4972 // values.
4973 //
4974 // See llvm bugzilla 46302.
4975 //
4976 // TODO: Theoretically we could use op-sel to use the high bits of the
4977 // 32-bit FP values.
4986 return AMDGPU::isPKFMACF16InlineConstant(Imm, ST.isGFX11Plus());
4991 return false;
4994 if (isInt<16>(Imm) || isUInt<16>(Imm)) {
4995 // A few special case instructions have 16-bit operands on subtargets
4996 // where 16-bit instructions are not legal.
4997 // TODO: Do the 32-bit immediates work? We shouldn't really need to handle
4998 // constants in these cases
4999 int16_t Trunc = static_cast<int16_t>(Imm);
5000 return ST.has16BitInsts() &&
5001 AMDGPU::isInlinableLiteralFP16(Trunc, ST.hasInv2PiInlineImm());
5002 }
5003
5004 return false;
5005 }
5008 if (isInt<16>(Imm) || isUInt<16>(Imm)) {
5009 int16_t Trunc = static_cast<int16_t>(Imm);
5010 return ST.has16BitInsts() &&
5011 AMDGPU::isInlinableLiteralBF16(Trunc, ST.hasInv2PiInlineImm());
5012 }
5013 return false;
5014 }
5019 return false;
5021 return isLegalAV64PseudoImm(Imm);
5024 // Always embedded in the instruction for free.
5025 return true;
5035 // Just ignore anything else.
5036 return false;
5037 default:
5038 llvm_unreachable("invalid operand type");
5039 }
5040}
5041
5042static bool compareMachineOp(const MachineOperand &Op0,
5043 const MachineOperand &Op1) {
5044 if (Op0.getType() != Op1.getType())
5045 return false;
5046
5047 switch (Op0.getType()) {
5049 return Op0.getReg() == Op1.getReg();
5051 return Op0.getImm() == Op1.getImm();
5052 default:
5053 llvm_unreachable("Didn't expect to be comparing these operand types");
5054 }
5055}
5056
5058 const MCOperandInfo &OpInfo) const {
5059 if (OpInfo.OperandType == MCOI::OPERAND_IMMEDIATE)
5060 return true;
5061
5062 if (!RI.opCanUseLiteralConstant(OpInfo.OperandType))
5063 return false;
5064
5065 if (!isVOP3(InstDesc) || !AMDGPU::isSISrcOperand(OpInfo))
5066 return true;
5067
5068 return ST.hasVOP3Literal();
5069}
5070
5071bool SIInstrInfo::isImmOperandLegal(const MCInstrDesc &InstDesc, unsigned OpNo,
5072 int64_t ImmVal) const {
5073 const unsigned Opc = InstDesc.getOpcode();
5074 int Src1Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1);
5075 if (Src1Idx != -1 && isDPP(Opc) && !ST.hasDPPSrc1SGPR() &&
5076 OpNo == static_cast<unsigned>(Src1Idx))
5077 return false;
5078
5079 const MCOperandInfo &OpInfo = InstDesc.operands()[OpNo];
5080 if (isInlineConstant(ImmVal, OpInfo.OperandType)) {
5081 if (isMAI(InstDesc) && ST.hasMFMAInlineLiteralBug() &&
5082 OpNo == (unsigned)AMDGPU::getNamedOperandIdx(InstDesc.getOpcode(),
5083 AMDGPU::OpName::src2))
5084 return false;
5085
5086 if (ST.hasBF16InlineConstFromUpperFP32() && isVOP1(Opc)) {
5087 if ((OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_BF16 ||
5088 OpInfo.OperandType == AMDGPU::OPERAND_REG_INLINE_C_BF16) &&
5089 isInlineConstant(ImmVal, OpInfo.OperandType))
5090 return false;
5091 }
5092
5093 return RI.opCanUseInlineConstant(OpInfo.OperandType);
5094 }
5095
5096 return isLiteralOperandLegal(InstDesc, OpInfo);
5097}
5098
5099bool SIInstrInfo::isImmOperandLegal(const MCInstrDesc &InstDesc, unsigned OpNo,
5100 const MachineOperand &MO) const {
5101 if (MO.isImm())
5102 return isImmOperandLegal(InstDesc, OpNo, MO.getImm());
5103
5104 assert((MO.isTargetIndex() || MO.isFI() || MO.isGlobal()) &&
5105 "unexpected imm-like operand kind");
5106 const MCOperandInfo &OpInfo = InstDesc.operands()[OpNo];
5107 return isLiteralOperandLegal(InstDesc, OpInfo);
5108}
5109
5111 // 2 32-bit inline constants packed into one.
5112 return AMDGPU::isInlinableLiteral32(Lo_32(Imm), ST.hasInv2PiInlineImm()) &&
5113 AMDGPU::isInlinableLiteral32(Hi_32(Imm), ST.hasInv2PiInlineImm());
5114}
5115
5116bool SIInstrInfo::hasVALU32BitEncoding(unsigned Opcode) const {
5117 // GFX90A does not have V_MUL_LEGACY_F32_e32.
5118 if (Opcode == AMDGPU::V_MUL_LEGACY_F32_e64 && ST.hasGFX90AInsts())
5119 return false;
5120
5121 int Op32 = AMDGPU::getVOPe32(Opcode);
5122 if (Op32 == -1)
5123 return false;
5124
5125 return pseudoToMCOpcode(Op32) != -1;
5126}
5127
5128/// Return true if \p MI is a VALU comparison, i.e. an instruction that writes
5129/// a lane mask with one bit per lane, and zeroes the bits of lanes that were
5130/// inactive when it executed.
5131///
5132/// TODO: Also handle the sdst result of V_ADD_CO_U32 and V_SUB_CO_U32 and
5133/// V_DIV_SCALE_F32.
5134static bool isVCmp(const SIInstrInfo &TII, const MachineInstr &MI) {
5135 if (TII.isVOPC(MI))
5136 return true;
5137 int Op32 = AMDGPU::getVOPe32(MI.getOpcode());
5138 return Op32 != -1 && TII.isVOPC(Op32);
5139}
5140
5142 const MachineRegisterInfo &MRI,
5143 unsigned Depth) const {
5144 assert(MRI.isSSA() && "isMaskedByExec requires SSA form");
5146 const MachineBasicBlock *MBB = Use.getParent();
5147
5148 // EXEC itself is trivially masked by EXEC.
5149 if (Reg == LMC.ExecReg)
5150 return true;
5151
5152 // Maximum depth of the def-use walk.
5153 constexpr unsigned MaxDepth = 6;
5154 if (Depth >= MaxDepth || !Reg.isVirtual())
5155 return false;
5156
5157 // Only look at definitions that can execute under the same EXEC mask as the
5158 // use.
5159 const MachineInstr *Def = MRI.getVRegDef(Reg);
5160 if (!Def || Def->getParent() != MBB)
5161 return false;
5162
5163 if (isVCmp(*this, *Def))
5164 return true;
5165
5166 // Recurse into an operand, which must be a whole register to say anything
5167 // about the whole lane mask.
5168 auto Recurse = [&](unsigned OpIdx) {
5169 const MachineOperand &MO = Def->getOperand(OpIdx);
5170 return MO.isReg() && !MO.getSubReg() &&
5171 isMaskedByExec(MO.getReg(), Use, MRI, Depth + 1);
5172 };
5173
5174 unsigned Opc = Def->getOpcode();
5175 if (Opc == AMDGPU::COPY && Recurse(1))
5176 return true;
5177 if (Opc == LMC.AndOpc && (Recurse(1) || Recurse(2)))
5178 return true;
5179 if (Opc == LMC.AndN2Opc && Recurse(1))
5180 return true;
5181 if ((Opc == LMC.OrOpc || Opc == LMC.XorOpc) && Recurse(1) && Recurse(2))
5182 return true;
5183 // TODO: Sometimes we encounter "reg = S_CSELECT -1, 0". If Reg has no other
5184 // uses this could be optimized to "reg = S_CSELECT $exec, 0".
5185
5186 return false;
5187}
5188
5189bool SIInstrInfo::hasModifiers(unsigned Opcode) const {
5190 // The src0_modifier operand is present on all instructions
5191 // that have modifiers.
5192
5193 return AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::src0_modifiers);
5194}
5195
5197 AMDGPU::OpName OpName) const {
5198 const MachineOperand *Mods = getNamedOperand(MI, OpName);
5199 return Mods && Mods->getImm();
5200}
5201
5203 return any_of(ModifierOpNames,
5204 [&](AMDGPU::OpName Name) { return hasModifiersSet(MI, Name); });
5205}
5206
5208 const MachineRegisterInfo &MRI) const {
5209 const MachineOperand *Src2 = getNamedOperand(MI, AMDGPU::OpName::src2);
5210 // Can't shrink instruction with three operands.
5211 if (Src2) {
5212 switch (MI.getOpcode()) {
5213 default: return false;
5214
5215 case AMDGPU::V_ADDC_U32_e64:
5216 case AMDGPU::V_SUBB_U32_e64:
5217 case AMDGPU::V_SUBBREV_U32_e64: {
5218 const MachineOperand *Src1
5219 = getNamedOperand(MI, AMDGPU::OpName::src1);
5220 if (!Src1->isReg() || !RI.isVGPR(MRI, Src1->getReg()))
5221 return false;
5222 // Additional verification is needed for sdst/src2.
5223 return true;
5224 }
5225 case AMDGPU::V_MAC_F16_e64:
5226 case AMDGPU::V_MAC_F32_e64:
5227 case AMDGPU::V_MAC_LEGACY_F32_e64:
5228 case AMDGPU::V_FMAC_F16_e64:
5229 case AMDGPU::V_FMAC_F16_t16_e64:
5230 case AMDGPU::V_FMAC_F16_fake16_e64:
5231 case AMDGPU::V_FMAC_F32_e64:
5232 case AMDGPU::V_FMAC_F64_e64:
5233 case AMDGPU::V_FMAC_LEGACY_F32_e64:
5234 if (!Src2->isReg() || !RI.isVGPR(MRI, Src2->getReg()) ||
5235 hasModifiersSet(MI, AMDGPU::OpName::src2_modifiers))
5236 return false;
5237 break;
5238
5239 case AMDGPU::V_CNDMASK_B32_e64:
5240 break;
5241 }
5242 }
5243
5244 const MachineOperand *Src1 = getNamedOperand(MI, AMDGPU::OpName::src1);
5245 if (Src1 && (!Src1->isReg() || !RI.isVGPR(MRI, Src1->getReg()) ||
5246 hasModifiersSet(MI, AMDGPU::OpName::src1_modifiers)))
5247 return false;
5248
5249 // Make sure src0 isn't using any modifiers.
5250 if (hasModifiersSet(MI, AMDGPU::OpName::src0_modifiers))
5251 return false;
5252
5253 // Can it be shrunk to a valid 32 bit opcode?
5254 if (!hasVALU32BitEncoding(MI.getOpcode()))
5255 return false;
5256
5257 const MachineOperand *Src0 = getNamedOperand(MI, AMDGPU::OpName::src0);
5258 if (Src0 && Src0->isImm()) {
5259 unsigned Op32 = AMDGPU::getVOPe32(MI.getOpcode());
5260 if (!isImmOperandLegal(
5261 get(Op32), AMDGPU::getNamedOperandIdx(Op32, AMDGPU::OpName::src0),
5262 *Src0))
5263 return false;
5264 }
5265
5266 // Check output modifiers
5267 return !hasModifiersSet(MI, AMDGPU::OpName::omod) &&
5268 !hasModifiersSet(MI, AMDGPU::OpName::clamp) &&
5269 !hasModifiersSet(MI, AMDGPU::OpName::byte_sel) &&
5270 // TODO: Can we avoid checking bound_ctrl/fi here?
5271 // They are only used by permlane*_swap special case.
5272 !hasModifiersSet(MI, AMDGPU::OpName::bound_ctrl) &&
5273 !hasModifiersSet(MI, AMDGPU::OpName::fi);
5274}
5275
5276// Set VCC operand with all flags from \p Orig, except for setting it as
5277// implicit.
5279 const MachineOperand &Orig) {
5280
5281 for (MachineOperand &Use : MI.implicit_operands()) {
5282 if (Use.isUse() &&
5283 (Use.getReg() == AMDGPU::VCC || Use.getReg() == AMDGPU::VCC_LO)) {
5284 Use.setIsUndef(Orig.isUndef());
5285 Use.setIsKill(Orig.isKill());
5286 return;
5287 }
5288 }
5289}
5290
5292 unsigned Op32) const {
5293 MachineBasicBlock *MBB = MI.getParent();
5294
5295 const MCInstrDesc &Op32Desc = get(Op32);
5296 MachineInstrBuilder Inst32 =
5297 BuildMI(*MBB, MI, MI.getDebugLoc(), Op32Desc)
5298 .setMIFlags(MI.getFlags());
5299
5300 // Add the dst operand if the 32-bit encoding also has an explicit $vdst.
5301 // For VOPC instructions, this is replaced by an implicit def of vcc.
5302
5303 // We assume the defs of the shrunk opcode are in the same order, and the
5304 // shrunk opcode loses the last def (SGPR def, in the VOP3->VOPC case).
5305 for (int I = 0, E = Op32Desc.getNumDefs(); I != E; ++I)
5306 Inst32.add(MI.getOperand(I));
5307
5308 const MachineOperand *Src2 = getNamedOperand(MI, AMDGPU::OpName::src2);
5309
5310 int Idx = MI.getNumExplicitDefs();
5311 for (const MachineOperand &Use : MI.explicit_uses()) {
5312 int OpTy = MI.getDesc().operands()[Idx++].OperandType;
5314 continue;
5315
5316 if (&Use == Src2) {
5317 if (AMDGPU::getNamedOperandIdx(Op32, AMDGPU::OpName::src2) == -1) {
5318 // In the case of V_CNDMASK_B32_e32, the explicit operand src2 is
5319 // replaced with an implicit read of vcc or vcc_lo. The implicit read
5320 // of vcc was already added during the initial BuildMI, but we
5321 // 1) may need to change vcc to vcc_lo to preserve the original register
5322 // 2) have to preserve the original flags.
5323 copyFlagsToImplicitVCC(*Inst32, *Src2);
5324 continue;
5325 }
5326 }
5327
5328 Inst32.add(Use);
5329 }
5330
5331 // FIXME: Losing implicit operands
5332 fixImplicitOperands(*Inst32);
5333
5334 // The explicit carry/result def is dropped in favor of an implicit VCC def;
5335 // preserve the dead flag.
5336 const MachineOperand *OldSDst = getNamedOperand(MI, AMDGPU::OpName::sdst);
5337 if (OldSDst && OldSDst->isDead()) {
5338 if (MachineOperand *NewVCC =
5339 Inst32->findRegisterDefOperand(RI.getVCC(), &RI))
5340 NewVCC->setIsDead();
5341 }
5342
5343 return Inst32;
5344}
5345
5347 // Null is free
5348 Register Reg = RegOp.getReg();
5349 if (Reg == AMDGPU::SGPR_NULL || Reg == AMDGPU::SGPR_NULL64)
5350 return false;
5351
5352 // SGPRs use the constant bus
5353
5354 // FIXME: implicit registers that are not part of the MCInstrDesc's implicit
5355 // physical register operands should also count, except for exec.
5356 if (RegOp.isImplicit())
5357 return Reg == AMDGPU::VCC || Reg == AMDGPU::VCC_LO || Reg == AMDGPU::M0;
5358
5359 // SGPRs use the constant bus
5360 return AMDGPU::SReg_32RegClass.contains(Reg) ||
5361 AMDGPU::SReg_64RegClass.contains(Reg);
5362}
5363
5365 const MachineRegisterInfo &MRI) const {
5366 Register Reg = RegOp.getReg();
5367 return Reg.isVirtual() ? RI.isSGPRClass(MRI.getRegClass(Reg))
5368 : physRegUsesConstantBus(RegOp);
5369}
5370
5372 const MachineOperand &MO,
5373 const MCOperandInfo &OpInfo) const {
5374 // Literal constants use the constant bus.
5375 if (!MO.isReg())
5376 return !isInlineConstant(MO, OpInfo);
5377
5378 Register Reg = MO.getReg();
5379 return Reg.isVirtual() ? RI.isSGPRClass(MRI.getRegClass(Reg))
5381}
5382
5384 for (const MachineOperand &MO : MI.implicit_operands()) {
5385 // We only care about reads.
5386 if (MO.isDef())
5387 continue;
5388
5389 switch (MO.getReg()) {
5390 case AMDGPU::VCC:
5391 case AMDGPU::VCC_LO:
5392 case AMDGPU::VCC_HI:
5393 case AMDGPU::M0:
5394 case AMDGPU::FLAT_SCR:
5395 return MO.getReg();
5396
5397 default:
5398 break;
5399 }
5400 }
5401
5402 return Register();
5403}
5404
5405static bool shouldReadExec(const MachineInstr &MI) {
5406 if (SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true)) {
5407 switch (MI.getOpcode()) {
5408 case AMDGPU::V_READLANE_B32:
5409 case AMDGPU::SI_RESTORE_S32_FROM_VGPR:
5410 case AMDGPU::V_WRITELANE_B32:
5411 case AMDGPU::SI_SPILL_S32_TO_VGPR:
5412 return false;
5413 }
5414
5415 return true;
5416 }
5417
5418 if (MI.isPreISelOpcode() ||
5419 SIInstrInfo::isGenericOpcode(MI.getOpcode()) ||
5422 return false;
5423
5424 return true;
5425}
5426
5427static bool isRegOrFI(const MachineOperand &MO) {
5428 return MO.isReg() || MO.isFI();
5429}
5430
5431static bool isSubRegOf(const SIRegisterInfo &TRI,
5432 const MachineOperand &SuperVec,
5433 const MachineOperand &SubReg) {
5434 if (SubReg.getReg().isPhysical())
5435 return TRI.isSubRegister(SuperVec.getReg(), SubReg.getReg());
5436
5437 return SubReg.getSubReg() != AMDGPU::NoSubRegister &&
5438 SubReg.getReg() == SuperVec.getReg();
5439}
5440
5441// Verify the illegal copy from vector register to SGPR for generic opcode COPY
5442bool SIInstrInfo::verifyCopy(const MachineInstr &MI,
5443 const MachineRegisterInfo &MRI,
5444 StringRef &ErrInfo) const {
5445 Register DstReg = MI.getOperand(0).getReg();
5446 Register SrcReg = MI.getOperand(1).getReg();
5447 // This is a check for copy from vector register to SGPR
5448 if (RI.isVectorRegister(MRI, SrcReg) && RI.isSGPRReg(MRI, DstReg)) {
5449 ErrInfo = "illegal copy from vector register to SGPR";
5450 return false;
5451 }
5452 return true;
5453}
5454
5456 StringRef &ErrInfo) const {
5457 uint32_t Opcode = MI.getOpcode();
5458 const MachineFunction *MF = MI.getMF();
5459 const MachineRegisterInfo &MRI = MF->getRegInfo();
5460
5461 // FIXME: At this point the COPY verify is done only for non-ssa forms.
5462 // Find a better property to recognize the point where instruction selection
5463 // is just done.
5464 // We can only enforce this check after SIFixSGPRCopies pass so that the
5465 // illegal copies are legalized and thereafter we don't expect a pass
5466 // inserting similar copies.
5467 if (!MRI.isSSA() && MI.isCopy())
5468 return verifyCopy(MI, MRI, ErrInfo);
5469
5470 if (SIInstrInfo::isGenericOpcode(Opcode))
5471 return true;
5472
5473 int Src0Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0);
5474 int Src1Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src1);
5475 int Src2Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src2);
5476 int Src3Idx = -1;
5477 if (Src0Idx == -1) {
5478 // VOPD V_DUAL_* instructions use different operand names.
5479 Src0Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0X);
5480 Src1Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vsrc1X);
5481 Src2Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0Y);
5482 Src3Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vsrc1Y);
5483 }
5484
5485 // Make sure the number of operands is correct.
5486 const MCInstrDesc &Desc = get(Opcode);
5487 if (!Desc.isVariadic() &&
5488 Desc.getNumOperands() != MI.getNumExplicitOperands()) {
5489 ErrInfo = "Instruction has wrong number of operands.";
5490 return false;
5491 }
5492
5493 if (MI.isInlineAsm()) {
5494 // Verify register classes for inlineasm constraints.
5495 for (unsigned I = InlineAsm::MIOp_FirstOperand, E = MI.getNumOperands();
5496 I != E; ++I) {
5497 const TargetRegisterClass *RC = MI.getRegClassConstraint(I, this, &RI);
5498 if (!RC)
5499 continue;
5500
5501 const MachineOperand &Op = MI.getOperand(I);
5502 if (!Op.isReg())
5503 continue;
5504
5505 Register Reg = Op.getReg();
5506 if (!Reg.isVirtual() && !RC->contains(Reg)) {
5507 ErrInfo = "inlineasm operand has incorrect register class.";
5508 return false;
5509 }
5510 }
5511
5512 return true;
5513 }
5514
5515 if (isImage(MI) && MI.memoperands_empty() && MI.mayLoadOrStore()) {
5516 ErrInfo = "missing memory operand from image instruction.";
5517 return false;
5518 }
5519
5520 // Make sure the register classes are correct.
5521 for (int i = 0, e = Desc.getNumOperands(); i != e; ++i) {
5522 const MachineOperand &MO = MI.getOperand(i);
5523 if (MO.isFPImm()) {
5524 ErrInfo = "FPImm Machine Operands are not supported. ISel should bitcast "
5525 "all fp values to integers.";
5526 return false;
5527 }
5528
5529 const MCOperandInfo &OpInfo = Desc.operands()[i];
5530
5531 switch (OpInfo.OperandType) {
5533 if (MI.getOperand(i).isImm() || MI.getOperand(i).isGlobal()) {
5534 ErrInfo = "Illegal immediate value for operand.";
5535 return false;
5536 }
5537 break;
5551 break;
5565 if (!MO.isReg() && (!MO.isImm() || !isInlineConstant(MI, i))) {
5566 ErrInfo = "Illegal immediate value for operand.";
5567 return false;
5568 }
5569 break;
5570 }
5575 if (ST.has64BitLiterals() && Desc.getSize() != 4 && MO.isImm() &&
5576 !isInlineConstant(MI, i) &&
5578 OpInfo.OperandType ==
5580 ErrInfo = "illegal 64-bit immediate value for operand.";
5581 return false;
5582 }
5583 break;
5586 if (!MI.getOperand(i).isImm() || !isInlineConstant(MI, i)) {
5587 ErrInfo = "Expected inline constant for operand.";
5588 return false;
5589 }
5590 break;
5593 break;
5598 // Check if this operand is an immediate.
5599 // FrameIndex operands will be replaced by immediates, so they are
5600 // allowed.
5601 if (!MI.getOperand(i).isImm() && !MI.getOperand(i).isFI()) {
5602 ErrInfo = "Expected immediate, but got non-immediate";
5603 return false;
5604 }
5605 break;
5609 break;
5610 default:
5611 if (OpInfo.isGenericType())
5612 continue;
5613 break;
5614 }
5615 }
5616
5617 // Verify SDWA
5618 if (isSDWA(MI)) {
5619 if (!ST.hasSDWA()) {
5620 ErrInfo = "SDWA is not supported on this target";
5621 return false;
5622 }
5623
5624 for (auto Op : {AMDGPU::OpName::src0_sel, AMDGPU::OpName::src1_sel,
5625 AMDGPU::OpName::dst_sel}) {
5626 const MachineOperand *MO = getNamedOperand(MI, Op);
5627 if (!MO)
5628 continue;
5629 int64_t Imm = MO->getImm();
5631 ErrInfo = "Invalid SDWA selection";
5632 return false;
5633 }
5634 }
5635
5636 int DstIdx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vdst);
5637
5638 for (int OpIdx : {DstIdx, Src0Idx, Src1Idx, Src2Idx}) {
5639 if (OpIdx == -1)
5640 continue;
5641 const MachineOperand &MO = MI.getOperand(OpIdx);
5642
5643 if (!ST.hasSDWAScalar()) {
5644 // Only VGPRS on VI
5645 if (!MO.isReg() || !RI.hasVGPRs(RI.getRegClassForReg(MRI, MO.getReg()))) {
5646 ErrInfo = "Only VGPRs allowed as operands in SDWA instructions on VI";
5647 return false;
5648 }
5649 } else {
5650 // No immediates on GFX9
5651 if (!MO.isReg()) {
5652 ErrInfo =
5653 "Only reg allowed as operands in SDWA instructions on GFX9+";
5654 return false;
5655 }
5656 }
5657 }
5658
5659 if (!ST.hasSDWAOmod()) {
5660 // No omod allowed on VI
5661 const MachineOperand *OMod = getNamedOperand(MI, AMDGPU::OpName::omod);
5662 if (OMod != nullptr &&
5663 (!OMod->isImm() || OMod->getImm() != 0)) {
5664 ErrInfo = "OMod not allowed in SDWA instructions on VI";
5665 return false;
5666 }
5667 }
5668
5669 if (Opcode == AMDGPU::V_CVT_F32_FP8_sdwa ||
5670 Opcode == AMDGPU::V_CVT_F32_BF8_sdwa ||
5671 Opcode == AMDGPU::V_CVT_PK_F32_FP8_sdwa ||
5672 Opcode == AMDGPU::V_CVT_PK_F32_BF8_sdwa) {
5673 const MachineOperand *Src0ModsMO =
5674 getNamedOperand(MI, AMDGPU::OpName::src0_modifiers);
5675 unsigned Mods = Src0ModsMO->getImm();
5676 if (Mods & SISrcMods::ABS || Mods & SISrcMods::NEG ||
5677 Mods & SISrcMods::SEXT) {
5678 ErrInfo = "sext, abs and neg are not allowed on this instruction";
5679 return false;
5680 }
5681 }
5682
5683 uint32_t BasicOpcode = AMDGPU::getBasicFromSDWAOp(Opcode);
5684 if (isVOPC(BasicOpcode)) {
5685 if (!ST.hasSDWASdst() && DstIdx != -1) {
5686 // Only vcc allowed as dst on VI for VOPC
5687 const MachineOperand &Dst = MI.getOperand(DstIdx);
5688 if (!Dst.isReg() || Dst.getReg() != AMDGPU::VCC) {
5689 ErrInfo = "Only VCC allowed as dst in SDWA instructions on VI";
5690 return false;
5691 }
5692 } else if (!ST.hasSDWAOutModsVOPC()) {
5693 // No clamp allowed on GFX9 for VOPC
5694 const MachineOperand *Clamp = getNamedOperand(MI, AMDGPU::OpName::clamp);
5695 if (Clamp && (!Clamp->isImm() || Clamp->getImm() != 0)) {
5696 ErrInfo = "Clamp not allowed in VOPC SDWA instructions on VI";
5697 return false;
5698 }
5699
5700 // No omod allowed on GFX9 for VOPC
5701 const MachineOperand *OMod = getNamedOperand(MI, AMDGPU::OpName::omod);
5702 if (OMod && (!OMod->isImm() || OMod->getImm() != 0)) {
5703 ErrInfo = "OMod not allowed in VOPC SDWA instructions on VI";
5704 return false;
5705 }
5706 }
5707 }
5708
5709 const MachineOperand *DstUnused = getNamedOperand(MI, AMDGPU::OpName::dst_unused);
5710 if (DstUnused && DstUnused->isImm() &&
5711 DstUnused->getImm() == AMDGPU::SDWA::UNUSED_PRESERVE) {
5712 const MachineOperand &Dst = MI.getOperand(DstIdx);
5713 if (!Dst.isReg() || !Dst.isTied()) {
5714 ErrInfo = "Dst register should have tied register";
5715 return false;
5716 }
5717
5718 const MachineOperand &TiedMO =
5719 MI.getOperand(MI.findTiedOperandIdx(DstIdx));
5720 if (!TiedMO.isReg() || !TiedMO.isImplicit() || !TiedMO.isUse()) {
5721 ErrInfo =
5722 "Dst register should be tied to implicit use of preserved register";
5723 return false;
5724 }
5725 if (TiedMO.getReg().isPhysical() && Dst.getReg() != TiedMO.getReg()) {
5726 ErrInfo = "Dst register should use same physical register as preserved";
5727 return false;
5728 }
5729 }
5730 }
5731
5732 if (isDPP(MI) && !ST.hasDPPSrc1SGPR() && Src1Idx != -1) {
5733 const MachineOperand &Src1MO = MI.getOperand(Src1Idx);
5734 if (Src1MO.isReg() && RI.isSGPRReg(MRI, Src1MO.getReg())) {
5735 ErrInfo = "DPP src1 cannot be SGPR on this subtarget";
5736 return false;
5737 }
5738 if (Src1MO.isImm()) {
5739 ErrInfo = "DPP src1 cannot be an immediate on this subtarget";
5740 return false;
5741 }
5742 }
5743
5744 // Verify MIMG / VIMAGE / VSAMPLE
5745 if (isImage(Opcode) && !MI.mayStore()) {
5746 // Ensure that the return type used is large enough for all the options
5747 // being used TFE/LWE require an extra result register.
5748 const MachineOperand *DMask = getNamedOperand(MI, AMDGPU::OpName::dmask);
5749 if (DMask) {
5750 uint64_t DMaskImm = DMask->getImm();
5751 uint32_t RegCount = isGather4(Opcode) ? 4 : llvm::popcount(DMaskImm);
5752 const MachineOperand *TFE = getNamedOperand(MI, AMDGPU::OpName::tfe);
5753 const MachineOperand *LWE = getNamedOperand(MI, AMDGPU::OpName::lwe);
5754 const MachineOperand *D16 = getNamedOperand(MI, AMDGPU::OpName::d16);
5755
5756 // Adjust for packed 16 bit values
5757 if (D16 && D16->getImm() && !ST.hasUnpackedD16VMem())
5758 RegCount = divideCeil(RegCount, 2);
5759
5760 // Adjust if using LWE or TFE
5761 if ((LWE && LWE->getImm()) || (TFE && TFE->getImm()))
5762 RegCount += 1;
5763
5764 const uint32_t DstIdx =
5765 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vdata);
5766 const MachineOperand &Dst = MI.getOperand(DstIdx);
5767 if (Dst.isReg()) {
5768 const TargetRegisterClass *DstRC = getOpRegClass(MI, DstIdx);
5769 uint32_t DstSize = RI.getRegSizeInBits(*DstRC) / 32;
5770 if (RegCount > DstSize) {
5771 ErrInfo = "Image instruction returns too many registers for dst "
5772 "register class";
5773 return false;
5774 }
5775 }
5776 }
5777 }
5778
5779 // Verify VOP*. Ignore multiple sgpr operands on writelane.
5780 if (isVALU(MI, /*AllowLDSDMA=*/false) &&
5781 Desc.getOpcode() != AMDGPU::V_WRITELANE_B32) {
5782 unsigned ConstantBusCount = 0;
5783 bool UsesLiteral = false;
5784 const MachineOperand *LiteralVal = nullptr;
5785
5786 int ImmIdx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::imm);
5787 if (ImmIdx != -1) {
5788 ++ConstantBusCount;
5789 UsesLiteral = true;
5790 LiteralVal = &MI.getOperand(ImmIdx);
5791 }
5792
5793 SmallVector<Register, 2> SGPRsUsed;
5794 Register SGPRUsed;
5795
5796 // Only look at the true operands. Only a real operand can use the constant
5797 // bus, and we don't want to check pseudo-operands like the source modifier
5798 // flags.
5799 for (int OpIdx : {Src0Idx, Src1Idx, Src2Idx, Src3Idx}) {
5800 if (OpIdx == -1)
5801 continue;
5802 const MachineOperand &MO = MI.getOperand(OpIdx);
5803 if (usesConstantBus(MRI, MO, MI.getDesc().operands()[OpIdx])) {
5804 if (MO.isReg()) {
5805 SGPRUsed = MO.getReg();
5806 if (!llvm::is_contained(SGPRsUsed, SGPRUsed)) {
5807 ++ConstantBusCount;
5808 SGPRsUsed.push_back(SGPRUsed);
5809 }
5810 } else if (!MO.isFI()) { // Treat FI like a register.
5811 if (!UsesLiteral) {
5812 ++ConstantBusCount;
5813 UsesLiteral = true;
5814 LiteralVal = &MO;
5815 } else if (!MO.isIdenticalTo(*LiteralVal)) {
5816 assert(isVOP2(MI) || isVOP3(MI));
5817 ErrInfo = "VOP2/VOP3 instruction uses more than one literal";
5818 return false;
5819 }
5820 }
5821 }
5822 }
5823
5824 SGPRUsed = findImplicitSGPRRead(MI);
5825 if (SGPRUsed) {
5826 // Implicit uses may safely overlap true operands
5827 if (llvm::all_of(SGPRsUsed, [this, SGPRUsed](unsigned SGPR) {
5828 return !RI.regsOverlap(SGPRUsed, SGPR);
5829 })) {
5830 ++ConstantBusCount;
5831 SGPRsUsed.push_back(SGPRUsed);
5832 }
5833 }
5834
5835 // v_writelane_b32 is an exception from constant bus restriction:
5836 // vsrc0 can be sgpr, const or m0 and lane select sgpr, m0 or inline-const
5837 if (ConstantBusCount > ST.getConstantBusLimit(Opcode) &&
5838 Opcode != AMDGPU::V_WRITELANE_B32) {
5839 ErrInfo = "VOP* instruction violates constant bus restriction";
5840 return false;
5841 }
5842
5843 if (isVOP3(MI) && UsesLiteral && !ST.hasVOP3Literal()) {
5844 ErrInfo = "VOP3 instruction uses literal";
5845 return false;
5846 }
5847 }
5848
5849 // Special case for writelane - this can break the multiple constant bus rule,
5850 // but still can't use more than one SGPR register
5851 if (Desc.getOpcode() == AMDGPU::V_WRITELANE_B32) {
5852 unsigned SGPRCount = 0;
5853 Register SGPRUsed;
5854
5855 for (int OpIdx : {Src0Idx, Src1Idx}) {
5856 if (OpIdx == -1)
5857 break;
5858
5859 const MachineOperand &MO = MI.getOperand(OpIdx);
5860
5861 if (usesConstantBus(MRI, MO, MI.getDesc().operands()[OpIdx])) {
5862 if (MO.isReg() && MO.getReg() != AMDGPU::M0) {
5863 if (MO.getReg() != SGPRUsed)
5864 ++SGPRCount;
5865 SGPRUsed = MO.getReg();
5866 }
5867 }
5868 if (SGPRCount > ST.getConstantBusLimit(Opcode)) {
5869 ErrInfo = "WRITELANE instruction violates constant bus restriction";
5870 return false;
5871 }
5872 }
5873 }
5874
5875 // Verify misc. restrictions on specific instructions.
5876 if (Desc.getOpcode() == AMDGPU::V_DIV_SCALE_F32_e64 ||
5877 Desc.getOpcode() == AMDGPU::V_DIV_SCALE_F64_e64) {
5878 const MachineOperand &Src0 = MI.getOperand(Src0Idx);
5879 const MachineOperand &Src1 = MI.getOperand(Src1Idx);
5880 const MachineOperand &Src2 = MI.getOperand(Src2Idx);
5881 if (Src0.isReg() && Src1.isReg() && Src2.isReg()) {
5882 if (!compareMachineOp(Src0, Src1) &&
5883 !compareMachineOp(Src0, Src2)) {
5884 ErrInfo = "v_div_scale_{f32|f64} require src0 = src1 or src2";
5885 return false;
5886 }
5887 }
5888 if ((getNamedOperand(MI, AMDGPU::OpName::src0_modifiers)->getImm() &
5889 SISrcMods::ABS) ||
5890 (getNamedOperand(MI, AMDGPU::OpName::src1_modifiers)->getImm() &
5891 SISrcMods::ABS) ||
5892 (getNamedOperand(MI, AMDGPU::OpName::src2_modifiers)->getImm() &
5893 SISrcMods::ABS)) {
5894 ErrInfo = "ABS not allowed in VOP3B instructions";
5895 return false;
5896 }
5897 }
5898
5899 if (isSOP2(MI) || isSOPC(MI)) {
5900 const MachineOperand &Src0 = MI.getOperand(Src0Idx);
5901 const MachineOperand &Src1 = MI.getOperand(Src1Idx);
5902
5903 if (!isRegOrFI(Src0) && !isRegOrFI(Src1) &&
5904 !isInlineConstant(Src0, Desc.operands()[Src0Idx]) &&
5905 !isInlineConstant(Src1, Desc.operands()[Src1Idx]) &&
5906 !Src0.isIdenticalTo(Src1)) {
5907 ErrInfo = "SOP2/SOPC instruction requires too many immediate constants";
5908 return false;
5909 }
5910 }
5911
5912 if (isSOPK(MI)) {
5913 const auto *Op = getNamedOperand(MI, AMDGPU::OpName::simm16);
5914 if (Desc.isBranch()) {
5915 if (!Op->isMBB()) {
5916 ErrInfo = "invalid branch target for SOPK instruction";
5917 return false;
5918 }
5919 } else {
5920 uint64_t Imm = Op->getImm();
5921 if (sopkIsZext(Opcode)) {
5922 if (!isUInt<16>(Imm)) {
5923 ErrInfo = "invalid immediate for SOPK instruction";
5924 return false;
5925 }
5926 } else {
5927 if (!isInt<16>(Imm)) {
5928 ErrInfo = "invalid immediate for SOPK instruction";
5929 return false;
5930 }
5931 }
5932 }
5933 }
5934
5935 if (Desc.getOpcode() == AMDGPU::V_MOVRELS_B32_e32 ||
5936 Desc.getOpcode() == AMDGPU::V_MOVRELS_B32_e64 ||
5937 Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e32 ||
5938 Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e64) {
5939 const bool IsDst = Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e32 ||
5940 Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e64;
5941
5942 const unsigned StaticNumOps =
5943 Desc.getNumOperands() + Desc.implicit_uses().size();
5944 const unsigned NumImplicitOps = IsDst ? 2 : 1;
5945
5946 // Require additional implicit operands. This allows a fixup done by the
5947 // post RA scheduler where the main implicit operand is killed and
5948 // implicit-defs are added for sub-registers that remain live after this
5949 // instruction.
5950 if (MI.getNumOperands() < StaticNumOps + NumImplicitOps) {
5951 ErrInfo = "missing implicit register operands";
5952 return false;
5953 }
5954
5955 const MachineOperand *Dst = getNamedOperand(MI, AMDGPU::OpName::vdst);
5956 if (IsDst) {
5957 if (!Dst->isUse()) {
5958 ErrInfo = "v_movreld_b32 vdst should be a use operand";
5959 return false;
5960 }
5961
5962 unsigned UseOpIdx;
5963 if (!MI.isRegTiedToUseOperand(StaticNumOps, &UseOpIdx) ||
5964 UseOpIdx != StaticNumOps + 1) {
5965 ErrInfo = "movrel implicit operands should be tied";
5966 return false;
5967 }
5968 }
5969
5970 const MachineOperand &Src0 = MI.getOperand(Src0Idx);
5971 const MachineOperand &ImpUse
5972 = MI.getOperand(StaticNumOps + NumImplicitOps - 1);
5973 if (!ImpUse.isReg() || !ImpUse.isUse() ||
5974 !isSubRegOf(RI, ImpUse, IsDst ? *Dst : Src0)) {
5975 ErrInfo = "src0 should be subreg of implicit vector use";
5976 return false;
5977 }
5978 }
5979
5980 // Make sure we aren't losing exec uses in the td files. This mostly requires
5981 // being careful when using let Uses to try to add other use registers.
5982 if (shouldReadExec(MI)) {
5983 if (!MI.hasRegisterImplicitUseOperand(AMDGPU::EXEC)) {
5984 ErrInfo = "VALU instruction does not implicitly read exec mask";
5985 return false;
5986 }
5987 }
5988
5989 if (isSMRD(MI)) {
5990 if (MI.mayStore() &&
5991 ST.getGeneration() == AMDGPUSubtarget::VOLCANIC_ISLANDS) {
5992 // The register offset form of scalar stores may only use m0 as the
5993 // soffset register.
5994 const MachineOperand *Soff = getNamedOperand(MI, AMDGPU::OpName::soffset);
5995 if (Soff && Soff->getReg() != AMDGPU::M0) {
5996 ErrInfo = "scalar stores must use m0 as offset register";
5997 return false;
5998 }
5999 }
6000 }
6001
6002 if (isFLAT(MI) && !ST.hasFlatInstOffsets()) {
6003 const MachineOperand *Offset = getNamedOperand(MI, AMDGPU::OpName::offset);
6004 if (Offset->getImm() != 0) {
6005 ErrInfo = "subtarget does not support offsets in flat instructions";
6006 return false;
6007 }
6008 }
6009
6010 if (isDS(MI) && !ST.hasGDS()) {
6011 const MachineOperand *GDSOp = getNamedOperand(MI, AMDGPU::OpName::gds);
6012 if (GDSOp && GDSOp->getImm() != 0) {
6013 ErrInfo = "GDS is not supported on this subtarget";
6014 return false;
6015 }
6016 }
6017
6018 if (isImage(MI)) {
6019 const MachineOperand *DimOp = getNamedOperand(MI, AMDGPU::OpName::dim);
6020 if (DimOp) {
6021 int VAddr0Idx = AMDGPU::getNamedOperandIdx(Opcode,
6022 AMDGPU::OpName::vaddr0);
6023 AMDGPU::OpName RSrcOpName =
6024 isMIMG(MI) ? AMDGPU::OpName::srsrc : AMDGPU::OpName::rsrc;
6025 int RsrcIdx = AMDGPU::getNamedOperandIdx(Opcode, RSrcOpName);
6026 const AMDGPU::MIMGInfo *Info = AMDGPU::getMIMGInfo(Opcode);
6027 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode =
6028 AMDGPU::getMIMGBaseOpcodeInfo(Info->BaseOpcode);
6029 const AMDGPU::MIMGDimInfo *Dim =
6031
6032 if (!Dim) {
6033 ErrInfo = "dim is out of range";
6034 return false;
6035 }
6036
6037 bool IsA16 = false;
6038 if (ST.hasR128A16()) {
6039 const MachineOperand *R128A16 = getNamedOperand(MI, AMDGPU::OpName::r128);
6040 IsA16 = R128A16->getImm() != 0;
6041 } else if (ST.hasA16()) {
6042 const MachineOperand *A16 = getNamedOperand(MI, AMDGPU::OpName::a16);
6043 IsA16 = A16->getImm() != 0;
6044 }
6045
6046 bool IsNSA = RsrcIdx - VAddr0Idx > 1;
6047
6048 unsigned AddrWords =
6049 AMDGPU::getAddrSizeMIMGOp(BaseOpcode, Dim, IsA16, ST.hasG16());
6050
6051 unsigned VAddrWords;
6052 if (IsNSA) {
6053 VAddrWords = RsrcIdx - VAddr0Idx;
6054 if (ST.hasPartialNSAEncoding() &&
6055 AddrWords > ST.getNSAMaxSize(isVSAMPLE(MI))) {
6056 unsigned LastVAddrIdx = RsrcIdx - 1;
6057 VAddrWords += getOpSize(MI, LastVAddrIdx) / 4 - 1;
6058 }
6059 } else {
6060 VAddrWords = getOpSize(MI, VAddr0Idx) / 4;
6061 if (AddrWords > 12)
6062 AddrWords = 16;
6063 }
6064
6065 if (VAddrWords != AddrWords) {
6066 LLVM_DEBUG(dbgs() << "bad vaddr size, expected " << AddrWords
6067 << " but got " << VAddrWords << "\n");
6068 ErrInfo = "bad vaddr size";
6069 return false;
6070 }
6071 }
6072 }
6073
6074 const MachineOperand *DppCt = getNamedOperand(MI, AMDGPU::OpName::dpp_ctrl);
6075 if (DppCt) {
6076 using namespace AMDGPU::DPP;
6077
6078 unsigned DC = DppCt->getImm();
6079 if (DC == DppCtrl::DPP_UNUSED1 || DC == DppCtrl::DPP_UNUSED2 ||
6080 DC == DppCtrl::DPP_UNUSED3 || DC > DppCtrl::DPP_LAST ||
6081 (DC >= DppCtrl::DPP_UNUSED4_FIRST && DC <= DppCtrl::DPP_UNUSED4_LAST) ||
6082 (DC >= DppCtrl::DPP_UNUSED5_FIRST && DC <= DppCtrl::DPP_UNUSED5_LAST) ||
6083 (DC >= DppCtrl::DPP_UNUSED6_FIRST && DC <= DppCtrl::DPP_UNUSED6_LAST) ||
6084 (DC >= DppCtrl::DPP_UNUSED7_FIRST && DC <= DppCtrl::DPP_UNUSED7_LAST) ||
6085 (DC >= DppCtrl::DPP_UNUSED8_FIRST && DC <= DppCtrl::DPP_UNUSED8_LAST)) {
6086 ErrInfo = "Invalid dpp_ctrl value";
6087 return false;
6088 }
6089 if (DC >= DppCtrl::WAVE_SHL1 && DC <= DppCtrl::WAVE_ROR1 &&
6090 !ST.hasDPPWavefrontShifts()) {
6091 ErrInfo = "Invalid dpp_ctrl value: "
6092 "wavefront shifts are not supported on GFX10+";
6093 return false;
6094 }
6095 if (DC >= DppCtrl::BCAST15 && DC <= DppCtrl::BCAST31 &&
6096 !ST.hasDPPBroadcasts()) {
6097 ErrInfo = "Invalid dpp_ctrl value: "
6098 "broadcasts are not supported on GFX10+";
6099 return false;
6100 }
6101 if (DC >= DppCtrl::ROW_SHARE_FIRST && DC <= DppCtrl::ROW_XMASK_LAST &&
6102 ST.getGeneration() < AMDGPUSubtarget::GFX10) {
6103 if (DC >= DppCtrl::ROW_NEWBCAST_FIRST &&
6104 DC <= DppCtrl::ROW_NEWBCAST_LAST &&
6105 !ST.hasGFX90AInsts()) {
6106 ErrInfo = "Invalid dpp_ctrl value: "
6107 "row_newbroadcast/row_share is not supported before "
6108 "GFX90A/GFX10";
6109 return false;
6110 }
6111 if (DC > DppCtrl::ROW_NEWBCAST_LAST || !ST.hasGFX90AInsts()) {
6112 ErrInfo = "Invalid dpp_ctrl value: "
6113 "row_share and row_xmask are not supported before GFX10";
6114 return false;
6115 }
6116 }
6117
6118 if (Opcode != AMDGPU::V_MOV_B64_DPP_PSEUDO &&
6120 ST.hasFeature(AMDGPU::FeatureDPALU_DPP) &&
6121 AMDGPU::isDPALU_DPP(Desc, *this, ST)) {
6122 ErrInfo = "Invalid dpp_ctrl value: "
6123 "DP ALU dpp only support row_newbcast";
6124 return false;
6125 }
6126 }
6127
6128 if ((MI.mayStore() || MI.mayLoad()) && !isVGPRSpill(MI)) {
6129 const MachineOperand *Dst = getNamedOperand(MI, AMDGPU::OpName::vdst);
6130 AMDGPU::OpName DataName =
6131 isDS(Opcode) ? AMDGPU::OpName::data0 : AMDGPU::OpName::vdata;
6132 const MachineOperand *Data = getNamedOperand(MI, DataName);
6133 const MachineOperand *Data2 = getNamedOperand(MI, AMDGPU::OpName::data1);
6134 if (Data && !Data->isReg())
6135 Data = nullptr;
6136
6137 if (!ST.hasGFX90AInsts()) {
6138 if ((Dst && RI.isAGPR(MRI, Dst->getReg())) ||
6139 (Data && RI.isAGPR(MRI, Data->getReg())) ||
6140 (Data2 && RI.isAGPR(MRI, Data2->getReg()))) {
6141 ErrInfo = "Invalid register class: "
6142 "agpr loads and stores not supported on this GPU";
6143 return false;
6144 }
6145 }
6146 }
6147
6148 if (ST.needsAlignedVGPRs()) {
6149 const auto isAlignedReg = [&MI, &MRI, this](AMDGPU::OpName OpName) -> bool {
6151 if (!Op)
6152 return true;
6153 Register Reg = Op->getReg();
6154 if (Reg.isPhysical())
6155 return !(RI.getHWRegIndex(Reg) & 1);
6156 const TargetRegisterClass &RC = *MRI.getRegClass(Reg);
6157 return RI.getRegSizeInBits(RC) > 32 && RI.isProperlyAlignedRC(RC) &&
6158 !(RI.getChannelFromSubReg(Op->getSubReg()) & 1);
6159 };
6160
6161 if (isMIMG(MI)) {
6162 if (!isAlignedReg(AMDGPU::OpName::vaddr)) {
6163 ErrInfo = "Subtarget requires even aligned vector registers "
6164 "for vaddr operand of image instructions";
6165 return false;
6166 }
6167 }
6168 }
6169
6170 if (Opcode == AMDGPU::V_ACCVGPR_WRITE_B32_e64 && !ST.hasGFX90AInsts()) {
6171 const MachineOperand *Src = getNamedOperand(MI, AMDGPU::OpName::src0);
6172 if (Src->isReg() && RI.isSGPRReg(MRI, Src->getReg())) {
6173 ErrInfo = "Invalid register class: "
6174 "v_accvgpr_write with an SGPR is not supported on this GPU";
6175 return false;
6176 }
6177 }
6178
6179 if (Desc.getOpcode() == AMDGPU::G_AMDGPU_WAVE_ADDRESS) {
6180 const MachineOperand &SrcOp = MI.getOperand(1);
6181 if (!SrcOp.isReg() || SrcOp.getReg().isVirtual()) {
6182 ErrInfo = "pseudo expects only physical SGPRs";
6183 return false;
6184 }
6185 }
6186
6187 if (const MachineOperand *CPol = getNamedOperand(MI, AMDGPU::OpName::cpol)) {
6188 if (CPol->getImm() & AMDGPU::CPol::SCAL) {
6189 if (!ST.hasScaleOffset()) {
6190 ErrInfo = "Subtarget does not support offset scaling";
6191 return false;
6192 }
6193 if (!AMDGPU::supportsScaleOffset(*this, MI.getOpcode())) {
6194 ErrInfo = "Instruction does not support offset scaling";
6195 return false;
6196 }
6197 }
6198 }
6199
6200 // See SIInstrInfo::isLegalSingleSGPRReadInstOperand for more information.
6201 if (AMDGPU::isSingleSGPRReadInst(Opcode)) {
6202 for (unsigned I = 0; I < 3; ++I) {
6204 return false;
6205 }
6206 }
6207
6208 if (ST.hasFlatScratchHiInB64InstHazard() && isSALU(MI) &&
6209 MI.readsRegister(AMDGPU::SRC_FLAT_SCRATCH_BASE_HI, nullptr)) {
6210 const MachineOperand *Dst = getNamedOperand(MI, AMDGPU::OpName::sdst);
6211 if ((Dst && RI.getRegClassForReg(MRI, Dst->getReg()) ==
6212 &AMDGPU::SReg_64RegClass) ||
6213 Opcode == AMDGPU::S_BITCMP0_B64 || Opcode == AMDGPU::S_BITCMP1_B64) {
6214 ErrInfo = "Instruction cannot read flat_scratch_base_hi";
6215 return false;
6216 }
6217 }
6218
6219 return true;
6220}
6221
6223 if (MI.getOpcode() == AMDGPU::S_MOV_B32) {
6224 const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
6225 return MI.getOperand(1).isReg() || RI.isAGPR(MRI, MI.getOperand(0).getReg())
6226 ? AMDGPU::COPY
6227 : AMDGPU::V_MOV_B32_e32;
6228 }
6229 return getVALUOp(MI.getOpcode());
6230}
6231
6232// It is more readable to list mapped opcodes on the same line.
6233// clang-format off
6234
6235unsigned SIInstrInfo::getVALUOp(unsigned Opc) const {
6236 switch (Opc) {
6237 default: return AMDGPU::INSTRUCTION_LIST_END;
6238 case AMDGPU::REG_SEQUENCE: return AMDGPU::REG_SEQUENCE;
6239 case AMDGPU::COPY: return AMDGPU::COPY;
6240 case AMDGPU::PHI: return AMDGPU::PHI;
6241 case AMDGPU::INSERT_SUBREG: return AMDGPU::INSERT_SUBREG;
6242 case AMDGPU::WQM: return AMDGPU::WQM;
6243 case AMDGPU::SOFT_WQM: return AMDGPU::SOFT_WQM;
6244 case AMDGPU::STRICT_WWM: return AMDGPU::STRICT_WWM;
6245 case AMDGPU::STRICT_WQM: return AMDGPU::STRICT_WQM;
6246 case AMDGPU::S_ADD_I32:
6247 return ST.hasAddNoCarryInsts() ? AMDGPU::V_ADD_U32_e64 : AMDGPU::V_ADD_CO_U32_e32;
6248 case AMDGPU::S_ADDC_U32:
6249 return AMDGPU::V_ADDC_U32_e32;
6250 case AMDGPU::S_SUB_I32:
6251 return ST.hasAddNoCarryInsts() ? AMDGPU::V_SUB_U32_e64 : AMDGPU::V_SUB_CO_U32_e32;
6252 // FIXME: These are not consistently handled, and selected when the carry is
6253 // used.
6254 case AMDGPU::S_ADD_U32:
6255 return AMDGPU::V_ADD_CO_U32_e32;
6256 case AMDGPU::S_SUB_U32:
6257 return AMDGPU::V_SUB_CO_U32_e32;
6258 case AMDGPU::S_ADD_U64_PSEUDO:
6259 return AMDGPU::V_ADD_U64_PSEUDO;
6260 case AMDGPU::S_SUB_U64_PSEUDO:
6261 return AMDGPU::V_SUB_U64_PSEUDO;
6262 case AMDGPU::S_SUBB_U32: return AMDGPU::V_SUBB_U32_e32;
6263 case AMDGPU::S_MUL_I32: return AMDGPU::V_MUL_LO_U32_e64;
6264 case AMDGPU::S_MUL_HI_U32: return AMDGPU::V_MUL_HI_U32_e64;
6265 case AMDGPU::S_MUL_HI_I32: return AMDGPU::V_MUL_HI_I32_e64;
6266 case AMDGPU::S_AND_B32: return AMDGPU::V_AND_B32_e64;
6267 case AMDGPU::S_OR_B32: return AMDGPU::V_OR_B32_e64;
6268 case AMDGPU::S_XOR_B32: return AMDGPU::V_XOR_B32_e64;
6269 case AMDGPU::S_XNOR_B32:
6270 return ST.hasDLInsts() ? AMDGPU::V_XNOR_B32_e64 : AMDGPU::INSTRUCTION_LIST_END;
6271 case AMDGPU::S_MIN_I32: return AMDGPU::V_MIN_I32_e64;
6272 case AMDGPU::S_MIN_U32: return AMDGPU::V_MIN_U32_e64;
6273 case AMDGPU::S_MAX_I32: return AMDGPU::V_MAX_I32_e64;
6274 case AMDGPU::S_MAX_U32: return AMDGPU::V_MAX_U32_e64;
6275 case AMDGPU::S_ASHR_I32: return AMDGPU::V_ASHR_I32_e32;
6276 case AMDGPU::S_ASHR_I64: return AMDGPU::V_ASHR_I64_e64;
6277 case AMDGPU::S_LSHL_B32: return AMDGPU::V_LSHL_B32_e32;
6278 case AMDGPU::S_LSHL_B64: return AMDGPU::V_LSHL_B64_e64;
6279 case AMDGPU::S_LSHR_B32: return AMDGPU::V_LSHR_B32_e32;
6280 case AMDGPU::S_LSHR_B64: return AMDGPU::V_LSHR_B64_e64;
6281 case AMDGPU::S_SEXT_I32_I8: return AMDGPU::V_BFE_I32_e64;
6282 case AMDGPU::S_SEXT_I32_I16: return AMDGPU::V_BFE_I32_e64;
6283 case AMDGPU::S_BFE_U32: return AMDGPU::V_BFE_U32_e64;
6284 case AMDGPU::S_BFE_I32: return AMDGPU::V_BFE_I32_e64;
6285 case AMDGPU::S_BFM_B32: return AMDGPU::V_BFM_B32_e64;
6286 case AMDGPU::S_BREV_B32: return AMDGPU::V_BFREV_B32_e32;
6287 case AMDGPU::S_NOT_B32: return AMDGPU::V_NOT_B32_e32;
6288 case AMDGPU::S_NOT_B64: return AMDGPU::V_NOT_B32_e32;
6289 case AMDGPU::S_CMP_EQ_I32: return AMDGPU::V_CMP_EQ_I32_e64;
6290 case AMDGPU::S_CMP_LG_I32: return AMDGPU::V_CMP_NE_I32_e64;
6291 case AMDGPU::S_CMP_GT_I32: return AMDGPU::V_CMP_GT_I32_e64;
6292 case AMDGPU::S_CMP_GE_I32: return AMDGPU::V_CMP_GE_I32_e64;
6293 case AMDGPU::S_CMP_LT_I32: return AMDGPU::V_CMP_LT_I32_e64;
6294 case AMDGPU::S_CMP_LE_I32: return AMDGPU::V_CMP_LE_I32_e64;
6295 case AMDGPU::S_CMP_EQ_U32: return AMDGPU::V_CMP_EQ_U32_e64;
6296 case AMDGPU::S_CMP_LG_U32: return AMDGPU::V_CMP_NE_U32_e64;
6297 case AMDGPU::S_CMP_GT_U32: return AMDGPU::V_CMP_GT_U32_e64;
6298 case AMDGPU::S_CMP_GE_U32: return AMDGPU::V_CMP_GE_U32_e64;
6299 case AMDGPU::S_CMP_LT_U32: return AMDGPU::V_CMP_LT_U32_e64;
6300 case AMDGPU::S_CMP_LE_U32: return AMDGPU::V_CMP_LE_U32_e64;
6301 case AMDGPU::S_CMP_EQ_U64: return AMDGPU::V_CMP_EQ_U64_e64;
6302 case AMDGPU::S_CMP_LG_U64: return AMDGPU::V_CMP_NE_U64_e64;
6303 case AMDGPU::S_BCNT1_I32_B32: return AMDGPU::V_BCNT_U32_B32_e64;
6304 case AMDGPU::S_FF1_I32_B32: return AMDGPU::V_FFBL_B32_e32;
6305 case AMDGPU::S_FLBIT_I32_B32: return AMDGPU::V_FFBH_U32_e32;
6306 case AMDGPU::S_FLBIT_I32: return AMDGPU::V_FFBH_I32_e64;
6307 case AMDGPU::S_CBRANCH_SCC0: return AMDGPU::S_CBRANCH_VCCZ;
6308 case AMDGPU::S_CBRANCH_SCC1: return AMDGPU::S_CBRANCH_VCCNZ;
6309 case AMDGPU::S_CVT_F32_I32: return AMDGPU::V_CVT_F32_I32_e64;
6310 case AMDGPU::S_CVT_F32_U32: return AMDGPU::V_CVT_F32_U32_e64;
6311 case AMDGPU::S_CVT_I32_F32: return AMDGPU::V_CVT_I32_F32_e64;
6312 case AMDGPU::S_CVT_U32_F32: return AMDGPU::V_CVT_U32_F32_e64;
6313 case AMDGPU::S_CVT_F32_F16:
6314 case AMDGPU::S_CVT_HI_F32_F16:
6315 return ST.useRealTrue16Insts() ? AMDGPU::V_CVT_F32_F16_t16_e64
6316 : AMDGPU::V_CVT_F32_F16_fake16_e64;
6317 case AMDGPU::S_CVT_F16_F32:
6318 return ST.useRealTrue16Insts() ? AMDGPU::V_CVT_F16_F32_t16_e64
6319 : AMDGPU::V_CVT_F16_F32_fake16_e64;
6320 case AMDGPU::S_CEIL_F32: return AMDGPU::V_CEIL_F32_e64;
6321 case AMDGPU::S_FLOOR_F32: return AMDGPU::V_FLOOR_F32_e64;
6322 case AMDGPU::S_TRUNC_F32: return AMDGPU::V_TRUNC_F32_e64;
6323 case AMDGPU::S_RNDNE_F32: return AMDGPU::V_RNDNE_F32_e64;
6324 case AMDGPU::S_CEIL_F16:
6325 return ST.useRealTrue16Insts() ? AMDGPU::V_CEIL_F16_t16_e64
6326 : AMDGPU::V_CEIL_F16_fake16_e64;
6327 case AMDGPU::S_FLOOR_F16:
6328 return ST.useRealTrue16Insts() ? AMDGPU::V_FLOOR_F16_t16_e64
6329 : AMDGPU::V_FLOOR_F16_fake16_e64;
6330 case AMDGPU::S_TRUNC_F16:
6331 return ST.useRealTrue16Insts() ? AMDGPU::V_TRUNC_F16_t16_e64
6332 : AMDGPU::V_TRUNC_F16_fake16_e64;
6333 case AMDGPU::S_RNDNE_F16:
6334 return ST.useRealTrue16Insts() ? AMDGPU::V_RNDNE_F16_t16_e64
6335 : AMDGPU::V_RNDNE_F16_fake16_e64;
6336 case AMDGPU::S_ADD_F32: return AMDGPU::V_ADD_F32_e64;
6337 case AMDGPU::S_SUB_F32: return AMDGPU::V_SUB_F32_e64;
6338 case AMDGPU::S_MIN_F32: return AMDGPU::V_MIN_F32_e64;
6339 case AMDGPU::S_MAX_F32: return AMDGPU::V_MAX_F32_e64;
6340 case AMDGPU::S_MINIMUM_F32: return AMDGPU::V_MINIMUM_F32_e64;
6341 case AMDGPU::S_MAXIMUM_F32: return AMDGPU::V_MAXIMUM_F32_e64;
6342 case AMDGPU::S_MUL_F32: return AMDGPU::V_MUL_F32_e64;
6343 case AMDGPU::S_ADD_F16:
6344 return ST.useRealTrue16Insts() ? AMDGPU::V_ADD_F16_t16_e64
6345 : AMDGPU::V_ADD_F16_fake16_e64;
6346 case AMDGPU::S_SUB_F16:
6347 return ST.useRealTrue16Insts() ? AMDGPU::V_SUB_F16_t16_e64
6348 : AMDGPU::V_SUB_F16_fake16_e64;
6349 case AMDGPU::S_MIN_F16:
6350 return ST.useRealTrue16Insts() ? AMDGPU::V_MIN_F16_t16_e64
6351 : AMDGPU::V_MIN_F16_fake16_e64;
6352 case AMDGPU::S_MAX_F16:
6353 return ST.useRealTrue16Insts() ? AMDGPU::V_MAX_F16_t16_e64
6354 : AMDGPU::V_MAX_F16_fake16_e64;
6355 case AMDGPU::S_MINIMUM_F16:
6356 return ST.useRealTrue16Insts() ? AMDGPU::V_MINIMUM_F16_t16_e64
6357 : AMDGPU::V_MINIMUM_F16_fake16_e64;
6358 case AMDGPU::S_MAXIMUM_F16:
6359 return ST.useRealTrue16Insts() ? AMDGPU::V_MAXIMUM_F16_t16_e64
6360 : AMDGPU::V_MAXIMUM_F16_fake16_e64;
6361 case AMDGPU::S_MUL_F16:
6362 return ST.useRealTrue16Insts() ? AMDGPU::V_MUL_F16_t16_e64
6363 : AMDGPU::V_MUL_F16_fake16_e64;
6364 case AMDGPU::S_CVT_PK_RTZ_F16_F32: return AMDGPU::V_CVT_PKRTZ_F16_F32_e64;
6365 case AMDGPU::S_FMAC_F32: return AMDGPU::V_FMAC_F32_e64;
6366 case AMDGPU::S_FMAC_F16:
6367 return ST.useRealTrue16Insts() ? AMDGPU::V_FMAC_F16_t16_e64
6368 : AMDGPU::V_FMAC_F16_fake16_e64;
6369 case AMDGPU::S_FMAMK_F32: return AMDGPU::V_FMAMK_F32;
6370 case AMDGPU::S_FMAAK_F32: return AMDGPU::V_FMAAK_F32;
6371 case AMDGPU::S_CMP_LT_F32: return AMDGPU::V_CMP_LT_F32_e64;
6372 case AMDGPU::S_CMP_EQ_F32: return AMDGPU::V_CMP_EQ_F32_e64;
6373 case AMDGPU::S_CMP_LE_F32: return AMDGPU::V_CMP_LE_F32_e64;
6374 case AMDGPU::S_CMP_GT_F32: return AMDGPU::V_CMP_GT_F32_e64;
6375 case AMDGPU::S_CMP_LG_F32: return AMDGPU::V_CMP_LG_F32_e64;
6376 case AMDGPU::S_CMP_GE_F32: return AMDGPU::V_CMP_GE_F32_e64;
6377 case AMDGPU::S_CMP_O_F32: return AMDGPU::V_CMP_O_F32_e64;
6378 case AMDGPU::S_CMP_U_F32: return AMDGPU::V_CMP_U_F32_e64;
6379 case AMDGPU::S_CMP_NGE_F32: return AMDGPU::V_CMP_NGE_F32_e64;
6380 case AMDGPU::S_CMP_NLG_F32: return AMDGPU::V_CMP_NLG_F32_e64;
6381 case AMDGPU::S_CMP_NGT_F32: return AMDGPU::V_CMP_NGT_F32_e64;
6382 case AMDGPU::S_CMP_NLE_F32: return AMDGPU::V_CMP_NLE_F32_e64;
6383 case AMDGPU::S_CMP_NEQ_F32: return AMDGPU::V_CMP_NEQ_F32_e64;
6384 case AMDGPU::S_CMP_NLT_F32: return AMDGPU::V_CMP_NLT_F32_e64;
6385 case AMDGPU::S_CMP_LT_F16:
6386 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_LT_F16_t16_e64
6387 : AMDGPU::V_CMP_LT_F16_fake16_e64;
6388 case AMDGPU::S_CMP_EQ_F16:
6389 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_EQ_F16_t16_e64
6390 : AMDGPU::V_CMP_EQ_F16_fake16_e64;
6391 case AMDGPU::S_CMP_LE_F16:
6392 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_LE_F16_t16_e64
6393 : AMDGPU::V_CMP_LE_F16_fake16_e64;
6394 case AMDGPU::S_CMP_GT_F16:
6395 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_GT_F16_t16_e64
6396 : AMDGPU::V_CMP_GT_F16_fake16_e64;
6397 case AMDGPU::S_CMP_LG_F16:
6398 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_LG_F16_t16_e64
6399 : AMDGPU::V_CMP_LG_F16_fake16_e64;
6400 case AMDGPU::S_CMP_GE_F16:
6401 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_GE_F16_t16_e64
6402 : AMDGPU::V_CMP_GE_F16_fake16_e64;
6403 case AMDGPU::S_CMP_O_F16:
6404 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_O_F16_t16_e64
6405 : AMDGPU::V_CMP_O_F16_fake16_e64;
6406 case AMDGPU::S_CMP_U_F16:
6407 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_U_F16_t16_e64
6408 : AMDGPU::V_CMP_U_F16_fake16_e64;
6409 case AMDGPU::S_CMP_NGE_F16:
6410 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NGE_F16_t16_e64
6411 : AMDGPU::V_CMP_NGE_F16_fake16_e64;
6412 case AMDGPU::S_CMP_NLG_F16:
6413 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NLG_F16_t16_e64
6414 : AMDGPU::V_CMP_NLG_F16_fake16_e64;
6415 case AMDGPU::S_CMP_NGT_F16:
6416 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NGT_F16_t16_e64
6417 : AMDGPU::V_CMP_NGT_F16_fake16_e64;
6418 case AMDGPU::S_CMP_NLE_F16:
6419 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NLE_F16_t16_e64
6420 : AMDGPU::V_CMP_NLE_F16_fake16_e64;
6421 case AMDGPU::S_CMP_NEQ_F16:
6422 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NEQ_F16_t16_e64
6423 : AMDGPU::V_CMP_NEQ_F16_fake16_e64;
6424 case AMDGPU::S_CMP_NLT_F16:
6425 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NLT_F16_t16_e64
6426 : AMDGPU::V_CMP_NLT_F16_fake16_e64;
6427 case AMDGPU::V_S_EXP_F32_e64: return AMDGPU::V_EXP_F32_e64;
6428 case AMDGPU::V_S_EXP_F16_e64:
6429 return ST.useRealTrue16Insts() ? AMDGPU::V_EXP_F16_t16_e64
6430 : AMDGPU::V_EXP_F16_fake16_e64;
6431 case AMDGPU::V_S_LOG_F32_e64: return AMDGPU::V_LOG_F32_e64;
6432 case AMDGPU::V_S_LOG_F16_e64:
6433 return ST.useRealTrue16Insts() ? AMDGPU::V_LOG_F16_t16_e64
6434 : AMDGPU::V_LOG_F16_fake16_e64;
6435 case AMDGPU::V_S_RCP_F32_e64: return AMDGPU::V_RCP_F32_e64;
6436 case AMDGPU::V_S_RCP_F16_e64:
6437 return ST.useRealTrue16Insts() ? AMDGPU::V_RCP_F16_t16_e64
6438 : AMDGPU::V_RCP_F16_fake16_e64;
6439 case AMDGPU::V_S_RSQ_F32_e64: return AMDGPU::V_RSQ_F32_e64;
6440 case AMDGPU::V_S_RSQ_F16_e64:
6441 return ST.useRealTrue16Insts() ? AMDGPU::V_RSQ_F16_t16_e64
6442 : AMDGPU::V_RSQ_F16_fake16_e64;
6443 case AMDGPU::V_S_SQRT_F32_e64: return AMDGPU::V_SQRT_F32_e64;
6444 case AMDGPU::V_S_SQRT_F16_e64:
6445 return ST.useRealTrue16Insts() ? AMDGPU::V_SQRT_F16_t16_e64
6446 : AMDGPU::V_SQRT_F16_fake16_e64;
6447 }
6449 "Unexpected scalar opcode without corresponding vector one!");
6450}
6451
6452// clang-format on
6453
6457 const DebugLoc &DL, Register Reg,
6458 bool IsSCCLive,
6459 SlotIndexes *Indexes) const {
6460 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
6461 const SIInstrInfo *TII = ST.getInstrInfo();
6463 if (IsSCCLive) {
6464 // Insert two move instructions, one to save the original value of EXEC and
6465 // the other to turn on all bits in EXEC. This is required as we can't use
6466 // the single instruction S_OR_SAVEEXEC that clobbers SCC.
6467 auto StoreExecMI = BuildMI(MBB, MBBI, DL, TII->get(LMC.MovOpc), Reg)
6469 auto FlipExecMI =
6470 BuildMI(MBB, MBBI, DL, TII->get(LMC.MovOpc), LMC.ExecReg).addImm(-1);
6471 if (Indexes) {
6472 Indexes->insertMachineInstrInMaps(*StoreExecMI);
6473 Indexes->insertMachineInstrInMaps(*FlipExecMI);
6474 }
6475 } else {
6476 auto SaveExec =
6477 BuildMI(MBB, MBBI, DL, TII->get(LMC.OrSaveExecOpc), Reg).addImm(-1);
6478 SaveExec->getOperand(3).setIsDead(); // Mark SCC as dead.
6479 if (Indexes)
6480 Indexes->insertMachineInstrInMaps(*SaveExec);
6481 }
6482}
6483
6486 const DebugLoc &DL, Register Reg,
6487 SlotIndexes *Indexes) const {
6489 auto ExecRestoreMI = BuildMI(MBB, MBBI, DL, get(LMC.MovOpc), LMC.ExecReg)
6490 .addReg(Reg, RegState::Kill);
6491 if (Indexes)
6492 Indexes->insertMachineInstrInMaps(*ExecRestoreMI);
6493}
6494
6498 "Not a whole wave func");
6499 MachineBasicBlock &MBB = *MF.begin();
6500 for (MachineInstr &MI : MBB)
6501 if (MI.getOpcode() == AMDGPU::SI_WHOLE_WAVE_FUNC_SETUP ||
6502 MI.getOpcode() == AMDGPU::G_AMDGPU_WHOLE_WAVE_FUNC_SETUP)
6503 return &MI;
6504
6505 llvm_unreachable("Couldn't find SI_SETUP_WHOLE_WAVE_FUNC instruction");
6506}
6507
6509 unsigned OpNo) const {
6510 const MCInstrDesc &Desc = get(MI.getOpcode());
6511 if (MI.isVariadic() || OpNo >= Desc.getNumOperands() ||
6512 Desc.operands()[OpNo].RegClass == -1) {
6513 Register Reg = MI.getOperand(OpNo).getReg();
6514
6515 if (Reg.isVirtual()) {
6516 const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
6517 return MRI.getRegClass(Reg);
6518 }
6519 return RI.getPhysRegBaseClass(Reg);
6520 }
6521
6522 int16_t RegClass = getOpRegClassID(Desc.operands()[OpNo]);
6523 return RegClass < 0 ? nullptr : RI.getRegClass(RegClass);
6524}
6525
6526// Convert VOP3 operand index to source number.
6527static unsigned VOP3OpIdxToSrcN(const MachineInstr &MI, unsigned OpIdx) {
6528 constexpr AMDGPU::OpName OpNames[] = {
6529 AMDGPU::OpName::src0, AMDGPU::OpName::src1, AMDGPU::OpName::src2};
6530
6531 for (auto [I, OpName] : enumerate(OpNames)) {
6532 int SrcIdx = AMDGPU::getNamedOperandIdx(MI.getOpcode(), OpNames[I]);
6533 if (static_cast<unsigned>(SrcIdx) == OpIdx)
6534 return I;
6535 }
6536
6537 return UINT_MAX;
6538}
6539
6542 MachineBasicBlock *MBB = MI.getParent();
6543 MachineOperand &MO = MI.getOperand(OpIdx);
6544 MachineRegisterInfo &MRI = MBB->getParent()->getRegInfo();
6545 unsigned RCID = getOpRegClassID(get(MI.getOpcode()).operands()[OpIdx]);
6546 const TargetRegisterClass *RC = RI.getRegClass(RCID);
6547 unsigned Size = RI.getRegSizeInBits(*RC);
6548 unsigned Opcode = (Size == 64) ? AMDGPU::V_MOV_B64_PSEUDO
6549 : Size == 16 ? AMDGPU::V_MOV_B16_t16_e64
6550 : AMDGPU::V_MOV_B32_e32;
6551 if (MO.isReg())
6552 Opcode = AMDGPU::COPY;
6553 else if (RI.isSGPRClass(RC))
6554 Opcode = (Size == 64) ? AMDGPU::S_MOV_B64 : AMDGPU::S_MOV_B32;
6555
6556 const TargetRegisterClass *VRC = RI.getEquivalentVGPRClass(RC);
6557 Register Reg = MRI.createVirtualRegister(VRC);
6558 DebugLoc DL = MBB->findDebugLoc(I);
6559
6560 if (Size == 128 && AMDGPU::isPackedSingleSGPR64BitInst(MI.getOpcode()) &&
6562 // Special case for V_PK_*64 instructions: these do not have OPSEL but SGPR
6563 // sources behave like OPSEL is set replicating low 64-bits into high. VGPR
6564 // sources in turn read actual 4 registers. To move operand from an SGPR to
6565 // a VGPR we need to replicate low half.
6566 // We also do not select immediates for these instructions so it always has
6567 // to be an SGPR register here.
6568 // Operands which are not legal as per isLegalSingleSGPRReadInstOperand()
6569 // sent here specifically to fix a non-splat SGPR and shall perform a full
6570 // copy.
6571
6572 const TargetRegisterClass *VRC64 = RI.getVGPRClassForBitWidth(64);
6573 Register Low64 = MRI.createVirtualRegister(VRC64);
6574 assert(MO.isReg() && RI.isSGPRReg(MRI, MO.getReg()));
6575 BuildMI(*MBB, I, DL, get(TargetOpcode::COPY), Low64)
6576 .addReg(MO.getReg(), {}, AMDGPU::sub0_sub1);
6577 BuildMI(*MBB, I, DL, get(TargetOpcode::REG_SEQUENCE), Reg)
6578 .addReg(Low64)
6579 .addImm(AMDGPU::sub0_sub1)
6580 .addReg(Low64, RegState::Kill)
6581 .addImm(AMDGPU::sub2_sub3);
6582 } else if (Opcode == AMDGPU::V_MOV_B16_t16_e64) {
6583 BuildMI(*MBB, I, DL, get(Opcode), Reg)
6584 .addImm(0) // src0_modifiers
6585 .add(MO)
6586 .addImm(0); // op_sel
6587 } else {
6588 BuildMI(*MBB, I, DL, get(Opcode), Reg).add(MO);
6589 }
6590
6591 MO.ChangeToRegister(Reg, false);
6592}
6593
6596 const MachineOperand &SuperReg, const TargetRegisterClass *SuperRC,
6597 unsigned SubIdx, const TargetRegisterClass *SubRC) const {
6598 if (!SuperReg.getReg().isVirtual())
6599 return RI.getSubReg(SuperReg.getReg(), SubIdx);
6600
6601 MachineBasicBlock *MBB = MI->getParent();
6602 const DebugLoc &DL = MI->getDebugLoc();
6603 Register SubReg = MRI.createVirtualRegister(SubRC);
6604
6605 unsigned NewSubIdx = RI.composeSubRegIndices(SuperReg.getSubReg(), SubIdx);
6606 BuildMI(*MBB, MI, DL, get(TargetOpcode::COPY), SubReg)
6607 .addReg(SuperReg.getReg(), {}, NewSubIdx);
6608 return SubReg;
6609}
6610
6613 const MachineOperand &Op, const TargetRegisterClass *SuperRC,
6614 unsigned SubIdx, const TargetRegisterClass *SubRC) const {
6615 if (Op.isImm()) {
6616 if (SubIdx == AMDGPU::sub0)
6617 return MachineOperand::CreateImm(static_cast<int32_t>(Op.getImm()));
6618 if (SubIdx == AMDGPU::sub1)
6619 return MachineOperand::CreateImm(static_cast<int32_t>(Op.getImm() >> 32));
6620
6621 llvm_unreachable("Unhandled register index for immediate");
6622 }
6623
6624 unsigned SubReg = buildExtractSubReg(MII, MRI, Op, SuperRC,
6625 SubIdx, SubRC);
6626 return MachineOperand::CreateReg(SubReg, false);
6627}
6628
6629// Change the order of operands from (0, 1, 2) to (0, 2, 1)
6630void SIInstrInfo::swapOperands(MachineInstr &Inst) const {
6631 assert(Inst.getNumExplicitOperands() == 3);
6632 MachineOperand Op1 = Inst.getOperand(1);
6633 Inst.removeOperand(1);
6634 Inst.addOperand(Op1);
6635}
6636
6638 const MCOperandInfo &OpInfo,
6639 const MachineOperand &MO) const {
6640 if (!MO.isReg())
6641 return false;
6642
6643 Register Reg = MO.getReg();
6644
6645 const TargetRegisterClass *DRC = RI.getRegClass(getOpRegClassID(OpInfo));
6646 if (Reg.isPhysical())
6647 return DRC->contains(Reg);
6648
6649 const TargetRegisterClass *RC = MRI.getRegClass(Reg);
6650
6651 if (MO.getSubReg()) {
6652 const TargetRegisterClass *SuperRC =
6653 RI.getLargestLegalSuperClass(RC, MRI.getMF());
6654 if (!SuperRC)
6655 return false;
6656 return RI.getMatchingSuperRegClass(SuperRC, DRC, MO.getSubReg()) != nullptr;
6657 }
6658
6659 return RI.getCommonSubClass(DRC, RC) != nullptr;
6660}
6661
6663 const MachineOperand &MO) const {
6664 const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
6665 const MCOperandInfo OpInfo = MI.getDesc().operands()[OpIdx];
6666 unsigned Opc = MI.getOpcode();
6667
6668 // See SIInstrInfo::isLegalSingleSGPRReadInstOperand for more information.
6669 if (MO.isReg() && RI.isSGPRReg(MRI, MO.getReg()) &&
6670 AMDGPU::isSingleSGPRReadInst(MI.getOpcode()) &&
6672 &MO))
6673 return false;
6674
6675 if (!isLegalRegOperand(MRI, OpInfo, MO))
6676 return false;
6677
6678 // check Accumulate GPR operand
6679 bool IsAGPR = RI.isAGPR(MRI, MO.getReg());
6680 if (IsAGPR && !ST.hasMAIInsts())
6681 return false;
6682 if (IsAGPR && (!ST.hasGFX90AInsts() || !MRI.reservedRegsFrozen()) &&
6683 (MI.mayLoad() || MI.mayStore() || isDS(Opc) || isMIMG(Opc)))
6684 return false;
6685 // Atomics should have both vdst and vdata either vgpr or agpr.
6686 const int VDstIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst);
6687 const int DataIdx = AMDGPU::getNamedOperandIdx(
6688 Opc, isDS(Opc) ? AMDGPU::OpName::data0 : AMDGPU::OpName::vdata);
6689 if ((int)OpIdx == VDstIdx && DataIdx != -1 &&
6690 MI.getOperand(DataIdx).isReg() &&
6691 RI.isAGPR(MRI, MI.getOperand(DataIdx).getReg()) != IsAGPR)
6692 return false;
6693 if ((int)OpIdx == DataIdx) {
6694 if (VDstIdx != -1 &&
6695 RI.isAGPR(MRI, MI.getOperand(VDstIdx).getReg()) != IsAGPR)
6696 return false;
6697 // DS instructions with 2 src operands also must have tied RC.
6698 const int Data1Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::data1);
6699 if (Data1Idx != -1 && MI.getOperand(Data1Idx).isReg() &&
6700 RI.isAGPR(MRI, MI.getOperand(Data1Idx).getReg()) != IsAGPR)
6701 return false;
6702 }
6703
6704 // Check V_ACCVGPR_WRITE_B32_e64
6705 if (Opc == AMDGPU::V_ACCVGPR_WRITE_B32_e64 && !ST.hasGFX90AInsts() &&
6706 (int)OpIdx == AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0) &&
6707 RI.isSGPRReg(MRI, MO.getReg()))
6708 return false;
6709
6710 if (ST.hasFlatScratchHiInB64InstHazard() &&
6711 MO.getReg() == AMDGPU::SRC_FLAT_SCRATCH_BASE_HI && isSALU(MI)) {
6712 if (const MachineOperand *Dst = getNamedOperand(MI, AMDGPU::OpName::sdst)) {
6713 if (AMDGPU::getRegBitWidth(*RI.getRegClassForReg(MRI, Dst->getReg())) ==
6714 64)
6715 return false;
6716 }
6717 if (Opc == AMDGPU::S_BITCMP0_B64 || Opc == AMDGPU::S_BITCMP1_B64)
6718 return false;
6719 }
6720 if (!ST.hasDPPSrc1SGPR() && isDPP(MI) && RI.isSGPRReg(MRI, MO.getReg()) &&
6721 (int)OpIdx == AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1))
6722 return false;
6723
6724 return true;
6725}
6726
6728 const MCOperandInfo &OpInfo,
6729 const MachineOperand &MO) const {
6730 if (MO.isReg())
6731 return isLegalRegOperand(MRI, OpInfo, MO);
6732
6733 // Handle non-register types that are treated like immediates.
6734 assert(MO.isImm() || MO.isTargetIndex() || MO.isFI() || MO.isGlobal());
6735 return true;
6736}
6737
6739 const MachineRegisterInfo &MRI, const MachineInstr &MI, unsigned SrcN,
6740 const MachineOperand *MO) const {
6741 constexpr unsigned NumOps = 3;
6742 constexpr AMDGPU::OpName OpNames[NumOps * 2] = {
6743 AMDGPU::OpName::src0, AMDGPU::OpName::src1,
6744 AMDGPU::OpName::src2, AMDGPU::OpName::src0_modifiers,
6745 AMDGPU::OpName::src1_modifiers, AMDGPU::OpName::src2_modifiers};
6746
6747 assert(SrcN < NumOps);
6748
6749 if (!MO) {
6750 int SrcIdx = AMDGPU::getNamedOperandIdx(MI.getOpcode(), OpNames[SrcN]);
6751 if (SrcIdx == -1)
6752 return true;
6753 MO = &MI.getOperand(SrcIdx);
6754 }
6755
6756 if (!MO->isReg() || !RI.isSGPRReg(MRI, MO->getReg()))
6757 return true;
6758
6759 int ModsIdx =
6760 AMDGPU::getNamedOperandIdx(MI.getOpcode(), OpNames[NumOps + SrcN]);
6761 if (ModsIdx == -1)
6762 return false;
6763
6764 unsigned Mods = MI.getOperand(ModsIdx).getImm();
6765 bool OpSel = Mods & SISrcMods::OP_SEL_0;
6766 bool OpSelHi = Mods & SISrcMods::OP_SEL_1;
6767
6768 return !OpSel && !OpSelHi;
6769}
6770
6771bool SIInstrInfo::isOperandLegal(const MachineInstr &MI, unsigned OpIdx,
6772 const MachineOperand *MO) const {
6773 const MachineFunction &MF = *MI.getMF();
6774 const MachineRegisterInfo &MRI = MF.getRegInfo();
6775 const MCInstrDesc &InstDesc = MI.getDesc();
6776 const MCOperandInfo &OpInfo = InstDesc.operands()[OpIdx];
6777 int64_t RegClass = getOpRegClassID(OpInfo);
6778 const TargetRegisterClass *DefinedRC =
6779 RegClass != -1 ? RI.getRegClass(RegClass) : nullptr;
6780 if (!MO)
6781 MO = &MI.getOperand(OpIdx);
6782
6783 const bool IsInlineConst = !MO->isReg() && isInlineConstant(*MO, OpInfo);
6784
6785 if (isVALU(MI, /*AllowLDSDMA=*/false) && !IsInlineConst &&
6786 usesConstantBus(MRI, *MO, OpInfo)) {
6787 const MachineOperand *UsedLiteral = nullptr;
6788
6789 int ConstantBusLimit = ST.getConstantBusLimit(MI.getOpcode());
6790 int LiteralLimit = !isVOP3(MI) || ST.hasVOP3Literal() ? 1 : 0;
6791
6792 // TODO: Be more permissive with frame indexes.
6793 if (!MO->isReg() && !isInlineConstant(*MO, OpInfo)) {
6794 if (!LiteralLimit--)
6795 return false;
6796
6797 UsedLiteral = MO;
6798 }
6799
6801 if (MO->isReg())
6802 SGPRsUsed.insert(RegSubRegPair(MO->getReg(), MO->getSubReg()));
6803
6804 for (unsigned i = 0, e = MI.getNumOperands(); i != e; ++i) {
6805 if (i == OpIdx)
6806 continue;
6807 const MachineOperand &Op = MI.getOperand(i);
6808 if (Op.isReg()) {
6809 if (Op.isUse()) {
6810 RegSubRegPair SGPR(Op.getReg(), Op.getSubReg());
6811 if (regUsesConstantBus(Op, MRI) && SGPRsUsed.insert(SGPR).second) {
6812 if (--ConstantBusLimit <= 0)
6813 return false;
6814 }
6815 }
6816 } else if (AMDGPU::isSISrcOperand(InstDesc.operands()[i]) &&
6817 !isInlineConstant(Op, InstDesc.operands()[i])) {
6818 // The same literal may be used multiple times.
6819 if (!UsedLiteral)
6820 UsedLiteral = &Op;
6821 else if (UsedLiteral->isIdenticalTo(Op))
6822 continue;
6823
6824 if (!LiteralLimit--)
6825 return false;
6826 if (--ConstantBusLimit <= 0)
6827 return false;
6828 }
6829 }
6830 } else if (!IsInlineConst && !MO->isReg() && isSALU(MI)) {
6831 // There can be at most one literal operand, but it can be repeated.
6832 for (unsigned i = 0, e = MI.getNumOperands(); i != e; ++i) {
6833 if (i == OpIdx)
6834 continue;
6835 const MachineOperand &Op = MI.getOperand(i);
6836 if (!Op.isReg() && !Op.isFI() && !Op.isRegMask() &&
6837 !isInlineConstant(Op, InstDesc.operands()[i]) &&
6838 !Op.isIdenticalTo(*MO))
6839 return false;
6840
6841 // Do not fold a non-inlineable and non-register operand into an
6842 // instruction that already has a frame index. The frame index handling
6843 // code could not handle well when a frame index co-exists with another
6844 // non-register operand, unless that operand is an inlineable immediate.
6845 if (Op.isFI())
6846 return false;
6847 }
6848 }
6849
6850 if (MO->isReg()) {
6851 if (!DefinedRC)
6852 return OpInfo.OperandType == MCOI::OPERAND_UNKNOWN;
6853 return isLegalRegOperand(MI, OpIdx, *MO);
6854 }
6855
6856 if (MO->isImm()) {
6857 uint64_t Imm = MO->getImm();
6858 bool Is64BitFPOp = OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_FP64 ||
6859 OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_V2FP64;
6860 bool Is64BitOp = Is64BitFPOp ||
6861 OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_INT64 ||
6862 OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_V2INT32 ||
6863 OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_V2FP32 ||
6864 OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_V2INT64;
6865 if (Is64BitOp &&
6866 !AMDGPU::isInlinableLiteral64(Imm, ST.hasInv2PiInlineImm())) {
6867 if (!AMDGPU::isValid32BitLiteral(Imm, Is64BitFPOp) &&
6868 (!ST.has64BitLiterals() || InstDesc.getSize() != 4))
6869 return false;
6870
6871 // FIXME: We can use sign extended 64-bit literals, but only for signed
6872 // operands. At the moment we do not know if an operand is signed.
6873 // Such operand will be encoded as its low 32 bits and then either
6874 // correctly sign extended or incorrectly zero extended by HW.
6875 // If 64-bit literals are supported and the literal will be encoded
6876 // as full 64 bit we still can use it.
6877 if (!Is64BitFPOp && (int32_t)Imm < 0 &&
6878 (!ST.has64BitLiterals() || AMDGPU::isValid32BitLiteral(Imm, false)))
6879 return false;
6880 }
6881 }
6882
6883 // Handle non-register types that are treated like immediates.
6884 assert(MO->isImm() || MO->isTargetIndex() || MO->isFI() || MO->isGlobal());
6885
6886 if (!DefinedRC) {
6887 // This operand expects an immediate.
6888 return true;
6889 }
6890
6891 return isImmOperandLegal(MI, OpIdx, *MO);
6892}
6893
6895 bool IsGFX950Only = ST.hasGFX950Insts();
6896 bool IsGFX940Only = ST.hasGFX940Insts();
6897
6898 if (!IsGFX950Only && !IsGFX940Only)
6899 return false;
6900
6901 if (!isVALU(MI, /*AllowLDSDMA=*/false))
6902 return false;
6903
6904 // V_COS, V_EXP, V_RCP, etc.
6905 if (isTRANS(MI))
6906 return true;
6907
6908 // DOT2, DOT2C, DOT4, etc.
6909 if (isDOT(MI))
6910 return true;
6911
6912 // MFMA, SMFMA
6913 if (isMFMA(MI))
6914 return true;
6915
6916 unsigned Opcode = MI.getOpcode();
6917 switch (Opcode) {
6918 case AMDGPU::V_CVT_PK_BF8_F32_e64:
6919 case AMDGPU::V_CVT_PK_FP8_F32_e64:
6920 case AMDGPU::V_MQSAD_PK_U16_U8_e64:
6921 case AMDGPU::V_MQSAD_U32_U8_e64:
6922 case AMDGPU::V_PK_ADD_F16:
6923 case AMDGPU::V_PK_ADD_F32:
6924 case AMDGPU::V_PK_ADD_I16:
6925 case AMDGPU::V_PK_ADD_U16:
6926 case AMDGPU::V_PK_ASHRREV_I16:
6927 case AMDGPU::V_PK_FMA_F16:
6928 case AMDGPU::V_PK_FMA_F32:
6929 case AMDGPU::V_PK_FMAC_F16_e32:
6930 case AMDGPU::V_PK_FMAC_F16_e64:
6931 case AMDGPU::V_PK_LSHLREV_B16:
6932 case AMDGPU::V_PK_LSHRREV_B16:
6933 case AMDGPU::V_PK_MAD_I16:
6934 case AMDGPU::V_PK_MAD_U16:
6935 case AMDGPU::V_PK_MAX_F16:
6936 case AMDGPU::V_PK_MAX_I16:
6937 case AMDGPU::V_PK_MAX_U16:
6938 case AMDGPU::V_PK_MIN_F16:
6939 case AMDGPU::V_PK_MIN_I16:
6940 case AMDGPU::V_PK_MIN_U16:
6941 case AMDGPU::V_PK_MOV_B32:
6942 case AMDGPU::V_PK_MUL_F16:
6943 case AMDGPU::V_PK_MUL_F32:
6944 case AMDGPU::V_PK_MUL_LO_U16:
6945 case AMDGPU::V_PK_SUB_I16:
6946 case AMDGPU::V_PK_SUB_U16:
6947 case AMDGPU::V_QSAD_PK_U16_U8_e64:
6948 return true;
6949 default:
6950 return false;
6951 }
6952}
6953
6955 MachineInstr &MI) const {
6956 unsigned Opc = MI.getOpcode();
6957 const MCInstrDesc &InstrDesc = get(Opc);
6958
6959 int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
6960 MachineOperand &Src0 = MI.getOperand(Src0Idx);
6961
6962 int Src1Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1);
6963 MachineOperand &Src1 = MI.getOperand(Src1Idx);
6964
6965 // If there is an implicit SGPR use such as VCC use for v_addc_u32/v_subb_u32
6966 // we need to only have one constant bus use before GFX10.
6967 bool HasImplicitSGPR = findImplicitSGPRRead(MI);
6968 if (HasImplicitSGPR && ST.getConstantBusLimit(Opc) <= 1 && Src0.isReg() &&
6969 RI.isSGPRReg(MRI, Src0.getReg()))
6970 legalizeOpWithMove(MI, Src0Idx);
6971
6972 // Special case: V_WRITELANE_B32 accepts only immediate or SGPR operands for
6973 // both the value to write (src0) and lane select (src1). Fix up non-SGPR
6974 // src0/src1 with V_READFIRSTLANE.
6975 if (Opc == AMDGPU::V_WRITELANE_B32) {
6976 const DebugLoc &DL = MI.getDebugLoc();
6977 if (Src0.isReg() && RI.isVGPR(MRI, Src0.getReg())) {
6978 Register Reg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
6979 BuildMI(*MI.getParent(), MI, DL, get(AMDGPU::V_READFIRSTLANE_B32), Reg)
6980 .add(Src0);
6981 Src0.ChangeToRegister(Reg, false);
6982 }
6983 if (Src1.isReg() && RI.isVGPR(MRI, Src1.getReg())) {
6984 Register Reg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
6985 const DebugLoc &DL = MI.getDebugLoc();
6986 BuildMI(*MI.getParent(), MI, DL, get(AMDGPU::V_READFIRSTLANE_B32), Reg)
6987 .add(Src1);
6988 Src1.ChangeToRegister(Reg, false);
6989 }
6990 return;
6991 }
6992
6993 // Special case: V_FMAC_F32 and V_FMAC_F16 have src2.
6994 if (Opc == AMDGPU::V_FMAC_F32_e32 || Opc == AMDGPU::V_FMAC_F16_e32) {
6995 int Src2Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2);
6996 if (!RI.isVGPR(MRI, MI.getOperand(Src2Idx).getReg()))
6997 legalizeOpWithMove(MI, Src2Idx);
6998 }
6999
7000 // VOP2 src0 instructions support all operand types, so we don't need to check
7001 // their legality. If src1 is already legal, we don't need to do anything.
7002 if (isLegalRegOperand(MRI, InstrDesc.operands()[Src1Idx], Src1))
7003 return;
7004
7005 // Special case: V_READLANE_B32 accepts only immediate or SGPR operands for
7006 // lane select. Fix up using V_READFIRSTLANE, since we assume that the lane
7007 // select is uniform.
7008 if (Opc == AMDGPU::V_READLANE_B32 && Src1.isReg() &&
7009 RI.isVGPR(MRI, Src1.getReg())) {
7010 Register Reg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7011 const DebugLoc &DL = MI.getDebugLoc();
7012 BuildMI(*MI.getParent(), MI, DL, get(AMDGPU::V_READFIRSTLANE_B32), Reg)
7013 .add(Src1);
7014 Src1.ChangeToRegister(Reg, false);
7015 return;
7016 }
7017
7018 // We do not use commuteInstruction here because it is too aggressive and will
7019 // commute if it is possible. We only want to commute here if it improves
7020 // legality. This can be called a fairly large number of times so don't waste
7021 // compile time pointlessly swapping and checking legality again.
7022 if (HasImplicitSGPR || !MI.isCommutable()) {
7023 legalizeOpWithMove(MI, Src1Idx);
7024 return;
7025 }
7026
7027 // If src0 can be used as src1, commuting will make the operands legal.
7028 // Otherwise we have to give up and insert a move.
7029 //
7030 // TODO: Other immediate-like operand kinds could be commuted if there was a
7031 // MachineOperand::ChangeTo* for them.
7032 if ((!Src1.isImm() && !Src1.isReg()) ||
7033 !isLegalRegOperand(MRI, InstrDesc.operands()[Src1Idx], Src0)) {
7034 legalizeOpWithMove(MI, Src1Idx);
7035 return;
7036 }
7037
7038 int CommutedOpc = commuteOpcode(MI);
7039 if (CommutedOpc == -1) {
7040 legalizeOpWithMove(MI, Src1Idx);
7041 return;
7042 }
7043
7044 MI.setDesc(get(CommutedOpc));
7045
7046 Register Src0Reg = Src0.getReg();
7047 unsigned Src0SubReg = Src0.getSubReg();
7048 bool Src0Kill = Src0.isKill();
7049
7050 if (Src1.isImm())
7051 Src0.ChangeToImmediate(Src1.getImm());
7052 else if (Src1.isReg()) {
7053 Src0.ChangeToRegister(Src1.getReg(), false, false, Src1.isKill());
7054 Src0.setSubReg(Src1.getSubReg());
7055 } else
7056 llvm_unreachable("Should only have register or immediate operands");
7057
7058 Src1.ChangeToRegister(Src0Reg, false, false, Src0Kill);
7059 Src1.setSubReg(Src0SubReg);
7061}
7062
7063// Legalize VOP3 operands. All operand types are supported for any operand
7064// but only one literal constant and only starting from GFX10.
7066 MachineInstr &MI) const {
7067 unsigned Opc = MI.getOpcode();
7068
7069 int VOP3Idx[3] = {
7070 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0),
7071 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1),
7072 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2)
7073 };
7074
7075 if (Opc == AMDGPU::V_PERMLANE16_B32_e64 ||
7076 Opc == AMDGPU::V_PERMLANEX16_B32_e64 ||
7077 Opc == AMDGPU::V_PERMLANE_BCAST_B32_e64 ||
7078 Opc == AMDGPU::V_PERMLANE_UP_B32_e64 ||
7079 Opc == AMDGPU::V_PERMLANE_DOWN_B32_e64 ||
7080 Opc == AMDGPU::V_PERMLANE_XOR_B32_e64 ||
7081 Opc == AMDGPU::V_PERMLANE_IDX_GEN_B32_e64) {
7082 // src1 and src2 must be scalar
7083 MachineOperand &Src1 = MI.getOperand(VOP3Idx[1]);
7084 const DebugLoc &DL = MI.getDebugLoc();
7085 if (Src1.isReg() && !RI.isSGPRClass(MRI.getRegClass(Src1.getReg()))) {
7086 Register Reg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7087 BuildMI(*MI.getParent(), MI, DL, get(AMDGPU::V_READFIRSTLANE_B32), Reg)
7088 .add(Src1);
7089 Src1.ChangeToRegister(Reg, false);
7090 }
7091 if (VOP3Idx[2] != -1) {
7092 MachineOperand &Src2 = MI.getOperand(VOP3Idx[2]);
7093 if (Src2.isReg() && !RI.isSGPRClass(MRI.getRegClass(Src2.getReg()))) {
7094 Register Reg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7095 BuildMI(*MI.getParent(), MI, DL, get(AMDGPU::V_READFIRSTLANE_B32), Reg)
7096 .add(Src2);
7097 Src2.ChangeToRegister(Reg, false);
7098 }
7099 }
7100 }
7101
7102 // Find the one SGPR operand we are allowed to use.
7103 int ConstantBusLimit = ST.getConstantBusLimit(Opc);
7104 int LiteralLimit = ST.hasVOP3Literal() ? 1 : 0;
7105 SmallDenseSet<unsigned> SGPRsUsed;
7106 Register SGPRReg = findUsedSGPR(MI, VOP3Idx);
7107 if (SGPRReg) {
7108 SGPRsUsed.insert(SGPRReg);
7109 --ConstantBusLimit;
7110 }
7111
7112 for (int Idx : VOP3Idx) {
7113 if (Idx == -1)
7114 break;
7115 MachineOperand &MO = MI.getOperand(Idx);
7116
7117 if (!MO.isReg()) {
7118 if (isInlineConstant(MO, get(Opc).operands()[Idx]))
7119 continue;
7120
7121 if (LiteralLimit > 0 && ConstantBusLimit > 0) {
7122 --LiteralLimit;
7123 --ConstantBusLimit;
7124 continue;
7125 }
7126
7127 --LiteralLimit;
7128 --ConstantBusLimit;
7129 legalizeOpWithMove(MI, Idx);
7130 continue;
7131 }
7132
7133 if (!RI.isSGPRClass(RI.getRegClassForReg(MRI, MO.getReg())))
7134 continue; // VGPRs are legal
7135
7136 // We can use one SGPR in each VOP3 instruction prior to GFX10
7137 // and two starting from GFX10.
7138 if (SGPRsUsed.count(MO.getReg()))
7139 continue;
7140 if (ConstantBusLimit > 0) {
7141 SGPRsUsed.insert(MO.getReg());
7142 --ConstantBusLimit;
7143 continue;
7144 }
7145
7146 // If we make it this far, then the operand is not legal and we must
7147 // legalize it.
7148 legalizeOpWithMove(MI, Idx);
7149 }
7150
7151 // Special case: V_FMAC_F32 and V_FMAC_F16 have src2 tied to vdst.
7152 if ((Opc == AMDGPU::V_FMAC_F32_e64 || Opc == AMDGPU::V_FMAC_F16_e64) &&
7153 !RI.isVGPR(MRI, MI.getOperand(VOP3Idx[2]).getReg()))
7154 legalizeOpWithMove(MI, VOP3Idx[2]);
7155
7156 // Fix the register class of single-sgpr-read instructions on gfx12+. See
7157 // SIInstrInfo::isLegalSingleSGPRReadInstOperand for more information.
7159 for (unsigned I = 0; I < 3; ++I) {
7160 if (!isLegalSingleSGPRReadInstOperand(MRI, MI, /*SrcN=*/I))
7161 legalizeOpWithMove(MI, VOP3Idx[I]);
7162 }
7163 }
7164}
7165
7168 const TargetRegisterClass *DstRC /*=nullptr*/) const {
7169 const TargetRegisterClass *VRC = MRI.getRegClass(SrcReg);
7170 const TargetRegisterClass *SRC = RI.getEquivalentSGPRClass(VRC);
7171 if (DstRC)
7172 SRC = RI.getCommonSubClass(SRC, DstRC);
7173
7174 Register DstReg = MRI.createVirtualRegister(SRC);
7175 unsigned SubRegs = RI.getRegSizeInBits(*VRC) / 32;
7176
7177 if (RI.hasAGPRs(VRC)) {
7178 VRC = RI.getEquivalentVGPRClass(VRC);
7179 Register NewSrcReg = MRI.createVirtualRegister(VRC);
7180 BuildMI(*UseMI.getParent(), UseMI, UseMI.getDebugLoc(),
7181 get(TargetOpcode::COPY), NewSrcReg)
7182 .addReg(SrcReg);
7183 SrcReg = NewSrcReg;
7184 }
7185
7186 if (SubRegs == 1) {
7187 BuildMI(*UseMI.getParent(), UseMI, UseMI.getDebugLoc(),
7188 get(AMDGPU::V_READFIRSTLANE_B32), DstReg)
7189 .addReg(SrcReg);
7190 return DstReg;
7191 }
7192
7194 for (unsigned i = 0; i < SubRegs; ++i) {
7195 Register SGPR = MRI.createVirtualRegister(&AMDGPU::SGPR_32RegClass);
7196 BuildMI(*UseMI.getParent(), UseMI, UseMI.getDebugLoc(),
7197 get(AMDGPU::V_READFIRSTLANE_B32), SGPR)
7198 .addReg(SrcReg, {}, RI.getSubRegFromChannel(i));
7199 SRegs.push_back(SGPR);
7200 }
7201
7203 BuildMI(*UseMI.getParent(), UseMI, UseMI.getDebugLoc(),
7204 get(AMDGPU::REG_SEQUENCE), DstReg);
7205 for (unsigned i = 0; i < SubRegs; ++i) {
7206 MIB.addReg(SRegs[i]);
7207 MIB.addImm(RI.getSubRegFromChannel(i));
7208 }
7209 return DstReg;
7210}
7211
7213 MachineInstr &MI) const {
7214
7215 // If the pointer is store in VGPRs, then we need to move them to
7216 // SGPRs using v_readfirstlane. This is safe because we only select
7217 // loads with uniform pointers to SMRD instruction so we know the
7218 // pointer value is uniform.
7219 MachineOperand *SBase = getNamedOperand(MI, AMDGPU::OpName::sbase);
7220 if (SBase && !RI.isSGPRClass(MRI.getRegClass(SBase->getReg()))) {
7221 Register SGPR = readlaneVGPRToSGPR(SBase->getReg(), MI, MRI);
7222 SBase->setReg(SGPR);
7223 }
7224 MachineOperand *SOff = getNamedOperand(MI, AMDGPU::OpName::soffset);
7225 if (SOff && !RI.isSGPRReg(MRI, SOff->getReg())) {
7226 Register SGPR = readlaneVGPRToSGPR(SOff->getReg(), MI, MRI);
7227 SOff->setReg(SGPR);
7228 }
7229}
7230
7232 unsigned Opc = Inst.getOpcode();
7233 int OldSAddrIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::saddr);
7234 if (OldSAddrIdx < 0)
7235 return false;
7236
7237 assert(isSegmentSpecificFLAT(Inst) || (isFLAT(Inst) && ST.hasFlatGVSMode()));
7238
7239 int NewOpc = AMDGPU::getGlobalVaddrOp(Opc);
7240 if (NewOpc < 0)
7242 if (NewOpc < 0)
7243 return false;
7244
7245 MachineRegisterInfo &MRI = Inst.getMF()->getRegInfo();
7246 MachineOperand &SAddr = Inst.getOperand(OldSAddrIdx);
7247 if (RI.isSGPRReg(MRI, SAddr.getReg()))
7248 return false;
7249
7250 int NewVAddrIdx = AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::vaddr);
7251 if (NewVAddrIdx < 0)
7252 return false;
7253
7254 int OldVAddrIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vaddr);
7255
7256 // Check vaddr, it shall be zero or absent.
7257 MachineInstr *VAddrDef = nullptr;
7258 if (OldVAddrIdx >= 0) {
7259 MachineOperand &VAddr = Inst.getOperand(OldVAddrIdx);
7260 VAddrDef = MRI.getUniqueVRegDef(VAddr.getReg());
7261 if (!VAddrDef || !VAddrDef->isMoveImmediate() ||
7262 !VAddrDef->getOperand(1).isImm() ||
7263 VAddrDef->getOperand(1).getImm() != 0)
7264 return false;
7265 }
7266
7267 const MCInstrDesc &NewDesc = get(NewOpc);
7268 Inst.setDesc(NewDesc);
7269
7270 // Callers expect iterator to be valid after this call, so modify the
7271 // instruction in place.
7272 if (OldVAddrIdx == NewVAddrIdx) {
7273 MachineOperand &NewVAddr = Inst.getOperand(NewVAddrIdx);
7274 // Clear use list from the old vaddr holding a zero register.
7275 MRI.removeRegOperandFromUseList(&NewVAddr);
7276 MRI.moveOperands(&NewVAddr, &SAddr, 1);
7277 Inst.removeOperand(OldSAddrIdx);
7278 // Update the use list with the pointer we have just moved from vaddr to
7279 // saddr position. Otherwise new vaddr will be missing from the use list.
7280 MRI.removeRegOperandFromUseList(&NewVAddr);
7281 MRI.addRegOperandToUseList(&NewVAddr);
7282 } else {
7283 assert(OldSAddrIdx == NewVAddrIdx);
7284
7285 if (OldVAddrIdx >= 0) {
7286 int NewVDstIn = AMDGPU::getNamedOperandIdx(NewOpc,
7287 AMDGPU::OpName::vdst_in);
7288
7289 // removeOperand doesn't try to fixup tied operand indexes at it goes, so
7290 // it asserts. Untie the operands for now and retie them afterwards.
7291 if (NewVDstIn != -1) {
7292 int OldVDstIn = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst_in);
7293 Inst.untieRegOperand(OldVDstIn);
7294 }
7295
7296 Inst.removeOperand(OldVAddrIdx);
7297
7298 if (NewVDstIn != -1) {
7299 int NewVDst = AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::vdst);
7300 Inst.tieOperands(NewVDst, NewVDstIn);
7301 }
7302 }
7303 }
7304
7305 if (VAddrDef && MRI.use_nodbg_empty(VAddrDef->getOperand(0).getReg()))
7306 VAddrDef->eraseFromParent();
7307
7308 return true;
7309}
7310
7311// FIXME: Remove this when SelectionDAG is obsoleted.
7313 MachineInstr &MI) const {
7314 if (!isSegmentSpecificFLAT(MI) && !ST.hasFlatGVSMode())
7315 return;
7316
7317 // Fixup SGPR operands in VGPRs. We only select these when the DAG divergence
7318 // thinks they are uniform, so a readfirstlane should be valid.
7319 MachineOperand *SAddr = getNamedOperand(MI, AMDGPU::OpName::saddr);
7320 if (!SAddr || RI.isSGPRClass(MRI.getRegClass(SAddr->getReg())))
7321 return;
7322
7324 return;
7325
7326 const TargetRegisterClass *DeclaredRC =
7327 getRegClass(MI.getDesc(), SAddr->getOperandNo());
7328
7329 Register ToSGPR = readlaneVGPRToSGPR(SAddr->getReg(), MI, MRI, DeclaredRC);
7330 SAddr->setReg(ToSGPR);
7331}
7332
7335 const TargetRegisterClass *DstRC,
7338 const DebugLoc &DL) const {
7339 Register OpReg = Op.getReg();
7340 unsigned OpSubReg = Op.getSubReg();
7341
7342 const TargetRegisterClass *OpRC = RI.getSubClassWithSubReg(
7343 RI.getRegClassForReg(MRI, OpReg), OpSubReg);
7344
7345 // Check if operand is already the correct register class.
7346 if (DstRC == OpRC)
7347 return;
7348
7349 Register DstReg = MRI.createVirtualRegister(DstRC);
7350 auto Copy = BuildMI(InsertMBB, I, DL, get(AMDGPU::COPY), DstReg)
7351 .addReg(OpReg, {}, OpSubReg);
7352 Op.setReg(DstReg);
7353 Op.setSubReg(AMDGPU::NoSubRegister);
7354
7355 MachineInstr *Def = MRI.getVRegDef(OpReg);
7356 if (!Def)
7357 return;
7358
7359 // Try to eliminate the copy if it is copying an immediate value.
7360 if (Def->isMoveImmediate() && DstRC != &AMDGPU::VReg_1RegClass)
7361 foldImmediate(*Copy, *Def, OpReg, &MRI);
7362
7363 bool ImpDef = Def->isImplicitDef();
7364 while (!ImpDef && Def && Def->isCopy()) {
7365 if (Def->getOperand(1).getReg().isPhysical())
7366 break;
7367 Def = MRI.getUniqueVRegDef(Def->getOperand(1).getReg());
7368 ImpDef = Def && Def->isImplicitDef();
7369 }
7370 if (!RI.isSGPRClass(DstRC) && !Copy->readsRegister(AMDGPU::EXEC, &RI) &&
7371 !ImpDef)
7372 Copy.addReg(AMDGPU::EXEC, RegState::Implicit);
7373}
7374
7375// Emit the actual waterfall loop, executing the wrapped instruction for each
7376// unique value of \p ScalarOps across all lanes. In the best case we execute 1
7377// iteration, in the worst case we execute 64 (once per lane).
7380 MachineBasicBlock &LoopBB, MachineBasicBlock &BodyBB, const DebugLoc &DL,
7381 ArrayRef<MachineOperand *> ScalarOps, ArrayRef<Register> PhySGPRs = {}) {
7382 MachineFunction &MF = *LoopBB.getParent();
7384 const SIRegisterInfo *TRI = ST.getRegisterInfo();
7386 const auto *BoolXExecRC = TRI->getWaveMaskRegClass();
7387
7388 // Emit v_cmpx_eq and s_andn2_wrexec when both instructions are
7389 // available. Otherwise, use the previous pattern of v_cmp_eq,
7390 // s_and_saveexec, and s_xor.
7391 bool UseNewExecInstructions =
7392 ST.hasNoSdstCMPX() && TII.pseudoToMCOpcode(LMC.AndN2WrExecOpc) != -1;
7393
7395 Register CondReg;
7396
7397 Register PhiExec;
7398 Register NewExec;
7399
7400 if (UseNewExecInstructions) {
7401 PhiExec = MRI.createVirtualRegister(BoolXExecRC);
7402 NewExec = MRI.createVirtualRegister(BoolXExecRC);
7403 Register InitExec = MRI.createVirtualRegister(BoolXExecRC);
7404 BuildMI(PredBB, PredBB.end(), DL, TII.get(LMC.MovOpc), InitExec)
7405 .addReg(LMC.ExecReg);
7406
7407 BuildMI(LoopBB, I, DL, TII.get(TargetOpcode::PHI), PhiExec)
7408 .addReg(InitExec)
7409 .addMBB(&PredBB)
7410 .addReg(NewExec)
7411 .addMBB(&BodyBB);
7412 }
7413
7414 // Placement of v_cmpx instructions (when index is longer than 64 bit)
7415 // involves a trade-off between register pressure and latency:
7416 // (a) Defering all v_cmpx after all v_readfirstlane may increase
7417 // register pressure because arguments and results of all
7418 // v_readfirstlane instructions must stay live until deferred v_cmpx use them.
7419 // (b) Interleaving v_cmpx with v_readfirstlanes may reduce live ranges and
7420 // increase latency by placing v_readfirstlane instructions
7421 // immediately before v_cmpx instruction that directly depend on it.
7422 ///
7423 // Emitting interleaved v_cmpx and v_readfirstlane requires
7424 // block splitting because v_cmpx changes EXEC mask and therefore for safety
7425 // v_cmpx needs to be treated as terminator until after register allocation
7426 // (spill placement) and instruction reordering.
7427 //
7428 // Current implementation defers v_cmpx and leaves other instruction
7429 // scheduling decisions to later passes, where register pressure is known or
7430 // easier to approximate.
7431 // Non-terminators (V_READFIRSTLANE and REG_SEQUENCE) are inserted before I;
7432 // v_cmpx instructions are inserted at the end of LoopBB.
7433 // After the first v_cmpx is emitted, I is updated to point to it
7434 // so subsequent non-terminators are inserted before all v_cmpx instructions.
7435 for (auto [Idx, ScalarOp] : enumerate(ScalarOps)) {
7436 unsigned RegSize = TRI->getRegSizeInBits(ScalarOp->getReg(), MRI);
7437 unsigned NumSubRegs = RegSize / 32;
7438 Register VScalarOp = ScalarOp->getReg();
7439
7440 const TargetRegisterClass *RFLSrcRC =
7441 TII.getRegClass(TII.get(AMDGPU::V_READFIRSTLANE_B32), 1);
7442
7443 if (NumSubRegs == 1) {
7444 const TargetRegisterClass *VScalarOpRC = MRI.getRegClass(VScalarOp);
7445 if (const TargetRegisterClass *Common =
7446 TRI->getCommonSubClass(VScalarOpRC, RFLSrcRC);
7447 Common != VScalarOpRC) {
7448 Register VRReg = MRI.createVirtualRegister(Common);
7449 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::COPY), VRReg).addReg(VScalarOp);
7450 VScalarOp = VRReg;
7451 }
7452 Register CurReg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7453
7454 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::V_READFIRSTLANE_B32), CurReg)
7455 .addReg(VScalarOp);
7456
7457 if (UseNewExecInstructions) {
7458 auto CmpxMI = BuildMI(LoopBB, LoopBB.end(), DL,
7459 TII.get(AMDGPU::V_CMPX_EQ_U32_nosdst_e32_term))
7460 .addReg(CurReg)
7461 .addReg(VScalarOp);
7462 if (I == LoopBB.end())
7463 I = CmpxMI.getInstr()->getIterator();
7464 } else {
7465 Register NewCondReg = MRI.createVirtualRegister(BoolXExecRC);
7466
7467 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::V_CMP_EQ_U32_e64), NewCondReg)
7468 .addReg(CurReg)
7469 .addReg(VScalarOp);
7470
7471 // Combine the comparison results with AND.
7472 if (!CondReg) { // First.
7473 CondReg = NewCondReg;
7474 } else { // If not the first, we create an AND.
7475 Register AndReg = MRI.createVirtualRegister(BoolXExecRC);
7476 BuildMI(LoopBB, I, DL, TII.get(LMC.AndOpc), AndReg)
7477 .addReg(CondReg)
7478 .addReg(NewCondReg)
7479 .setOperandDead(3);
7480 CondReg = AndReg;
7481 }
7482 }
7483
7484 // Update ScalarOp operand to use the SGPR ScalarOp.
7485 if (PhySGPRs.empty() || !PhySGPRs[Idx].isValid())
7486 ScalarOp->setReg(CurReg);
7487 else {
7488 // Insert into the same block of use
7489 BuildMI(*ScalarOp->getParent()->getParent(), ScalarOp->getParent(), DL,
7490 TII.get(AMDGPU::COPY), PhySGPRs[Idx])
7491 .addReg(CurReg);
7492 ScalarOp->setReg(PhySGPRs[Idx]);
7493 }
7494 ScalarOp->setIsKill();
7495 } else {
7496 SmallVector<Register, 8> ReadlanePieces;
7497 RegState VScalarOpUndef = getUndefRegState(ScalarOp->isUndef());
7498 assert(NumSubRegs % 2 == 0 && NumSubRegs <= 32 &&
7499 "Unhandled register size");
7500
7501 for (unsigned Idx = 0; Idx < NumSubRegs; Idx += 2) {
7502 Register CurRegLo =
7503 MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7504 Register CurRegHi =
7505 MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7506
7507 // Read the next variant <- also loop target.
7508 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::V_READFIRSTLANE_B32), CurRegLo)
7509 .addReg(VScalarOp, VScalarOpUndef, TRI->getSubRegFromChannel(Idx));
7510
7511 // Read the next variant <- also loop target.
7512 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::V_READFIRSTLANE_B32), CurRegHi)
7513 .addReg(VScalarOp, VScalarOpUndef,
7514 TRI->getSubRegFromChannel(Idx + 1));
7515
7516 ReadlanePieces.push_back(CurRegLo);
7517 ReadlanePieces.push_back(CurRegHi);
7518
7519 // Comparison is to be done as 64-bit.
7520 Register CurReg = MRI.createVirtualRegister(&AMDGPU::SGPR_64RegClass);
7521 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::REG_SEQUENCE), CurReg)
7522 .addReg(CurRegLo)
7523 .addImm(AMDGPU::sub0)
7524 .addReg(CurRegHi)
7525 .addImm(AMDGPU::sub1);
7526
7527 unsigned SubReg =
7528 NumSubRegs <= 2 ? 0 : TRI->getSubRegFromChannel(Idx, 2);
7529
7530 if (UseNewExecInstructions) {
7531 auto CmpxMI = BuildMI(LoopBB, LoopBB.end(), DL,
7532 TII.get(AMDGPU::V_CMPX_EQ_U64_nosdst_e32_term))
7533 .addReg(CurReg)
7534 .addReg(VScalarOp, VScalarOpUndef, SubReg);
7535 if (I == LoopBB.end())
7536 I = CmpxMI.getInstr()->getIterator();
7537 } else {
7538 Register NewCondReg = MRI.createVirtualRegister(BoolXExecRC);
7539 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::V_CMP_EQ_U64_e64), NewCondReg)
7540 .addReg(CurReg)
7541 .addReg(VScalarOp, VScalarOpUndef, SubReg);
7542
7543 // Combine the comparison results with AND.
7544 if (!CondReg) { // First.
7545 CondReg = NewCondReg;
7546 } else { // If not the first, we create an AND.
7547 Register AndReg = MRI.createVirtualRegister(BoolXExecRC);
7548 BuildMI(LoopBB, I, DL, TII.get(LMC.AndOpc), AndReg)
7549 .addReg(CondReg)
7550 .addReg(NewCondReg)
7551 .setOperandDead(3);
7552 CondReg = AndReg;
7553 }
7554 }
7555 } // End for loop.
7556
7557 const auto *SScalarOpRC =
7558 TRI->getEquivalentSGPRClass(MRI.getRegClass(VScalarOp));
7559 Register SScalarOp = MRI.createVirtualRegister(SScalarOpRC);
7560
7561 // Build scalar ScalarOp.
7562 auto Merge =
7563 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::REG_SEQUENCE), SScalarOp);
7564 unsigned Channel = 0;
7565 for (Register Piece : ReadlanePieces) {
7566 Merge.addReg(Piece).addImm(TRI->getSubRegFromChannel(Channel++));
7567 }
7568
7569 // Update ScalarOp operand to use the SGPR ScalarOp.
7570 if (PhySGPRs.empty() || !PhySGPRs[Idx].isValid())
7571 ScalarOp->setReg(SScalarOp);
7572 else {
7573 BuildMI(*ScalarOp->getParent()->getParent(), ScalarOp->getParent(), DL,
7574 TII.get(AMDGPU::COPY), PhySGPRs[Idx])
7575 .addReg(SScalarOp);
7576 ScalarOp->setReg(PhySGPRs[Idx]);
7577 }
7578 ScalarOp->setIsKill();
7579 }
7580 }
7581
7582 // AndSaveExecOpc modifies EXEC but can't be isTerminator=1: terminators
7583 // that define virtual registers aren't supported.
7584 Register SaveExec;
7585 if (!UseNewExecInstructions) {
7586 SaveExec = MRI.createVirtualRegister(BoolXExecRC);
7587 MRI.setSimpleHint(SaveExec, CondReg);
7588
7589 // Update EXEC to matching lanes, saving original to SaveExec.
7590 BuildMI(LoopBB, I, DL, TII.get(LMC.AndSaveExecOpc), SaveExec)
7591 .addReg(CondReg, RegState::Kill)
7592 .setOperandDead(3);
7593 }
7594
7595 // The original instruction is here; we insert the terminators after it.
7596 I = BodyBB.end();
7597
7598 if (UseNewExecInstructions) {
7599 // Compute the remaining lanes into a plain virtual register and write EXEC
7600 // from a terminator, so spill code for NewExec is placed before EXEC
7601 // changes. SIOptimizeExecMasking opportunistically folds the pair back
7602 // into S_ANDN2_WREXEC after register allocation.
7603 MRI.setSimpleHint(NewExec, PhiExec);
7604 BuildMI(BodyBB, I, DL, TII.get(LMC.AndN2Opc), NewExec)
7605 .addReg(PhiExec)
7606 .addReg(LMC.ExecReg)
7607 .setOperandDead(3);
7608 BuildMI(BodyBB, I, DL, TII.get(LMC.MovTermOpc), LMC.ExecReg)
7609 .addReg(NewExec);
7610 } else {
7611 // Update EXEC, switch all done bits to 0 and all todo bits to 1.
7612 BuildMI(BodyBB, I, DL, TII.get(LMC.XorTermOpc), LMC.ExecReg)
7613 .addReg(LMC.ExecReg)
7614 .addReg(SaveExec)
7615 .setOperandDead(3);
7616 }
7617
7618 BuildMI(BodyBB, I, DL, TII.get(AMDGPU::SI_WATERFALL_LOOP)).addMBB(&LoopBB);
7619}
7620
7621// Build a waterfall loop around \p MI, replacing the VGPR \p ScalarOp register
7622// with SGPRs by iterating over all unique values across all lanes.
7623// Returns the loop basic block that now contains \p MI.
7624static MachineBasicBlock *
7628 MachineBasicBlock::iterator Begin = nullptr,
7629 MachineBasicBlock::iterator End = nullptr,
7630 ArrayRef<Register> PhySGPRs = {}) {
7631 assert((PhySGPRs.empty() || PhySGPRs.size() == ScalarOps.size()) &&
7632 "Physical SGPRs must be empty or match the number of scalar operands");
7634 MachineFunction &MF = *MBB.getParent();
7636 const SIRegisterInfo *TRI = ST.getRegisterInfo();
7637 MachineRegisterInfo &MRI = MF.getRegInfo();
7638 if (!Begin.isValid())
7639 Begin = &MI;
7640 if (!End.isValid()) {
7641 End = &MI;
7642 ++End;
7643 }
7644 const DebugLoc &DL = MI.getDebugLoc();
7646 const auto *BoolXExecRC = TRI->getWaveMaskRegClass();
7647
7648 // Save SCC. Waterfall Loop may overwrite SCC.
7649 Register SaveSCCReg;
7650
7651 // FIXME: We should maintain SCC liveness while doing the FixSGPRCopies walk
7652 // rather than unlimited scan everywhere
7653 bool SCCNotDead =
7654 MBB.computeRegisterLiveness(TRI, AMDGPU::SCC, MI,
7655 std::numeric_limits<unsigned>::max()) !=
7657 if (SCCNotDead) {
7658 SaveSCCReg = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
7659 BuildMI(MBB, Begin, DL, TII.get(AMDGPU::S_CSELECT_B32), SaveSCCReg)
7660 .addImm(1)
7661 .addImm(0);
7662 }
7663
7664 Register SaveExec = MRI.createVirtualRegister(BoolXExecRC);
7665
7666 // Save the EXEC mask
7667 BuildMI(MBB, Begin, DL, TII.get(LMC.MovOpc), SaveExec).addReg(LMC.ExecReg);
7668
7669 // Killed uses in the instruction we are waterfalling around will be
7670 // incorrect due to the added control-flow.
7672 ++AfterMI;
7673 for (auto I = Begin; I != AfterMI; I++) {
7674 for (auto &MO : I->all_uses())
7675 MRI.clearKillFlags(MO.getReg());
7676 }
7677
7678 // To insert the loop we need to split the block. Move everything after this
7679 // point to a new block, and insert a new empty block between the two.
7682 MachineBasicBlock *RemainderBB = MF.CreateMachineBasicBlock();
7684 ++MBBI;
7685
7686 MF.insert(MBBI, LoopBB);
7687 MF.insert(MBBI, BodyBB);
7688 MF.insert(MBBI, RemainderBB);
7689
7690 LoopBB->addSuccessor(BodyBB);
7691 BodyBB->addSuccessor(LoopBB);
7692 BodyBB->addSuccessor(RemainderBB);
7693
7694 // Move Begin to MI to the BodyBB, and the remainder of the block to
7695 // RemainderBB.
7696 RemainderBB->transferSuccessorsAndUpdatePHIs(&MBB);
7697 RemainderBB->splice(RemainderBB->begin(), &MBB, End, MBB.end());
7698 BodyBB->splice(BodyBB->begin(), &MBB, Begin, MBB.end());
7699
7700 MBB.addSuccessor(LoopBB);
7701
7702 // Update dominators. We know that MBB immediately dominates LoopBB, that
7703 // LoopBB immediately dominates BodyBB, and BodyBB immediately dominates
7704 // RemainderBB. RemainderBB immediately dominates all of the successors
7705 // transferred to it from MBB that MBB used to properly dominate.
7706 if (MDT) {
7707 MDT->addNewBlock(LoopBB, &MBB);
7708 MDT->addNewBlock(BodyBB, LoopBB);
7709 MDT->addNewBlock(RemainderBB, BodyBB);
7710 for (auto &Succ : RemainderBB->successors()) {
7711 if (MDT->properlyDominates(&MBB, Succ)) {
7712 MDT->changeImmediateDominator(Succ, RemainderBB);
7713 }
7714 }
7715 }
7716
7717 emitLoadScalarOpsFromVGPRLoop(TII, MRI, MBB, *LoopBB, *BodyBB, DL, ScalarOps,
7718 PhySGPRs);
7719
7720 MachineBasicBlock::iterator First = RemainderBB->begin();
7721 // Restore SCC
7722 if (SCCNotDead) {
7723 BuildMI(*RemainderBB, First, DL, TII.get(AMDGPU::S_CMP_LG_U32))
7724 .addReg(SaveSCCReg, RegState::Kill)
7725 .addImm(0);
7726 }
7727
7728 // Restore the EXEC mask
7729 BuildMI(*RemainderBB, First, DL, TII.get(LMC.MovOpc), LMC.ExecReg)
7730 .addReg(SaveExec);
7731 return BodyBB;
7732}
7733
7734// Extract pointer from Rsrc and return a zero-value Rsrc replacement.
7735static std::tuple<unsigned, unsigned>
7737 MachineBasicBlock &MBB = *MI.getParent();
7738 MachineFunction &MF = *MBB.getParent();
7739 MachineRegisterInfo &MRI = MF.getRegInfo();
7740
7741 // Extract the ptr from the resource descriptor.
7742 unsigned RsrcPtr =
7743 TII.buildExtractSubReg(MI, MRI, Rsrc, &AMDGPU::VReg_128RegClass,
7744 AMDGPU::sub0_sub1, &AMDGPU::VReg_64RegClass);
7745
7746 // Create an empty resource descriptor
7747 Register Zero64 = MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
7748 Register SRsrcFormatLo = MRI.createVirtualRegister(&AMDGPU::SGPR_32RegClass);
7749 Register SRsrcFormatHi = MRI.createVirtualRegister(&AMDGPU::SGPR_32RegClass);
7750 Register NewSRsrc = MRI.createVirtualRegister(&AMDGPU::SGPR_128RegClass);
7751 uint64_t RsrcDataFormat = TII.getDefaultRsrcDataFormat();
7752
7753 // Zero64 = 0
7754 BuildMI(MBB, MI, MI.getDebugLoc(), TII.get(AMDGPU::S_MOV_B64), Zero64)
7755 .addImm(0);
7756
7757 // SRsrcFormatLo = RSRC_DATA_FORMAT{31-0}
7758 BuildMI(MBB, MI, MI.getDebugLoc(), TII.get(AMDGPU::S_MOV_B32), SRsrcFormatLo)
7759 .addImm(Lo_32(RsrcDataFormat));
7760
7761 // SRsrcFormatHi = RSRC_DATA_FORMAT{63-32}
7762 BuildMI(MBB, MI, MI.getDebugLoc(), TII.get(AMDGPU::S_MOV_B32), SRsrcFormatHi)
7763 .addImm(Hi_32(RsrcDataFormat));
7764
7765 // NewSRsrc = {Zero64, SRsrcFormat}
7766 BuildMI(MBB, MI, MI.getDebugLoc(), TII.get(AMDGPU::REG_SEQUENCE), NewSRsrc)
7767 .addReg(Zero64)
7768 .addImm(AMDGPU::sub0_sub1)
7769 .addReg(SRsrcFormatLo)
7770 .addImm(AMDGPU::sub2)
7771 .addReg(SRsrcFormatHi)
7772 .addImm(AMDGPU::sub3);
7773
7774 return std::tuple(RsrcPtr, NewSRsrc);
7775}
7776
7779 MachineDominatorTree *MDT) const {
7780 MachineFunction &MF = *MI.getMF();
7781 MachineRegisterInfo &MRI = MF.getRegInfo();
7782 MachineBasicBlock *CreatedBB = nullptr;
7783
7784 // Legalize True16
7785 if (ST.useRealTrue16Insts())
7787
7788 // Legalize VOP2
7789 if (isVOP2(MI) || isVOPC(MI)) {
7791 return CreatedBB;
7792 }
7793
7794 // Legalize VOP3
7795 if (isVOP3(MI)) {
7797 return CreatedBB;
7798 }
7799
7800 // Legalize SMRD
7801 if (isSMRD(MI)) {
7803 return CreatedBB;
7804 }
7805
7806 // Legalize FLAT
7807 if (isFLAT(MI)) {
7809 return CreatedBB;
7810 }
7811
7812 // Legalize PHI
7813 // The register class of the operands must be the same type as the register
7814 // class of the output.
7815 if (MI.getOpcode() == AMDGPU::PHI) {
7816 const TargetRegisterClass *VRC = getOpRegClass(MI, 0);
7817 assert(!RI.isSGPRClass(VRC));
7818
7819 // Update all the operands so they have the same type.
7820 for (unsigned I = 1, E = MI.getNumOperands(); I != E; I += 2) {
7821 MachineOperand &Op = MI.getOperand(I);
7822 if (!Op.isReg() || !Op.getReg().isVirtual())
7823 continue;
7824
7825 // MI is a PHI instruction.
7826 MachineBasicBlock *InsertBB = MI.getOperand(I + 1).getMBB();
7828
7829 // Avoid creating no-op copies with the same src and dst reg class. These
7830 // confuse some of the machine passes.
7831 legalizeGenericOperand(*InsertBB, Insert, VRC, Op, MRI, MI.getDebugLoc());
7832 }
7833 }
7834
7835 // REG_SEQUENCE doesn't really require operand legalization, but if one has a
7836 // VGPR dest type and SGPR sources, insert copies so all operands are
7837 // VGPRs. This seems to help operand folding / the register coalescer.
7838 if (MI.getOpcode() == AMDGPU::REG_SEQUENCE) {
7839 MachineBasicBlock *MBB = MI.getParent();
7840 const TargetRegisterClass *DstRC = getOpRegClass(MI, 0);
7841 if (RI.hasVGPRs(DstRC)) {
7842 // Update all the operands so they are VGPR register classes. These may
7843 // not be the same register class because REG_SEQUENCE supports mixing
7844 // subregister index types e.g. sub0_sub1 + sub2 + sub3
7845 for (unsigned I = 1, E = MI.getNumOperands(); I != E; I += 2) {
7846 MachineOperand &Op = MI.getOperand(I);
7847 if (!Op.isReg() || !Op.getReg().isVirtual())
7848 continue;
7849
7850 const TargetRegisterClass *OpRC = MRI.getRegClass(Op.getReg());
7851 const TargetRegisterClass *VRC = RI.getEquivalentVGPRClass(OpRC);
7852 if (VRC == OpRC)
7853 continue;
7854
7855 legalizeGenericOperand(*MBB, MI, VRC, Op, MRI, MI.getDebugLoc());
7856 Op.setIsKill();
7857 }
7858 }
7859
7860 return CreatedBB;
7861 }
7862
7863 // Legalize INSERT_SUBREG
7864 // src0 must have the same register class as dst
7865 if (MI.getOpcode() == AMDGPU::INSERT_SUBREG) {
7866 Register Dst = MI.getOperand(0).getReg();
7867 Register Src0 = MI.getOperand(1).getReg();
7868 const TargetRegisterClass *DstRC = MRI.getRegClass(Dst);
7869 const TargetRegisterClass *Src0RC = MRI.getRegClass(Src0);
7870 if (DstRC != Src0RC) {
7871 MachineBasicBlock *MBB = MI.getParent();
7872 MachineOperand &Op = MI.getOperand(1);
7873 legalizeGenericOperand(*MBB, MI, DstRC, Op, MRI, MI.getDebugLoc());
7874 }
7875 return CreatedBB;
7876 }
7877
7878 // Legalize SI_INIT_M0
7879 if (MI.getOpcode() == AMDGPU::SI_INIT_M0) {
7880 MachineOperand &Src = MI.getOperand(0);
7881 if (Src.isReg() && RI.hasVectorRegisters(MRI.getRegClass(Src.getReg())))
7882 Src.setReg(readlaneVGPRToSGPR(Src.getReg(), MI, MRI));
7883 return CreatedBB;
7884 }
7885
7886 // Legalize S_BITREPLICATE, S_QUADMASK and S_WQM
7887 if (MI.getOpcode() == AMDGPU::S_BITREPLICATE_B64_B32 ||
7888 MI.getOpcode() == AMDGPU::S_QUADMASK_B32 ||
7889 MI.getOpcode() == AMDGPU::S_QUADMASK_B64 ||
7890 MI.getOpcode() == AMDGPU::S_WQM_B32 ||
7891 MI.getOpcode() == AMDGPU::S_WQM_B64 ||
7892 MI.getOpcode() == AMDGPU::S_INVERSE_BALLOT_U32 ||
7893 MI.getOpcode() == AMDGPU::S_INVERSE_BALLOT_U64) {
7894 MachineOperand &Src = MI.getOperand(1);
7895 if (Src.isReg() && RI.hasVectorRegisters(MRI.getRegClass(Src.getReg())))
7896 Src.setReg(readlaneVGPRToSGPR(Src.getReg(), MI, MRI));
7897 return CreatedBB;
7898 }
7899
7900 // Legalize MIMG/VIMAGE/VSAMPLE and MUBUF/MTBUF for shaders.
7901 //
7902 // Shaders only generate MUBUF/MTBUF instructions via intrinsics or via
7903 // scratch memory access. In both cases, the legalization never involves
7904 // conversion to the addr64 form.
7906 (isMUBUF(MI) || isMTBUF(MI)))) {
7907 AMDGPU::OpName RSrcOpName = (isVIMAGE(MI) || isVSAMPLE(MI))
7908 ? AMDGPU::OpName::rsrc
7909 : AMDGPU::OpName::srsrc;
7910 MachineOperand *SRsrc = getNamedOperand(MI, RSrcOpName);
7911 if (SRsrc && !RI.isSGPRClass(MRI.getRegClass(SRsrc->getReg())))
7912 CreatedBB = generateWaterFallLoop(*this, MI, {SRsrc}, MDT);
7913
7914 AMDGPU::OpName SampOpName =
7915 isMIMG(MI) ? AMDGPU::OpName::ssamp : AMDGPU::OpName::samp;
7916 MachineOperand *SSamp = getNamedOperand(MI, SampOpName);
7917 if (SSamp && !RI.isSGPRClass(MRI.getRegClass(SSamp->getReg())))
7918 CreatedBB = generateWaterFallLoop(*this, MI, {SSamp}, MDT);
7919
7920 return CreatedBB;
7921 }
7922
7923 // Legalize SI_CALL
7924 if (MI.getOpcode() == AMDGPU::SI_CALL_ISEL) {
7925 MachineOperand *Dest = &MI.getOperand(0);
7926 if (!RI.isSGPRClass(MRI.getRegClass(Dest->getReg()))) {
7927 createWaterFallForSiCall(&MI, MDT, {Dest});
7928 }
7929 }
7930
7931 // Legalize s_sleep_var.
7932 if (MI.getOpcode() == AMDGPU::S_SLEEP_VAR) {
7933 const DebugLoc &DL = MI.getDebugLoc();
7934 Register Reg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7935 int Src0Idx =
7936 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::src0);
7937 MachineOperand &Src0 = MI.getOperand(Src0Idx);
7938 BuildMI(*MI.getParent(), MI, DL, get(AMDGPU::V_READFIRSTLANE_B32), Reg)
7939 .add(Src0);
7940 Src0.ChangeToRegister(Reg, false);
7941 return nullptr;
7942 }
7943
7944 // Legalize TENSOR_LOAD_TO_LDS_d2/_d4, TENSOR_STORE_FROM_LDS_d2/_d4. All their
7945 // operands are scalar.
7946 if (MI.getOpcode() == AMDGPU::TENSOR_LOAD_TO_LDS_d2 ||
7947 MI.getOpcode() == AMDGPU::TENSOR_LOAD_TO_LDS_d4 ||
7948 MI.getOpcode() == AMDGPU::TENSOR_STORE_FROM_LDS_d2 ||
7949 MI.getOpcode() == AMDGPU::TENSOR_STORE_FROM_LDS_d4) {
7950 for (MachineOperand &Src : MI.explicit_operands()) {
7951 if (Src.isReg() && RI.hasVectorRegisters(MRI.getRegClass(Src.getReg())))
7952 Src.setReg(readlaneVGPRToSGPR(Src.getReg(), MI, MRI));
7953 }
7954 return CreatedBB;
7955 }
7956
7957 // Legalize MUBUF instructions.
7958 bool isSoffsetLegal = true;
7959 int SoffsetIdx =
7960 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::soffset);
7961 if (SoffsetIdx != -1) {
7962 MachineOperand *Soffset = &MI.getOperand(SoffsetIdx);
7963 if (Soffset->isReg() && Soffset->getReg().isVirtual() &&
7964 !RI.isSGPRClass(MRI.getRegClass(Soffset->getReg()))) {
7965 isSoffsetLegal = false;
7966 }
7967 }
7968
7969 bool isRsrcLegal = true;
7970 int RsrcIdx =
7971 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::srsrc);
7972 if (RsrcIdx != -1) {
7973 MachineOperand *Rsrc = &MI.getOperand(RsrcIdx);
7974 if (Rsrc->isReg() && !RI.isSGPRReg(MRI, Rsrc->getReg()))
7975 isRsrcLegal = false;
7976 }
7977
7978 // The operands are legal.
7979 if (isRsrcLegal && isSoffsetLegal)
7980 return CreatedBB;
7981
7982 if (!isRsrcLegal) {
7983 // Legalize a VGPR Rsrc
7984 //
7985 // If the instruction is _ADDR64, we can avoid a waterfall by extracting
7986 // the base pointer from the VGPR Rsrc, adding it to the VAddr, then using
7987 // a zero-value SRsrc.
7988 //
7989 // If the instruction is _OFFSET (both idxen and offen disabled), and we
7990 // support ADDR64 instructions, we can convert to ADDR64 and do the same as
7991 // above.
7992 //
7993 // Otherwise we are on non-ADDR64 hardware, and/or we have
7994 // idxen/offen/bothen and we fall back to a waterfall loop.
7995
7996 MachineOperand *Rsrc = &MI.getOperand(RsrcIdx);
7997 MachineBasicBlock &MBB = *MI.getParent();
7998
7999 MachineOperand *VAddr = getNamedOperand(MI, AMDGPU::OpName::vaddr);
8000 if (VAddr && AMDGPU::getIfAddr64Inst(MI.getOpcode()) != -1) {
8001 // This is already an ADDR64 instruction so we need to add the pointer
8002 // extracted from the resource descriptor to the current value of VAddr.
8003 Register NewVAddrLo = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
8004 Register NewVAddrHi = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
8005 Register NewVAddr = MRI.createVirtualRegister(&AMDGPU::VReg_64RegClass);
8006
8007 const auto *BoolXExecRC = RI.getWaveMaskRegClass();
8008 Register CondReg0 = MRI.createVirtualRegister(BoolXExecRC);
8009 Register CondReg1 = MRI.createVirtualRegister(BoolXExecRC);
8010
8011 unsigned RsrcPtr, NewSRsrc;
8012 std::tie(RsrcPtr, NewSRsrc) = extractRsrcPtr(*this, MI, *Rsrc);
8013
8014 // NewVaddrLo = RsrcPtr:sub0 + VAddr:sub0
8015 const DebugLoc &DL = MI.getDebugLoc();
8016 BuildMI(MBB, MI, DL, get(AMDGPU::V_ADD_CO_U32_e64), NewVAddrLo)
8017 .addDef(CondReg0)
8018 .addReg(RsrcPtr, {}, AMDGPU::sub0)
8019 .addReg(VAddr->getReg(), {}, AMDGPU::sub0)
8020 .addImm(0);
8021
8022 // NewVaddrHi = RsrcPtr:sub1 + VAddr:sub1
8023 BuildMI(MBB, MI, DL, get(AMDGPU::V_ADDC_U32_e64), NewVAddrHi)
8024 .addDef(CondReg1, RegState::Dead)
8025 .addReg(RsrcPtr, {}, AMDGPU::sub1)
8026 .addReg(VAddr->getReg(), {}, AMDGPU::sub1)
8027 .addReg(CondReg0, RegState::Kill)
8028 .addImm(0);
8029
8030 // NewVaddr = {NewVaddrHi, NewVaddrLo}
8031 BuildMI(MBB, MI, MI.getDebugLoc(), get(AMDGPU::REG_SEQUENCE), NewVAddr)
8032 .addReg(NewVAddrLo)
8033 .addImm(AMDGPU::sub0)
8034 .addReg(NewVAddrHi)
8035 .addImm(AMDGPU::sub1);
8036
8037 VAddr->setReg(NewVAddr);
8038 Rsrc->setReg(NewSRsrc);
8039 } else if (!VAddr && ST.hasAddr64()) {
8040 // This instructions is the _OFFSET variant, so we need to convert it to
8041 // ADDR64.
8042 assert(ST.getGeneration() < AMDGPUSubtarget::VOLCANIC_ISLANDS &&
8043 "FIXME: Need to emit flat atomics here");
8044
8045 unsigned RsrcPtr, NewSRsrc;
8046 std::tie(RsrcPtr, NewSRsrc) = extractRsrcPtr(*this, MI, *Rsrc);
8047
8048 Register NewVAddr = MRI.createVirtualRegister(&AMDGPU::VReg_64RegClass);
8049 MachineOperand *VData = getNamedOperand(MI, AMDGPU::OpName::vdata);
8050 MachineOperand *Offset = getNamedOperand(MI, AMDGPU::OpName::offset);
8051 MachineOperand *SOffset = getNamedOperand(MI, AMDGPU::OpName::soffset);
8052 unsigned Addr64Opcode = AMDGPU::getAddr64Inst(MI.getOpcode());
8053
8054 // Atomics with return have an additional tied operand and are
8055 // missing some of the special bits.
8056 MachineOperand *VDataIn = getNamedOperand(MI, AMDGPU::OpName::vdata_in);
8057 MachineInstr *Addr64;
8058
8059 if (!VDataIn) {
8060 // Regular buffer load / store.
8062 BuildMI(MBB, MI, MI.getDebugLoc(), get(Addr64Opcode))
8063 .add(*VData)
8064 .addReg(NewVAddr)
8065 .addReg(NewSRsrc)
8066 .add(*SOffset)
8067 .add(*Offset);
8068
8069 if (const MachineOperand *CPol =
8070 getNamedOperand(MI, AMDGPU::OpName::cpol)) {
8071 MIB.addImm(CPol->getImm());
8072 }
8073
8074 if (const MachineOperand *TFE =
8075 getNamedOperand(MI, AMDGPU::OpName::tfe)) {
8076 MIB.addImm(TFE->getImm());
8077 }
8078
8079 MIB.addImm(getNamedImmOperand(MI, AMDGPU::OpName::swz));
8080
8081 MIB.cloneMemRefs(MI);
8082 Addr64 = MIB;
8083 } else {
8084 // Atomics with return.
8085 Addr64 = BuildMI(MBB, MI, MI.getDebugLoc(), get(Addr64Opcode))
8086 .add(*VData)
8087 .add(*VDataIn)
8088 .addReg(NewVAddr)
8089 .addReg(NewSRsrc)
8090 .add(*SOffset)
8091 .add(*Offset)
8092 .addImm(getNamedImmOperand(MI, AMDGPU::OpName::cpol))
8093 .cloneMemRefs(MI);
8094 }
8095
8096 MI.removeFromParent();
8097
8098 // NewVaddr = {NewVaddrHi, NewVaddrLo}
8099 BuildMI(MBB, Addr64, Addr64->getDebugLoc(), get(AMDGPU::REG_SEQUENCE),
8100 NewVAddr)
8101 .addReg(RsrcPtr, {}, AMDGPU::sub0)
8102 .addImm(AMDGPU::sub0)
8103 .addReg(RsrcPtr, {}, AMDGPU::sub1)
8104 .addImm(AMDGPU::sub1);
8105 } else {
8106 // Legalize a VGPR Rsrc and soffset together.
8107 if (!isSoffsetLegal) {
8108 MachineOperand *Soffset = getNamedOperand(MI, AMDGPU::OpName::soffset);
8109 CreatedBB = generateWaterFallLoop(*this, MI, {Rsrc, Soffset}, MDT);
8110 return CreatedBB;
8111 }
8112 CreatedBB = generateWaterFallLoop(*this, MI, {Rsrc}, MDT);
8113 return CreatedBB;
8114 }
8115 }
8116
8117 // Legalize a VGPR soffset.
8118 if (!isSoffsetLegal) {
8119 MachineOperand *Soffset = getNamedOperand(MI, AMDGPU::OpName::soffset);
8120 CreatedBB = generateWaterFallLoop(*this, MI, {Soffset}, MDT);
8121 return CreatedBB;
8122 }
8123 return CreatedBB;
8124}
8125
8127 if (InSet.insert(MI).second)
8128 InstrList.push_back(MI);
8129 // Add MBUF instructiosn to deferred list.
8130 int RsrcIdx =
8131 AMDGPU::getNamedOperandIdx(MI->getOpcode(), AMDGPU::OpName::srsrc);
8132 if (RsrcIdx != -1) {
8133 DeferredList.insert(MI);
8134 }
8135}
8136
8138 return DeferredList.contains(MI);
8139}
8140
8141// Legalize size mismatches between 16bit and 32bit registers in v2s copy
8142// lowering (change sgpr to vgpr).
8143// This is mainly caused by 16bit SALU and 16bit VALU using reg with different
8144// size. Need to legalize the size of the operands during the vgpr lowering
8145// chain. This can be removed after we have sgpr16 in place
8147 MachineRegisterInfo &MRI) const {
8148 if (!ST.useRealTrue16Insts())
8149 return;
8150
8151 unsigned Opcode = MI.getOpcode();
8152 MachineBasicBlock *MBB = MI.getParent();
8153 // Legalize operands and check for size mismatch
8154 if (OpIdx >= MI.getNumExplicitOperands() ||
8155 OpIdx >= get(Opcode).getNumOperands() ||
8156 get(Opcode).operands()[OpIdx].RegClass == -1)
8157 return;
8158
8159 MachineOperand &Op = MI.getOperand(OpIdx);
8160 if (!Op.isReg() || !Op.getReg().isVirtual() || Op.isDef())
8161 return;
8162
8163 const TargetRegisterClass *CurrRC = MRI.getRegClass(Op.getReg());
8164 if (!RI.isVGPRClass(CurrRC))
8165 return;
8166
8167 int16_t RCID = getOpRegClassID(get(Opcode).operands()[OpIdx]);
8168 const TargetRegisterClass *ExpectedRC = RI.getRegClass(RCID);
8169 if (RI.getMatchingSuperRegClass(CurrRC, ExpectedRC, AMDGPU::lo16)) {
8170 // Default to the lo16 only if the subregister is not specified.
8171 if (Op.getSubReg() == AMDGPU::NoSubRegister)
8172 Op.setSubReg(AMDGPU::lo16);
8173 return;
8174 }
8175
8176 const TargetRegisterClass *CurrSRC =
8177 RI.getSubRegisterClass(CurrRC, Op.getSubReg());
8178 if (RI.getMatchingSuperRegClass(ExpectedRC, CurrSRC, AMDGPU::lo16)) {
8179 const DebugLoc &DL = MI.getDebugLoc();
8180 Register NewDstReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
8181 Register Undef = MRI.createVirtualRegister(&AMDGPU::VGPR_16RegClass);
8182 BuildMI(*MBB, MI, DL, get(AMDGPU::IMPLICIT_DEF), Undef);
8183 BuildMI(*MBB, MI, DL, get(AMDGPU::REG_SEQUENCE), NewDstReg)
8184 .addReg(Op.getReg(), {}, Op.getSubReg())
8185 .addImm(AMDGPU::lo16)
8186 .addReg(Undef)
8187 .addImm(AMDGPU::hi16);
8188 Op.setReg(NewDstReg);
8189 Op.setSubReg(AMDGPU::NoSubRegister);
8190 }
8191}
8193 MachineRegisterInfo &MRI) const {
8194 for (unsigned OpIdx = 0; OpIdx < MI.getNumExplicitOperands(); OpIdx++)
8195 legalizeOperandsVALUt16(MI, OpIdx, MRI);
8196}
8197
8201 ArrayRef<Register> PhySGPRs) const {
8202 assert(MI->getOpcode() == AMDGPU::SI_CALL_ISEL &&
8203 "This only handle waterfall for SI_CALL_ISEL");
8204 // Move everything between ADJCALLSTACKUP and ADJCALLSTACKDOWN and
8205 // following copies, we also need to move copies from and to physical
8206 // registers into the loop block.
8207 // Also move the copies to physical registers into the loop block
8208 MachineBasicBlock &MBB = *MI->getParent();
8210 while (Start->getOpcode() != AMDGPU::ADJCALLSTACKUP)
8211 --Start;
8213 while (End->getOpcode() != AMDGPU::ADJCALLSTACKDOWN)
8214 ++End;
8215
8216 // Also include following copies of the return value
8217 ++End;
8218 while (End != MBB.end() && End->isCopy() &&
8219 MI->definesRegister(End->getOperand(1).getReg(), &RI))
8220 ++End;
8221
8222 generateWaterFallLoop(*this, *MI, ScalarOps, MDT, Start, End, PhySGPRs);
8223}
8224
8226 MachineDominatorTree *MDT) const {
8228 DenseMap<MachineInstr *, bool> V2SPhyCopiesToErase;
8229 while (!Worklist.empty()) {
8230 MachineInstr &Inst = *Worklist.top();
8231 Worklist.erase_top();
8232 // Skip MachineInstr in the deferred list.
8233 if (Worklist.isDeferred(&Inst))
8234 continue;
8235 moveToVALUImpl(Worklist, MDT, Inst, WaterFalls, V2SPhyCopiesToErase);
8236 }
8237
8238 // Deferred list of instructions will be processed once
8239 // all the MachineInstr in the worklist are done.
8240 for (MachineInstr *Inst : Worklist.getDeferredList()) {
8241 moveToVALUImpl(Worklist, MDT, *Inst, WaterFalls, V2SPhyCopiesToErase);
8242 assert(Worklist.empty() &&
8243 "Deferred MachineInstr are not supposed to re-populate worklist");
8244 }
8245
8246 for (auto &Entry : WaterFalls) {
8247 if (Entry.first->getOpcode() == AMDGPU::SI_CALL_ISEL)
8248 createWaterFallForSiCall(Entry.first, MDT, Entry.second.MOs,
8249 Entry.second.SGPRs);
8250 }
8251
8252 for (std::pair<MachineInstr *, bool> Entry : V2SPhyCopiesToErase)
8253 if (Entry.second)
8254 Entry.first->eraseFromParent();
8255}
8257 MachineRegisterInfo &MRI, Register DstReg, MachineInstr &Inst) const {
8258 // If it's a copy of a VGPR to a physical SGPR, insert a V_READFIRSTLANE and
8259 // hope for the best.
8260 const TargetRegisterClass *DstRC = RI.getRegClassForReg(MRI, DstReg);
8261 ArrayRef<int16_t> SubRegIndices = RI.getRegSplitParts(DstRC, 4);
8262 if (SubRegIndices.size() <= 1) {
8263 Register NewDst = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
8264 BuildMI(*Inst.getParent(), &Inst, Inst.getDebugLoc(),
8265 get(AMDGPU::V_READFIRSTLANE_B32), NewDst)
8266 .add(Inst.getOperand(1));
8267 BuildMI(*Inst.getParent(), &Inst, Inst.getDebugLoc(), get(AMDGPU::COPY),
8268 DstReg)
8269 .addReg(NewDst);
8270 } else {
8272 for (int16_t Indice : SubRegIndices) {
8273 Register NewDst = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
8274 BuildMI(*Inst.getParent(), &Inst, Inst.getDebugLoc(),
8275 get(AMDGPU::V_READFIRSTLANE_B32), NewDst)
8276 .addReg(Inst.getOperand(1).getReg(), {}, Indice);
8277
8278 DstRegs.push_back(NewDst);
8279 }
8281 BuildMI(*Inst.getParent(), &Inst, Inst.getDebugLoc(),
8282 get(AMDGPU::REG_SEQUENCE), DstReg);
8283 for (unsigned i = 0; i < SubRegIndices.size(); ++i) {
8284 MIB.addReg(DstRegs[i]);
8285 MIB.addImm(RI.getSubRegFromChannel(i));
8286 }
8287 }
8288}
8289
8291 SIInstrWorklist &Worklist, Register DstReg, MachineInstr &Inst,
8294 DenseMap<MachineInstr *, bool> &V2SPhyCopiesToErase) const {
8295 if (DstReg == AMDGPU::M0) {
8296 createReadFirstLaneFromCopyToPhysReg(MRI, DstReg, Inst);
8297 V2SPhyCopiesToErase.try_emplace(&Inst, true);
8298 return;
8299 }
8300 Register SrcReg = Inst.getOperand(1).getReg();
8303 // Only search current block since phyreg's def & use cannot cross
8304 // blocks when MF.NoPhi = false.
8305 while (++I != E) {
8306 // For SI_CALL_ISEL users, replace the phys SGPR with the VGPR source
8307 // and record the operand for later waterfall loop generation.
8308 if (I->getOpcode() == AMDGPU::SI_CALL_ISEL) {
8309 MachineInstr *UseMI = &*I;
8310 for (unsigned i = 0; i < UseMI->getNumOperands(); ++i) {
8311 if (UseMI->getOperand(i).isReg() &&
8312 UseMI->getOperand(i).getReg() == DstReg) {
8313 MachineOperand *MO = &UseMI->getOperand(i);
8314 MO->setReg(SrcReg);
8315 V2PhysSCopyInfo &V2SCopyInfo = WaterFalls[UseMI];
8316 V2SCopyInfo.MOs.push_back(MO);
8317 V2SCopyInfo.SGPRs.push_back(DstReg);
8318 V2SPhyCopiesToErase.try_emplace(&Inst, true);
8319 }
8320 }
8321 } else if (I->getOpcode() == AMDGPU::SI_RETURN_TO_EPILOG &&
8322 I->getOperand(0).isReg() &&
8323 I->getOperand(0).getReg() == DstReg) {
8324 createReadFirstLaneFromCopyToPhysReg(MRI, DstReg, Inst);
8325 V2SPhyCopiesToErase.try_emplace(&Inst, true);
8326 } else if (I->readsRegister(DstReg, &RI)) {
8327 // COPY cannot be erased if other type of inst uses it.
8328 V2SPhyCopiesToErase[&Inst] = false;
8329 }
8330 if (I->findRegisterDefOperand(DstReg, &RI))
8331 break;
8332 }
8333}
8334
8336 SIInstrWorklist &Worklist, MachineDominatorTree *MDT, MachineInstr &Inst,
8338 DenseMap<MachineInstr *, bool> &V2SPhyCopiesToErase) const {
8339
8341 if (!MBB)
8342 return;
8343 MachineRegisterInfo &MRI = MBB->getParent()->getRegInfo();
8344 unsigned Opcode = Inst.getOpcode();
8345 unsigned NewOpcode = getVALUOp(Inst);
8346 const DebugLoc &DL = Inst.getDebugLoc();
8347
8348 // Handle some special cases
8349 switch (Opcode) {
8350 default:
8351 break;
8352 case AMDGPU::S_ADD_I32:
8353 case AMDGPU::S_SUB_I32: {
8354 // FIXME: The u32 versions currently selected use the carry.
8355 bool Changed;
8356 MachineBasicBlock *CreatedBBTmp = nullptr;
8357 std::tie(Changed, CreatedBBTmp) = moveScalarAddSub(Worklist, Inst, MDT);
8358 if (Changed)
8359 return;
8360
8361 // Default handling
8362 break;
8363 }
8364
8365 case AMDGPU::S_MUL_U64:
8366 if (ST.useVMulU64Inst()) {
8367 NewOpcode = AMDGPU::V_MUL_U64_e64;
8368 break;
8369 }
8370 // Split s_mul_u64 in 32-bit vector multiplications.
8371 splitScalarSMulU64(Worklist, Inst, MDT);
8372 Inst.eraseFromParent();
8373 return;
8374
8375 case AMDGPU::S_MUL_U64_U32_PSEUDO:
8376 case AMDGPU::S_MUL_I64_I32_PSEUDO:
8377 // This is a special case of s_mul_u64 where all the operands are either
8378 // zero extended or sign extended.
8379 splitScalarSMulPseudo(Worklist, Inst, MDT);
8380 Inst.eraseFromParent();
8381 return;
8382
8383 case AMDGPU::S_AND_B64:
8384 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_AND_B32, MDT);
8385 Inst.eraseFromParent();
8386 return;
8387
8388 case AMDGPU::S_OR_B64:
8389 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_OR_B32, MDT);
8390 Inst.eraseFromParent();
8391 return;
8392
8393 case AMDGPU::S_XOR_B64:
8394 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_XOR_B32, MDT);
8395 Inst.eraseFromParent();
8396 return;
8397
8398 case AMDGPU::S_NAND_B64:
8399 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_NAND_B32, MDT);
8400 Inst.eraseFromParent();
8401 return;
8402
8403 case AMDGPU::S_NOR_B64:
8404 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_NOR_B32, MDT);
8405 Inst.eraseFromParent();
8406 return;
8407
8408 case AMDGPU::S_XNOR_B64:
8409 if (ST.hasDLInsts())
8410 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_XNOR_B32, MDT);
8411 else
8412 splitScalar64BitXnor(Worklist, Inst, MDT);
8413 Inst.eraseFromParent();
8414 return;
8415
8416 case AMDGPU::S_ANDN2_B64:
8417 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_ANDN2_B32, MDT);
8418 Inst.eraseFromParent();
8419 return;
8420
8421 case AMDGPU::S_ORN2_B64:
8422 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_ORN2_B32, MDT);
8423 Inst.eraseFromParent();
8424 return;
8425
8426 case AMDGPU::S_BREV_B64:
8427 splitScalar64BitUnaryOp(Worklist, Inst, AMDGPU::S_BREV_B32, true);
8428 Inst.eraseFromParent();
8429 return;
8430
8431 case AMDGPU::S_NOT_B64:
8432 splitScalar64BitUnaryOp(Worklist, Inst, AMDGPU::S_NOT_B32);
8433 Inst.eraseFromParent();
8434 return;
8435
8436 case AMDGPU::S_BCNT1_I32_B64:
8437 splitScalar64BitBCNT(Worklist, Inst);
8438 Inst.eraseFromParent();
8439 return;
8440
8441 case AMDGPU::S_BFE_I64:
8442 splitScalar64BitBFE(Worklist, Inst);
8443 Inst.eraseFromParent();
8444 return;
8445
8446 case AMDGPU::S_FLBIT_I32_B64:
8447 splitScalar64BitCountOp(Worklist, Inst, AMDGPU::V_FFBH_U32_e32);
8448 Inst.eraseFromParent();
8449 return;
8450 case AMDGPU::S_FF1_I32_B64:
8451 splitScalar64BitCountOp(Worklist, Inst, AMDGPU::V_FFBL_B32_e32);
8452 Inst.eraseFromParent();
8453 return;
8454
8455 case AMDGPU::S_LSHL_B32:
8456 if (ST.hasOnlyRevVALUShifts()) {
8457 NewOpcode = AMDGPU::V_LSHLREV_B32_e64;
8458 swapOperands(Inst);
8459 }
8460 break;
8461 case AMDGPU::S_ASHR_I32:
8462 if (ST.hasOnlyRevVALUShifts()) {
8463 NewOpcode = AMDGPU::V_ASHRREV_I32_e64;
8464 swapOperands(Inst);
8465 }
8466 break;
8467 case AMDGPU::S_LSHR_B32:
8468 if (ST.hasOnlyRevVALUShifts()) {
8469 NewOpcode = AMDGPU::V_LSHRREV_B32_e64;
8470 swapOperands(Inst);
8471 }
8472 break;
8473 case AMDGPU::S_LSHL_B64:
8474 if (ST.hasOnlyRevVALUShifts()) {
8475 NewOpcode = ST.getGeneration() >= AMDGPUSubtarget::GFX12
8476 ? AMDGPU::V_LSHLREV_B64_pseudo_e64
8477 : AMDGPU::V_LSHLREV_B64_e64;
8478 swapOperands(Inst);
8479 }
8480 break;
8481 case AMDGPU::S_ASHR_I64:
8482 if (ST.hasOnlyRevVALUShifts()) {
8483 NewOpcode = AMDGPU::V_ASHRREV_I64_e64;
8484 swapOperands(Inst);
8485 }
8486 break;
8487 case AMDGPU::S_LSHR_B64:
8488 if (ST.hasOnlyRevVALUShifts()) {
8489 NewOpcode = AMDGPU::V_LSHRREV_B64_e64;
8490 swapOperands(Inst);
8491 }
8492 break;
8493
8494 case AMDGPU::S_ABS_I32:
8495 lowerScalarAbs(Worklist, Inst);
8496 Inst.eraseFromParent();
8497 return;
8498
8499 case AMDGPU::S_ABSDIFF_I32:
8500 lowerScalarAbsDiff(Worklist, Inst);
8501 Inst.eraseFromParent();
8502 return;
8503
8504 case AMDGPU::S_CBRANCH_SCC0:
8505 case AMDGPU::S_CBRANCH_SCC1: {
8506 // Clear unused bits of vcc
8507 Register CondReg = Inst.getOperand(1).getReg();
8508 bool IsSCC = CondReg == AMDGPU::SCC;
8510 BuildMI(*MBB, Inst, Inst.getDebugLoc(), get(LMC.AndOpc), LMC.VccReg)
8511 .addReg(LMC.ExecReg)
8512 .addReg(IsSCC ? LMC.VccReg : CondReg)
8513 .setOperandDead(3); // implicit-def $scc
8514 Inst.removeOperand(1);
8515 } break;
8516
8517 case AMDGPU::S_BFE_U64:
8518 case AMDGPU::S_BFM_B64:
8519 llvm_unreachable("Moving this op to VALU not implemented");
8520
8521 case AMDGPU::S_PACK_LL_B32_B16:
8522 case AMDGPU::S_PACK_LH_B32_B16:
8523 case AMDGPU::S_PACK_HL_B32_B16:
8524 case AMDGPU::S_PACK_HH_B32_B16:
8525 movePackToVALU(Worklist, MRI, Inst);
8526 Inst.eraseFromParent();
8527 return;
8528
8529 case AMDGPU::S_XNOR_B32:
8530 lowerScalarXnor(Worklist, Inst);
8531 Inst.eraseFromParent();
8532 return;
8533
8534 case AMDGPU::S_NAND_B32:
8535 splitScalarNotBinop(Worklist, Inst, AMDGPU::S_AND_B32);
8536 Inst.eraseFromParent();
8537 return;
8538
8539 case AMDGPU::S_NOR_B32:
8540 splitScalarNotBinop(Worklist, Inst, AMDGPU::S_OR_B32);
8541 Inst.eraseFromParent();
8542 return;
8543
8544 case AMDGPU::S_ANDN2_B32:
8545 splitScalarBinOpN2(Worklist, Inst, AMDGPU::S_AND_B32);
8546 Inst.eraseFromParent();
8547 return;
8548
8549 case AMDGPU::S_ORN2_B32:
8550 splitScalarBinOpN2(Worklist, Inst, AMDGPU::S_OR_B32);
8551 Inst.eraseFromParent();
8552 return;
8553
8554 // TODO: remove as soon as everything is ready
8555 // to replace VGPR to SGPR copy with V_READFIRSTLANEs.
8556 // S_ADD/SUB_CO_PSEUDO as well as S_UADDO/USUBO_PSEUDO
8557 // can only be selected from the uniform SDNode.
8558 case AMDGPU::S_ADD_CO_PSEUDO:
8559 case AMDGPU::S_SUB_CO_PSEUDO: {
8560 unsigned Opc = (Inst.getOpcode() == AMDGPU::S_ADD_CO_PSEUDO)
8561 ? AMDGPU::V_ADDC_U32_e64
8562 : AMDGPU::V_SUBB_U32_e64;
8563 const auto *CarryRC = RI.getWaveMaskRegClass();
8564
8565 Register CarryInReg = Inst.getOperand(4).getReg();
8566 if (!MRI.constrainRegClass(CarryInReg, CarryRC)) {
8567 Register NewCarryReg = MRI.createVirtualRegister(CarryRC);
8568 BuildMI(*MBB, Inst, Inst.getDebugLoc(), get(AMDGPU::COPY), NewCarryReg)
8569 .addReg(CarryInReg);
8570 }
8571
8572 Register CarryOutReg = Inst.getOperand(1).getReg();
8573
8574 Register DestReg = MRI.createVirtualRegister(RI.getEquivalentVGPRClass(
8575 MRI.getRegClass(Inst.getOperand(0).getReg())));
8576 MachineInstr *CarryOp =
8577 BuildMI(*MBB, &Inst, Inst.getDebugLoc(), get(Opc), DestReg)
8578 .addReg(CarryOutReg, RegState::Define)
8579 .add(Inst.getOperand(2))
8580 .add(Inst.getOperand(3))
8581 .addReg(CarryInReg)
8582 .addImm(0);
8583 legalizeOperands(*CarryOp);
8584 MRI.replaceRegWith(Inst.getOperand(0).getReg(), DestReg);
8585 addUsersToMoveToVALUWorklist(DestReg, MRI, Worklist);
8586 Inst.eraseFromParent();
8587 }
8588 return;
8589 case AMDGPU::S_UADDO_PSEUDO:
8590 case AMDGPU::S_USUBO_PSEUDO: {
8591 MachineOperand &Dest0 = Inst.getOperand(0);
8592 MachineOperand &Dest1 = Inst.getOperand(1);
8593 MachineOperand &Src0 = Inst.getOperand(2);
8594 MachineOperand &Src1 = Inst.getOperand(3);
8595
8596 unsigned Opc = (Inst.getOpcode() == AMDGPU::S_UADDO_PSEUDO)
8597 ? AMDGPU::V_ADD_CO_U32_e64
8598 : AMDGPU::V_SUB_CO_U32_e64;
8599 const TargetRegisterClass *NewRC =
8600 RI.getEquivalentVGPRClass(MRI.getRegClass(Dest0.getReg()));
8601 Register DestReg = MRI.createVirtualRegister(NewRC);
8602 MachineInstr *NewInstr = BuildMI(*MBB, &Inst, DL, get(Opc), DestReg)
8603 .addReg(Dest1.getReg(), RegState::Define)
8604 .add(Src0)
8605 .add(Src1)
8606 .addImm(0); // clamp bit
8607
8608 legalizeOperands(*NewInstr, MDT);
8609 MRI.replaceRegWith(Dest0.getReg(), DestReg);
8610 addUsersToMoveToVALUWorklist(DestReg, MRI, Worklist);
8611 Inst.eraseFromParent();
8612 }
8613 return;
8614 case AMDGPU::S_LSHL1_ADD_U32:
8615 case AMDGPU::S_LSHL2_ADD_U32:
8616 case AMDGPU::S_LSHL3_ADD_U32:
8617 case AMDGPU::S_LSHL4_ADD_U32: {
8618 MachineOperand &Dest = Inst.getOperand(0);
8619 MachineOperand &Src0 = Inst.getOperand(1);
8620 MachineOperand &Src1 = Inst.getOperand(2);
8621 unsigned ShiftAmt = (Opcode == AMDGPU::S_LSHL1_ADD_U32 ? 1
8622 : Opcode == AMDGPU::S_LSHL2_ADD_U32 ? 2
8623 : Opcode == AMDGPU::S_LSHL3_ADD_U32 ? 3
8624 : 4);
8625
8626 const TargetRegisterClass *NewRC =
8627 RI.getEquivalentVGPRClass(MRI.getRegClass(Dest.getReg()));
8628 Register DestReg = MRI.createVirtualRegister(NewRC);
8629 MachineInstr *NewInstr =
8630 BuildMI(*MBB, &Inst, DL, get(AMDGPU::V_LSHL_ADD_U32_e64), DestReg)
8631 .add(Src0)
8632 .addImm(ShiftAmt)
8633 .add(Src1);
8634
8635 legalizeOperands(*NewInstr, MDT);
8636 MRI.replaceRegWith(Dest.getReg(), DestReg);
8637 addUsersToMoveToVALUWorklist(DestReg, MRI, Worklist);
8638 Inst.eraseFromParent();
8639 }
8640 return;
8641 case AMDGPU::S_CSELECT_B32:
8642 case AMDGPU::S_CSELECT_B64:
8643 lowerSelect(Worklist, Inst, MDT);
8644 Inst.eraseFromParent();
8645 return;
8646 case AMDGPU::S_CMP_EQ_I32:
8647 case AMDGPU::S_CMP_LG_I32:
8648 case AMDGPU::S_CMP_GT_I32:
8649 case AMDGPU::S_CMP_GE_I32:
8650 case AMDGPU::S_CMP_LT_I32:
8651 case AMDGPU::S_CMP_LE_I32:
8652 case AMDGPU::S_CMP_EQ_U32:
8653 case AMDGPU::S_CMP_LG_U32:
8654 case AMDGPU::S_CMP_GT_U32:
8655 case AMDGPU::S_CMP_GE_U32:
8656 case AMDGPU::S_CMP_LT_U32:
8657 case AMDGPU::S_CMP_LE_U32:
8658 case AMDGPU::S_CMP_EQ_U64:
8659 case AMDGPU::S_CMP_LG_U64:
8660 case AMDGPU::S_CMP_LT_F32:
8661 case AMDGPU::S_CMP_EQ_F32:
8662 case AMDGPU::S_CMP_LE_F32:
8663 case AMDGPU::S_CMP_GT_F32:
8664 case AMDGPU::S_CMP_LG_F32:
8665 case AMDGPU::S_CMP_GE_F32:
8666 case AMDGPU::S_CMP_O_F32:
8667 case AMDGPU::S_CMP_U_F32:
8668 case AMDGPU::S_CMP_NGE_F32:
8669 case AMDGPU::S_CMP_NLG_F32:
8670 case AMDGPU::S_CMP_NGT_F32:
8671 case AMDGPU::S_CMP_NLE_F32:
8672 case AMDGPU::S_CMP_NEQ_F32:
8673 case AMDGPU::S_CMP_NLT_F32: {
8674 Register CondReg = MRI.createVirtualRegister(RI.getWaveMaskRegClass());
8675 auto NewInstr =
8676 BuildMI(*MBB, Inst, Inst.getDebugLoc(), get(NewOpcode), CondReg)
8677 .setMIFlags(Inst.getFlags());
8678 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::src0_modifiers) >=
8679 0) {
8680 NewInstr
8681 .addImm(0) // src0_modifiers
8682 .add(Inst.getOperand(0)) // src0
8683 .addImm(0) // src1_modifiers
8684 .add(Inst.getOperand(1)) // src1
8685 .addImm(0); // clamp
8686 } else {
8687 NewInstr.add(Inst.getOperand(0)).add(Inst.getOperand(1));
8688 }
8689 legalizeOperands(*NewInstr, MDT);
8690 int SCCIdx = Inst.findRegisterDefOperandIdx(AMDGPU::SCC, /*TRI=*/nullptr);
8691 const MachineOperand &SCCOp = Inst.getOperand(SCCIdx);
8692 addSCCDefUsersToVALUWorklist(SCCOp, Inst, Worklist, CondReg);
8693 Inst.eraseFromParent();
8694 return;
8695 }
8696 case AMDGPU::S_CMP_LT_F16:
8697 case AMDGPU::S_CMP_EQ_F16:
8698 case AMDGPU::S_CMP_LE_F16:
8699 case AMDGPU::S_CMP_GT_F16:
8700 case AMDGPU::S_CMP_LG_F16:
8701 case AMDGPU::S_CMP_GE_F16:
8702 case AMDGPU::S_CMP_O_F16:
8703 case AMDGPU::S_CMP_U_F16:
8704 case AMDGPU::S_CMP_NGE_F16:
8705 case AMDGPU::S_CMP_NLG_F16:
8706 case AMDGPU::S_CMP_NGT_F16:
8707 case AMDGPU::S_CMP_NLE_F16:
8708 case AMDGPU::S_CMP_NEQ_F16:
8709 case AMDGPU::S_CMP_NLT_F16: {
8710 Register CondReg = MRI.createVirtualRegister(RI.getWaveMaskRegClass());
8711 auto NewInstr =
8712 BuildMI(*MBB, Inst, Inst.getDebugLoc(), get(NewOpcode), CondReg)
8713 .setMIFlags(Inst.getFlags());
8714 if (AMDGPU::hasNamedOperand(NewOpcode, AMDGPU::OpName::src0_modifiers)) {
8715 NewInstr
8716 .addImm(0) // src0_modifiers
8717 .add(Inst.getOperand(0)) // src0
8718 .addImm(0) // src1_modifiers
8719 .add(Inst.getOperand(1)) // src1
8720 .addImm(0); // clamp
8721 if (AMDGPU::hasNamedOperand(NewOpcode, AMDGPU::OpName::op_sel))
8722 NewInstr.addImm(0); // op_sel0
8723 } else {
8724 NewInstr
8725 .add(Inst.getOperand(0))
8726 .add(Inst.getOperand(1));
8727 }
8728 legalizeOperands(*NewInstr, MDT);
8729 int SCCIdx = Inst.findRegisterDefOperandIdx(AMDGPU::SCC, /*TRI=*/nullptr);
8730 const MachineOperand &SCCOp = Inst.getOperand(SCCIdx);
8731 addSCCDefUsersToVALUWorklist(SCCOp, Inst, Worklist, CondReg);
8732 Inst.eraseFromParent();
8733 return;
8734 }
8735 case AMDGPU::S_CVT_HI_F32_F16: {
8736 Register TmpReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
8737 Register NewDst = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
8738 if (ST.useRealTrue16Insts()) {
8739 BuildMI(*MBB, Inst, DL, get(AMDGPU::COPY), TmpReg)
8740 .add(Inst.getOperand(1));
8741 BuildMI(*MBB, Inst, DL, get(NewOpcode), NewDst)
8742 .addImm(0) // src0_modifiers
8743 .addReg(TmpReg, {}, AMDGPU::hi16)
8744 .addImm(0) // clamp
8745 .addImm(0) // omod
8746 .addImm(0); // op_sel0
8747 } else {
8748 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_LSHRREV_B32_e64), TmpReg)
8749 .addImm(16)
8750 .add(Inst.getOperand(1));
8751 BuildMI(*MBB, Inst, DL, get(NewOpcode), NewDst)
8752 .addImm(0) // src0_modifiers
8753 .addReg(TmpReg)
8754 .addImm(0) // clamp
8755 .addImm(0); // omod
8756 }
8757
8758 MRI.replaceRegWith(Inst.getOperand(0).getReg(), NewDst);
8759 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8760 Inst.eraseFromParent();
8761 return;
8762 }
8763 case AMDGPU::S_MINIMUM_F32:
8764 case AMDGPU::S_MAXIMUM_F32: {
8765 Register NewDst = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
8766 MachineInstr *NewInstr = BuildMI(*MBB, Inst, DL, get(NewOpcode), NewDst)
8767 .addImm(0) // src0_modifiers
8768 .add(Inst.getOperand(1))
8769 .addImm(0) // src1_modifiers
8770 .add(Inst.getOperand(2))
8771 .addImm(0) // clamp
8772 .addImm(0); // omod
8773 MRI.replaceRegWith(Inst.getOperand(0).getReg(), NewDst);
8774
8775 legalizeOperands(*NewInstr, MDT);
8776 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8777 Inst.eraseFromParent();
8778 return;
8779 }
8780 case AMDGPU::S_MINIMUM_F16:
8781 case AMDGPU::S_MAXIMUM_F16: {
8782 Register NewDst = MRI.createVirtualRegister(ST.useRealTrue16Insts()
8783 ? &AMDGPU::VGPR_16RegClass
8784 : &AMDGPU::VGPR_32RegClass);
8785 MachineInstr *NewInstr = BuildMI(*MBB, Inst, DL, get(NewOpcode), NewDst)
8786 .addImm(0) // src0_modifiers
8787 .add(Inst.getOperand(1))
8788 .addImm(0) // src1_modifiers
8789 .add(Inst.getOperand(2))
8790 .addImm(0) // clamp
8791 .addImm(0) // omod
8792 .addImm(0); // opsel0
8793 MRI.replaceRegWith(Inst.getOperand(0).getReg(), NewDst);
8794 legalizeOperands(*NewInstr, MDT);
8795 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8796 Inst.eraseFromParent();
8797 return;
8798 }
8799 case AMDGPU::V_S_EXP_F16_e64:
8800 case AMDGPU::V_S_LOG_F16_e64:
8801 case AMDGPU::V_S_RCP_F16_e64:
8802 case AMDGPU::V_S_RSQ_F16_e64:
8803 case AMDGPU::V_S_SQRT_F16_e64: {
8804 Register NewDst = MRI.createVirtualRegister(ST.useRealTrue16Insts()
8805 ? &AMDGPU::VGPR_16RegClass
8806 : &AMDGPU::VGPR_32RegClass);
8807 auto NewInstr = BuildMI(*MBB, Inst, DL, get(NewOpcode), NewDst)
8808 .add(Inst.getOperand(1)) // src0_modifiers
8809 .add(Inst.getOperand(2))
8810 .add(Inst.getOperand(3)) // clamp
8811 .add(Inst.getOperand(4)) // omod
8812 .setMIFlags(Inst.getFlags());
8813 if (AMDGPU::hasNamedOperand(NewOpcode, AMDGPU::OpName::op_sel))
8814 NewInstr.addImm(0); // opsel0
8815 MRI.replaceRegWith(Inst.getOperand(0).getReg(), NewDst);
8816 legalizeOperands(*NewInstr, MDT);
8817 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8818 Inst.eraseFromParent();
8819 return;
8820 }
8821 }
8822
8823 if (NewOpcode == AMDGPU::INSTRUCTION_LIST_END) {
8824 // We cannot move this instruction to the VALU, so we should try to
8825 // legalize its operands instead.
8826 legalizeOperands(Inst, MDT);
8827 return;
8828 }
8829 // Handle converting generic instructions like COPY-to-SGPR into
8830 // COPY-to-VGPR.
8831 if (NewOpcode == Opcode) {
8832 Register DstReg = Inst.getOperand(0).getReg();
8833 const TargetRegisterClass *NewDstRC = getDestEquivalentVGPRClass(Inst);
8834
8835 if (Inst.isCopy() && DstReg.isPhysical() &&
8836 Inst.getOperand(1).getReg().isVirtual()) {
8837 handleCopyToPhysHelper(Worklist, DstReg, Inst, MRI, WaterFalls,
8838 V2SPhyCopiesToErase);
8839 return;
8840 }
8841
8842 if (Inst.isCopy() && Inst.getOperand(1).getReg().isVirtual()) {
8843 Register NewDstReg = Inst.getOperand(1).getReg();
8844 const TargetRegisterClass *SrcRC = RI.getRegClassForReg(MRI, NewDstReg);
8845 if (const TargetRegisterClass *CommonRC =
8846 RI.getCommonSubClass(NewDstRC, SrcRC)) {
8847 // Instead of creating a copy where src and dst are the same register
8848 // class, we just replace all uses of dst with src. These kinds of
8849 // copies interfere with the heuristics MachineSink uses to decide
8850 // whether or not to split a critical edge. Since the pass assumes
8851 // that copies will end up as machine instructions and not be
8852 // eliminated.
8853 addUsersToMoveToVALUWorklist(DstReg, MRI, Worklist);
8854 unsigned SrcSubReg = Inst.getOperand(1).getSubReg();
8855 bool IsUndef = Inst.getOperand(1).isUndef();
8856 for (MachineOperand &UseMO :
8857 make_early_inc_range(MRI.use_operands(DstReg))) {
8858 UseMO.setSubReg(
8859 RI.composeSubRegIndices(SrcSubReg, UseMO.getSubReg()));
8860 UseMO.setReg(NewDstReg);
8861 if (IsUndef)
8862 UseMO.setIsUndef();
8863 }
8864 MRI.clearKillFlags(NewDstReg);
8865
8866 if (!MRI.constrainRegClass(NewDstReg, CommonRC))
8867 llvm_unreachable("failed to constrain register");
8868
8869 Inst.eraseFromParent();
8870
8871 for (MachineOperand &UseMO :
8872 make_early_inc_range(MRI.use_operands(NewDstReg))) {
8873 MachineInstr &UseMI = *UseMO.getParent();
8874
8875 // Legalize t16 operands since replaceReg is called after
8876 // addUsersToVALU.
8878
8879 unsigned OpIdx = UseMI.getOperandNo(&UseMO);
8880 if (const TargetRegisterClass *OpRC =
8881 getRegClass(UseMI.getDesc(), OpIdx))
8882 MRI.constrainRegClass(NewDstReg, OpRC);
8883 }
8884
8885 return;
8886 }
8887 }
8888
8889 // If this is a v2s copy between 16bit and 32bit reg,
8890 // replace vgpr copy to reg_sequence/extract_subreg
8891 // This can be remove after we have sgpr16 in place
8892 if (ST.useRealTrue16Insts() && Inst.isCopy() &&
8893 Inst.getOperand(1).getReg().isVirtual() &&
8894 RI.isVGPR(MRI, Inst.getOperand(1).getReg())) {
8895 const TargetRegisterClass *SrcRegRC = getOpRegClass(Inst, 1);
8896 if (RI.getMatchingSuperRegClass(NewDstRC, SrcRegRC, AMDGPU::lo16)) {
8897 Register NewDstReg = MRI.createVirtualRegister(NewDstRC);
8898 Register Undef = MRI.createVirtualRegister(&AMDGPU::VGPR_16RegClass);
8899 BuildMI(*Inst.getParent(), &Inst, Inst.getDebugLoc(),
8900 get(AMDGPU::IMPLICIT_DEF), Undef);
8901 BuildMI(*Inst.getParent(), &Inst, Inst.getDebugLoc(),
8902 get(AMDGPU::REG_SEQUENCE), NewDstReg)
8903 .addReg(Inst.getOperand(1).getReg())
8904 .addImm(AMDGPU::lo16)
8905 .addReg(Undef)
8906 .addImm(AMDGPU::hi16);
8907 Inst.eraseFromParent();
8908 MRI.replaceRegWith(DstReg, NewDstReg);
8909 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
8910 return;
8911 } else if (RI.getMatchingSuperRegClass(SrcRegRC, NewDstRC,
8912 AMDGPU::lo16)) {
8913 Inst.getOperand(1).setSubReg(AMDGPU::lo16);
8914 Register NewDstReg = MRI.createVirtualRegister(NewDstRC);
8915 MRI.replaceRegWith(DstReg, NewDstReg);
8916 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
8917 return;
8918 }
8919 }
8920
8921 Register NewDstReg = MRI.createVirtualRegister(NewDstRC);
8922 MRI.replaceRegWith(DstReg, NewDstReg);
8923 legalizeOperands(Inst, MDT);
8924 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
8925 return;
8926 }
8927
8928 // Use the new VALU Opcode.
8929 auto NewInstr = BuildMI(*MBB, Inst, Inst.getDebugLoc(), get(NewOpcode))
8930 .setMIFlags(Inst.getFlags());
8931 if (isVOP3(NewOpcode) && !isVOP3(Opcode)) {
8932 // Intersperse VOP3 modifiers among the SALU operands.
8933 NewInstr->addOperand(Inst.getOperand(0));
8934 if (AMDGPU::getNamedOperandIdx(NewOpcode,
8935 AMDGPU::OpName::src0_modifiers) >= 0)
8936 NewInstr.addImm(0);
8937 if (AMDGPU::hasNamedOperand(NewOpcode, AMDGPU::OpName::src0)) {
8938 const MachineOperand &Src = Inst.getOperand(1);
8939 NewInstr->addOperand(Src);
8940 }
8941
8942 if (Opcode == AMDGPU::S_SEXT_I32_I8 || Opcode == AMDGPU::S_SEXT_I32_I16) {
8943 // We are converting these to a BFE, so we need to add the missing
8944 // operands for the size and offset.
8945 unsigned Size = (Opcode == AMDGPU::S_SEXT_I32_I8) ? 8 : 16;
8946 NewInstr.addImm(0);
8947 NewInstr.addImm(Size);
8948 } else if (Opcode == AMDGPU::S_BCNT1_I32_B32) {
8949 // The VALU version adds the second operand to the result, so insert an
8950 // extra 0 operand.
8951 NewInstr.addImm(0);
8952 } else if (Opcode == AMDGPU::S_BFE_I32 || Opcode == AMDGPU::S_BFE_U32) {
8953 const MachineOperand &OffsetWidthOp = Inst.getOperand(2);
8954 // If we need to move this to VGPRs, we need to unpack the second
8955 // operand back into the 2 separate ones for bit offset and width.
8956 assert(OffsetWidthOp.isImm() &&
8957 "Scalar BFE is only implemented for constant width and offset");
8958 uint32_t Imm = OffsetWidthOp.getImm();
8959
8960 uint32_t Offset = Imm & 0x3f; // Extract bits [5:0].
8961 uint32_t BitWidth = (Imm & 0x7f0000) >> 16; // Extract bits [22:16].
8962 NewInstr.addImm(Offset);
8963 NewInstr.addImm(BitWidth);
8964 } else {
8965 if (AMDGPU::getNamedOperandIdx(NewOpcode,
8966 AMDGPU::OpName::src1_modifiers) >= 0)
8967 NewInstr.addImm(0);
8968 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::src1) >= 0)
8969 NewInstr->addOperand(Inst.getOperand(2));
8970 if (AMDGPU::getNamedOperandIdx(NewOpcode,
8971 AMDGPU::OpName::src2_modifiers) >= 0)
8972 NewInstr.addImm(0);
8973 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::src2) >= 0)
8974 NewInstr->addOperand(Inst.getOperand(3));
8975 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::clamp) >= 0)
8976 NewInstr.addImm(0);
8977 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::omod) >= 0)
8978 NewInstr.addImm(0);
8979 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::op_sel) >= 0)
8980 NewInstr.addImm(0);
8981 }
8982 } else {
8983 // Just copy the SALU operands.
8984 for (const MachineOperand &Op : Inst.explicit_operands())
8985 NewInstr->addOperand(Op);
8986 }
8987
8988 // Remove any references to SCC. Vector instructions can't read from it, and
8989 // We're just about to add the implicit use / defs of VCC, and we don't want
8990 // both.
8991 bool DeadSCCDef = false;
8992 for (MachineOperand &Op : Inst.implicit_operands()) {
8993 if (Op.getReg() == AMDGPU::SCC) {
8994 // Only propagate through live-def of SCC.
8995 if (Op.isDef()) {
8996 if (Op.isDead())
8997 DeadSCCDef = true;
8998 else
8999 addSCCDefUsersToVALUWorklist(Op, Inst, Worklist);
9000 continue;
9001 }
9002
9003 addSCCDefsToVALUWorklist(NewInstr, Worklist);
9004 }
9005 }
9006 Inst.eraseFromParent();
9007 Register NewDstReg;
9008 if (NewInstr->getOperand(0).isReg() && NewInstr->getOperand(0).isDef()) {
9009 Register DstReg = NewInstr->getOperand(0).getReg();
9010 assert(DstReg.isVirtual());
9011 // Update the destination register class.
9012 const TargetRegisterClass *NewDstRC = getDestEquivalentVGPRClass(*NewInstr);
9013 assert(NewDstRC);
9014 NewDstReg = MRI.createVirtualRegister(NewDstRC);
9015 MRI.replaceRegWith(DstReg, NewDstReg);
9016 }
9017 fixImplicitOperands(*NewInstr);
9018
9019 if (DeadSCCDef) {
9020 // A scalar op with a dead SCC def lowers to a VALU op whose VCC def will
9021 // also be dead.
9022 if (MachineOperand *VCCDef =
9023 NewInstr->findRegisterDefOperand(RI.getVCC(), &RI))
9024 VCCDef->setIsDead();
9025 }
9026
9027 // Legalize the operands
9028 legalizeOperands(*NewInstr, MDT);
9029 if (NewDstReg)
9030 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
9031}
9032
9033// Add/sub require special handling to deal with carry outs.
9034std::pair<bool, MachineBasicBlock *>
9035SIInstrInfo::moveScalarAddSub(SIInstrWorklist &Worklist, MachineInstr &Inst,
9036 MachineDominatorTree *MDT) const {
9037 if (ST.hasAddNoCarryInsts()) {
9038 // Assume there is no user of scc since we don't select this in that case.
9039 // Since scc isn't used, it doesn't really matter if the i32 or u32 variant
9040 // is used.
9041
9042 MachineBasicBlock &MBB = *Inst.getParent();
9043 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9044
9045 Register OldDstReg = Inst.getOperand(0).getReg();
9046 Register ResultReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9047
9048 unsigned Opc = Inst.getOpcode();
9049 assert(Opc == AMDGPU::S_ADD_I32 || Opc == AMDGPU::S_SUB_I32);
9050
9051 unsigned NewOpc = Opc == AMDGPU::S_ADD_I32 ?
9052 AMDGPU::V_ADD_U32_e64 : AMDGPU::V_SUB_U32_e64;
9053
9054 assert(Inst.getOperand(3).getReg() == AMDGPU::SCC);
9055 Inst.removeOperand(3);
9056
9057 Inst.setDesc(get(NewOpc));
9058 Inst.addOperand(MachineOperand::CreateImm(0)); // clamp bit
9059 Inst.addImplicitDefUseOperands(*MBB.getParent());
9060 MRI.replaceRegWith(OldDstReg, ResultReg);
9061 MachineBasicBlock *NewBB = legalizeOperands(Inst, MDT);
9062
9063 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9064 return std::pair(true, NewBB);
9065 }
9066
9067 return std::pair(false, nullptr);
9068}
9069
9070void SIInstrInfo::lowerSelect(SIInstrWorklist &Worklist, MachineInstr &Inst,
9071 MachineDominatorTree *MDT) const {
9072
9073 MachineBasicBlock &MBB = *Inst.getParent();
9074 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9075 MachineBasicBlock::iterator MII = Inst;
9076 const DebugLoc &DL = Inst.getDebugLoc();
9077
9078 MachineOperand &Dest = Inst.getOperand(0);
9079 MachineOperand &Src0 = Inst.getOperand(1);
9080 MachineOperand &Src1 = Inst.getOperand(2);
9081 MachineOperand &Cond = Inst.getOperand(3);
9082
9083 Register CondReg = Cond.getReg();
9084 bool IsSCC = (CondReg == AMDGPU::SCC);
9085
9086 // Remove S_CSELECT instructions that we previously inserted to feed the SCC
9087 // condition output from S_CMP into the SGPR condition input of V_CNDMASK. If
9088 // the S_CMP has been promoted to V_CMP then we can feed its SGPR condition
9089 // output directly into the V_CNDMASK.
9090 if (!IsSCC && Src0.isImm() && (Src0.getImm() == -1) && Src1.isImm() &&
9091 (Src1.getImm() == 0)) {
9092 for (MachineOperand &UseMO :
9094 MachineInstr &UseMI = *UseMO.getParent();
9095 switch (UseMI.getOpcode()) {
9096 case AMDGPU::V_CNDMASK_B16_fake16_e32:
9097 case AMDGPU::V_CNDMASK_B16_fake16_e64:
9098 case AMDGPU::V_CNDMASK_B16_t16_e32:
9099 case AMDGPU::V_CNDMASK_B16_t16_e64:
9100 case AMDGPU::V_CNDMASK_B32_e32:
9101 case AMDGPU::V_CNDMASK_B32_e64:
9102 case AMDGPU::V_CNDMASK_B64_PSEUDO:
9103 if (UseMO.isImplicit() ||
9104 &UseMO == getNamedOperand(UseMI, AMDGPU::OpName::src2))
9105 UseMO.setReg(CondReg);
9106 }
9107 }
9108 if (MRI.use_nodbg_empty(Dest.getReg()))
9109 return;
9110 }
9111
9112 Register NewCondReg = CondReg;
9113 if (IsSCC) {
9114 const TargetRegisterClass *TC = RI.getWaveMaskRegClass();
9115 NewCondReg = MRI.createVirtualRegister(TC);
9116
9117 // Now look for the closest SCC def if it is a copy
9118 // replacing the CondReg with the COPY source register
9119 bool CopyFound = false;
9120 for (MachineInstr &CandI :
9122 Inst.getParent()->rend())) {
9123 if (CandI.findRegisterDefOperandIdx(AMDGPU::SCC, &RI, false, false) !=
9124 -1) {
9125 if (CandI.isCopy() && CandI.getOperand(0).getReg() == AMDGPU::SCC) {
9126 BuildMI(MBB, MII, DL, get(AMDGPU::COPY), NewCondReg)
9127 .addReg(CandI.getOperand(1).getReg());
9128 CopyFound = true;
9129 }
9130 break;
9131 }
9132 }
9133 if (!CopyFound) {
9134 // SCC def is not a copy
9135 // Insert a trivial select instead of creating a copy, because a copy from
9136 // SCC would semantically mean just copying a single bit, but we may need
9137 // the result to be a vector condition mask that needs preserving.
9138 unsigned Opcode =
9139 ST.isWave64() ? AMDGPU::S_CSELECT_B64 : AMDGPU::S_CSELECT_B32;
9140 auto NewSelect =
9141 BuildMI(MBB, MII, DL, get(Opcode), NewCondReg).addImm(-1).addImm(0);
9142 NewSelect->getOperand(3).setIsUndef(Cond.isUndef());
9143 }
9144 }
9145
9146 Register NewDestReg = MRI.createVirtualRegister(
9147 RI.getEquivalentVGPRClass(MRI.getRegClass(Dest.getReg())));
9148 MachineInstr *NewInst;
9149 if (Inst.getOpcode() == AMDGPU::S_CSELECT_B32) {
9150 NewInst = BuildMI(MBB, MII, DL, get(AMDGPU::V_CNDMASK_B32_e64), NewDestReg)
9151 .addImm(0)
9152 .add(Src1) // False
9153 .addImm(0)
9154 .add(Src0) // True
9155 .addReg(NewCondReg);
9156 } else {
9157 NewInst =
9158 BuildMI(MBB, MII, DL, get(AMDGPU::V_CNDMASK_B64_PSEUDO), NewDestReg)
9159 .add(Src1) // False
9160 .add(Src0) // True
9161 .addReg(NewCondReg);
9162 }
9163 MRI.replaceRegWith(Dest.getReg(), NewDestReg);
9164 legalizeOperands(*NewInst, MDT);
9165 addUsersToMoveToVALUWorklist(NewDestReg, MRI, Worklist);
9166}
9167
9168void SIInstrInfo::lowerScalarAbs(SIInstrWorklist &Worklist,
9169 MachineInstr &Inst) const {
9170 MachineBasicBlock &MBB = *Inst.getParent();
9171 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9172 MachineBasicBlock::iterator MII = Inst;
9173 const DebugLoc &DL = Inst.getDebugLoc();
9174
9175 MachineOperand &Dest = Inst.getOperand(0);
9176 MachineOperand &Src = Inst.getOperand(1);
9177 Register TmpReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9178 Register ResultReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9179
9180 bool HasCarryOut = !ST.hasAddNoCarryInsts();
9181 unsigned SubOp =
9182 HasCarryOut ? AMDGPU::V_SUB_CO_U32_e32 : AMDGPU::V_SUB_U32_e32;
9183
9184 MachineInstrBuilder Sub =
9185 BuildMI(MBB, MII, DL, get(SubOp), TmpReg).addImm(0).addReg(Src.getReg());
9186 if (HasCarryOut)
9187 Sub.setOperandDead(3); // Dead vcc
9188
9189 BuildMI(MBB, MII, DL, get(AMDGPU::V_MAX_I32_e64), ResultReg)
9190 .addReg(Src.getReg())
9191 .addReg(TmpReg);
9192
9193 MRI.replaceRegWith(Dest.getReg(), ResultReg);
9194 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9195}
9196
9197void SIInstrInfo::lowerScalarAbsDiff(SIInstrWorklist &Worklist,
9198 MachineInstr &Inst) const {
9199 MachineBasicBlock &MBB = *Inst.getParent();
9200 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9201 MachineBasicBlock::iterator MII = Inst;
9202 const DebugLoc &DL = Inst.getDebugLoc();
9203
9204 MachineOperand &Dest = Inst.getOperand(0);
9205 MachineOperand &Src1 = Inst.getOperand(1);
9206 MachineOperand &Src2 = Inst.getOperand(2);
9207 Register SubResultReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9208 Register TmpReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9209 Register ResultReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9210
9211 bool HasCarryOut = !ST.hasAddNoCarryInsts();
9212 unsigned SubOp =
9213 HasCarryOut ? AMDGPU::V_SUB_CO_U32_e32 : AMDGPU::V_SUB_U32_e32;
9214
9215 MachineInstrBuilder Sub1 = BuildMI(MBB, MII, DL, get(SubOp), SubResultReg)
9216 .addReg(Src1.getReg())
9217 .addReg(Src2.getReg());
9218
9219 MachineInstrBuilder Sub2 =
9220 BuildMI(MBB, MII, DL, get(SubOp), TmpReg).addImm(0).addReg(SubResultReg);
9221
9222 if (HasCarryOut) {
9223 Sub1.setOperandDead(3); // Dead vcc
9224 Sub2.setOperandDead(3); // Dead vcc
9225 }
9226
9227 BuildMI(MBB, MII, DL, get(AMDGPU::V_MAX_I32_e64), ResultReg)
9228 .addReg(SubResultReg)
9229 .addReg(TmpReg);
9230
9231 MRI.replaceRegWith(Dest.getReg(), ResultReg);
9232 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9233}
9234
9235void SIInstrInfo::lowerScalarXnor(SIInstrWorklist &Worklist,
9236 MachineInstr &Inst) const {
9237 MachineBasicBlock &MBB = *Inst.getParent();
9238 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9239 MachineBasicBlock::iterator MII = Inst;
9240 const DebugLoc &DL = Inst.getDebugLoc();
9241
9242 MachineOperand &Dest = Inst.getOperand(0);
9243 MachineOperand &Src0 = Inst.getOperand(1);
9244 MachineOperand &Src1 = Inst.getOperand(2);
9245
9246 if (ST.hasDLInsts()) {
9247 Register NewDest = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9248 legalizeGenericOperand(MBB, MII, &AMDGPU::VGPR_32RegClass, Src0, MRI, DL);
9249 legalizeGenericOperand(MBB, MII, &AMDGPU::VGPR_32RegClass, Src1, MRI, DL);
9250
9251 BuildMI(MBB, MII, DL, get(AMDGPU::V_XNOR_B32_e64), NewDest)
9252 .add(Src0)
9253 .add(Src1);
9254
9255 MRI.replaceRegWith(Dest.getReg(), NewDest);
9256 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9257 } else {
9258 // Using the identity !(x ^ y) == (!x ^ y) == (x ^ !y), we can
9259 // invert either source and then perform the XOR. If either source is a
9260 // scalar register, then we can leave the inversion on the scalar unit to
9261 // achieve a better distribution of scalar and vector instructions.
9262 bool Src0IsSGPR = Src0.isReg() &&
9263 RI.isSGPRClass(MRI.getRegClass(Src0.getReg()));
9264 bool Src1IsSGPR = Src1.isReg() &&
9265 RI.isSGPRClass(MRI.getRegClass(Src1.getReg()));
9266 MachineInstr *Xor;
9267 Register Temp = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
9268 Register NewDest = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
9269
9270 // Build a pair of scalar instructions and add them to the work list.
9271 // The next iteration over the work list will lower these to the vector
9272 // unit as necessary.
9273 if (Src0IsSGPR) {
9274 BuildMI(MBB, MII, DL, get(AMDGPU::S_NOT_B32), Temp).add(Src0);
9275 Xor = BuildMI(MBB, MII, DL, get(AMDGPU::S_XOR_B32), NewDest)
9276 .addReg(Temp)
9277 .add(Src1);
9278 } else if (Src1IsSGPR) {
9279 BuildMI(MBB, MII, DL, get(AMDGPU::S_NOT_B32), Temp).add(Src1);
9280 Xor = BuildMI(MBB, MII, DL, get(AMDGPU::S_XOR_B32), NewDest)
9281 .add(Src0)
9282 .addReg(Temp);
9283 } else {
9284 Xor = BuildMI(MBB, MII, DL, get(AMDGPU::S_XOR_B32), Temp)
9285 .add(Src0)
9286 .add(Src1);
9287 MachineInstr *Not =
9288 BuildMI(MBB, MII, DL, get(AMDGPU::S_NOT_B32), NewDest).addReg(Temp);
9289 Worklist.insert(Not);
9290 }
9291
9292 MRI.replaceRegWith(Dest.getReg(), NewDest);
9293
9294 Worklist.insert(Xor);
9295
9296 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9297 }
9298}
9299
9300void SIInstrInfo::splitScalarNotBinop(SIInstrWorklist &Worklist,
9301 MachineInstr &Inst,
9302 unsigned Opcode) const {
9303 MachineBasicBlock &MBB = *Inst.getParent();
9304 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9305 MachineBasicBlock::iterator MII = Inst;
9306 const DebugLoc &DL = Inst.getDebugLoc();
9307
9308 MachineOperand &Dest = Inst.getOperand(0);
9309 MachineOperand &Src0 = Inst.getOperand(1);
9310 MachineOperand &Src1 = Inst.getOperand(2);
9311
9312 Register NewDest = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
9313 Register Interm = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
9314
9315 MachineInstr &Op = *BuildMI(MBB, MII, DL, get(Opcode), Interm)
9316 .add(Src0)
9317 .add(Src1);
9318
9319 MachineInstr &Not = *BuildMI(MBB, MII, DL, get(AMDGPU::S_NOT_B32), NewDest)
9320 .addReg(Interm);
9321
9322 Worklist.insert(&Op);
9323 Worklist.insert(&Not);
9324
9325 MRI.replaceRegWith(Dest.getReg(), NewDest);
9326 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9327}
9328
9329void SIInstrInfo::splitScalarBinOpN2(SIInstrWorklist &Worklist,
9330 MachineInstr &Inst,
9331 unsigned Opcode) const {
9332 MachineBasicBlock &MBB = *Inst.getParent();
9333 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9334 MachineBasicBlock::iterator MII = Inst;
9335 const DebugLoc &DL = Inst.getDebugLoc();
9336
9337 MachineOperand &Dest = Inst.getOperand(0);
9338 MachineOperand &Src0 = Inst.getOperand(1);
9339 MachineOperand &Src1 = Inst.getOperand(2);
9340
9341 Register NewDest = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
9342 Register Interm = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
9343
9344 MachineInstr &Not = *BuildMI(MBB, MII, DL, get(AMDGPU::S_NOT_B32), Interm)
9345 .add(Src1);
9346
9347 MachineInstr &Op = *BuildMI(MBB, MII, DL, get(Opcode), NewDest)
9348 .add(Src0)
9349 .addReg(Interm);
9350
9351 Worklist.insert(&Not);
9352 Worklist.insert(&Op);
9353
9354 MRI.replaceRegWith(Dest.getReg(), NewDest);
9355 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9356}
9357
9358void SIInstrInfo::splitScalar64BitUnaryOp(SIInstrWorklist &Worklist,
9359 MachineInstr &Inst, unsigned Opcode,
9360 bool Swap) const {
9361 MachineBasicBlock &MBB = *Inst.getParent();
9362 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9363
9364 MachineOperand &Dest = Inst.getOperand(0);
9365 MachineOperand &Src0 = Inst.getOperand(1);
9366 const DebugLoc &DL = Inst.getDebugLoc();
9367
9368 MachineBasicBlock::iterator MII = Inst;
9369
9370 const MCInstrDesc &InstDesc = get(Opcode);
9371 const TargetRegisterClass *Src0RC = Src0.isReg() ?
9372 MRI.getRegClass(Src0.getReg()) :
9373 &AMDGPU::SGPR_32RegClass;
9374
9375 const TargetRegisterClass *Src0SubRC =
9376 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9377
9378 MachineOperand SrcReg0Sub0 = buildExtractSubRegOrImm(MII, MRI, Src0, Src0RC,
9379 AMDGPU::sub0, Src0SubRC);
9380
9381 const TargetRegisterClass *DestRC = MRI.getRegClass(Dest.getReg());
9382 const TargetRegisterClass *NewDestRC = RI.getEquivalentVGPRClass(DestRC);
9383 const TargetRegisterClass *NewDestSubRC =
9384 RI.getSubRegisterClass(NewDestRC, AMDGPU::sub0);
9385
9386 Register DestSub0 = MRI.createVirtualRegister(NewDestSubRC);
9387 MachineInstr &LoHalf = *BuildMI(MBB, MII, DL, InstDesc, DestSub0).add(SrcReg0Sub0);
9388
9389 MachineOperand SrcReg0Sub1 = buildExtractSubRegOrImm(MII, MRI, Src0, Src0RC,
9390 AMDGPU::sub1, Src0SubRC);
9391
9392 Register DestSub1 = MRI.createVirtualRegister(NewDestSubRC);
9393 MachineInstr &HiHalf = *BuildMI(MBB, MII, DL, InstDesc, DestSub1).add(SrcReg0Sub1);
9394
9395 if (Swap)
9396 std::swap(DestSub0, DestSub1);
9397
9398 Register FullDestReg = MRI.createVirtualRegister(NewDestRC);
9399 BuildMI(MBB, MII, DL, get(TargetOpcode::REG_SEQUENCE), FullDestReg)
9400 .addReg(DestSub0)
9401 .addImm(AMDGPU::sub0)
9402 .addReg(DestSub1)
9403 .addImm(AMDGPU::sub1);
9404
9405 MRI.replaceRegWith(Dest.getReg(), FullDestReg);
9406
9407 Worklist.insert(&LoHalf);
9408 Worklist.insert(&HiHalf);
9409
9410 // We don't need to legalizeOperands here because for a single operand, src0
9411 // will support any kind of input.
9412
9413 // Move all users of this moved value.
9414 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9415}
9416
9417// There is not a vector equivalent of s_mul_u64. For this reason, we need to
9418// split the s_mul_u64 in 32-bit vector multiplications.
9419void SIInstrInfo::splitScalarSMulU64(SIInstrWorklist &Worklist,
9420 MachineInstr &Inst,
9421 MachineDominatorTree *MDT) const {
9422 MachineBasicBlock &MBB = *Inst.getParent();
9423 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9424
9425 Register FullDestReg = MRI.createVirtualRegister(&AMDGPU::VReg_64RegClass);
9426 Register DestSub0 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9427 Register DestSub1 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9428
9429 MachineOperand &Dest = Inst.getOperand(0);
9430 MachineOperand &Src0 = Inst.getOperand(1);
9431 MachineOperand &Src1 = Inst.getOperand(2);
9432 const DebugLoc &DL = Inst.getDebugLoc();
9433 MachineBasicBlock::iterator MII = Inst;
9434
9435 const TargetRegisterClass *Src0RC = MRI.getRegClass(Src0.getReg());
9436 const TargetRegisterClass *Src1RC = MRI.getRegClass(Src1.getReg());
9437 const TargetRegisterClass *Src0SubRC =
9438 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9439 if (RI.isSGPRClass(Src0SubRC))
9440 Src0SubRC = RI.getEquivalentVGPRClass(Src0SubRC);
9441 const TargetRegisterClass *Src1SubRC =
9442 RI.getSubRegisterClass(Src1RC, AMDGPU::sub0);
9443 if (RI.isSGPRClass(Src1SubRC))
9444 Src1SubRC = RI.getEquivalentVGPRClass(Src1SubRC);
9445
9446 // First, we extract the low 32-bit and high 32-bit values from each of the
9447 // operands.
9448 MachineOperand Op0L =
9449 buildExtractSubRegOrImm(MII, MRI, Src0, Src0RC, AMDGPU::sub0, Src0SubRC);
9450 MachineOperand Op1L =
9451 buildExtractSubRegOrImm(MII, MRI, Src1, Src1RC, AMDGPU::sub0, Src1SubRC);
9452 MachineOperand Op0H =
9453 buildExtractSubRegOrImm(MII, MRI, Src0, Src0RC, AMDGPU::sub1, Src0SubRC);
9454 MachineOperand Op1H =
9455 buildExtractSubRegOrImm(MII, MRI, Src1, Src1RC, AMDGPU::sub1, Src1SubRC);
9456
9457 // The multilication is done as follows:
9458 //
9459 // Op1H Op1L
9460 // * Op0H Op0L
9461 // --------------------
9462 // Op1H*Op0L Op1L*Op0L
9463 // + Op1H*Op0H Op1L*Op0H
9464 // -----------------------------------------
9465 // (Op1H*Op0L + Op1L*Op0H + carry) Op1L*Op0L
9466 //
9467 // We drop Op1H*Op0H because the result of the multiplication is a 64-bit
9468 // value and that would overflow.
9469 // The low 32-bit value is Op1L*Op0L.
9470 // The high 32-bit value is Op1H*Op0L + Op1L*Op0H + carry (from Op1L*Op0L).
9471
9472 Register Op1L_Op0H_Reg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9473 MachineInstr *Op1L_Op0H =
9474 BuildMI(MBB, MII, DL, get(AMDGPU::V_MUL_LO_U32_e64), Op1L_Op0H_Reg)
9475 .add(Op1L)
9476 .add(Op0H);
9477
9478 Register Op1H_Op0L_Reg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9479 MachineInstr *Op1H_Op0L =
9480 BuildMI(MBB, MII, DL, get(AMDGPU::V_MUL_LO_U32_e64), Op1H_Op0L_Reg)
9481 .add(Op1H)
9482 .add(Op0L);
9483
9484 Register CarryReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9485 MachineInstr *Carry =
9486 BuildMI(MBB, MII, DL, get(AMDGPU::V_MUL_HI_U32_e64), CarryReg)
9487 .add(Op1L)
9488 .add(Op0L);
9489
9490 MachineInstr *LoHalf =
9491 BuildMI(MBB, MII, DL, get(AMDGPU::V_MUL_LO_U32_e64), DestSub0)
9492 .add(Op1L)
9493 .add(Op0L);
9494
9495 Register AddReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9496 MachineInstr *Add = BuildMI(MBB, MII, DL, get(AMDGPU::V_ADD_U32_e32), AddReg)
9497 .addReg(Op1L_Op0H_Reg)
9498 .addReg(Op1H_Op0L_Reg);
9499
9500 MachineInstr *HiHalf =
9501 BuildMI(MBB, MII, DL, get(AMDGPU::V_ADD_U32_e32), DestSub1)
9502 .addReg(AddReg)
9503 .addReg(CarryReg);
9504
9505 BuildMI(MBB, MII, DL, get(TargetOpcode::REG_SEQUENCE), FullDestReg)
9506 .addReg(DestSub0)
9507 .addImm(AMDGPU::sub0)
9508 .addReg(DestSub1)
9509 .addImm(AMDGPU::sub1);
9510
9511 MRI.replaceRegWith(Dest.getReg(), FullDestReg);
9512
9513 // Try to legalize the operands in case we need to swap the order to keep it
9514 // valid.
9515 legalizeOperands(*Op1L_Op0H, MDT);
9516 legalizeOperands(*Op1H_Op0L, MDT);
9517 legalizeOperands(*Carry, MDT);
9518 legalizeOperands(*LoHalf, MDT);
9519 legalizeOperands(*Add, MDT);
9520 legalizeOperands(*HiHalf, MDT);
9521
9522 // Move all users of this moved value.
9523 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9524}
9525
9526// Lower S_MUL_U64_U32_PSEUDO/S_MUL_I64_I32_PSEUDO in two 32-bit vector
9527// multiplications.
9528void SIInstrInfo::splitScalarSMulPseudo(SIInstrWorklist &Worklist,
9529 MachineInstr &Inst,
9530 MachineDominatorTree *MDT) const {
9531 MachineBasicBlock &MBB = *Inst.getParent();
9532 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9533
9534 Register FullDestReg = MRI.createVirtualRegister(&AMDGPU::VReg_64RegClass);
9535 Register DestSub0 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9536 Register DestSub1 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9537
9538 MachineOperand &Dest = Inst.getOperand(0);
9539 MachineOperand &Src0 = Inst.getOperand(1);
9540 MachineOperand &Src1 = Inst.getOperand(2);
9541 const DebugLoc &DL = Inst.getDebugLoc();
9542 MachineBasicBlock::iterator MII = Inst;
9543
9544 const TargetRegisterClass *Src0RC = MRI.getRegClass(Src0.getReg());
9545 const TargetRegisterClass *Src1RC = MRI.getRegClass(Src1.getReg());
9546 const TargetRegisterClass *Src0SubRC =
9547 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9548 if (RI.isSGPRClass(Src0SubRC))
9549 Src0SubRC = RI.getEquivalentVGPRClass(Src0SubRC);
9550 const TargetRegisterClass *Src1SubRC =
9551 RI.getSubRegisterClass(Src1RC, AMDGPU::sub0);
9552 if (RI.isSGPRClass(Src1SubRC))
9553 Src1SubRC = RI.getEquivalentVGPRClass(Src1SubRC);
9554
9555 // First, we extract the low 32-bit and high 32-bit values from each of the
9556 // operands.
9557 MachineOperand Op0L =
9558 buildExtractSubRegOrImm(MII, MRI, Src0, Src0RC, AMDGPU::sub0, Src0SubRC);
9559 MachineOperand Op1L =
9560 buildExtractSubRegOrImm(MII, MRI, Src1, Src1RC, AMDGPU::sub0, Src1SubRC);
9561
9562 unsigned Opc = Inst.getOpcode();
9563 unsigned NewOpc = Opc == AMDGPU::S_MUL_U64_U32_PSEUDO
9564 ? AMDGPU::V_MUL_HI_U32_e64
9565 : AMDGPU::V_MUL_HI_I32_e64;
9566 MachineInstr *HiHalf =
9567 BuildMI(MBB, MII, DL, get(NewOpc), DestSub1).add(Op1L).add(Op0L);
9568
9569 MachineInstr *LoHalf =
9570 BuildMI(MBB, MII, DL, get(AMDGPU::V_MUL_LO_U32_e64), DestSub0)
9571 .add(Op1L)
9572 .add(Op0L);
9573
9574 BuildMI(MBB, MII, DL, get(TargetOpcode::REG_SEQUENCE), FullDestReg)
9575 .addReg(DestSub0)
9576 .addImm(AMDGPU::sub0)
9577 .addReg(DestSub1)
9578 .addImm(AMDGPU::sub1);
9579
9580 MRI.replaceRegWith(Dest.getReg(), FullDestReg);
9581
9582 // Try to legalize the operands in case we need to swap the order to keep it
9583 // valid.
9584 legalizeOperands(*HiHalf, MDT);
9585 legalizeOperands(*LoHalf, MDT);
9586
9587 // Move all users of this moved value.
9588 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9589}
9590
9591void SIInstrInfo::splitScalar64BitBinaryOp(SIInstrWorklist &Worklist,
9592 MachineInstr &Inst, unsigned Opcode,
9593 MachineDominatorTree *MDT) const {
9594 MachineBasicBlock &MBB = *Inst.getParent();
9595 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9596
9597 MachineOperand &Dest = Inst.getOperand(0);
9598 MachineOperand &Src0 = Inst.getOperand(1);
9599 MachineOperand &Src1 = Inst.getOperand(2);
9600 const DebugLoc &DL = Inst.getDebugLoc();
9601
9602 MachineBasicBlock::iterator MII = Inst;
9603
9604 const MCInstrDesc &InstDesc = get(Opcode);
9605 const TargetRegisterClass *Src0RC = Src0.isReg() ?
9606 MRI.getRegClass(Src0.getReg()) :
9607 &AMDGPU::SGPR_32RegClass;
9608
9609 const TargetRegisterClass *Src0SubRC =
9610 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9611 const TargetRegisterClass *Src1RC = Src1.isReg() ?
9612 MRI.getRegClass(Src1.getReg()) :
9613 &AMDGPU::SGPR_32RegClass;
9614
9615 const TargetRegisterClass *Src1SubRC =
9616 RI.getSubRegisterClass(Src1RC, AMDGPU::sub0);
9617
9618 MachineOperand SrcReg0Sub0 = buildExtractSubRegOrImm(MII, MRI, Src0, Src0RC,
9619 AMDGPU::sub0, Src0SubRC);
9620 MachineOperand SrcReg1Sub0 = buildExtractSubRegOrImm(MII, MRI, Src1, Src1RC,
9621 AMDGPU::sub0, Src1SubRC);
9622 MachineOperand SrcReg0Sub1 = buildExtractSubRegOrImm(MII, MRI, Src0, Src0RC,
9623 AMDGPU::sub1, Src0SubRC);
9624 MachineOperand SrcReg1Sub1 = buildExtractSubRegOrImm(MII, MRI, Src1, Src1RC,
9625 AMDGPU::sub1, Src1SubRC);
9626
9627 const TargetRegisterClass *DestRC = MRI.getRegClass(Dest.getReg());
9628 const TargetRegisterClass *NewDestRC = RI.getEquivalentVGPRClass(DestRC);
9629 const TargetRegisterClass *NewDestSubRC =
9630 RI.getSubRegisterClass(NewDestRC, AMDGPU::sub0);
9631
9632 Register DestSub0 = MRI.createVirtualRegister(NewDestSubRC);
9633 MachineInstr &LoHalf = *BuildMI(MBB, MII, DL, InstDesc, DestSub0)
9634 .add(SrcReg0Sub0)
9635 .add(SrcReg1Sub0);
9636
9637 Register DestSub1 = MRI.createVirtualRegister(NewDestSubRC);
9638 MachineInstr &HiHalf = *BuildMI(MBB, MII, DL, InstDesc, DestSub1)
9639 .add(SrcReg0Sub1)
9640 .add(SrcReg1Sub1);
9641
9642 Register FullDestReg = MRI.createVirtualRegister(NewDestRC);
9643 BuildMI(MBB, MII, DL, get(TargetOpcode::REG_SEQUENCE), FullDestReg)
9644 .addReg(DestSub0)
9645 .addImm(AMDGPU::sub0)
9646 .addReg(DestSub1)
9647 .addImm(AMDGPU::sub1);
9648
9649 MRI.replaceRegWith(Dest.getReg(), FullDestReg);
9650
9651 Worklist.insert(&LoHalf);
9652 Worklist.insert(&HiHalf);
9653
9654 // Move all users of this moved value.
9655 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9656}
9657
9658void SIInstrInfo::splitScalar64BitXnor(SIInstrWorklist &Worklist,
9659 MachineInstr &Inst,
9660 MachineDominatorTree *MDT) const {
9661 MachineBasicBlock &MBB = *Inst.getParent();
9662 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9663
9664 MachineOperand &Dest = Inst.getOperand(0);
9665 MachineOperand &Src0 = Inst.getOperand(1);
9666 MachineOperand &Src1 = Inst.getOperand(2);
9667 const DebugLoc &DL = Inst.getDebugLoc();
9668
9669 MachineBasicBlock::iterator MII = Inst;
9670
9671 const TargetRegisterClass *DestRC = MRI.getRegClass(Dest.getReg());
9672
9673 Register Interm = MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
9674
9675 MachineOperand* Op0;
9676 MachineOperand* Op1;
9677
9678 if (Src0.isReg() && RI.isSGPRReg(MRI, Src0.getReg())) {
9679 Op0 = &Src0;
9680 Op1 = &Src1;
9681 } else {
9682 Op0 = &Src1;
9683 Op1 = &Src0;
9684 }
9685
9686 BuildMI(MBB, MII, DL, get(AMDGPU::S_NOT_B64), Interm)
9687 .add(*Op0);
9688
9689 Register NewDest = MRI.createVirtualRegister(DestRC);
9690
9691 MachineInstr &Xor = *BuildMI(MBB, MII, DL, get(AMDGPU::S_XOR_B64), NewDest)
9692 .addReg(Interm)
9693 .add(*Op1);
9694
9695 MRI.replaceRegWith(Dest.getReg(), NewDest);
9696
9697 Worklist.insert(&Xor);
9698}
9699
9700void SIInstrInfo::splitScalar64BitBCNT(SIInstrWorklist &Worklist,
9701 MachineInstr &Inst) const {
9702 MachineBasicBlock &MBB = *Inst.getParent();
9703 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9704
9705 MachineBasicBlock::iterator MII = Inst;
9706 const DebugLoc &DL = Inst.getDebugLoc();
9707
9708 MachineOperand &Dest = Inst.getOperand(0);
9709 MachineOperand &Src = Inst.getOperand(1);
9710
9711 const MCInstrDesc &InstDesc = get(AMDGPU::V_BCNT_U32_B32_e64);
9712 const TargetRegisterClass *SrcRC = Src.isReg() ?
9713 MRI.getRegClass(Src.getReg()) :
9714 &AMDGPU::SGPR_32RegClass;
9715
9716 Register MidReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9717 Register ResultReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9718
9719 const TargetRegisterClass *SrcSubRC =
9720 RI.getSubRegisterClass(SrcRC, AMDGPU::sub0);
9721
9722 MachineOperand SrcRegSub0 = buildExtractSubRegOrImm(MII, MRI, Src, SrcRC,
9723 AMDGPU::sub0, SrcSubRC);
9724 MachineOperand SrcRegSub1 = buildExtractSubRegOrImm(MII, MRI, Src, SrcRC,
9725 AMDGPU::sub1, SrcSubRC);
9726
9727 BuildMI(MBB, MII, DL, InstDesc, MidReg).add(SrcRegSub0).addImm(0);
9728
9729 BuildMI(MBB, MII, DL, InstDesc, ResultReg).add(SrcRegSub1).addReg(MidReg);
9730
9731 MRI.replaceRegWith(Dest.getReg(), ResultReg);
9732
9733 // We don't need to legalize operands here. src0 for either instruction can be
9734 // an SGPR, and the second input is unused or determined here.
9735 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9736}
9737
9738void SIInstrInfo::splitScalar64BitBFE(SIInstrWorklist &Worklist,
9739 MachineInstr &Inst) const {
9740 MachineBasicBlock &MBB = *Inst.getParent();
9741 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9742 MachineBasicBlock::iterator MII = Inst;
9743 const DebugLoc &DL = Inst.getDebugLoc();
9744
9745 MachineOperand &Dest = Inst.getOperand(0);
9746 uint32_t Imm = Inst.getOperand(2).getImm();
9747 uint32_t Offset = Imm & 0x3f; // Extract bits [5:0].
9748 uint32_t BitWidth = (Imm & 0x7f0000) >> 16; // Extract bits [22:16].
9749
9750 (void) Offset;
9751
9752 // Only sext_inreg cases handled.
9753 assert(Inst.getOpcode() == AMDGPU::S_BFE_I64 && BitWidth <= 32 &&
9754 Offset == 0 && "Not implemented");
9755
9756 if (BitWidth < 32) {
9757 Register MidRegLo = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9758 Register MidRegHi = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9759 Register ResultReg = MRI.createVirtualRegister(&AMDGPU::VReg_64RegClass);
9760
9761 BuildMI(MBB, MII, DL, get(AMDGPU::V_BFE_I32_e64), MidRegLo)
9762 .addReg(Inst.getOperand(1).getReg(), {}, AMDGPU::sub0)
9763 .addImm(0)
9764 .addImm(BitWidth);
9765
9766 BuildMI(MBB, MII, DL, get(AMDGPU::V_ASHRREV_I32_e32), MidRegHi)
9767 .addImm(31)
9768 .addReg(MidRegLo);
9769
9770 BuildMI(MBB, MII, DL, get(TargetOpcode::REG_SEQUENCE), ResultReg)
9771 .addReg(MidRegLo)
9772 .addImm(AMDGPU::sub0)
9773 .addReg(MidRegHi)
9774 .addImm(AMDGPU::sub1);
9775
9776 MRI.replaceRegWith(Dest.getReg(), ResultReg);
9777 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9778 return;
9779 }
9780
9781 MachineOperand &Src = Inst.getOperand(1);
9782 Register TmpReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9783 Register ResultReg = MRI.createVirtualRegister(&AMDGPU::VReg_64RegClass);
9784
9785 BuildMI(MBB, MII, DL, get(AMDGPU::V_ASHRREV_I32_e64), TmpReg)
9786 .addImm(31)
9787 .addReg(Src.getReg(), {}, AMDGPU::sub0);
9788
9789 BuildMI(MBB, MII, DL, get(TargetOpcode::REG_SEQUENCE), ResultReg)
9790 .addReg(Src.getReg(), {}, AMDGPU::sub0)
9791 .addImm(AMDGPU::sub0)
9792 .addReg(TmpReg)
9793 .addImm(AMDGPU::sub1);
9794
9795 MRI.replaceRegWith(Dest.getReg(), ResultReg);
9796 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9797}
9798
9799void SIInstrInfo::splitScalar64BitCountOp(SIInstrWorklist &Worklist,
9800 MachineInstr &Inst, unsigned Opcode,
9801 MachineDominatorTree *MDT) const {
9802 // (S_FLBIT_I32_B64 hi:lo) ->
9803 // -> (umin (V_FFBH_U32_e32 hi), (or (V_FFBH_U32_e32 lo), 32))
9804 // (S_FF1_I32_B64 hi:lo) ->
9805 // ->(umin (or (V_FFBL_B32_e32 hi), 32) (V_FFBL_B32_e32 lo))
9806
9807 MachineBasicBlock &MBB = *Inst.getParent();
9808 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9809 MachineBasicBlock::iterator MII = Inst;
9810 const DebugLoc &DL = Inst.getDebugLoc();
9811
9812 MachineOperand &Dest = Inst.getOperand(0);
9813 MachineOperand &Src = Inst.getOperand(1);
9814
9815 const MCInstrDesc &InstDesc = get(Opcode);
9816
9817 bool IsCtlz = Opcode == AMDGPU::V_FFBH_U32_e32;
9818
9819 const TargetRegisterClass *SrcRC =
9820 Src.isReg() ? MRI.getRegClass(Src.getReg()) : &AMDGPU::SGPR_32RegClass;
9821 const TargetRegisterClass *SrcSubRC =
9822 RI.getSubRegisterClass(SrcRC, AMDGPU::sub0);
9823
9824 MachineOperand SrcRegSub0 =
9825 buildExtractSubRegOrImm(MII, MRI, Src, SrcRC, AMDGPU::sub0, SrcSubRC);
9826 MachineOperand SrcRegSub1 =
9827 buildExtractSubRegOrImm(MII, MRI, Src, SrcRC, AMDGPU::sub1, SrcSubRC);
9828
9829 Register MidReg1 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9830 Register MidReg2 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9831 Register MidReg3 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9832 Register MidReg4 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9833
9834 BuildMI(MBB, MII, DL, InstDesc, MidReg1).add(SrcRegSub0);
9835
9836 BuildMI(MBB, MII, DL, InstDesc, MidReg2).add(SrcRegSub1);
9837
9838 BuildMI(MBB, MII, DL, get(AMDGPU::V_OR_B32_e32), MidReg3)
9839 .addImm(32)
9840 .addReg(IsCtlz ? MidReg1 : MidReg2);
9841
9842 BuildMI(MBB, MII, DL, get(AMDGPU::V_MIN_U32_e64), MidReg4)
9843 .addReg(MidReg3)
9844 .addReg(IsCtlz ? MidReg2 : MidReg1);
9845
9846 MRI.replaceRegWith(Dest.getReg(), MidReg4);
9847
9848 addUsersToMoveToVALUWorklist(MidReg4, MRI, Worklist);
9849}
9850
9851void SIInstrInfo::addUsersToMoveToVALUWorklist(
9852 Register DstReg, MachineRegisterInfo &MRI,
9853 SIInstrWorklist &Worklist) const {
9854 for (MachineOperand &MO : make_early_inc_range(MRI.use_operands(DstReg))) {
9855 MachineInstr &UseMI = *MO.getParent();
9856
9857 unsigned OpNo = 0;
9858
9859 switch (UseMI.getOpcode()) {
9860 case AMDGPU::COPY:
9861 case AMDGPU::WQM:
9862 case AMDGPU::SOFT_WQM:
9863 case AMDGPU::STRICT_WWM:
9864 case AMDGPU::STRICT_WQM:
9865 case AMDGPU::REG_SEQUENCE:
9866 case AMDGPU::PHI:
9867 case AMDGPU::INSERT_SUBREG:
9868 break;
9869 default:
9870 OpNo = MO.getOperandNo();
9871 break;
9872 }
9873
9874 const TargetRegisterClass *OpRC = getOpRegClass(UseMI, OpNo);
9875 MRI.constrainRegClass(DstReg, OpRC);
9876
9877 if (!RI.hasVectorRegisters(OpRC))
9878 Worklist.insert(&UseMI);
9879 else
9880 // Legalization could change user list.
9881 legalizeOperandsVALUt16(UseMI, OpNo, MRI);
9882 }
9883}
9884
9885void SIInstrInfo::movePackToVALU(SIInstrWorklist &Worklist,
9887 MachineInstr &Inst) const {
9888 Register ResultReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9889 MachineBasicBlock *MBB = Inst.getParent();
9890 MachineOperand &Src0 = Inst.getOperand(1);
9891 MachineOperand &Src1 = Inst.getOperand(2);
9892 const DebugLoc &DL = Inst.getDebugLoc();
9893
9894 if (ST.useRealTrue16Insts()) {
9895 Register SrcReg0, SrcReg1;
9896 if (!Src0.isReg() || !RI.isVGPR(MRI, Src0.getReg())) {
9897 SrcReg0 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9898 BuildMI(*MBB, Inst, DL,
9899 get(Src0.isImm() ? AMDGPU::V_MOV_B32_e32 : AMDGPU::COPY), SrcReg0)
9900 .add(Src0);
9901 } else {
9902 SrcReg0 = Src0.getReg();
9903 }
9904
9905 if (!Src1.isReg() || !RI.isVGPR(MRI, Src1.getReg())) {
9906 SrcReg1 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9907 BuildMI(*MBB, Inst, DL,
9908 get(Src1.isImm() ? AMDGPU::V_MOV_B32_e32 : AMDGPU::COPY), SrcReg1)
9909 .add(Src1);
9910 } else {
9911 SrcReg1 = Src1.getReg();
9912 }
9913
9914 bool isSrc0Reg16 = MRI.constrainRegClass(SrcReg0, &AMDGPU::VGPR_16RegClass);
9915 bool isSrc1Reg16 = MRI.constrainRegClass(SrcReg1, &AMDGPU::VGPR_16RegClass);
9916
9917 auto NewMI = BuildMI(*MBB, Inst, DL, get(AMDGPU::REG_SEQUENCE), ResultReg);
9918 switch (Inst.getOpcode()) {
9919 case AMDGPU::S_PACK_LL_B32_B16:
9920 NewMI
9921 .addReg(SrcReg0, {},
9922 isSrc0Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9923 .addImm(AMDGPU::lo16)
9924 .addReg(SrcReg1, {},
9925 isSrc1Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9926 .addImm(AMDGPU::hi16);
9927 break;
9928 case AMDGPU::S_PACK_LH_B32_B16:
9929 NewMI
9930 .addReg(SrcReg0, {},
9931 isSrc0Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9932 .addImm(AMDGPU::lo16)
9933 .addReg(SrcReg1, {}, AMDGPU::hi16)
9934 .addImm(AMDGPU::hi16);
9935 break;
9936 case AMDGPU::S_PACK_HL_B32_B16:
9937 NewMI.addReg(SrcReg0, {}, AMDGPU::hi16)
9938 .addImm(AMDGPU::lo16)
9939 .addReg(SrcReg1, {},
9940 isSrc1Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9941 .addImm(AMDGPU::hi16);
9942 break;
9943 case AMDGPU::S_PACK_HH_B32_B16:
9944 NewMI.addReg(SrcReg0, {}, AMDGPU::hi16)
9945 .addImm(AMDGPU::lo16)
9946 .addReg(SrcReg1, {}, AMDGPU::hi16)
9947 .addImm(AMDGPU::hi16);
9948 break;
9949 default:
9950 llvm_unreachable("unhandled s_pack_* instruction");
9951 }
9952
9953 MachineOperand &Dest = Inst.getOperand(0);
9954 MRI.replaceRegWith(Dest.getReg(), ResultReg);
9955 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9956 return;
9957 }
9958
9959 switch (Inst.getOpcode()) {
9960 case AMDGPU::S_PACK_LL_B32_B16: {
9961 Register ImmReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9962 Register TmpReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9963
9964 // FIXME: Can do a lot better if we know the high bits of src0 or src1 are
9965 // 0.
9966 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_MOV_B32_e32), ImmReg)
9967 .addImm(0xffff);
9968
9969 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_AND_B32_e64), TmpReg)
9970 .addReg(ImmReg, RegState::Kill)
9971 .add(Src0);
9972
9973 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_LSHL_OR_B32_e64), ResultReg)
9974 .add(Src1)
9975 .addImm(16)
9976 .addReg(TmpReg, RegState::Kill);
9977 break;
9978 }
9979 case AMDGPU::S_PACK_LH_B32_B16: {
9980 Register ImmReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9981 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_MOV_B32_e32), ImmReg)
9982 .addImm(0xffff);
9983 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_BFI_B32_e64), ResultReg)
9984 .addReg(ImmReg, RegState::Kill)
9985 .add(Src0)
9986 .add(Src1);
9987 break;
9988 }
9989 case AMDGPU::S_PACK_HL_B32_B16: {
9990 Register TmpReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9991 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_LSHRREV_B32_e64), TmpReg)
9992 .addImm(16)
9993 .add(Src0);
9994 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_LSHL_OR_B32_e64), ResultReg)
9995 .add(Src1)
9996 .addImm(16)
9997 .addReg(TmpReg, RegState::Kill);
9998 break;
9999 }
10000 case AMDGPU::S_PACK_HH_B32_B16: {
10001 Register ImmReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
10002 Register TmpReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
10003 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_LSHRREV_B32_e64), TmpReg)
10004 .addImm(16)
10005 .add(Src0);
10006 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_MOV_B32_e32), ImmReg)
10007 .addImm(0xffff0000);
10008 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_AND_OR_B32_e64), ResultReg)
10009 .add(Src1)
10010 .addReg(ImmReg, RegState::Kill)
10011 .addReg(TmpReg, RegState::Kill);
10012 break;
10013 }
10014 default:
10015 llvm_unreachable("unhandled s_pack_* instruction");
10016 }
10017
10018 MachineOperand &Dest = Inst.getOperand(0);
10019 MRI.replaceRegWith(Dest.getReg(), ResultReg);
10020 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
10021}
10022
10023void SIInstrInfo::addSCCDefUsersToVALUWorklist(const MachineOperand &Op,
10024 MachineInstr &SCCDefInst,
10025 SIInstrWorklist &Worklist,
10026 Register NewCond) const {
10027
10028 // Ensure that def inst defines SCC, which is still live.
10029 assert(Op.isReg() && Op.getReg() == AMDGPU::SCC && Op.isDef() &&
10030 !Op.isDead() && Op.getParent() == &SCCDefInst);
10031 SmallVector<MachineInstr *, 4> CopyToDelete;
10032 // This assumes that all the users of SCC are in the same block
10033 // as the SCC def.
10034 for (MachineInstr &MI : // Skip the def inst itself.
10035 make_range(std::next(MachineBasicBlock::iterator(SCCDefInst)),
10036 SCCDefInst.getParent()->end())) {
10037 // Check if SCC is used first.
10038 int SCCIdx = MI.findRegisterUseOperandIdx(AMDGPU::SCC, &RI, false);
10039 if (SCCIdx != -1) {
10040 if (MI.isCopy()) {
10041 MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
10042 Register DestReg = MI.getOperand(0).getReg();
10043
10044 MRI.replaceRegWith(DestReg, NewCond);
10045 CopyToDelete.push_back(&MI);
10046 } else {
10047
10048 if (NewCond.isValid())
10049 MI.getOperand(SCCIdx).setReg(NewCond);
10050
10051 Worklist.insert(&MI);
10052 }
10053 }
10054 // Exit if we find another SCC def.
10055 if (MI.findRegisterDefOperandIdx(AMDGPU::SCC, &RI, false, false) != -1)
10056 break;
10057 }
10058 for (auto &Copy : CopyToDelete)
10059 Copy->eraseFromParent();
10060}
10061
10062// Instructions that use SCC may be converted to VALU instructions. When that
10063// happens, the SCC register is changed to VCC_LO. The instruction that defines
10064// SCC must be changed to an instruction that defines VCC. This function makes
10065// sure that the instruction that defines SCC is added to the moveToVALU
10066// worklist.
10067void SIInstrInfo::addSCCDefsToVALUWorklist(MachineInstr *SCCUseInst,
10068 SIInstrWorklist &Worklist) const {
10069 // Look for a preceding instruction that either defines VCC or SCC. If VCC
10070 // then there is nothing to do because the defining instruction has been
10071 // converted to a VALU already. If SCC then that instruction needs to be
10072 // converted to a VALU.
10073 for (MachineInstr &MI :
10074 make_range(std::next(MachineBasicBlock::reverse_iterator(SCCUseInst)),
10075 SCCUseInst->getParent()->rend())) {
10076 if (MI.modifiesRegister(AMDGPU::VCC, &RI))
10077 break;
10078 if (MI.definesRegister(AMDGPU::SCC, &RI)) {
10079 Worklist.insert(&MI);
10080 break;
10081 }
10082 }
10083}
10084
10085const TargetRegisterClass *SIInstrInfo::getDestEquivalentVGPRClass(
10086 const MachineInstr &Inst) const {
10087 const TargetRegisterClass *NewDstRC = getOpRegClass(Inst, 0);
10088
10089 switch (Inst.getOpcode()) {
10090 // For target instructions, getOpRegClass just returns the virtual register
10091 // class associated with the operand, so we need to find an equivalent VGPR
10092 // register class in order to move the instruction to the VALU.
10093 case AMDGPU::COPY:
10094 case AMDGPU::PHI:
10095 case AMDGPU::REG_SEQUENCE:
10096 case AMDGPU::INSERT_SUBREG:
10097 case AMDGPU::WQM:
10098 case AMDGPU::SOFT_WQM:
10099 case AMDGPU::STRICT_WWM:
10100 case AMDGPU::STRICT_WQM: {
10101 const TargetRegisterClass *SrcRC = getOpRegClass(Inst, 1);
10102 if (RI.isAGPRClass(SrcRC)) {
10103 if (RI.isAGPRClass(NewDstRC))
10104 return nullptr;
10105
10106 switch (Inst.getOpcode()) {
10107 case AMDGPU::PHI:
10108 case AMDGPU::REG_SEQUENCE:
10109 case AMDGPU::INSERT_SUBREG:
10110 NewDstRC = RI.getEquivalentAGPRClass(NewDstRC);
10111 break;
10112 default:
10113 NewDstRC = RI.getEquivalentVGPRClass(NewDstRC);
10114 }
10115
10116 if (!NewDstRC)
10117 return nullptr;
10118 } else {
10119 if (!RI.isSGPRClass(NewDstRC) || NewDstRC == &AMDGPU::VReg_1RegClass)
10120 return nullptr;
10121
10122 NewDstRC = RI.getEquivalentVGPRClass(NewDstRC);
10123 if (!NewDstRC)
10124 return nullptr;
10125 }
10126
10127 return NewDstRC;
10128 }
10129 default:
10130 return NewDstRC;
10131 }
10132}
10133
10134// Find the one SGPR operand we are allowed to use.
10135Register SIInstrInfo::findUsedSGPR(const MachineInstr &MI,
10136 int OpIndices[3]) const {
10137 const MCInstrDesc &Desc = MI.getDesc();
10138
10139 // Find the one SGPR operand we are allowed to use.
10140 //
10141 // First we need to consider the instruction's operand requirements before
10142 // legalizing. Some operands are required to be SGPRs, such as implicit uses
10143 // of VCC, but we are still bound by the constant bus requirement to only use
10144 // one.
10145 //
10146 // If the operand's class is an SGPR, we can never move it.
10147
10148 Register SGPRReg = findImplicitSGPRRead(MI);
10149 if (SGPRReg)
10150 return SGPRReg;
10151
10152 Register UsedSGPRs[3] = {Register()};
10153 const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
10154
10155 for (unsigned i = 0; i < 3; ++i) {
10156 int Idx = OpIndices[i];
10157 if (Idx == -1)
10158 break;
10159
10160 const MachineOperand &MO = MI.getOperand(Idx);
10161 if (!MO.isReg())
10162 continue;
10163
10164 // Is this operand statically required to be an SGPR based on the operand
10165 // constraints?
10166 const TargetRegisterClass *OpRC =
10167 RI.getRegClass(getOpRegClassID(Desc.operands()[Idx]));
10168 bool IsRequiredSGPR = RI.isSGPRClass(OpRC);
10169 if (IsRequiredSGPR)
10170 return MO.getReg();
10171
10172 // If this could be a VGPR or an SGPR, Check the dynamic register class.
10173 Register Reg = MO.getReg();
10174 const TargetRegisterClass *RegRC = MRI.getRegClass(Reg);
10175 if (RI.isSGPRClass(RegRC))
10176 UsedSGPRs[i] = Reg;
10177 }
10178
10179 // We don't have a required SGPR operand, so we have a bit more freedom in
10180 // selecting operands to move.
10181
10182 // Try to select the most used SGPR. If an SGPR is equal to one of the
10183 // others, we choose that.
10184 //
10185 // e.g.
10186 // V_FMA_F32 v0, s0, s0, s0 -> No moves
10187 // V_FMA_F32 v0, s0, s1, s0 -> Move s1
10188
10189 // TODO: If some of the operands are 64-bit SGPRs and some 32, we should
10190 // prefer those.
10191
10192 if (UsedSGPRs[0]) {
10193 if (UsedSGPRs[0] == UsedSGPRs[1] || UsedSGPRs[0] == UsedSGPRs[2])
10194 SGPRReg = UsedSGPRs[0];
10195 }
10196
10197 if (!SGPRReg && UsedSGPRs[1]) {
10198 if (UsedSGPRs[1] == UsedSGPRs[2])
10199 SGPRReg = UsedSGPRs[1];
10200 }
10201
10202 return SGPRReg;
10203}
10204
10206 AMDGPU::OpName OperandName) const {
10207 if (OperandName == AMDGPU::OpName::NUM_OPERAND_NAMES)
10208 return nullptr;
10209
10210 int Idx = AMDGPU::getNamedOperandIdx(MI.getOpcode(), OperandName);
10211 if (Idx == -1)
10212 return nullptr;
10213
10214 return &MI.getOperand(Idx);
10215}
10216
10218 if (ST.getGeneration() >= AMDGPUSubtarget::GFX10) {
10219 int64_t Format = ST.getGeneration() >= AMDGPUSubtarget::GFX11
10222 return (Format << 44) |
10223 (1ULL << 56) | // RESOURCE_LEVEL = 1
10224 (3ULL << 60); // OOB_SELECT = 3
10225 }
10226
10227 uint64_t RsrcDataFormat = AMDGPU::RSRC_DATA_FORMAT;
10228 if (ST.isAmdHsaOS()) {
10229 // Set ATC = 1. GFX9 doesn't have this bit.
10230 if (ST.getGeneration() <= AMDGPUSubtarget::VOLCANIC_ISLANDS)
10231 RsrcDataFormat |= (1ULL << 56);
10232
10233 // Set MTYPE = 2 (MTYPE_UC = uncached). GFX9 doesn't have this.
10234 // BTW, it disables TC L2 and therefore decreases performance.
10235 if (ST.getGeneration() == AMDGPUSubtarget::VOLCANIC_ISLANDS)
10236 RsrcDataFormat |= (2ULL << 59);
10237 }
10238
10239 return RsrcDataFormat;
10240}
10241
10243 uint64_t Rsrc23 = getDefaultRsrcDataFormat() |
10245 0xffffffff; // Size;
10246
10247 // GFX9 doesn't have ELEMENT_SIZE.
10248 if (ST.getGeneration() <= AMDGPUSubtarget::VOLCANIC_ISLANDS) {
10249 uint64_t EltSizeValue = Log2_32(ST.getMaxPrivateElementSize(true)) - 1;
10250 Rsrc23 |= EltSizeValue << AMDGPU::RSRC_ELEMENT_SIZE_SHIFT;
10251 }
10252
10253 // IndexStride = 64 / 32.
10254 uint64_t IndexStride = ST.isWave64() ? 3 : 2;
10255 Rsrc23 |= IndexStride << AMDGPU::RSRC_INDEX_STRIDE_SHIFT;
10256
10257 // If TID_ENABLE is set, DATA_FORMAT specifies stride bits [14:17].
10258 // Clear them unless we want a huge stride.
10259 if (ST.getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS &&
10260 ST.getGeneration() <= AMDGPUSubtarget::GFX9)
10261 Rsrc23 &= ~AMDGPU::RSRC_DATA_FORMAT;
10262
10263 return Rsrc23;
10264}
10265
10267 unsigned Opc = MI.getOpcode();
10268
10269 return isSMRD(Opc);
10270}
10271
10273 return get(Opc).mayLoad() &&
10274 (isMUBUF(Opc) || isMTBUF(Opc) || isMIMG(Opc) || isFLAT(Opc));
10275}
10276
10278 TypeSize &MemBytes) const {
10279 const MachineOperand *Addr = getNamedOperand(MI, AMDGPU::OpName::vaddr);
10280 if (!Addr || !Addr->isFI())
10281 return Register();
10282
10283 assert(!MI.memoperands_empty() &&
10284 (*MI.memoperands_begin())->getAddrSpace() == AMDGPUAS::PRIVATE_ADDRESS);
10285
10286 FrameIndex = Addr->getIndex();
10287
10288 int VDataIdx =
10289 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::vdata);
10290 MemBytes = TypeSize::getFixed(getOpSize(MI.getOpcode(), VDataIdx));
10291 return MI.getOperand(VDataIdx).getReg();
10292}
10293
10295 TypeSize &MemBytes) const {
10296 const MachineOperand *Addr = getNamedOperand(MI, AMDGPU::OpName::addr);
10297 assert(Addr && Addr->isFI());
10298 FrameIndex = Addr->getIndex();
10299
10300 int DataIdx =
10301 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::data);
10302 MemBytes = TypeSize::getFixed(getOpSize(MI.getOpcode(), DataIdx));
10303 return MI.getOperand(DataIdx).getReg();
10304}
10305
10307 int &FrameIndex,
10308 TypeSize &MemBytes) const {
10309 if (!MI.mayLoad())
10310 return Register();
10311
10312 if (isMUBUF(MI) || isVGPRSpill(MI))
10313 return isStackAccess(MI, FrameIndex, MemBytes);
10314
10315 if (isSGPRSpill(MI))
10316 return isSGPRStackAccess(MI, FrameIndex, MemBytes);
10317
10318 return Register();
10319}
10320
10322 int &FrameIndex,
10323 TypeSize &MemBytes) const {
10324 if (!MI.mayStore())
10325 return Register();
10326
10327 if (isMUBUF(MI) || isVGPRSpill(MI))
10328 return isStackAccess(MI, FrameIndex, MemBytes);
10329
10330 if (isSGPRSpill(MI))
10331 return isSGPRStackAccess(MI, FrameIndex, MemBytes);
10332
10333 return Register();
10334}
10335
10337 unsigned Opc = MI.getOpcode();
10339 unsigned DescSize = Desc.getSize();
10340
10341 // If we have a definitive size, we can use it. Otherwise we need to inspect
10342 // the operands to know the size.
10343 if (isFixedSize(MI)) {
10344 unsigned Size = DescSize;
10345
10346 // If we hit the buggy offset, an extra nop will be inserted in MC so
10347 // estimate the worst case.
10348 if (MI.isBranch() && ST.hasOffset3fBug())
10349 Size += 4;
10350
10351 return Size;
10352 }
10353
10354 // Instructions may have a 32-bit literal encoded after them. Check
10355 // operands that could ever be literals.
10356 if (isVALU(MI, /*AllowLDSDMA=*/false) || isSALU(MI)) {
10357 if (isDPP(MI))
10358 return DescSize;
10359 bool HasLiteral = false;
10360 unsigned LiteralSize = 4;
10361 for (int I = 0, E = MI.getNumExplicitOperands(); I != E; ++I) {
10362 const MachineOperand &Op = MI.getOperand(I);
10363 const MCOperandInfo &OpInfo = Desc.operands()[I];
10364 if (!Op.isReg() && !isInlineConstant(Op, OpInfo)) {
10365 HasLiteral = true;
10366 if (ST.has64BitLiterals()) {
10367 switch (OpInfo.OperandType) {
10368 default:
10369 break;
10372 if (!AMDGPU::isValid32BitLiteral(Op.getImm(), true))
10373 LiteralSize = 8;
10374 break;
10377 // A 32-bit literal is only valid when the value fits in BOTH signed
10378 // and unsigned 32-bit ranges [0, 2^31-1], matching the MC code
10379 // emitter's getLit64Encoding logic. This is because of the lack of
10380 // abilility to tell signedness of the literal, therefore we need to
10381 // be conservative and assume values outside this range require a
10382 // 64-bit literal encoding (8 bytes).
10383 if (!Op.isImm() || !isInt<32>(Op.getImm()) ||
10384 !isUInt<32>(Op.getImm()))
10385 LiteralSize = 8;
10386 break;
10387 }
10388 }
10389 break;
10390 }
10391 }
10392 return HasLiteral ? DescSize + LiteralSize : DescSize;
10393 }
10394
10395 // Check whether we have extra NSA words.
10396 if (isMIMG(MI)) {
10397 int VAddr0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vaddr0);
10398 if (VAddr0Idx < 0)
10399 return 8;
10400
10401 int RSrcIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::srsrc);
10402 return 8 + 4 * ((RSrcIdx - VAddr0Idx + 2) / 4);
10403 }
10404
10405 switch (Opc) {
10406 case TargetOpcode::BUNDLE:
10407 return getInstBundleSize(MI);
10408 case TargetOpcode::INLINEASM:
10409 case TargetOpcode::INLINEASM_BR: {
10410 const MachineFunction *MF = MI.getMF();
10411 const char *AsmStr = MI.getOperand(0).getSymbolName();
10412 return getInlineAsmLength(AsmStr, MF->getTarget().getMCAsmInfo(), &ST);
10413 }
10414 default:
10415 if (MI.isMetaInstruction())
10416 return 0;
10417
10418 // If D16 Pseudo inst, get correct MC code size
10419 const auto *D16Info = AMDGPU::getT16D16Helper(Opc);
10420 if (D16Info) {
10421 // Assume d16_lo/hi inst are always in same size
10422 unsigned LoInstOpcode = D16Info->LoOp;
10423 const MCInstrDesc &Desc = getMCOpcodeFromPseudo(LoInstOpcode);
10424 DescSize = Desc.getSize();
10425 }
10426
10427 // If FMA Pseudo inst, get correct MC code size
10428 if (Opc == AMDGPU::V_FMA_MIX_F16_t16 || Opc == AMDGPU::V_FMA_MIX_BF16_t16) {
10429 // All potential lowerings are the same size; arbitrarily pick one.
10430 const MCInstrDesc &Desc = getMCOpcodeFromPseudo(AMDGPU::V_FMA_MIXLO_F16);
10431 DescSize = Desc.getSize();
10432 }
10433
10434 return DescSize;
10435 }
10436}
10437
10440 if (MI.isBranch() && ST.hasOffset3fBug())
10441 return InstSizeVerifyMode::NoVerify;
10442 return InstSizeVerifyMode::ExactSize;
10443}
10444
10446 if (!isFLAT(MI))
10447 return false;
10448
10449 if (MI.memoperands_empty())
10450 return true;
10451
10452 for (const MachineMemOperand *MMO : MI.memoperands()) {
10454 return true;
10455 }
10456 return false;
10457}
10458
10461 static const std::pair<int, const char *> TargetIndices[] = {
10462 {AMDGPU::TI_CONSTDATA_START, "amdgpu-constdata-start"},
10463 {AMDGPU::TI_SCRATCH_RSRC_DWORD0, "amdgpu-scratch-rsrc-dword0"},
10464 {AMDGPU::TI_SCRATCH_RSRC_DWORD1, "amdgpu-scratch-rsrc-dword1"},
10465 {AMDGPU::TI_SCRATCH_RSRC_DWORD2, "amdgpu-scratch-rsrc-dword2"},
10466 {AMDGPU::TI_SCRATCH_RSRC_DWORD3, "amdgpu-scratch-rsrc-dword3"}};
10467 return ArrayRef(TargetIndices);
10468}
10469
10470/// This is used by the post-RA scheduler (SchedulePostRAList.cpp). The
10471/// post-RA version of misched uses CreateTargetMIHazardRecognizer.
10474 const ScheduleDAG *DAG) const {
10475 return new GCNHazardRecognizer(DAG->MF);
10476}
10477
10478/// This is the hazard recognizer used at -O0 by the PostRAHazardRecognizer
10479/// pass.
10486
10487// Called during:
10488// - pre-RA scheduling and post-RA scheduling
10491 const ScheduleDAGMI *DAG) const {
10492 // Borrowed from Arm Target
10493 // We would like to restrict this hazard recognizer to only
10494 // post-RA scheduling; we can tell that we're post-RA because we don't
10495 // track VRegLiveness.
10496 if (!DAG->hasVRegLiveness())
10497 return new GCNHazardRecognizer(DAG->MF);
10499}
10500
10501std::pair<unsigned, unsigned>
10503 return std::pair(TF & MO_MASK, TF & ~MO_MASK);
10504}
10505
10508 static const std::pair<unsigned, const char *> TargetFlags[] = {
10509 {MO_GOTPCREL, "amdgpu-gotprel"},
10510 {MO_GOTPCREL32_LO, "amdgpu-gotprel32-lo"},
10511 {MO_GOTPCREL32_HI, "amdgpu-gotprel32-hi"},
10512 {MO_GOTPCREL64, "amdgpu-gotprel64"},
10513 {MO_REL32_LO, "amdgpu-rel32-lo"},
10514 {MO_REL32_HI, "amdgpu-rel32-hi"},
10515 {MO_REL64, "amdgpu-rel64"},
10516 {MO_ABS32_LO, "amdgpu-abs32-lo"},
10517 {MO_ABS32_HI, "amdgpu-abs32-hi"},
10518 {MO_ABS64, "amdgpu-abs64"},
10519 };
10520
10521 return ArrayRef(TargetFlags);
10522}
10523
10526 static const std::pair<MachineMemOperand::Flags, const char *> TargetFlags[] =
10527 {
10528 {MONoClobber, "amdgpu-noclobber"},
10529 {MOLastUse, "amdgpu-last-use"},
10530 {MOCooperative, "amdgpu-cooperative"},
10531 {MOThreadPrivate, "amdgpu-thread-private"},
10532 };
10533
10534 return ArrayRef(TargetFlags);
10535}
10536
10538 const MachineFunction &MF) const {
10540 assert(SrcReg.isVirtual());
10541 if (MFI->checkFlag(SrcReg, AMDGPU::VirtRegFlag::WWM_REG))
10542 return AMDGPU::WWM_COPY;
10543
10544 return AMDGPU::COPY;
10545}
10546
10548 uint32_t Opcode = MI.getOpcode();
10549 // Check if it is SGPR spill or wwm-register spill Opcode.
10550 if (isSGPRSpill(Opcode) || isWWMRegSpillOpcode(Opcode))
10551 return true;
10552
10553 const MachineFunction *MF = MI.getMF();
10554 const MachineRegisterInfo &MRI = MF->getRegInfo();
10556
10557 // See if this is Liverange split instruction inserted for SGPR or
10558 // wwm-register. The implicit def inserted for wwm-registers should also be
10559 // included as they can appear at the bb begin.
10560 bool IsLRSplitInst = MI.getFlag(MachineInstr::LRSplit);
10561 if (!IsLRSplitInst && Opcode != AMDGPU::IMPLICIT_DEF)
10562 return false;
10563
10564 Register Reg = MI.getOperand(0).getReg();
10565 if (RI.isSGPRClass(RI.getRegClassForReg(MRI, Reg)))
10566 return IsLRSplitInst;
10567
10568 return MFI->isWWMReg(Reg);
10569}
10570
10572 Register Reg) const {
10573 // We need to handle instructions which may be inserted during register
10574 // allocation to handle the prolog. The initial prolog instruction may have
10575 // been separated from the start of the block by spills and copies inserted
10576 // needed by the prolog. However, the insertions for scalar registers can
10577 // always be placed at the BB top as they are independent of the exec mask
10578 // value.
10579 bool IsNullOrVectorRegister = true;
10580 if (Reg) {
10581 const MachineFunction *MF = MI.getMF();
10582 const MachineRegisterInfo &MRI = MF->getRegInfo();
10583 IsNullOrVectorRegister = !RI.isSGPRClass(RI.getRegClassForReg(MRI, Reg));
10584 }
10585
10586 return IsNullOrVectorRegister &&
10587 (canAddToBBProlog(MI) ||
10588 (!MI.isTerminator() && MI.getOpcode() != AMDGPU::COPY &&
10589 MI.modifiesRegister(AMDGPU::EXEC, &RI)));
10590}
10591
10595 const DebugLoc &DL,
10596 Register DestReg) const {
10597 if (ST.hasAddNoCarryInsts())
10598 return BuildMI(MBB, I, DL, get(AMDGPU::V_ADD_U32_e64), DestReg);
10599
10600 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
10601 Register UnusedCarry = MRI.createVirtualRegister(RI.getBoolRC());
10602 MRI.setRegAllocationHint(UnusedCarry, 0, RI.getVCC());
10603
10604 return BuildMI(MBB, I, DL, get(AMDGPU::V_ADD_CO_U32_e64), DestReg)
10605 .addReg(UnusedCarry, RegState::Define | RegState::Dead);
10606}
10607
10610 const DebugLoc &DL,
10611 Register DestReg,
10612 RegScavenger &RS) const {
10613 if (ST.hasAddNoCarryInsts())
10614 return BuildMI(MBB, I, DL, get(AMDGPU::V_ADD_U32_e32), DestReg);
10615
10616 // If available, prefer to use vcc.
10617 Register UnusedCarry = !RS.isRegUsed(AMDGPU::VCC)
10618 ? Register(RI.getVCC())
10619 : RS.scavengeRegisterBackwards(
10620 *RI.getBoolRC(), I, /* RestoreAfter */ false,
10621 0, /* AllowSpill */ false);
10622
10623 // TODO: Users need to deal with this.
10624 if (!UnusedCarry.isValid())
10625 return MachineInstrBuilder();
10626
10627 return BuildMI(MBB, I, DL, get(AMDGPU::V_ADD_CO_U32_e64), DestReg)
10628 .addReg(UnusedCarry, RegState::Define | RegState::Dead);
10629}
10630
10631bool SIInstrInfo::isKillTerminator(unsigned Opcode) {
10632 switch (Opcode) {
10633 case AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR:
10634 case AMDGPU::SI_KILL_I1_TERMINATOR:
10635 return true;
10636 default:
10637 return false;
10638 }
10639}
10640
10642 switch (Opcode) {
10643 case AMDGPU::SI_KILL_F32_COND_IMM_PSEUDO:
10644 return get(AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR);
10645 case AMDGPU::SI_KILL_I1_PSEUDO:
10646 return get(AMDGPU::SI_KILL_I1_TERMINATOR);
10647 default:
10648 llvm_unreachable("invalid opcode, expected SI_KILL_*_PSEUDO");
10649 }
10650}
10651
10653 return Imm <= getMaxMUBUFImmOffset(ST);
10654}
10655
10657 // GFX12 field is non-negative 24-bit signed byte offset.
10658 const unsigned OffsetBits =
10659 ST.getGeneration() >= AMDGPUSubtarget::GFX12 ? 23 : 12;
10660 return (1 << OffsetBits) - 1;
10661}
10662
10664 if (!ST.isWave32())
10665 return;
10666
10667 if (MI.isInlineAsm())
10668 return;
10669
10670 if (MI.getNumOperands() < MI.getDesc().getNumOperands())
10671 return;
10672
10673 for (auto &Op : MI.implicit_operands()) {
10674 if (Op.isReg() && Op.getReg() == AMDGPU::VCC)
10675 Op.setReg(AMDGPU::VCC_LO);
10676 }
10677}
10678
10680 if (!isSMRD(MI))
10681 return false;
10682
10683 // Check that it is using a buffer resource.
10684 int Idx = AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::sbase);
10685 if (Idx == -1) // e.g. s_memtime
10686 return false;
10687
10688 const int16_t RCID = getOpRegClassID(MI.getDesc().operands()[Idx]);
10689 return RI.getRegClass(RCID)->hasSubClassEq(&AMDGPU::SGPR_128RegClass);
10690}
10691
10692// Given Imm, split it into the values to put into the SOffset and ImmOffset
10693// fields in an MUBUF instruction. Return false if it is not possible (due to a
10694// hardware bug needing a workaround).
10695//
10696// The required alignment ensures that individual address components remain
10697// aligned if they are aligned to begin with. It also ensures that additional
10698// offsets within the given alignment can be added to the resulting ImmOffset.
10700 uint32_t &ImmOffset, Align Alignment) const {
10701 const uint64_t MaxOffset = SIInstrInfo::getMaxMUBUFImmOffset(ST);
10702 const uint32_t MaxImm = alignDown(MaxOffset, Alignment.value());
10703 uint32_t Overflow = 0;
10704
10705 if (Imm > MaxImm) {
10706 if (Imm <= MaxImm + 64) {
10707 // Use an SOffset inline constant for 4..64
10708 Overflow = Imm - MaxImm;
10709 Imm = MaxImm;
10710 } else {
10711 // Try to keep the same value in SOffset for adjacent loads, so that
10712 // the corresponding register contents can be re-used.
10713 //
10714 // Load values with all low-bits (except for alignment bits) set into
10715 // SOffset, so that a larger range of values can be covered using
10716 // s_movk_i32.
10717 //
10718 // Atomic operations fail to work correctly when individual address
10719 // components are unaligned, even if their sum is aligned.
10720 uint32_t High = (Imm + Alignment.value()) & ~MaxOffset;
10721 uint32_t Low = (Imm + Alignment.value()) & MaxOffset;
10722 Imm = Low;
10723 Overflow = High - Alignment.value();
10724 }
10725 }
10726
10727 if (Overflow > 0) {
10728 // There is a hardware bug in SI and CI which prevents address clamping in
10729 // MUBUF instructions from working correctly with SOffsets. The immediate
10730 // offset is unaffected.
10731 if (ST.getGeneration() <= AMDGPUSubtarget::SEA_ISLANDS)
10732 return false;
10733
10734 // It is not possible to set immediate in SOffset field on some targets.
10735 if (ST.hasRestrictedSOffset())
10736 return false;
10737 }
10738
10739 ImmOffset = Imm;
10740 SOffset = Overflow;
10741 return true;
10742}
10743
10744// Depending on the used address space and instructions, some immediate offsets
10745// are allowed and some are not.
10746// Pre-GFX12, flat instruction offsets can only be non-negative, global and
10747// scratch instruction offsets can also be negative. On GFX12, offsets can be
10748// negative for all variants.
10749//
10750// There are several bugs related to these offsets:
10751// On gfx10.1, flat instructions that go into the global address space cannot
10752// use an offset.
10753//
10754// For scratch instructions, the address can be either an SGPR or a VGPR.
10755// The following offsets can be used, depending on the architecture (x means
10756// cannot be used):
10757// +----------------------------+------+------+
10758// | Address-Mode | SGPR | VGPR |
10759// +----------------------------+------+------+
10760// | gfx9 | | |
10761// | negative, 4-aligned offset | x | ok |
10762// | negative, unaligned offset | x | ok |
10763// +----------------------------+------+------+
10764// | gfx10 | | |
10765// | negative, 4-aligned offset | ok | ok |
10766// | negative, unaligned offset | ok | x |
10767// +----------------------------+------+------+
10768// | gfx10.3 | | |
10769// | negative, 4-aligned offset | ok | ok |
10770// | negative, unaligned offset | ok | ok |
10771// +----------------------------+------+------+
10772//
10773// This function ignores the addressing mode, so if an offset cannot be used in
10774// one addressing mode, it is considered illegal.
10775bool SIInstrInfo::isLegalFLATOffset(int64_t Offset, unsigned AddrSpace,
10776 AMDGPU::FlatAddrSpace FlatVariant) const {
10777 // TODO: Should 0 be special cased?
10778 if (!ST.hasFlatInstOffsets())
10779 return false;
10780
10782 if (ST.hasFlatSegmentOffsetBug() && FlatVariant == FlatAddrSpace::FLAT &&
10783 (AddrSpace == AMDGPUAS::FLAT_ADDRESS ||
10784 AddrSpace == AMDGPUAS::GLOBAL_ADDRESS))
10785 return false;
10786
10787 if (ST.hasNegativeUnalignedScratchOffsetBug() &&
10788 FlatVariant == FlatAddrSpace::FlatScratch && Offset < 0 &&
10789 (Offset % 4) != 0) {
10790 return false;
10791 }
10792
10793 bool AllowNegative = allowNegativeFlatOffset(FlatVariant);
10794 unsigned N = AMDGPU::getNumFlatOffsetBits(ST);
10795 return isIntN(N, Offset) && (AllowNegative || Offset >= 0);
10796}
10797
10798// See comment on SIInstrInfo::isLegalFLATOffset for what is legal and what not.
10799std::pair<int64_t, int64_t>
10800SIInstrInfo::splitFlatOffset(int64_t COffsetVal, unsigned AddrSpace,
10801 AMDGPU::FlatAddrSpace FlatVariant) const {
10802 int64_t RemainderOffset = COffsetVal;
10803 int64_t ImmField = 0;
10804
10805 bool AllowNegative = allowNegativeFlatOffset(FlatVariant);
10806 const unsigned NumBits = AMDGPU::getNumFlatOffsetBits(ST) - 1;
10807
10808 if (AllowNegative) {
10809 // Use signed division by a power of two to truncate towards 0.
10810 int64_t D = 1LL << NumBits;
10811 RemainderOffset = (COffsetVal / D) * D;
10812 ImmField = COffsetVal - RemainderOffset;
10813
10814 if (ST.hasNegativeUnalignedScratchOffsetBug() &&
10815 FlatVariant == AMDGPU::FlatAddrSpace::FlatScratch && ImmField < 0 &&
10816 (ImmField % 4) != 0) {
10817 // Make ImmField a multiple of 4
10818 RemainderOffset += ImmField % 4;
10819 ImmField -= ImmField % 4;
10820 }
10821 } else if (COffsetVal >= 0) {
10822 ImmField = COffsetVal & maskTrailingOnes<uint64_t>(NumBits);
10823 RemainderOffset = COffsetVal - ImmField;
10824 }
10825
10826 assert(isLegalFLATOffset(ImmField, AddrSpace, FlatVariant));
10827 assert(RemainderOffset + ImmField == COffsetVal);
10828 return {ImmField, RemainderOffset};
10829}
10830
10832 AMDGPU::FlatAddrSpace FlatVariant) const {
10833 if (ST.hasNegativeScratchOffsetBug() &&
10835 return false;
10836
10837 return FlatVariant != AMDGPU::FlatAddrSpace::FLAT || AMDGPU::isGFX12Plus(ST);
10838}
10839
10840static unsigned subtargetEncodingFamily(const GCNSubtarget &ST) {
10841 switch (ST.getGeneration()) {
10842 default:
10843 break;
10846 return SIEncodingFamily::SI;
10848 // The GFX80 encoding family only contains buffer instructions with unpacked
10849 // D16 data; pseudoToMCOpcode falls back on VI for everything else.
10850 // TODO: remove this when we discard GFX80 encoding.
10851 return ST.hasUnpackedD16VMem() ? SIEncodingFamily::GFX80
10854 return SIEncodingFamily::VI;
10858 return ST.hasGFX11_7Insts() ? SIEncodingFamily::GFX1170
10861 return ST.hasGFX1250Insts() ? SIEncodingFamily::GFX1250
10865 }
10866 llvm_unreachable("Unknown subtarget generation!");
10867}
10868
10869bool SIInstrInfo::isAsmOnlyOpcode(int MCOp) const {
10870 switch(MCOp) {
10871 // These opcodes use indirect register addressing so
10872 // they need special handling by codegen (currently missing).
10873 // Therefore it is too risky to allow these opcodes
10874 // to be selected by dpp combiner or sdwa peepholer.
10875 case AMDGPU::V_MOVRELS_B32_dpp_gfx10:
10876 case AMDGPU::V_MOVRELS_B32_sdwa_gfx10:
10877 case AMDGPU::V_MOVRELD_B32_dpp_gfx10:
10878 case AMDGPU::V_MOVRELD_B32_sdwa_gfx10:
10879 case AMDGPU::V_MOVRELSD_B32_dpp_gfx10:
10880 case AMDGPU::V_MOVRELSD_B32_sdwa_gfx10:
10881 case AMDGPU::V_MOVRELSD_2_B32_dpp_gfx10:
10882 case AMDGPU::V_MOVRELSD_2_B32_sdwa_gfx10:
10883 return true;
10884 default:
10885 return false;
10886 }
10887}
10888
10889#define GENERATE_RENAMED_GFX9_CASES(OPCODE) \
10890 case OPCODE##_dpp: \
10891 case OPCODE##_e32: \
10892 case OPCODE##_e64: \
10893 case OPCODE##_e64_dpp: \
10894 case OPCODE##_sdwa:
10895
10896static bool isRenamedInGFX9(int Opcode) {
10897 switch (Opcode) {
10898 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_ADDC_U32)
10899 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_ADD_CO_U32)
10900 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_ADD_U32)
10901 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_SUBBREV_U32)
10902 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_SUBB_U32)
10903 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_SUBREV_CO_U32)
10904 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_SUBREV_U32)
10905 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_SUB_CO_U32)
10906 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_SUB_U32)
10907 //
10908 case AMDGPU::V_DIV_FIXUP_F16_gfx9_e64:
10909 case AMDGPU::V_DIV_FIXUP_F16_gfx9_fake16_e64:
10910 case AMDGPU::V_FMA_F16_gfx9_e64:
10911 case AMDGPU::V_FMA_F16_gfx9_fake16_e64:
10912 case AMDGPU::V_INTERP_P2_F16:
10913 case AMDGPU::V_MAD_F16_e64:
10914 case AMDGPU::V_MAD_U16_e64:
10915 case AMDGPU::V_MAD_I16_e64:
10916 return true;
10917 default:
10918 return false;
10919 }
10920}
10921
10922int SIInstrInfo::pseudoToMCOpcode(int Opcode) const {
10923 assert(Opcode == (int)SIInstrInfo::getNonSoftWaitcntOpcode(Opcode) &&
10924 "SIInsertWaitcnts should have promoted soft waitcnt instructions!");
10925
10926 unsigned Gen = subtargetEncodingFamily(ST);
10927
10928 if (ST.getGeneration() == AMDGPUSubtarget::GFX9 && isRenamedInGFX9(Opcode))
10930
10931 if (SIInstrFlags::isSDWA(get(Opcode))) {
10932 switch (ST.getGeneration()) {
10933 default:
10935 break;
10938 break;
10941 break;
10942 }
10943 }
10944
10945 if (isMAI(Opcode)) {
10946 int MFMAOp = AMDGPU::getMFMAEarlyClobberOp(Opcode);
10947 if (MFMAOp != -1)
10948 Opcode = MFMAOp;
10949 }
10950
10951 int32_t MCOp = AMDGPU::getMCOpcode(Opcode, Gen);
10952
10953 // Only buffer instructions with unpacked D16 data have a GFX80 encoding.
10954 // Anything else on such a subtarget uses the plain VI encoding.
10955 // TODO: remove this when we discard GFX80 encoding.
10956 if (MCOp == AMDGPU::INSTRUCTION_LIST_END && Gen == SIEncodingFamily::GFX80)
10958
10959 if (MCOp == AMDGPU::INSTRUCTION_LIST_END && ST.hasGFX11_7Insts())
10961
10962 if (MCOp == AMDGPU::INSTRUCTION_LIST_END && ST.hasGFX1250Insts())
10964
10965 // -1 means that Opcode is already a native instruction.
10966 if (MCOp == -1)
10967 return Opcode;
10968
10969 if (ST.hasGFX90AInsts()) {
10970 uint32_t NMCOp = AMDGPU::INSTRUCTION_LIST_END;
10971 if (ST.hasGFX940Insts())
10973 if (NMCOp == AMDGPU::INSTRUCTION_LIST_END)
10975 if (NMCOp == AMDGPU::INSTRUCTION_LIST_END)
10977 if (NMCOp != AMDGPU::INSTRUCTION_LIST_END)
10978 MCOp = NMCOp;
10979 }
10980
10981 // INSTRUCTION_LIST_END means that Opcode is a pseudo instruction that has no
10982 // encoding in the given subtarget generation.
10983 if (MCOp == AMDGPU::INSTRUCTION_LIST_END)
10984 return -1;
10985
10986 if (isAsmOnlyOpcode(MCOp))
10987 return -1;
10988
10989 return MCOp;
10990}
10991
10992static
10994 assert(RegOpnd.isReg());
10995 return RegOpnd.isUndef() ? TargetInstrInfo::RegSubRegPair() :
10996 getRegSubRegPair(RegOpnd);
10997}
10998
11001 assert(MI.isRegSequence());
11002 for (unsigned I = 0, E = (MI.getNumOperands() - 1)/ 2; I < E; ++I)
11003 if (MI.getOperand(1 + 2 * I + 1).getImm() == SubReg) {
11004 auto &RegOp = MI.getOperand(1 + 2 * I);
11005 return getRegOrUndef(RegOp);
11006 }
11008}
11009
11010// Try to find the definition of reg:subreg in subreg-manipulation pseudos
11011// Following a subreg of reg:subreg isn't supported
11014 if (!RSR.SubReg)
11015 return false;
11016 switch (MI.getOpcode()) {
11017 default: break;
11018 case AMDGPU::REG_SEQUENCE:
11019 RSR = getRegSequenceSubReg(MI, RSR.SubReg);
11020 return true;
11021 // EXTRACT_SUBREG ins't supported as this would follow a subreg of subreg
11022 case AMDGPU::INSERT_SUBREG:
11023 if (RSR.SubReg == (unsigned)MI.getOperand(3).getImm())
11024 // inserted the subreg we're looking for
11025 RSR = getRegOrUndef(MI.getOperand(2));
11026 else { // the subreg in the rest of the reg
11027 auto R1 = getRegOrUndef(MI.getOperand(1));
11028 if (R1.SubReg) // subreg of subreg isn't supported
11029 return false;
11030 RSR.Reg = R1.Reg;
11031 }
11032 return true;
11033 }
11034 return false;
11035}
11036
11038 const MachineRegisterInfo &MRI) {
11039 assert(MRI.isSSA());
11040 if (!P.Reg.isVirtual())
11041 return nullptr;
11042
11043 auto RSR = P;
11044 auto *DefInst = MRI.getVRegDef(RSR.Reg);
11045 while (auto *MI = DefInst) {
11046 DefInst = nullptr;
11047 switch (MI->getOpcode()) {
11048 case AMDGPU::COPY:
11049 case AMDGPU::V_MOV_B32_e32: {
11050 auto &Op1 = MI->getOperand(1);
11051 if (Op1.isReg() && Op1.getReg().isVirtual()) {
11052 if (Op1.isUndef())
11053 return nullptr;
11054 RSR = getRegSubRegPair(Op1);
11055 DefInst = MRI.getVRegDef(RSR.Reg);
11056 }
11057 break;
11058 }
11059 default:
11060 if (followSubRegDef(*MI, RSR)) {
11061 if (!RSR.Reg)
11062 return nullptr;
11063 DefInst = MRI.getVRegDef(RSR.Reg);
11064 }
11065 }
11066 if (!DefInst)
11067 return MI;
11068 }
11069 return nullptr;
11070}
11071
11073 Register VReg,
11074 const MachineInstr &DefMI,
11075 const MachineInstr &UseMI) {
11076 assert(MRI.isSSA() && "Must be run on SSA");
11077
11078 auto *TRI = MRI.getTargetRegisterInfo();
11079 auto *DefBB = DefMI.getParent();
11080
11081 // Don't bother searching between blocks, although it is possible this block
11082 // doesn't modify exec.
11083 if (UseMI.getParent() != DefBB)
11084 return true;
11085
11086 const int MaxInstScan = 20;
11087 int NumInst = 0;
11088
11089 // Stop scan at the use.
11090 auto E = UseMI.getIterator();
11091 for (auto I = std::next(DefMI.getIterator()); I != E; ++I) {
11092 if (I->isDebugInstr())
11093 continue;
11094
11095 if (++NumInst > MaxInstScan)
11096 return true;
11097
11098 if (I->modifiesRegister(AMDGPU::EXEC, TRI))
11099 return true;
11100 }
11101
11102 return false;
11103}
11104
11106 Register VReg,
11107 const MachineInstr &DefMI) {
11108 assert(MRI.isSSA() && "Must be run on SSA");
11109
11110 auto *TRI = MRI.getTargetRegisterInfo();
11111 auto *DefBB = DefMI.getParent();
11112
11113 const int MaxUseScan = 10;
11114 int NumUse = 0;
11115
11116 for (auto &Use : MRI.use_nodbg_operands(VReg)) {
11117 auto &UseInst = *Use.getParent();
11118 // Don't bother searching between blocks, although it is possible this block
11119 // doesn't modify exec.
11120 if (UseInst.getParent() != DefBB || UseInst.isPHI())
11121 return true;
11122
11123 if (++NumUse > MaxUseScan)
11124 return true;
11125 }
11126
11127 if (NumUse == 0)
11128 return false;
11129
11130 const int MaxInstScan = 20;
11131 int NumInst = 0;
11132
11133 // Stop scan when we have seen all the uses.
11134 for (auto I = std::next(DefMI.getIterator()); ; ++I) {
11135 assert(I != DefBB->end());
11136
11137 if (I->isDebugInstr())
11138 continue;
11139
11140 if (++NumInst > MaxInstScan)
11141 return true;
11142
11143 for (const MachineOperand &Op : I->operands()) {
11144 // We don't check reg masks here as they're used only on calls:
11145 // 1. EXEC is only considered const within one BB
11146 // 2. Call should be a terminator instruction if present in a BB
11147
11148 if (!Op.isReg())
11149 continue;
11150
11151 Register Reg = Op.getReg();
11152 if (Op.isUse()) {
11153 if (Reg == VReg && --NumUse == 0)
11154 return false;
11155 } else if (TRI->regsOverlap(Reg, AMDGPU::EXEC))
11156 return true;
11157 }
11158 }
11159}
11160
11163 const DebugLoc &DL, Register Src, Register Dst) const {
11164 auto Cur = MBB.begin();
11165 if (Cur != MBB.end())
11166 do {
11167 if (!Cur->isPHI() && Cur->readsRegister(Dst, /*TRI=*/nullptr))
11168 return BuildMI(MBB, Cur, DL, get(TargetOpcode::COPY), Dst).addReg(Src);
11169 ++Cur;
11170 } while (Cur != MBB.end() && Cur != LastPHIIt);
11171
11172 return TargetInstrInfo::createPHIDestinationCopy(MBB, LastPHIIt, DL, Src,
11173 Dst);
11174}
11175
11178 const DebugLoc &DL, Register Src, unsigned SrcSubReg, Register Dst) const {
11179 if (InsPt != MBB.end() &&
11180 (InsPt->getOpcode() == AMDGPU::SI_IF ||
11181 InsPt->getOpcode() == AMDGPU::SI_ELSE ||
11182 InsPt->getOpcode() == AMDGPU::SI_IF_BREAK) &&
11183 InsPt->definesRegister(Src, /*TRI=*/nullptr)) {
11184 InsPt++;
11185 return BuildMI(MBB, InsPt, DL,
11186 get(AMDGPU::LaneMaskConstants::get(ST).MovTermOpc), Dst)
11187 .addReg(Src, {}, SrcSubReg)
11188 .addReg(AMDGPU::EXEC, RegState::Implicit);
11189 }
11190 return TargetInstrInfo::createPHISourceCopy(MBB, InsPt, DL, Src, SrcSubReg,
11191 Dst);
11192}
11193
11194bool llvm::SIInstrInfo::isWave32() const { return ST.isWave32(); }
11195
11197 const MachineInstr &SecondMI) const {
11198 for (const auto &Use : SecondMI.all_uses()) {
11199 if (Use.isReg() && FirstMI.modifiesRegister(Use.getReg(), &RI))
11200 return true;
11201 }
11202 return false;
11203}
11204
11205/// If OpX is multicycle, anti-dependencies are not allowed.
11206/// isDPMACCInstruction was not designed for VOPD, but it is fit for the
11207/// purpose.
11209 const MachineInstr &OpX) const {
11211}
11212
11215 ArrayRef<unsigned> Ops, int FrameIndex,
11216 MachineInstr *&CopyMI, LiveIntervals *LIS,
11217 VirtRegMap *VRM) const {
11218 // This is a bit of a hack (copied from AArch64). Consider this instruction:
11219 //
11220 // %0:sreg_32 = COPY $m0
11221 //
11222 // We explicitly chose SReg_32 for the virtual register so such a copy might
11223 // be eliminated by RegisterCoalescer. However, that may not be possible, and
11224 // %0 may even spill. We can't spill $m0 normally (it would require copying to
11225 // a numbered SGPR anyway), and since it is in the SReg_32 register class,
11226 // TargetInstrInfo::foldMemoryOperand() is going to try.
11227 // A similar issue also exists with spilling and reloading $exec registers.
11228 //
11229 // To prevent that, constrain the %0 register class here.
11230 if (isFullCopyInstr(MI)) {
11231 Register DstReg = MI.getOperand(0).getReg();
11232 Register SrcReg = MI.getOperand(1).getReg();
11233 if ((DstReg.isVirtual() || SrcReg.isVirtual()) &&
11234 (DstReg.isVirtual() != SrcReg.isVirtual())) {
11235 MachineRegisterInfo &MRI = MF.getRegInfo();
11236 Register VirtReg = DstReg.isVirtual() ? DstReg : SrcReg;
11237 const TargetRegisterClass *RC = MRI.getRegClass(VirtReg);
11238 if (RC->hasSuperClassEq(&AMDGPU::SReg_32RegClass)) {
11239 MRI.constrainRegClass(VirtReg, &AMDGPU::SReg_32_XM0_XEXECRegClass);
11240 return nullptr;
11241 }
11242 if (RC->hasSuperClassEq(&AMDGPU::SReg_64RegClass)) {
11243 MRI.constrainRegClass(VirtReg, &AMDGPU::SReg_64_XEXECRegClass);
11244 return nullptr;
11245 }
11246 }
11247 }
11248
11249 return nullptr;
11250}
11251
11253 const MachineInstr &MI,
11254 unsigned *PredCost) const {
11255 if (MI.isBundle()) {
11257 MachineBasicBlock::const_instr_iterator E(MI.getParent()->instr_end());
11258 unsigned Lat = 0, Count = 0;
11259 for (++I; I != E && I->isBundledWithPred(); ++I) {
11260 ++Count;
11261 Lat = std::max(Lat, SchedModel.computeInstrLatency(&*I));
11262 }
11263 return Lat + Count - 1;
11264 }
11265
11266 return SchedModel.computeInstrLatency(&MI);
11267}
11268
11270 if (!ST.hasGFX1250VALUBlockingCycles())
11271 return 0;
11273}
11274
11275unsigned
11277 if (const auto *Entry = AMDGPU::getGFX1250BlockingCyclesInfo(MI.getOpcode()))
11278 return Entry->GFX1250BlockingCycles;
11279 return 0;
11280}
11281
11282const MachineOperand &
11284 if (const MachineOperand *CallAddrOp =
11285 getNamedOperand(MI, AMDGPU::OpName::src0))
11286 return *CallAddrOp;
11288}
11289
11292 const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
11293 unsigned Opcode = MI.getOpcode();
11294
11295 auto HandleAddrSpaceCast = [this, &MRI](const MachineInstr &MI) {
11296 Register Dst = MI.getOperand(0).getReg();
11297 Register Src = MI.getOperand(1).getReg();
11298 LLT DstTy = MRI.getType(Dst);
11299 LLT SrcTy = MRI.getType(Src);
11300 unsigned DstAS = DstTy.getAddressSpace();
11301 unsigned SrcAS = SrcTy.getAddressSpace();
11302 return SrcAS == AMDGPUAS::PRIVATE_ADDRESS &&
11303 DstAS == AMDGPUAS::FLAT_ADDRESS &&
11304 ST.hasGloballyAddressableScratch()
11307 };
11308
11309 // If the target supports globally addressable scratch, the mapping from
11310 // scratch memory to the flat aperture changes therefore an address space cast
11311 // is no longer uniform.
11312 if (Opcode == TargetOpcode::G_ADDRSPACE_CAST)
11313 return HandleAddrSpaceCast(MI);
11314
11315 if (auto *GI = dyn_cast<GIntrinsic>(&MI)) {
11316 auto IID = GI->getIntrinsicID();
11321
11322 switch (IID) {
11323 case Intrinsic::amdgcn_if:
11324 case Intrinsic::amdgcn_else:
11325 // FIXME: Uniform if second result
11326 break;
11327 }
11328
11330 }
11331
11332 // Loads from the private and flat address spaces are divergent, because
11333 // threads can execute the load instruction with the same inputs and get
11334 // different results.
11335 //
11336 // All other loads are not divergent, because if threads issue loads with the
11337 // same arguments, they will always get the same result.
11338 if (Opcode == AMDGPU::G_LOAD || Opcode == AMDGPU::G_ZEXTLOAD ||
11339 Opcode == AMDGPU::G_SEXTLOAD) {
11340 if (MI.memoperands_empty())
11341 return ValueUniformity::NeverUniform; // conservative assumption
11342
11343 if (llvm::any_of(MI.memoperands(), [](const MachineMemOperand *mmo) {
11344 return mmo->getAddrSpace() == AMDGPUAS::PRIVATE_ADDRESS ||
11345 mmo->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS;
11346 })) {
11347 // At least one MMO in a non-global address space.
11349 }
11351 }
11352
11353 if (SIInstrInfo::isGenericAtomicRMWOpcode(Opcode) ||
11354 Opcode == AMDGPU::G_ATOMIC_CMPXCHG ||
11355 Opcode == AMDGPU::G_ATOMIC_CMPXCHG_WITH_SUCCESS ||
11356 AMDGPU::isGenericAtomic(Opcode)) {
11358 }
11359
11360 // Result is computed from uniform SP and uniform wave-wide max size.
11361 if (Opcode == TargetOpcode::G_DYN_STACKALLOC)
11363
11364 if (Opcode == AMDGPU::G_AMDGPU_WHOLE_WAVE_FUNC_SETUP)
11366
11368}
11369
11371 if (!Formatter)
11372 Formatter = std::make_unique<AMDGPUMIRFormatter>(ST);
11373 return Formatter.get();
11374}
11375
11377
11378 if (isNeverUniform(MI))
11380
11381 unsigned opcode = MI.getOpcode();
11382 if (opcode == AMDGPU::V_READLANE_B32 ||
11383 opcode == AMDGPU::V_READFIRSTLANE_B32 ||
11384 opcode == AMDGPU::SI_RESTORE_S32_FROM_VGPR)
11386
11387 // If any of defs is divergent, report as NeverUniform. isUniformReg will
11388 // calculate in more detail for each def from its reg class, if available.
11389 if (MI.isInlineAsm()) {
11390 for (const MachineOperand &MO : MI.operands()) {
11391 if (!MO.isReg() || !MO.isDef())
11392 continue;
11393 const TargetRegisterClass *RC =
11394 MI.getRegClassConstraint(MO.getOperandNo(), this, &RI);
11395 if (!RC || !RI.isSGPRClass(RC))
11397 }
11398 }
11399
11400 if (isCopyInstr(MI)) {
11401 const MachineOperand &srcOp = MI.getOperand(1);
11402 if (srcOp.isReg() && srcOp.getReg().isPhysical()) {
11403 const TargetRegisterClass *regClass =
11404 RI.getPhysRegBaseClass(srcOp.getReg());
11405 return RI.isSGPRClass(regClass) ? ValueUniformity::AlwaysUniform
11407 }
11409 }
11410
11411 // GMIR handling
11412 if (MI.isPreISelOpcode())
11414
11415 // Atomics are divergent because they are executed sequentially: when an
11416 // atomic operation refers to the same address in each thread, then each
11417 // thread after the first sees the value written by the previous thread as
11418 // original value.
11419
11420 if (isAtomic(MI))
11422
11423 // Loads from the private and flat address spaces are divergent, because
11424 // threads can execute the load instruction with the same inputs and get
11425 // different results.
11426 if (isFLAT(MI) && MI.mayLoad()) {
11427 if (MI.memoperands_empty())
11428 return ValueUniformity::NeverUniform; // conservative assumption
11429
11430 if (llvm::any_of(MI.memoperands(), [](const MachineMemOperand *mmo) {
11431 return mmo->getAddrSpace() == AMDGPUAS::PRIVATE_ADDRESS ||
11432 mmo->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS;
11433 })) {
11434 // At least one MMO in a non-global address space.
11436 }
11437
11439 }
11440
11441 const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
11442 const AMDGPURegisterBankInfo *RBI = ST.getRegBankInfo();
11443
11444 // FIXME: It's conceptually broken to report this for an instruction, and not
11445 // a specific def operand. For inline asm in particular, there could be mixed
11446 // uniform and divergent results.
11447 for (unsigned I = 0, E = MI.getNumOperands(); I != E; ++I) {
11448 const MachineOperand &SrcOp = MI.getOperand(I);
11449 if (!SrcOp.isReg())
11450 continue;
11451
11452 Register Reg = SrcOp.getReg();
11453 if (!Reg || !SrcOp.readsReg())
11454 continue;
11455
11456 // If RegBank is null, this is unassigned or an unallocatable special
11457 // register, which are all scalars.
11458 const RegisterBank *RegBank = RBI->getRegBank(Reg, MRI, RI);
11459 if (RegBank && RegBank->getID() != AMDGPU::SGPRRegBankID)
11461 }
11462
11463 // TODO: Uniformity check condtions above can be rearranged for more
11464 // redability
11465
11466 // TODO: amdgcn.{ballot, [if]cmp} should be AlwaysUniform, but they are
11467 // currently turned into no-op COPYs by SelectionDAG ISel and are
11468 // therefore no longer recognizable.
11469
11471}
11472
11474 switch (MF.getFunction().getCallingConv()) {
11476 return 1;
11478 return 2;
11480 return 3;
11484 const Function &F = MF.getFunction();
11485 F.getContext().diagnose(DiagnosticInfoUnsupported(
11486 F, "ds_ordered_count unsupported for this calling conv"));
11487 [[fallthrough]];
11488 }
11491 case CallingConv::C:
11492 case CallingConv::Fast:
11493 default:
11494 // Assume other calling conventions are various compute callable functions
11495 return 0;
11496 }
11497}
11498
11500 Register &SrcReg2, int64_t &CmpMask,
11501 int64_t &CmpValue) const {
11502 if (!MI.getOperand(0).isReg() || MI.getOperand(0).getSubReg())
11503 return false;
11504
11505 switch (MI.getOpcode()) {
11506 default:
11507 break;
11508 case AMDGPU::S_CMP_EQ_U32:
11509 case AMDGPU::S_CMP_EQ_I32:
11510 case AMDGPU::S_CMP_LG_U32:
11511 case AMDGPU::S_CMP_LG_I32:
11512 case AMDGPU::S_CMP_LT_U32:
11513 case AMDGPU::S_CMP_LT_I32:
11514 case AMDGPU::S_CMP_GT_U32:
11515 case AMDGPU::S_CMP_GT_I32:
11516 case AMDGPU::S_CMP_LE_U32:
11517 case AMDGPU::S_CMP_LE_I32:
11518 case AMDGPU::S_CMP_GE_U32:
11519 case AMDGPU::S_CMP_GE_I32:
11520 case AMDGPU::S_CMP_EQ_U64:
11521 case AMDGPU::S_CMP_LG_U64:
11522 SrcReg = MI.getOperand(0).getReg();
11523 if (MI.getOperand(1).isReg()) {
11524 if (MI.getOperand(1).getSubReg())
11525 return false;
11526 SrcReg2 = MI.getOperand(1).getReg();
11527 CmpValue = 0;
11528 } else if (MI.getOperand(1).isImm()) {
11529 SrcReg2 = Register();
11530 CmpValue = MI.getOperand(1).getImm();
11531 } else {
11532 return false;
11533 }
11534 CmpMask = ~0;
11535 return true;
11536 case AMDGPU::S_CMPK_EQ_U32:
11537 case AMDGPU::S_CMPK_EQ_I32:
11538 case AMDGPU::S_CMPK_LG_U32:
11539 case AMDGPU::S_CMPK_LG_I32:
11540 case AMDGPU::S_CMPK_LT_U32:
11541 case AMDGPU::S_CMPK_LT_I32:
11542 case AMDGPU::S_CMPK_GT_U32:
11543 case AMDGPU::S_CMPK_GT_I32:
11544 case AMDGPU::S_CMPK_LE_U32:
11545 case AMDGPU::S_CMPK_LE_I32:
11546 case AMDGPU::S_CMPK_GE_U32:
11547 case AMDGPU::S_CMPK_GE_I32:
11548 SrcReg = MI.getOperand(0).getReg();
11549 SrcReg2 = Register();
11550 CmpValue = MI.getOperand(1).getImm();
11551 CmpMask = ~0;
11552 return true;
11553 }
11554
11555 return false;
11556}
11557
11559 for (MachineBasicBlock *S : MBB->successors()) {
11560 if (S->isLiveIn(AMDGPU::SCC))
11561 return false;
11562 }
11563 return true;
11564}
11565
11566// Invert all uses of SCC following SCCDef because SCCDef may be deleted and
11567// (incoming SCC) = !(SCC defined by SCCDef).
11568// Return true if all uses can be re-written, false otherwise.
11569bool SIInstrInfo::invertSCCUse(MachineInstr *SCCDef) const {
11570 MachineBasicBlock *MBB = SCCDef->getParent();
11571 SmallVector<MachineInstr *> InvertInstr;
11572 bool SCCIsDead = false;
11573
11574 // Scan instructions for SCC uses that need to be inverted until SCC is dead.
11575 constexpr unsigned ScanLimit = 12;
11576 unsigned Count = 0;
11577 for (MachineInstr &MI :
11578 make_range(std::next(MachineBasicBlock::iterator(SCCDef)), MBB->end())) {
11579 if (++Count > ScanLimit)
11580 return false;
11581 if (MI.readsRegister(AMDGPU::SCC, &RI)) {
11582 if (MI.getOpcode() == AMDGPU::S_CSELECT_B32 ||
11583 MI.getOpcode() == AMDGPU::S_CSELECT_B64 ||
11584 MI.getOpcode() == AMDGPU::S_CBRANCH_SCC0 ||
11585 MI.getOpcode() == AMDGPU::S_CBRANCH_SCC1)
11586 InvertInstr.push_back(&MI);
11587 else
11588 return false;
11589 }
11590 if (MI.definesRegister(AMDGPU::SCC, &RI)) {
11591 SCCIsDead = true;
11592 break;
11593 }
11594 }
11595 if (!SCCIsDead && isSCCDeadOnExit(MBB))
11596 SCCIsDead = true;
11597
11598 // SCC may have more uses. Can't invert all of them.
11599 if (!SCCIsDead)
11600 return false;
11601
11602 // Invert uses
11603 for (MachineInstr *MI : InvertInstr) {
11604 if (MI->getOpcode() == AMDGPU::S_CSELECT_B32 ||
11605 MI->getOpcode() == AMDGPU::S_CSELECT_B64) {
11606 swapOperands(*MI);
11607 } else if (MI->getOpcode() == AMDGPU::S_CBRANCH_SCC0 ||
11608 MI->getOpcode() == AMDGPU::S_CBRANCH_SCC1) {
11609 MI->setDesc(get(MI->getOpcode() == AMDGPU::S_CBRANCH_SCC0
11610 ? AMDGPU::S_CBRANCH_SCC1
11611 : AMDGPU::S_CBRANCH_SCC0));
11612 } else {
11613 llvm_unreachable("SCC used but no inversion handling");
11614 }
11615 }
11616 return true;
11617}
11618
11619// SCC is already valid after SCCValid.
11620// SCCRedefine will redefine SCC to the same value already available after
11621// SCCValid. If there are no intervening SCC conflicts delete SCCRedefine and
11622// update kill/dead flags if necessary.
11623bool SIInstrInfo::optimizeSCC(MachineInstr *SCCValid, MachineInstr *SCCRedefine,
11624 bool NeedInversion) const {
11625 MachineInstr *KillsSCC = nullptr;
11626 if (SCCValid->getParent() != SCCRedefine->getParent())
11627 return false;
11628 for (MachineInstr &MI : make_range(std::next(SCCValid->getIterator()),
11629 SCCRedefine->getIterator())) {
11630 if (MI.modifiesRegister(AMDGPU::SCC, &RI))
11631 return false;
11632 if (MI.killsRegister(AMDGPU::SCC, &RI))
11633 KillsSCC = &MI;
11634 }
11635 if (NeedInversion && !invertSCCUse(SCCRedefine))
11636 return false;
11637 if (MachineOperand *SccDef =
11638 SCCValid->findRegisterDefOperand(AMDGPU::SCC, /*TRI=*/nullptr))
11639 SccDef->setIsDead(false);
11640 if (KillsSCC)
11641 KillsSCC->clearRegisterKills(AMDGPU::SCC, /*TRI=*/nullptr);
11642 SCCRedefine->eraseFromParent();
11643 return true;
11644}
11645
11646/// If \p Sel is an S_CSELECT* of two different constants A and B, return them,
11647/// truncated to the width of the select.
11648static std::optional<std::pair<int64_t, int64_t>>
11650 const MachineInstr &Sel) {
11651 unsigned Opc = Sel.getOpcode();
11652 if (Opc != AMDGPU::S_CSELECT_B32 && Opc != AMDGPU::S_CSELECT_B64)
11653 return {};
11654 std::optional<int64_t> A =
11655 TII.getImmOrMaterializedImm(MRI, Sel.getOperand(1));
11656 if (!A)
11657 return {};
11658 std::optional<int64_t> B =
11659 TII.getImmOrMaterializedImm(MRI, Sel.getOperand(2));
11660 if (!B)
11661 return {};
11662 if (Opc == AMDGPU::S_CSELECT_B32) {
11663 A = Lo_32(*A);
11664 B = Lo_32(*B);
11665 }
11666 if (*A == *B)
11667 return {};
11668 return std::pair(*A, *B);
11669}
11670
11671static bool setsSCCIfResultIsZero(const MachineInstr &Def, bool &NeedInversion,
11672 unsigned &NewDefOpc) {
11673 // S_ADD_U32 X, 1 sets SCC on carryout which can only happen if result==0.
11674 // S_ADD_I32 X, 1 can be converted to S_ADD_U32 X, 1 if SCC is dead.
11675 if (Def.getOpcode() != AMDGPU::S_ADD_I32 &&
11676 Def.getOpcode() != AMDGPU::S_ADD_U32)
11677 return false;
11678 const MachineOperand &AddSrc1 = Def.getOperand(1);
11679 const MachineOperand &AddSrc2 = Def.getOperand(2);
11680 const MachineRegisterInfo &MRI = Def.getMF()->getRegInfo();
11681 const SIInstrInfo *TII = static_cast<const SIInstrInfo *>(
11682 Def.getMF()->getSubtarget().getInstrInfo());
11683
11684 auto Imm1 = TII->getImmOrMaterializedImm(MRI, AddSrc1);
11685 auto Imm2 = TII->getImmOrMaterializedImm(MRI, AddSrc2);
11686 if ((!Imm1 || *Imm1 != 1) && (!Imm2 || *Imm2 != 1))
11687 return false;
11688
11689 if (Def.getOpcode() == AMDGPU::S_ADD_I32) {
11690 const MachineOperand *SccDef =
11691 Def.findRegisterDefOperand(AMDGPU::SCC, /*TRI=*/nullptr);
11692 if (!SccDef->isDead())
11693 return false;
11694 NewDefOpc = AMDGPU::S_ADD_U32;
11695 }
11696 NeedInversion = !NeedInversion;
11697 return true;
11698}
11699
11701 Register SrcReg2, int64_t CmpMask,
11702 int64_t CmpValue,
11703 const MachineRegisterInfo *MRI) const {
11704 if (!SrcReg || SrcReg.isPhysical())
11705 return false;
11706
11707 if (SrcReg2) {
11708 auto ImmOpt = getImmOrMaterializedImm(*MRI, SrcReg2);
11709 if (!ImmOpt)
11710 return false;
11711 CmpValue = *ImmOpt;
11712 }
11713
11714 const auto optimizeCmpSelect = [&CmpInstr, SrcReg, CmpValue, MRI,
11715 this](bool NeedInversion) -> bool {
11716 MachineInstr *Def = MRI->getVRegDef(SrcReg);
11717 if (!Def)
11718 return false;
11719
11720 unsigned NewDefOpc = Def->getOpcode();
11721 if (auto Consts = getSelectConstants(*this, *MRI, *Def)) {
11722 // sX = S_CSELECT* A, B with A != B, so sX == A exactly when SCC was set.
11723 // Comparing sX with A or B recomputes SCC or its inverse:
11724 //
11725 // s_cmp_eq_* sX, A => SCC s_cmp_lg_* sX, A => !SCC
11726 // s_cmp_eq_* sX, B => !SCC s_cmp_lg_* sX, B => SCC
11727 auto [A, B] = *Consts;
11728 int64_t C = Def->getOpcode() == AMDGPU::S_CSELECT_B32 ? Lo_32(CmpValue)
11729 : CmpValue;
11730 if (C == A)
11731 NeedInversion = !NeedInversion;
11732 else if (C != B)
11733 return false;
11734 } else {
11735 // For S_OP that set SCC = DST!=0, do the transformation
11736 //
11737 // s_cmp_[lg|eq]_* (S_OP ...), 0 => (S_OP ...)
11738 //
11739 // For (S_OP ...) that set SCC = DST==0, invert NeedInversion and
11740 // do the transformation:
11741 //
11742 // s_cmp_[lg|eq]_* (S_OP ...), 0 => (S_OP ...)
11743 if (CmpValue != 0 ||
11744 (!setsSCCIfResultIsNonZero(*Def) &&
11745 !setsSCCIfResultIsZero(*Def, NeedInversion, NewDefOpc)))
11746 return false;
11747 }
11748
11749 if (!optimizeSCC(Def, &CmpInstr, NeedInversion))
11750 return false;
11751
11752 if (NewDefOpc != Def->getOpcode())
11753 Def->setDesc(get(NewDefOpc));
11754
11755 // If s_or_b32 result, sY, is unused (i.e. it is effectively a 64-bit
11756 // s_cmp_lg of a register pair) and the inputs are the hi and lo-halves of a
11757 // 64-bit select then delete s_or_b32 in the sequence:
11758 // sX = s_cselect_b64 A, B (A != B, one of them 0)
11759 // sLo = copy sX.sub0
11760 // sHi = copy sX.sub1
11761 // sY = s_or_b32 sLo, sHi
11762 if (Def->getOpcode() == AMDGPU::S_OR_B32 &&
11763 MRI->use_nodbg_empty(Def->getOperand(0).getReg())) {
11764 const MachineOperand &OrOpnd1 = Def->getOperand(1);
11765 const MachineOperand &OrOpnd2 = Def->getOperand(2);
11766 if (OrOpnd1.isReg() && OrOpnd2.isReg()) {
11767 MachineInstr *Def1 = MRI->getVRegDef(OrOpnd1.getReg());
11768 MachineInstr *Def2 = MRI->getVRegDef(OrOpnd2.getReg());
11769 if (Def1 && Def1->getOpcode() == AMDGPU::COPY && Def2 &&
11770 Def2->getOpcode() == AMDGPU::COPY && Def1->getOperand(1).isReg() &&
11771 Def2->getOperand(1).isReg() &&
11772 Def1->getOperand(1).getSubReg() == AMDGPU::sub0 &&
11773 Def2->getOperand(1).getSubReg() == AMDGPU::sub1 &&
11774 Def1->getOperand(1).getReg() == Def2->getOperand(1).getReg()) {
11775 if (MachineInstr *Select =
11776 MRI->getVRegDef(Def1->getOperand(1).getReg())) {
11777 if (auto Consts = getSelectConstants(*this, *MRI, *Select)) {
11778 auto [A, B] = *Consts;
11779 if (A == 0 || B == 0)
11780 optimizeSCC(Select, Def, /*NeedInversion=*/A == 0);
11781 }
11782 }
11783 }
11784 }
11785 }
11786 return true;
11787 };
11788
11789 const auto optimizeCmpAnd = [&CmpInstr, SrcReg, CmpValue, MRI,
11790 this](int64_t ExpectedValue, unsigned SrcSize,
11791 bool IsReversible, bool IsSigned) -> bool {
11792 // s_cmp_eq_u32 (s_and_b32 $src, 1 << n), 1 << n => s_and_b32 $src, 1 << n
11793 // s_cmp_eq_i32 (s_and_b32 $src, 1 << n), 1 << n => s_and_b32 $src, 1 << n
11794 // s_cmp_ge_u32 (s_and_b32 $src, 1 << n), 1 << n => s_and_b32 $src, 1 << n
11795 // s_cmp_ge_i32 (s_and_b32 $src, 1 << n), 1 << n => s_and_b32 $src, 1 << n
11796 // s_cmp_eq_u64 (s_and_b64 $src, 1 << n), 1 << n => s_and_b64 $src, 1 << n
11797 // s_cmp_lg_u32 (s_and_b32 $src, 1 << n), 0 => s_and_b32 $src, 1 << n
11798 // s_cmp_lg_i32 (s_and_b32 $src, 1 << n), 0 => s_and_b32 $src, 1 << n
11799 // s_cmp_gt_u32 (s_and_b32 $src, 1 << n), 0 => s_and_b32 $src, 1 << n
11800 // s_cmp_gt_i32 (s_and_b32 $src, 1 << n), 0 => s_and_b32 $src, 1 << n
11801 // s_cmp_lg_u64 (s_and_b64 $src, 1 << n), 0 => s_and_b64 $src, 1 << n
11802 //
11803 // Signed ge/gt are not used for the sign bit.
11804 //
11805 // If result of the AND is unused except in the compare:
11806 // s_and_b(32|64) $src, 1 << n => s_bitcmp1_b(32|64) $src, n
11807 //
11808 // s_cmp_eq_u32 (s_and_b32 $src, 1 << n), 0 => s_bitcmp0_b32 $src, n
11809 // s_cmp_eq_i32 (s_and_b32 $src, 1 << n), 0 => s_bitcmp0_b32 $src, n
11810 // s_cmp_eq_u64 (s_and_b64 $src, 1 << n), 0 => s_bitcmp0_b64 $src, n
11811 // s_cmp_lg_u32 (s_and_b32 $src, 1 << n), 1 << n => s_bitcmp0_b32 $src, n
11812 // s_cmp_lg_i32 (s_and_b32 $src, 1 << n), 1 << n => s_bitcmp0_b32 $src, n
11813 // s_cmp_lg_u64 (s_and_b64 $src, 1 << n), 1 << n => s_bitcmp0_b64 $src, n
11814
11815 MachineInstr *Def = MRI->getVRegDef(SrcReg);
11816 if (!Def)
11817 return false;
11818
11819 if (Def->getOpcode() != AMDGPU::S_AND_B32 &&
11820 Def->getOpcode() != AMDGPU::S_AND_B64)
11821 return false;
11822
11823 int64_t Mask;
11824 const auto isMask = [&Mask, SrcSize, MRI,
11825 this](const MachineOperand *MO) -> bool {
11826 auto ImmOpt = this->getImmOrMaterializedImm(*MRI, *MO);
11827 if (!ImmOpt)
11828 return false;
11829 Mask = *ImmOpt;
11830 Mask &= maxUIntN(SrcSize);
11831 return isPowerOf2_64(Mask);
11832 };
11833
11834 MachineOperand *SrcOp = &Def->getOperand(1);
11835 if (isMask(SrcOp))
11836 SrcOp = &Def->getOperand(2);
11837 else if (isMask(&Def->getOperand(2)))
11838 SrcOp = &Def->getOperand(1);
11839 else
11840 return false;
11841
11842 // A valid Mask is required to have a single bit set, hence a non-zero and
11843 // power-of-two value. This verifies that we will not do 64-bit shift below.
11844 assert(llvm::has_single_bit<uint64_t>(Mask) && "Invalid mask.");
11845 unsigned BitNo = llvm::countr_zero((uint64_t)Mask);
11846 if (IsSigned && BitNo == SrcSize - 1)
11847 return false;
11848
11849 ExpectedValue <<= BitNo;
11850
11851 bool IsReversedCC = false;
11852 if (CmpValue != ExpectedValue) {
11853 if (!IsReversible)
11854 return false;
11855 IsReversedCC = CmpValue == (ExpectedValue ^ Mask);
11856 if (!IsReversedCC)
11857 return false;
11858 }
11859
11860 Register DefReg = Def->getOperand(0).getReg();
11861 if (IsReversedCC && !MRI->hasOneNonDBGUse(DefReg))
11862 return false;
11863
11864 if (!optimizeSCC(Def, &CmpInstr, /*NeedInversion=*/false))
11865 return false;
11866
11867 if (!MRI->use_nodbg_empty(DefReg)) {
11868 assert(!IsReversedCC);
11869 return true;
11870 }
11871
11872 // Replace AND with unused result with a S_BITCMP.
11873 MachineBasicBlock *MBB = Def->getParent();
11874
11875 unsigned NewOpc = (SrcSize == 32) ? IsReversedCC ? AMDGPU::S_BITCMP0_B32
11876 : AMDGPU::S_BITCMP1_B32
11877 : IsReversedCC ? AMDGPU::S_BITCMP0_B64
11878 : AMDGPU::S_BITCMP1_B64;
11879
11880 BuildMI(*MBB, Def, Def->getDebugLoc(), get(NewOpc))
11881 .add(*SrcOp)
11882 .addImm(BitNo);
11883 Def->eraseFromParent();
11884
11885 return true;
11886 };
11887
11888 switch (CmpInstr.getOpcode()) {
11889 default:
11890 break;
11891 case AMDGPU::S_CMP_EQ_U32:
11892 case AMDGPU::S_CMP_EQ_I32:
11893 case AMDGPU::S_CMPK_EQ_U32:
11894 case AMDGPU::S_CMPK_EQ_I32:
11895 return optimizeCmpAnd(1, 32, true, false) ||
11896 optimizeCmpSelect(/*NeedInversion=*/true);
11897 case AMDGPU::S_CMP_GE_U32:
11898 case AMDGPU::S_CMPK_GE_U32:
11899 return optimizeCmpAnd(1, 32, false, false);
11900 case AMDGPU::S_CMP_GE_I32:
11901 case AMDGPU::S_CMPK_GE_I32:
11902 return optimizeCmpAnd(1, 32, false, true);
11903 case AMDGPU::S_CMP_EQ_U64:
11904 return optimizeCmpAnd(1, 64, true, false) ||
11905 optimizeCmpSelect(/*NeedInversion=*/true);
11906 case AMDGPU::S_CMP_LG_U32:
11907 case AMDGPU::S_CMP_LG_I32:
11908 case AMDGPU::S_CMPK_LG_U32:
11909 case AMDGPU::S_CMPK_LG_I32:
11910 return optimizeCmpAnd(0, 32, true, false) ||
11911 optimizeCmpSelect(/*NeedInversion=*/false);
11912 case AMDGPU::S_CMP_GT_U32:
11913 case AMDGPU::S_CMPK_GT_U32:
11914 return optimizeCmpAnd(0, 32, false, false);
11915 case AMDGPU::S_CMP_GT_I32:
11916 case AMDGPU::S_CMPK_GT_I32:
11917 return optimizeCmpAnd(0, 32, false, true);
11918 case AMDGPU::S_CMP_LG_U64:
11919 return optimizeCmpAnd(0, 64, true, false) ||
11920 optimizeCmpSelect(/*NeedInversion=*/false);
11921 }
11922
11923 return false;
11924}
11925
11927 AMDGPU::OpName OpName) const {
11928 if (!ST.needsAlignedVGPRs())
11929 return;
11930
11931 int OpNo = AMDGPU::getNamedOperandIdx(MI.getOpcode(), OpName);
11932 if (OpNo < 0)
11933 return;
11934 MachineOperand &Op = MI.getOperand(OpNo);
11935 if (getOpSize(MI, OpNo) > 4)
11936 return;
11937
11938 // Add implicit aligned super-reg to force alignment on the data operand.
11939 const DebugLoc &DL = MI.getDebugLoc();
11940 MachineBasicBlock *BB = MI.getParent();
11942 Register DataReg = Op.getReg();
11943 bool IsAGPR = RI.isAGPR(MRI, DataReg);
11945 IsAGPR ? &AMDGPU::AGPR_32RegClass : &AMDGPU::VGPR_32RegClass);
11946 BuildMI(*BB, MI, DL, get(AMDGPU::IMPLICIT_DEF), Undef);
11947 Register NewVR =
11948 MRI.createVirtualRegister(IsAGPR ? &AMDGPU::AReg_64_Align2RegClass
11949 : &AMDGPU::VReg_64_Align2RegClass);
11950 BuildMI(*BB, MI, DL, get(AMDGPU::REG_SEQUENCE), NewVR)
11951 .addReg(DataReg, {}, Op.getSubReg())
11952 .addImm(AMDGPU::sub0)
11953 .addReg(Undef)
11954 .addImm(AMDGPU::sub1);
11955 Op.setReg(NewVR);
11956 Op.setSubReg(AMDGPU::sub0);
11957 MI.addOperand(MachineOperand::CreateReg(NewVR, false, true));
11958}
11959
11961 if (!SchedModel.hasInstrSchedModel())
11962 return 0;
11963
11964 // The repeat rate is the throughput-limiting resource occupancy: the largest
11965 // number of cycles any written processor resource is held.
11966 const MCSchedClassDesc *SCDesc = SchedModel.resolveSchedClass(&MI);
11967 unsigned RepeatRate = 0;
11969 PI = SchedModel.getWriteProcResBegin(SCDesc),
11970 PE = SchedModel.getWriteProcResEnd(SCDesc);
11971 PI != PE; ++PI) {
11972 RepeatRate = std::max(RepeatRate, (unsigned)PI->ReleaseAtCycle);
11973 }
11974
11975 return RepeatRate;
11976}
11977
11979 if (isIGLP(*MI))
11980 return false;
11981
11983}
11984
11986 if (!isWMMA(MI) && !isSWMMAC(MI))
11987 return false;
11988
11989 if (ST.hasGFX1250Insts())
11990 return AMDGPU::getWMMAIsXDL(MI.getOpcode());
11991
11992 return true;
11993}
11994
11996 unsigned Opcode = MI.getOpcode();
11997
11998 if (AMDGPU::isGFX12Plus(ST))
11999 return isDOT(MI) || isXDLWMMA(MI);
12000
12001 if (!isMAI(MI) || isDGEMM(Opcode) ||
12002 Opcode == AMDGPU::V_ACCVGPR_WRITE_B32_e64 ||
12003 Opcode == AMDGPU::V_ACCVGPR_READ_B32_e64)
12004 return false;
12005
12006 if (!ST.hasGFX940Insts())
12007 return true;
12008
12009 return AMDGPU::getMAIIsGFX940XDL(Opcode);
12010}
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
static const TargetRegisterClass * getRegClass(const MachineInstr &MI, Register Reg)
unsigned RegSize
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
Contains the definition of a TargetInstrInfo class that is common to all AMD GPUs.
AMDGPU Register Bank Select
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
AMD GCN specific subclass of TargetSubtarget.
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static bool isUndef(const MachineInstr &MI)
TargetInstrInfo::RegSubRegPair RegSubRegPair
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
uint64_t High
uint64_t IntrinsicInst * II
#define P(N)
R600 Clause Merge
const SmallVectorImpl< MachineOperand > MachineBasicBlock * TBB
const SmallVectorImpl< MachineOperand > & Cond
This file declares the machine register scavenger class.
static cl::opt< bool > Fix16BitCopies("amdgpu-fix-16-bit-physreg-copies", cl::desc("Fix copies between 32 and 16 bit registers by extending to 32 bit"), cl::init(true), cl::ReallyHidden)
static void expandSGPRCopy(const SIInstrInfo &TII, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg, bool KillSrc, const TargetRegisterClass *RC, bool Forward)
static std::optional< std::pair< int64_t, int64_t > > getSelectConstants(const SIInstrInfo &TII, const MachineRegisterInfo &MRI, const MachineInstr &Sel)
If Sel is an S_CSELECT* of two different constants A and B, return them, truncated to the width of th...
static unsigned getNewFMAInst(const GCNSubtarget &ST, unsigned Opc)
static unsigned getIndirectSGPRWriteMovRelPseudo32(unsigned VecSize)
static bool compareMachineOp(const MachineOperand &Op0, const MachineOperand &Op1)
static bool isStride64(unsigned Opc)
static MachineBasicBlock * generateWaterFallLoop(const SIInstrInfo &TII, MachineInstr &MI, ArrayRef< MachineOperand * > ScalarOps, MachineDominatorTree *MDT, MachineBasicBlock::iterator Begin=nullptr, MachineBasicBlock::iterator End=nullptr, ArrayRef< Register > PhySGPRs={})
#define GENERATE_RENAMED_GFX9_CASES(OPCODE)
static std::tuple< unsigned, unsigned > extractRsrcPtr(const SIInstrInfo &TII, MachineInstr &MI, MachineOperand &Rsrc)
static unsigned VOP3OpIdxToSrcN(const MachineInstr &MI, unsigned OpIdx)
static bool followSubRegDef(MachineInstr &MI, TargetInstrInfo::RegSubRegPair &RSR)
static unsigned getIndirectSGPRWriteMovRelPseudo64(unsigned VecSize)
static MachineInstr * swapImmOperands(MachineInstr &MI, MachineOperand &NonRegOp1, MachineOperand &NonRegOp2)
static void copyFlagsToImplicitVCC(MachineInstr &MI, const MachineOperand &Orig)
static bool offsetsDoNotOverlap(LocationSize WidthA, int OffsetA, LocationSize WidthB, int OffsetB)
static void indirectCopyToAGPR(const SIInstrInfo &TII, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg, bool KillSrc, RegScavenger &RS, bool RegsOverlap, Register ImpUseSuperReg=Register())
Handle copying from SGPR to AGPR, or from AGPR to AGPR on GFX908.
static unsigned getWWMRegSpillSaveOpcode(unsigned Size, bool IsVectorSuperClass)
static bool memOpsHaveSameBaseOperands(ArrayRef< const MachineOperand * > BaseOps1, ArrayRef< const MachineOperand * > BaseOps2)
static unsigned getWWMRegSpillRestoreOpcode(unsigned Size, bool IsVectorSuperClass)
static unsigned getSGPRSpillSaveOpcode(unsigned Size, bool NeedsCFI)
static bool setsSCCIfResultIsZero(const MachineInstr &Def, bool &NeedInversion, unsigned &NewDefOpc)
static bool isSCCDeadOnExit(MachineBasicBlock *MBB)
static unsigned getIndirectVGPRWriteMovRelPseudoOpc(unsigned VecSize)
static unsigned subtargetEncodingFamily(const GCNSubtarget &ST)
static void preserveCondRegFlags(MachineOperand &CondReg, const MachineOperand &OrigCond)
static Register findImplicitSGPRRead(const MachineInstr &MI)
static unsigned getNewFMAAKInst(const GCNSubtarget &ST, unsigned Opc)
static cl::opt< unsigned > BranchOffsetBits("amdgpu-s-branch-bits", cl::ReallyHidden, cl::init(16), cl::desc("Restrict range of branch instructions (DEBUG)"))
static unsigned getAVSpillSaveOpcode(unsigned Size, bool NeedsCFI)
static bool memOpsHaveSameBasePtr(const MachineInstr &MI1, ArrayRef< const MachineOperand * > BaseOps1, const MachineInstr &MI2, ArrayRef< const MachineOperand * > BaseOps2)
static unsigned getSGPRSpillRestoreOpcode(unsigned Size)
static bool isRegOrFI(const MachineOperand &MO)
static bool isVCmp(const SIInstrInfo &TII, const MachineInstr &MI)
Return true if MI is a VALU comparison, i.e.
static unsigned getVGPRSpillSaveOpcode(unsigned Size, bool NeedsCFI)
static constexpr AMDGPU::OpName ModifierOpNames[]
static void reportIllegalCopy(const SIInstrInfo *TII, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg, bool KillSrc, const char *Msg="illegal VGPR to SGPR copy")
static MachineInstr * swapRegAndNonRegOperand(MachineInstr &MI, MachineOperand &RegOp, MachineOperand &NonRegOp)
static bool shouldReadExec(const MachineInstr &MI)
static unsigned getNewFMAMKInst(const GCNSubtarget &ST, unsigned Opc)
static bool isRenamedInGFX9(int Opcode)
static TargetInstrInfo::RegSubRegPair getRegOrUndef(const MachineOperand &RegOpnd)
static std::tuple< unsigned, unsigned, unsigned > splitGlobalAddressRelocFlags(const GCNSubtarget &ST, const MachineOperand &SrcOp)
static bool changesVGPRIndexingMode(const MachineInstr &MI)
static bool isSubRegOf(const SIRegisterInfo &TRI, const MachineOperand &SuperVec, const MachineOperand &SubReg)
static bool nodesHaveSameOperandValue(SDNode *N0, SDNode *N1, AMDGPU::OpName OpName)
Returns true if both nodes have the same value for the given operand Op, or if both nodes do not have...
static unsigned getNumOperandsNoGlue(SDNode *Node)
static bool canRemat(const MachineInstr &MI)
static unsigned getAVSpillRestoreOpcode(unsigned Size)
static void emitLoadScalarOpsFromVGPRLoop(const SIInstrInfo &TII, MachineRegisterInfo &MRI, MachineBasicBlock &PredBB, MachineBasicBlock &LoopBB, MachineBasicBlock &BodyBB, const DebugLoc &DL, ArrayRef< MachineOperand * > ScalarOps, ArrayRef< Register > PhySGPRs={})
static unsigned getVGPRSpillRestoreOpcode(unsigned Size)
Interface definition for SIInstrInfo.
bool IsDead
const char * Msg
This file contains some templates that are useful if you are working with the STL at all.
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
#define LLVM_DEBUG(...)
Definition Debug.h:119
static const LaneMaskConstants & get(const GCNSubtarget &ST)
static LLVM_ABI Semantics SemanticsToEnum(const llvm::fltSemantics &Sem)
Definition APFloat.cpp:185
Class for arbitrary precision integers.
Definition APInt.h:78
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1582
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
const T & front() const
Get the first element.
Definition ArrayRef.h:144
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
This class is the base class for the comparison instructions.
Definition InstrTypes.h:728
uint64_t getZExtValue() const
Opaque handle to a cycle within a GenericCycleInfo that wraps the cycle's preorder index.
A debug info location.
Definition DebugLoc.h:126
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
Definition DenseMap.h:872
Diagnostic information for unsupported feature in backend.
void changeImmediateDominator(DomTreeNodeBase< NodeT > *N, DomTreeNodeBase< NodeT > *NewIDom)
changeImmediateDominator - This method is used to update the dominator tree information when a node's...
DomTreeNodeBase< NodeT > * addNewBlock(NodeT *BB, NodeT *DomBB)
Add a new node to the dominator tree information.
bool properlyDominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
properlyDominates - Returns true iff A dominates B and A != B.
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
void getExitingBlocks(CycleRef C, SmallVectorImpl< BlockT * > &TmpStorage) const
Return all blocks of C that have a successor outside of C.
CycleRef getParentCycle(CycleRef C) const
bool contains(CycleRef Outer, CycleRef Inner) const
Returns true iff Outer contains Inner. O(1). Non-strict.
CycleRef getCycle(const BlockT *Block) const
Find the innermost cycle containing Block.
Itinerary data supplied by a subtarget to be used by a target.
constexpr unsigned getAddressSpace() const
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LiveInterval - This class represents the liveness of a register, or stack slot.
bool hasInterval(Register Reg) const
SlotIndex getInstructionIndex(const MachineInstr &Instr) const
Returns the base index of the given instruction.
LiveInterval & getInterval(Register Reg)
LLVM_ABI bool shrinkToUses(LiveInterval *li, SmallVectorImpl< MachineInstr * > *dead=nullptr)
After removing some uses of a register, shrink its live range to just the remaining uses.
SlotIndex ReplaceMachineInstrInMaps(MachineInstr &MI, MachineInstr &NewMI)
This class represents the liveness of a register, stack slot, etc.
bool hasValue() const
static LocationSize precise(uint64_t Value)
TypeSize getValue() const
static const MCBinaryExpr * createAnd(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:347
static const MCBinaryExpr * createAShr(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:417
static const MCBinaryExpr * createSub(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:427
static LLVM_ABI const MCConstantExpr * create(int64_t Value, MCContext &Ctx, bool PrintInHex=false, unsigned SizeInBytes=0)
Definition MCExpr.cpp:212
Describe properties that are true of each instruction in the target description file.
unsigned getNumOperands() const
Return the number of declared MachineOperands for this MachineInstruction.
ArrayRef< MCOperandInfo > operands() const
unsigned getNumDefs() const
Return the number of MachineOperands that are register definitions.
unsigned getSize() const
Return the number of bytes in the encoding of this instruction, or zero if the encoding size cannot b...
ArrayRef< MCPhysReg > implicit_uses() const
Return a list of registers that are potentially read by any instance of this machine instruction.
unsigned getOpcode() const
Return the opcode number for this descriptor.
This holds information about one operand of a machine instruction, indicating the register class for ...
Definition MCInstrDesc.h:88
uint8_t OperandType
Information about the type of the operand.
int16_t RegClass
This specifies the register class enumeration of the operand if the operand is a register.
Definition MCInstrDesc.h:94
bool hasSuperClassEq(const MCRegisterClass *RC) const
Returns true if RC is a super-class of or equal to this class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
static const MCSymbolRefExpr * create(const MCSymbol *Symbol, MCContext &Ctx, SMLoc Loc=SMLoc())
Definition MCExpr.h:213
MCSymbol - Instances of this class represent a symbol name in the MC file, and MCSymbols are created ...
Definition MCSymbol.h:42
LLVM_ABI void setVariableValue(const MCExpr *Value)
Definition MCSymbol.cpp:50
Helper class for constructing bundles of MachineInstrs.
MachineBasicBlock::instr_iterator begin() const
Return an iterator to the first bundled instruction.
MIBundleBuilder & append(MachineInstr *MI)
Insert MI into MBB by appending it to the instructions in the bundle.
MIRFormater - Interface to format MIR operand based on target.
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
LLVM_ABI MCSymbol * getSymbol() const
Return the MCSymbol for this basic block.
void push_back(MachineInstr *MI)
LLVM_ABI LivenessQueryResult computeRegisterLiveness(const TargetRegisterInfo *TRI, MCRegister Reg, const_iterator Before, unsigned Neighborhood=10) const
Return whether (physical) register Reg has been defined and not killed as of just before Before.
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
MachineInstrBundleIterator< MachineInstr, true > reverse_iterator
Instructions::const_iterator const_instr_iterator
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
iterator_range< succ_iterator > successors()
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
MachineInstrBundleIterator< MachineInstr > iterator
@ LQR_Dead
Register is known to be fully dead.
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
bool isImmutableObjectIndex(int ObjectIdx) const
Returns true if the specified index corresponds to an immutable object.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
void push_back(MachineBasicBlock *MBB)
MCContext & getContext() const
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
MachineBasicBlock * CreateMachineBasicBlock(const BasicBlock *BB=nullptr, std::optional< UniqueBBID > BBID=std::nullopt)
CreateMachineInstr - Allocate a new MachineInstr.
void insert(iterator MBBI, MachineBasicBlock *MBB)
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addSym(MCSymbol *Sym, unsigned char TargetFlags=0) const
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addGlobalAddress(const GlobalValue *GV, int64_t Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
const MachineInstrBuilder & copyImplicitOps(const MachineInstr &OtherMI) const
Copy all the implicit operands from OtherMI onto this one.
const MachineInstrBuilder & addMemOperand(MachineMemOperand *MMO) const
MachineInstr * getInstr() const
If conversion operators fail, use this method to get the MachineInstr explicitly.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
bool mayLoadOrStore(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read or modify memory.
bool isCopy() const
const MachineBasicBlock * getParent() const
LLVM_ABI void addImplicitDefUseOperands(MachineFunction &MF)
Add all implicit def and use operands to this instruction.
bool isBundle() const
LLVM_ABI void addOperand(MachineFunction &MF, const MachineOperand &Op)
Add the specified operand to the instruction.
LLVM_ABI unsigned getNumExplicitOperands() const
Returns the number of non-implicit operands.
mop_range implicit_operands()
bool modifiesRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr modifies (fully define or partially define) the specified register.
bool mayLoad(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read memory.
LLVM_ABI bool hasUnmodeledSideEffects() const
Return true if this instruction has side effects that are not modeled by mayLoad / mayStore,...
void untieRegOperand(unsigned OpIdx)
Break any tie involving OpIdx.
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
LLVM_ABI void eraseFromBundle()
Unlink 'this' from its basic block and delete it.
bool hasOneMemOperand() const
Return true if this instruction has exactly one MachineMemOperand.
mop_range explicit_operands()
LLVM_ABI void tieOperands(unsigned DefIdx, unsigned UseIdx)
Add a tie between the register operands at DefIdx and UseIdx.
mmo_iterator memoperands_begin() const
Access to memory operands of the instruction.
LLVM_ABI bool hasOrderedMemoryRef() const
Return true if this instruction may have an ordered or volatile memory reference, or if the informati...
LLVM_ABI const MachineFunction * getMF() const
Return the function that contains the basic block that this instruction belongs to.
ArrayRef< MachineMemOperand * > memoperands() const
Access to memory operands of the instruction.
bool mayStore(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly modify memory.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
bool isMoveImmediate(QueryType Type=IgnoreBundle) const
Return true if this instruction is a move immediate (including conditional moves) instruction.
LLVM_ABI void removeOperand(unsigned OpNo)
Erase an operand from an instruction, leaving it with one fewer operand than it started with.
filtered_mop_range all_uses()
Returns an iterator range over all operands that are (explicit or implicit) register uses.
LLVM_ABI void setPostInstrSymbol(MachineFunction &MF, MCSymbol *Symbol)
Set a symbol that will be emitted just after the instruction itself.
LLVM_ABI void clearRegisterKills(Register Reg, const TargetRegisterInfo *RegInfo)
Clear all kill flags affecting Reg.
const MachineOperand & getOperand(unsigned i) const
uint32_t getFlags() const
Return the MI flags bitvector.
LLVM_ABI int findRegisterDefOperandIdx(Register Reg, const TargetRegisterInfo *TRI, bool isDead=false, bool Overlap=false) const
Returns the operand index that is a def of the specified register or -1 if it is not found.
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
MachineOperand * findRegisterDefOperand(Register Reg, const TargetRegisterInfo *TRI, bool isDead=false, bool Overlap=false)
Wrapper for findRegisterDefOperandIdx, it returns a pointer to the MachineOperand rather than an inde...
A description of a memory reference used in the backend.
unsigned getAddrSpace() const
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
MachineOperand class - Representation of each machine instruction operand.
void setSubReg(unsigned subReg)
unsigned getSubReg() const
LLVM_ABI unsigned getOperandNo() const
Returns the index of this operand in the instruction that it belongs to.
const GlobalValue * getGlobal() const
LLVM_ABI void ChangeToFrameIndex(int Idx, unsigned TargetFlags=0)
Replace this operand with a frame index.
void setImm(int64_t immVal)
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
void setIsDead(bool Val=true)
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
LLVM_ABI void ChangeToImmediate(int64_t ImmVal, unsigned TargetFlags=0)
ChangeToImmediate - Replace this operand with a new immediate operand of the specified value.
LLVM_ABI void ChangeToGA(const GlobalValue *GV, int64_t Offset, unsigned TargetFlags=0)
ChangeToGA - Replace this operand with a new global address operand.
void setIsKill(bool Val=true)
LLVM_ABI void ChangeToRegister(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isDebug=false)
ChangeToRegister - Replace this operand with a new register operand of the specified value.
void setOffset(int64_t Offset)
unsigned getTargetFlags() const
static MachineOperand CreateImm(int64_t Val)
bool isGlobal() const
isGlobal - Tests if this is a MO_GlobalAddress operand.
MachineOperandType getType() const
getType - Returns the MachineOperandType for this operand.
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
bool isTargetIndex() const
isTargetIndex - Tests if this is a MO_TargetIndex operand.
void setTargetFlags(unsigned F)
bool isFI() const
isFI - Tests if this is a MO_FrameIndex operand.
LLVM_ABI bool isIdenticalTo(const MachineOperand &Other) const
Returns true if this operand is identical to the specified operand except for liveness related flags ...
@ MO_Immediate
Immediate operand.
@ MO_Register
Register operand.
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
int64_t getOffset() const
Return the offset from the symbol in this operand.
bool isFPImm() const
isFPImm - Tests if this is a MO_FPImmediate operand.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
iterator_range< use_nodbg_iterator > use_nodbg_operands(Register Reg) const
bool use_nodbg_empty(Register RegNo) const
use_nodbg_empty - Return true if there are no non-Debug instructions using the specified register.
LLVM_ABI void moveOperands(MachineOperand *Dst, MachineOperand *Src, unsigned NumOps)
Move NumOps operands from Src to Dst, updating use-def lists as needed.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
bool reservedRegsFrozen() const
reservedRegsFrozen - Returns true after freezeReservedRegs() was called to ensure the set of reserved...
LLVM_ABI void clearVirtRegs()
clearVirtRegs - Remove all virtual registers (after physreg assignment).
void setRegAllocationHint(Register VReg, unsigned Type, Register PrefReg)
setRegAllocationHint - Specify a register allocation hint for the specified virtual register.
const MachineFunction & getMF() const
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
void setSimpleHint(Register VReg, Register PrefReg)
Specify the preferred (target independent) register allocation hint for the specified virtual registe...
const TargetRegisterInfo * getTargetRegisterInfo() const
LLVM_ABI bool isConstantPhysReg(MCRegister PhysReg) const
Returns true if PhysReg is unallocatable and constant throughout the function.
LLVM_ABI Register cloneVirtualRegister(Register VReg, StringRef Name="")
Create and return a new virtual register in the function with the same attributes as the given regist...
LLVM_ABI const TargetRegisterClass * constrainRegClass(Register Reg, const TargetRegisterClass *RC, unsigned MinNumRegs=0)
constrainRegClass - Constrain the register class of the specified virtual register to be a common sub...
iterator_range< use_iterator > use_operands(Register Reg) const
LLVM_ABI void removeRegOperandFromUseList(MachineOperand *MO)
Remove MO from its use-def list.
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
LLVM_ABI void addRegOperandToUseList(MachineOperand *MO)
Add MO to the linked list of operands for its register.
LLVM_ABI LLVM_READONLY MachineInstr * getUniqueVRegDef(Register Reg) const
getUniqueVRegDef - Return the unique machine instr that defines the specified virtual register or nul...
const RegisterBank & getRegBank(unsigned ID)
Get the register bank identified by ID.
This class implements the register bank concept.
unsigned getID() const
Get the identifier of this register bank.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
MCRegister asMCReg() const
Utility to check-convert this value to a MCRegister.
Definition Register.h:107
constexpr bool isValid() const
Definition Register.h:112
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
Definition Register.h:79
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Definition Register.h:83
Represents one node in the SelectionDAG.
bool isMachineOpcode() const
Test if this node has a post-isel opcode, directly corresponding to a MachineInstr opcode.
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
unsigned getMachineOpcode() const
This may only be called if isMachineOpcode returns true.
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
bool isLegalMUBUFImmOffset(unsigned Imm) const
bool isInlineConstant(const APInt &Imm) const
void legalizeOperandsVOP3(MachineRegisterInfo &MRI, MachineInstr &MI) const
Fix operands in MI to satisfy constant bus requirements.
bool canAddToBBProlog(const MachineInstr &MI) const
static bool isDS(const MachineInstr &MI)
MachineBasicBlock * legalizeOperands(MachineInstr &MI, MachineDominatorTree *MDT=nullptr) const
Legalize all operands in this instruction.
bool areLoadsFromSameBasePtr(SDNode *Load0, SDNode *Load1, int64_t &Offset0, int64_t &Offset1) const override
unsigned getLiveRangeSplitOpcode(Register Reg, const MachineFunction &MF) const override
bool getMemOperandsWithOffsetWidth(const MachineInstr &LdSt, SmallVectorImpl< const MachineOperand * > &BaseOps, int64_t &Offset, bool &OffsetIsScalable, LocationSize &Width, const TargetRegisterInfo *TRI) const final
unsigned getInstSizeInBytes(const MachineInstr &MI) const override
static bool isNeverUniform(const MachineInstr &MI)
bool isXDLWMMA(const MachineInstr &MI) const
bool isBasicBlockPrologue(const MachineInstr &MI, Register Reg=Register()) const override
uint64_t getDefaultRsrcDataFormat() const
static bool isSOPP(const MachineInstr &MI)
bool mayAccessScratch(const MachineInstr &MI) const
bool isIGLP(unsigned Opcode) const
static bool isFLATScratch(const MachineInstr &MI)
bool isLegalFLATOffset(int64_t Offset, unsigned AddrSpace, AMDGPU::FlatAddrSpace FlatVariant) const
Returns if Offset is legal for the subtarget as the offset to a FLAT encoded instruction with the giv...
const MCInstrDesc & getIndirectRegWriteMovRelPseudo(unsigned VecSize, unsigned EltSize, bool IsSGPR) const
MachineInstrBuilder getAddNoCarry(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register DestReg) const
Return a partially built integer add instruction without carry.
bool mayAccessFlatAddressSpace(const MachineInstr &MI) const
bool shouldScheduleLoadsNear(SDNode *Load0, SDNode *Load1, int64_t Offset0, int64_t Offset1, unsigned NumLoads) const override
bool splitMUBUFOffset(uint32_t Imm, uint32_t &SOffset, uint32_t &ImmOffset, Align Alignment=Align(4)) const
bool isIgnorableUse(const MachineInstr &MI, unsigned OpIdx) const override
ArrayRef< std::pair< unsigned, const char * > > getSerializableDirectMachineOperandTargetFlags() const override
void moveToVALU(SIInstrWorklist &Worklist, MachineDominatorTree *MDT) const
Replace the instructions opcode with the equivalent VALU opcode.
static bool isSMRD(const MachineInstr &MI)
unsigned getGFX1250BlockingCyclesTable(const MachineInstr &MI) const
GFX1250 blocking-cycles table lookup with no occupancy subtarget gate.
void restoreExec(MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, Register Reg, SlotIndexes *Indexes=nullptr) const
void storeRegToStackSlotCFI(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register SrcReg, bool isKill, int FrameIndex, const TargetRegisterClass *RC) const
bool usesConstantBus(const MachineRegisterInfo &MRI, const MachineOperand &MO, const MCOperandInfo &OpInfo) const
Returns true if this operand uses the constant bus.
static unsigned getMaxMUBUFImmOffset(const GCNSubtarget &ST)
static unsigned getFoldableCopySrcIdx(const MachineInstr &MI)
unsigned getOpSize(uint32_t Opcode, unsigned OpNo) const
Return the size in bytes of the operand OpNo on the given.
void legalizeOperandsFLAT(MachineRegisterInfo &MRI, MachineInstr &MI) const
bool optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg, Register SrcReg2, int64_t CmpMask, int64_t CmpValue, const MachineRegisterInfo *MRI) const override
static std::optional< int64_t > extractSubregFromImm(int64_t ImmVal, unsigned SubRegIndex)
Return the extracted immediate value in a subregister use from a constant materialized in a super reg...
Register isStoreToStackSlot(const MachineInstr &MI, int &FrameIndex) const override
static bool isMTBUF(const MachineInstr &MI)
const MCInstrDesc & getIndirectGPRIDXPseudo(unsigned VecSize, bool IsIndirectSrc) const
static bool isDGEMM(unsigned Opcode)
static bool isEXP(const MachineInstr &MI)
static bool isSALU(const MachineInstr &MI)
static bool setsSCCIfResultIsNonZero(const MachineInstr &MI)
const MIRFormatter * getMIRFormatter() const override
static bool isXcntDrain(const MachineInstr &MI)
True if MI implicitly drains XCNT.
void legalizeGenericOperand(MachineBasicBlock &InsertMBB, MachineBasicBlock::iterator I, const TargetRegisterClass *DstRC, MachineOperand &Op, MachineRegisterInfo &MRI, const DebugLoc &DL) const
MachineInstr * buildShrunkInst(MachineInstr &MI, unsigned NewOpcode) const
static bool isVOP2(const MachineInstr &MI)
bool analyzeBranch(MachineBasicBlock &MBB, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, SmallVectorImpl< MachineOperand > &Cond, bool AllowModify=false) const override
static bool isSDWA(const MachineInstr &MI)
const MCInstrDesc & getKillTerminatorFromPseudo(unsigned Opcode) const
void insertNoops(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, unsigned Quantity) const override
static bool isGather4(const MachineInstr &MI)
MachineInstr * getWholeWaveFunctionSetup(MachineFunction &MF) const
bool isLegalVSrcOperand(const MachineRegisterInfo &MRI, const MCOperandInfo &OpInfo, const MachineOperand &MO) const
Check if MO would be a valid operand for the given operand definition OpInfo.
static bool isDOT(const MachineInstr &MI)
std::unique_ptr< PipelinerLoopInfo > analyzeLoopForPipelining(MachineBasicBlock *LoopBB) const override
InstSizeVerifyMode getInstSizeVerifyMode(const MachineInstr &MI) const override
MachineInstr * createPHISourceCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, unsigned SrcSubReg, Register Dst) const override
bool hasModifiers(unsigned Opcode) const
Return true if this instruction has any modifiers.
bool shouldClusterMemOps(ArrayRef< const MachineOperand * > BaseOps1, int64_t Offset1, bool OffsetIsScalable1, ArrayRef< const MachineOperand * > BaseOps2, int64_t Offset2, bool OffsetIsScalable2, unsigned ClusterSize, unsigned NumBytes) const override
static bool isSWMMAC(const MachineInstr &MI)
ScheduleHazardRecognizer * CreateTargetMIHazardRecognizer(const InstrItineraryData *II, const ScheduleDAGMI *DAG) const override
bool isWave32() const
bool isHighLatencyDef(int Opc) const override
void legalizeOpWithMove(MachineInstr &MI, unsigned OpIdx) const
Legalize the OpIndex operand of this instruction by inserting a MOV.
bool reverseBranchCondition(SmallVectorImpl< MachineOperand > &Cond) const override
static bool isVOPC(const MachineInstr &MI)
void removeModOperands(MachineInstr &MI) const
unsigned getRepeatRate(const MachineInstr &MI) const
Get the repeat rate for a VALU instruction from the scheduling model.
unsigned getVectorRegSpillRestoreOpcode(Register Reg, const TargetRegisterClass *RC, unsigned Size, const SIMachineFunctionInfo &MFI) const
bool isLegalSingleSGPRReadInstOperand(const MachineRegisterInfo &MRI, const MachineInstr &MI, unsigned SrcN, const MachineOperand *MO=nullptr) const
Check if MO would be a legal operand for a single-SGPR-read instruction.
bool isXDL(const MachineInstr &MI) const
Register isStackAccess(const MachineInstr &MI, int &FrameIndex, TypeSize &MemBytes) const
static bool isVIMAGE(const MachineInstr &MI)
void enforceOperandRCAlignment(MachineInstr &MI, AMDGPU::OpName OpName) const
static bool isSOP2(const MachineInstr &MI)
static bool isGWS(const MachineInstr &MI)
bool hasRAWDependency(const MachineInstr &FirstMI, const MachineInstr &SecondMI) const
bool isLegalAV64PseudoImm(uint64_t Imm) const
Check if this immediate value can be used for AV_MOV_B64_IMM_PSEUDO.
bool isNeverCoissue(MachineInstr &MI) const
static bool isBUF(const MachineInstr &MI)
bool isNonCommutableDPP(const MachineInstr &MI) const
void handleCopyToPhysHelper(SIInstrWorklist &Worklist, Register DstReg, MachineInstr &Inst, MachineRegisterInfo &MRI, DenseMap< MachineInstr *, V2PhysSCopyInfo > &WaterFalls, DenseMap< MachineInstr *, bool > &V2SPhyCopiesToErase) const
bool hasModifiersSet(const MachineInstr &MI, AMDGPU::OpName OpName) const
bool isLegalToSwap(const MachineInstr &MI, unsigned fromIdx, unsigned toIdx) const
static bool isFLATGlobal(const MachineInstr &MI)
MachineInstr * foldMemoryOperandImpl(MachineFunction &MF, MachineInstr &MI, ArrayRef< unsigned > Ops, int FrameIndex, MachineInstr *&CopyMI, LiveIntervals *LIS=nullptr, VirtRegMap *VRM=nullptr) const override
bool isGlobalMemoryObject(const MachineInstr *MI) const override
static bool isVSAMPLE(const MachineInstr &MI)
bool isBufferSMRD(const MachineInstr &MI) const
static bool isKillTerminator(unsigned Opcode)
bool isVOPDAntidependencyAllowed(const MachineInstr &MI) const
If OpX is multicycle, anti-dependencies are not allowed.
bool findCommutedOpIndices(const MachineInstr &MI, unsigned &SrcOpIdx0, unsigned &SrcOpIdx1) const override
void insertScratchExecCopy(MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, Register Reg, bool IsSCCLive, SlotIndexes *Indexes=nullptr) const
bool hasVALU32BitEncoding(unsigned Opcode) const
Return true if this 64-bit VALU instruction has a 32-bit encoding.
unsigned getBlockingCycles(const MachineInstr &MI) const
unsigned getMovOpcode(const TargetRegisterClass *DstRC) const
Register isSGPRStackAccess(const MachineInstr &MI, int &FrameIndex, TypeSize &MemBytes) const
unsigned buildExtractSubReg(MachineBasicBlock::iterator MI, MachineRegisterInfo &MRI, const MachineOperand &SuperReg, const TargetRegisterClass *SuperRC, unsigned SubIdx, const TargetRegisterClass *SubRC) const
void legalizeOperandsVOP2(MachineRegisterInfo &MRI, MachineInstr &MI) const
Legalize operands in MI by either commuting it or inserting a copy of src1.
static bool isVPermPk16(unsigned Opcode)
static bool isVALU(const MachineInstr &MI, bool AllowLDSDMA)
bool foldImmediate(MachineInstr &UseMI, MachineInstr &DefMI, Register Reg, MachineRegisterInfo *MRI) const final
static bool isTRANS(const MachineInstr &MI)
static bool isImage(const MachineInstr &MI)
static bool isSOPK(const MachineInstr &MI)
const TargetRegisterClass * getOpRegClass(const MachineInstr &MI, unsigned OpNo) const
Return the correct register class for OpNo.
MachineBasicBlock * insertSimulatedTrap(MachineRegisterInfo &MRI, MachineBasicBlock &MBB, MachineInstr &MI, const DebugLoc &DL) const
Build instructions that simulate the behavior of a s_trap 2 instructions for hardware (namely,...
static unsigned getNonSoftWaitcntOpcode(unsigned Opcode)
static unsigned getDSShaderTypeValue(const MachineFunction &MF)
static bool isFoldableCopy(const MachineInstr &MI)
static bool isMUBUF(const MachineInstr &MI)
bool expandPostRAPseudo(MachineInstr &MI) const override
bool analyzeCompare(const MachineInstr &MI, Register &SrcReg, Register &SrcReg2, int64_t &CmpMask, int64_t &CmpValue) const override
void createWaterFallForSiCall(MachineInstr *MI, MachineDominatorTree *MDT, ArrayRef< MachineOperand * > ScalarOps, ArrayRef< Register > PhySGPRs={}) const
Wrapper function for generating waterfall for instruction MI This function take into consideration of...
void loadRegFromStackSlot(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg, int FrameIndex, const TargetRegisterClass *RC, Register VReg, unsigned SubReg=0, MachineInstr::MIFlag Flags=MachineInstr::NoFlags) const override
static bool isSegmentSpecificFLAT(const MachineInstr &MI)
bool isReMaterializableImpl(const MachineInstr &MI) const override
static bool isVOP3(const MCInstrDesc &Desc)
Register isLoadFromStackSlot(const MachineInstr &MI, int &FrameIndex) const override
bool physRegUsesConstantBus(const MachineOperand &Reg) const
void insertSelect(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register DstReg, ArrayRef< MachineOperand > Cond, Register TrueReg, Register FalseReg) const override
bool mayAccessVMEMThroughFlat(const MachineInstr &MI) const
static bool isDPP(const MachineInstr &MI)
bool analyzeBranchImpl(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, SmallVectorImpl< MachineOperand > &Cond, bool AllowModify) const
static bool isMFMA(const MachineInstr &MI)
bool isLowLatencyInstruction(const MachineInstr &MI) const
std::optional< DestSourcePair > isCopyInstrImpl(const MachineInstr &MI) const override
If the specific machine instruction is a instruction that moves/copies value from one register to ano...
void mutateAndCleanupImplicit(MachineInstr &MI, const MCInstrDesc &NewDesc) const
ValueUniformity getGenericValueUniformity(const MachineInstr &MI) const
static bool isMAI(const MCInstrDesc &Desc)
static bool isSrc1DPPRevOpcode(const GCNSubtarget &ST, uint32_t Opcode)
void reMaterialize(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg, unsigned SubIdx, const MachineInstr &Orig, LaneBitmask UsedLanes=LaneBitmask::getAll()) const override
static bool usesLGKM_CNT(const MachineInstr &MI)
void legalizeOperandsVALUt16(MachineInstr &Inst, MachineRegisterInfo &MRI) const
Fix operands in Inst to fix 16bit SALU to VALU lowering.
bool isImmOperandLegal(const MCInstrDesc &InstDesc, unsigned OpNo, const MachineOperand &MO) const
bool canShrink(const MachineInstr &MI, const MachineRegisterInfo &MRI) const
const MachineOperand & getCalleeOperand(const MachineInstr &MI) const override
bool isAsmOnlyOpcode(int MCOp) const
Check if this instruction should only be used by assembler.
bool isAlwaysGDS(uint32_t Opcode) const
static bool isVGPRSpill(const MachineInstr &MI)
ScheduleHazardRecognizer * CreateTargetPostRAHazardRecognizer(const InstrItineraryData *II, const ScheduleDAG *DAG) const override
This is used by the post-RA scheduler (SchedulePostRAList.cpp).
bool verifyInstruction(const MachineInstr &MI, StringRef &ErrInfo) const override
unsigned getInstrLatency(const InstrItineraryData *ItinData, const MachineInstr &MI, unsigned *PredCost=nullptr) const override
unsigned getVectorRegSpillSaveOpcode(Register Reg, const TargetRegisterClass *RC, unsigned Size, const SIMachineFunctionInfo &MFI, bool NeedsCFI) const
int64_t getNamedImmOperand(const MachineInstr &MI, AMDGPU::OpName OperandName) const
Get required immediate operand.
ArrayRef< std::pair< int, const char * > > getSerializableTargetIndices() const override
bool regUsesConstantBus(const MachineOperand &Reg, const MachineRegisterInfo &MRI) const
static bool isMIMG(const MachineInstr &MI)
MachineOperand buildExtractSubRegOrImm(MachineBasicBlock::iterator MI, MachineRegisterInfo &MRI, const MachineOperand &SuperReg, const TargetRegisterClass *SuperRC, unsigned SubIdx, const TargetRegisterClass *SubRC) const
bool isSchedulingBoundary(const MachineInstr &MI, const MachineBasicBlock *MBB, const MachineFunction &MF) const override
bool isLegalRegOperand(const MachineRegisterInfo &MRI, const MCOperandInfo &OpInfo, const MachineOperand &MO) const
Check if MO (a register operand) is a legal register for the given operand description or operand ind...
static unsigned getNumWaitStates(const MachineInstr &MI)
Return the number of wait states that result from executing this instruction.
unsigned getVALUOp(const MachineInstr &MI) const
static bool modifiesModeRegister(const MachineInstr &MI)
Return true if the instruction modifies the mode register.q.
Register readlaneVGPRToSGPR(Register SrcReg, MachineInstr &UseMI, MachineRegisterInfo &MRI, const TargetRegisterClass *DstRC=nullptr) const
Copy a value from a VGPR (SrcReg) to SGPR.
bool hasDivergentBranch(const MachineBasicBlock *MBB) const
Return whether the block terminate with divergent branch.
std::pair< int64_t, int64_t > splitFlatOffset(int64_t COffsetVal, unsigned AddrSpace, AMDGPU::FlatAddrSpace FlatVariant) const
Split COffsetVal into {immediate offset field, remainder offset} values.
unsigned removeBranch(MachineBasicBlock &MBB, int *BytesRemoved=nullptr) const override
void fixImplicitOperands(MachineInstr &MI) const
bool moveFlatAddrToVGPR(MachineInstr &Inst) const
Change SADDR form of a FLAT Inst to its VADDR form if saddr operand was moved to VGPR.
void copyPhysReg(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, Register DestReg, Register SrcReg, bool KillSrc, bool RenamableDest=false, bool RenamableSrc=false) const override
bool isMaskedByExec(Register Reg, const MachineInstr &Use, const MachineRegisterInfo &MRI, unsigned Depth=0) const
Return true if Reg is a lane mask that already has 0 in every bit corresponding to a lane that is ina...
void createReadFirstLaneFromCopyToPhysReg(MachineRegisterInfo &MRI, Register DstReg, MachineInstr &Inst) const
bool swapSourceModifiers(MachineInstr &MI, MachineOperand &Src0, AMDGPU::OpName Src0OpName, MachineOperand &Src1, AMDGPU::OpName Src1OpName) const
MachineBasicBlock * getBranchDestBlock(const MachineInstr &MI) const override
bool hasUnwantedEffectsWhenEXECEmpty(const MachineInstr &MI) const
This function is used to determine if an instruction can be safely executed under EXEC = 0 without ha...
bool getConstValDefinedInReg(const MachineInstr &MI, const Register Reg, int64_t &ImmVal) const override
static bool isAtomic(const MachineInstr &MI)
bool canInsertSelect(const MachineBasicBlock &MBB, ArrayRef< MachineOperand > Cond, Register DstReg, Register TrueReg, Register FalseReg, int &CondCycles, int &TrueCycles, int &FalseCycles) const override
bool isLiteralOperandLegal(const MCInstrDesc &InstDesc, const MCOperandInfo &OpInfo) const
static bool isWWMRegSpillOpcode(uint32_t Opcode)
static bool sopkIsZext(unsigned Opcode)
static bool isSGPRSpill(const MachineInstr &MI)
static bool isWMMA(const MachineInstr &MI)
ArrayRef< std::pair< MachineMemOperand::Flags, const char * > > getSerializableMachineMemOperandTargetFlags() const override
bool mayReadEXEC(const MachineRegisterInfo &MRI, const MachineInstr &MI) const
Returns true if the instruction could potentially depend on the value of exec.
void legalizeOperandsSMRD(MachineRegisterInfo &MRI, MachineInstr &MI) const
bool isBranchOffsetInRange(unsigned BranchOpc, int64_t BrOffset) const override
unsigned insertBranch(MachineBasicBlock &MBB, MachineBasicBlock *TBB, MachineBasicBlock *FBB, ArrayRef< MachineOperand > Cond, const DebugLoc &DL, int *BytesAdded=nullptr) const override
void insertNoop(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI) const override
std::pair< MachineInstr *, MachineInstr * > expandMovDPP64(MachineInstr &MI) const
static bool isSOPC(const MachineInstr &MI)
static bool isFLAT(const MachineInstr &MI)
bool isBarrier(unsigned Opcode) const
MachineInstr * commuteInstructionImpl(MachineInstr &MI, bool NewMI, unsigned OpIdx0, unsigned OpIdx1) const override
bool mayAccessLDSThroughFlat(const MachineInstr &MI, bool TgSplit) const
int pseudoToMCOpcode(int Opcode) const
Return a target-specific opcode if Opcode is a pseudo instruction.
const MCInstrDesc & getMCOpcodeFromPseudo(unsigned Opcode) const
Return the descriptor of the target-specific machine instruction that corresponds to the specified ps...
static bool usesVM_CNT(const MachineInstr &MI)
MachineInstr * createPHIDestinationCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, Register Dst) const override
static bool isFixedSize(const MachineInstr &MI)
bool isSafeToSink(MachineInstr &MI, MachineBasicBlock *SuccToSinkTo, MachineCycleInfo *CI) const override
LLVM_READONLY int commuteOpcode(unsigned Opc) const
ValueUniformity getValueUniformity(const MachineInstr &MI) const final
uint64_t getScratchRsrcWords23() const
LLVM_READONLY MachineOperand * getNamedOperand(MachineInstr &MI, AMDGPU::OpName OperandName) const
Returns the operand named Op.
MachineInstr * convertToThreeAddress(MachineInstr &MI, LiveIntervals *LIS) const override
std::pair< unsigned, unsigned > decomposeMachineOperandsTargetFlags(unsigned TF) const override
bool areMemAccessesTriviallyDisjoint(const MachineInstr &MIa, const MachineInstr &MIb) const override
bool isOperandLegal(const MachineInstr &MI, unsigned OpIdx, const MachineOperand *MO=nullptr) const
Check if MO is a legal operand if it was the OpIdx Operand for MI.
void storeRegToStackSlot(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register SrcReg, bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg, MachineInstr::MIFlag Flags=MachineInstr::NoFlags) const override
bool allowNegativeFlatOffset(AMDGPU::FlatAddrSpace FlatVariant) const
Returns true if negative offsets are allowed for the given FlatVariant.
void moveToVALUImpl(SIInstrWorklist &Worklist, MachineDominatorTree *MDT, MachineInstr &Inst, DenseMap< MachineInstr *, V2PhysSCopyInfo > &WaterFalls, DenseMap< MachineInstr *, bool > &V2SPhyCopiesToErase) const
static bool isLDSDMA(const MachineInstr &MI)
static bool isVOP1(const MachineInstr &MI)
SIInstrInfo(const GCNSubtarget &ST)
std::optional< int64_t > getImmOrMaterializedImm(const MachineRegisterInfo &MRI, const MachineOperand &Op, MachineInstr **DefMI=nullptr) const
void insertIndirectBranch(MachineBasicBlock &MBB, MachineBasicBlock &NewDestBB, MachineBasicBlock &RestoreBB, const DebugLoc &DL, int64_t BrOffset, RegScavenger *RS) const override
bool hasAnyModifiersSet(const MachineInstr &MI) const
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
void setHasSpilledVGPRs(bool Spill=true)
bool isWWMReg(Register Reg) const
bool checkFlag(Register Reg, uint8_t Flag) const
void setHasSpilledSGPRs(bool Spill=true)
unsigned getScratchReservedForDynamicVGPRs() const
static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs=1)
ArrayRef< int16_t > getRegSplitParts(const TargetRegisterClass *RC, unsigned EltSize) const
unsigned getHWRegIndex(MCRegister Reg) const
bool isSGPRReg(const MachineRegisterInfo &MRI, Register Reg) const
unsigned getRegPressureLimit(const TargetRegisterClass *RC, MachineFunction &MF) const override
unsigned getChannelFromSubReg(unsigned SubReg) const
static bool isSGPRClass(const TargetRegisterClass *RC)
static bool isAGPRClass(const TargetRegisterClass *RC)
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
virtual bool hasVRegLiveness() const
Return true if this DAG supports VReg liveness and RegPressure.
MachineFunction & MF
Machine function.
HazardRecognizer - This determines whether or not an instruction can be issued this cycle,...
SlotIndex - An opaque wrapper around machine indexes.
Definition SlotIndexes.h:66
SlotIndex getRegSlot(bool EC=false) const
Returns the register use/def slot in the current instruction for a normal or early-clobber def.
SlotIndexes pass.
SlotIndex insertMachineInstrInMaps(MachineInstr &MI, bool Late=false)
Insert the given machine instruction into the mapping.
Implements a dense probed hash-table based set with some number of buckets stored inline.
Definition DenseSet.h:293
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
int64_t getImm() const
Register getReg() const
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Object returned by analyzeLoopForPipelining.
virtual ScheduleHazardRecognizer * CreateTargetMIHazardRecognizer(const InstrItineraryData *, const ScheduleDAGMI *DAG) const
Allocate and return a hazard recognizer to use for this target when scheduling the machine instructio...
virtual MachineInstr * createPHIDestinationCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, Register Dst) const
During PHI eleimination lets target to make necessary checks and insert the copy to the PHI destinati...
virtual const MachineOperand & getCalleeOperand(const MachineInstr &MI) const
Returns the callee operand from the given MI.
virtual void reMaterialize(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg, unsigned SubIdx, const MachineInstr &Orig, LaneBitmask UsedLanes=LaneBitmask::getAll()) const
Re-issue the specified 'original' instruction at the specific location targeting a new destination re...
virtual MachineInstr * createPHISourceCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, unsigned SrcSubReg, Register Dst) const
During PHI eleimination lets target to make necessary checks and insert the copy to the PHI destinati...
virtual MachineInstr * commuteInstructionImpl(MachineInstr &MI, bool NewMI, unsigned OpIdx1, unsigned OpIdx2) const
This method commutes the operands of the given machine instruction MI.
virtual bool isGlobalMemoryObject(const MachineInstr *MI) const
Returns true if MI is an instruction we are unable to reason about (like a call or something with unm...
virtual bool expandPostRAPseudo(MachineInstr &MI) const
This function is called for all pseudo instructions that remain after register allocation.
const MCAsmInfo & getMCAsmInfo() const
Return target specific asm information.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
const MCWriteProcResEntry * ProcResIter
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
std::pair< iterator, bool > insert(const ValueT &V)
Definition DenseSet.h:209
size_type count(const_arg_type_t< ValueT > V) const
Return 1 if the specified key is in the set, 0 otherwise.
Definition DenseSet.h:187
self_iterator getIterator()
Definition ilist_node.h:123
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ REGION_ADDRESS
Address space for region memory. (GDS)
@ LOCAL_ADDRESS
Address space for local memory.
@ FLAT_ADDRESS
Address space for flat memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
@ PRIVATE_ADDRESS
Address space for private memory.
unsigned encodeFieldSaSdst(unsigned Encoded, unsigned SaSdst)
bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi)
const uint64_t RSRC_DATA_FORMAT
bool isPKFMACF16InlineConstant(uint32_t Literal, bool IsGFX11Plus)
LLVM_READONLY const MIMGInfo * getMIMGInfo(unsigned Opc)
bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi)
bool getWMMAIsXDL(unsigned Opc)
unsigned mapWMMA2AddrTo3AddrOpcode(unsigned Opc)
bool isInlinableLiteralV2I16(uint32_t Literal)
bool isDPMACCInstruction(unsigned Opc)
bool isHi16Reg(MCRegister Reg, const MCRegisterInfo &MRI)
bool isInlinableLiteralV2BF16(uint32_t Literal)
LLVM_READONLY int32_t getCommuteRev(uint32_t Opcode)
LLVM_READONLY int32_t getCommuteOrig(uint32_t Opcode)
unsigned getNumFlatOffsetBits(const MCSubtargetInfo &ST)
For pre-GFX12 FLAT instructions the offset must be positive; MSB is ignored and forced to zero.
bool isGFX12Plus(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2F16(uint32_t Literal)
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
bool isValid32BitLiteral(uint64_t Val, bool IsFP64)
LLVM_READONLY int32_t getGlobalVaddrOp(uint32_t Opcode)
LLVM_READNONE bool isLegalDPALU_DPPControl(const MCSubtargetInfo &ST, unsigned DC)
LLVM_READONLY int32_t getMFMAEarlyClobberOp(uint32_t Opcode)
bool getMAIIsGFX940XDL(unsigned Opc)
const uint64_t RSRC_ELEMENT_SIZE_SHIFT
bool isIntrinsicAlwaysUniform(unsigned IntrID)
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
bool isPackedSingleSGPR64BitInst(unsigned Opc)
The opcode is a packed 64-bit instruction which only reads low 64 bits of a scalar operand and propag...
LLVM_READONLY int32_t getIfAddr64Inst(uint32_t Opcode)
Check if Opcode is an Addr64 opcode.
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfoByEncoding(uint8_t DimEnc)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
const uint64_t RSRC_TID_ENABLE
LLVM_READONLY int32_t getVOPe32(uint32_t Opcode)
bool isIntrinsicSourceOfDivergence(unsigned IntrID)
constexpr bool isSISrcOperand(const MCOperandInfo &OpInfo)
Is this an AMDGPU specific source operand?
bool isGenericAtomic(unsigned Opc)
LLVM_READNONE bool isInlinableIntLiteral(int64_t Literal)
Is this literal inlinable, and not one of the values intended for floating point values.
unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode, const MIMGDimInfo *Dim, bool IsA16, bool IsG16Supported)
LLVM_READONLY int32_t getAddr64Inst(uint32_t Opcode)
int32_t getMCOpcode(uint32_t Opcode, unsigned Gen)
@ OPERAND_REG_IMM_V2FP64
Definition SIDefines.h:441
@ OPERAND_KIMM32
Operand with 32-bit immediate that uses the constant bus.
Definition SIDefines.h:459
@ OPERAND_REG_IMM_INT64
Definition SIDefines.h:426
@ OPERAND_REG_IMM_V2FP16
Definition SIDefines.h:434
@ OPERAND_REG_INLINE_C_FP64
Definition SIDefines.h:450
@ OPERAND_REG_IMM_NOINLINE_FP16
Definition SIDefines.h:432
@ OPERAND_REG_INLINE_C_BF16
Definition SIDefines.h:447
@ OPERAND_REG_INLINE_C_V2BF16
Definition SIDefines.h:452
@ OPERAND_REG_IMM_V2INT64
Definition SIDefines.h:437
@ OPERAND_REG_IMM_V2INT16
Definition SIDefines.h:436
@ OPERAND_REG_IMM_BF16
Definition SIDefines.h:430
@ OPERAND_REG_IMM_INT32
Operands with register, 32-bit, or 64-bit immediate.
Definition SIDefines.h:425
@ OPERAND_REG_IMM_V2BF16
Definition SIDefines.h:433
@ OPERAND_REG_IMM_FP16
Definition SIDefines.h:431
@ OPERAND_REG_IMM_V2FP16_SPLAT
Definition SIDefines.h:435
@ OPERAND_REG_INLINE_C_INT64
Definition SIDefines.h:446
@ OPERAND_REG_INLINE_C_INT16
Operands with register or inline constant.
Definition SIDefines.h:444
@ OPERAND_REG_IMM_NOINLINE_V2FP16
Definition SIDefines.h:438
@ OPERAND_REG_IMM_FP64
Definition SIDefines.h:429
@ OPERAND_REG_INLINE_C_V2FP16
Definition SIDefines.h:453
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
Definition SIDefines.h:464
@ OPERAND_REG_INLINE_AC_FP32
Definition SIDefines.h:465
@ OPERAND_REG_IMM_V2INT32
Definition SIDefines.h:439
@ OPERAND_SDWA_VOPC_DST
Definition SIDefines.h:476
@ OPERAND_REG_IMM_FP32
Definition SIDefines.h:428
@ OPERAND_REG_INLINE_C_FP32
Definition SIDefines.h:449
@ OPERAND_REG_INLINE_C_INT32
Definition SIDefines.h:445
@ OPERAND_REG_INLINE_C_V2INT16
Definition SIDefines.h:451
@ OPERAND_INLINE_C_AV64_PSEUDO
Definition SIDefines.h:470
@ OPERAND_REG_IMM_V2FP32
Definition SIDefines.h:440
@ OPERAND_REG_INLINE_AC_FP64
Definition SIDefines.h:466
@ OPERAND_REG_INLINE_C_FP16
Definition SIDefines.h:448
@ OPERAND_REG_IMM_INT16
Definition SIDefines.h:427
@ OPERAND_INLINE_SPLIT_BARRIER_INT32
Definition SIDefines.h:456
LLVM_READONLY int32_t getBasicFromSDWAOp(uint32_t Opcode)
bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII, const MCSubtargetInfo &ST)
@ TI_SCRATCH_RSRC_DWORD1
Definition AMDGPU.h:686
@ TI_SCRATCH_RSRC_DWORD3
Definition AMDGPU.h:688
@ TI_SCRATCH_RSRC_DWORD0
Definition AMDGPU.h:685
@ TI_SCRATCH_RSRC_DWORD2
Definition AMDGPU.h:687
@ TI_CONSTDATA_START
Definition AMDGPU.h:684
bool isSingleSGPRReadInst(unsigned Opc)
Packed instructions that read a single SGPR for SGPR operands, except for 64-bit elements which read ...
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
const uint64_t RSRC_INDEX_STRIDE_SHIFT
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
LLVM_READONLY int32_t getFlatScratchInstSVfromSS(uint32_t Opcode)
bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi)
LLVM_READNONE constexpr bool isGraphics(CallingConv::ID CC)
bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi)
Is this literal inlinable.
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
@ AMDGPU_VS
Used for Mesa vertex shaders, or AMDPAL last shader stage before rasterization (vertex shader if tess...
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ AMDGPU_HS
Used for Mesa/AMDPAL hull shaders (= tessellation control shaders).
@ AMDGPU_GS
Used for Mesa/AMDPAL geometry shaders.
@ AMDGPU_PS
Used for Mesa/AMDPAL pixel shaders.
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ AMDGPU_ES
Used for AMDPAL shader stage before geometry shader if geometry is in use.
@ AMDGPU_LS
Used for AMDPAL vertex shader if tessellation is in use.
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
@ OPERAND_GENERIC_4
Definition MCInstrDesc.h:71
@ OPERAND_GENERIC_2
Definition MCInstrDesc.h:69
@ OPERAND_GENERIC_1
Definition MCInstrDesc.h:68
@ OPERAND_GENERIC_3
Definition MCInstrDesc.h:70
@ OPERAND_IMMEDIATE
Definition MCInstrDesc.h:61
@ OPERAND_GENERIC_0
Definition MCInstrDesc.h:67
@ OPERAND_GENERIC_5
Definition MCInstrDesc.h:72
Not(const Pred &P) -> Not< Pred >
constexpr bool isSDWA(const T &...O)
Definition SIDefines.h:249
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
@ Low
Lower the current thread's priority such that it does not affect foreground tasks significantly.
Definition Threading.h:280
@ Offset
Definition DWP.cpp:577
LLVM_ABI void finalizeBundle(MachineBasicBlock &MBB, MachineBasicBlock::instr_iterator FirstMI, MachineBasicBlock::instr_iterator LastMI)
finalizeBundle - Finalize a machine instruction bundle which includes a sequence of instructions star...
TargetInstrInfo::RegSubRegPair getRegSubRegPair(const MachineOperand &O)
Create RegSubRegPair from a register MachineOperand.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
constexpr uint64_t maxUIntN(uint64_t N)
Gets the maximum value for a N-bit unsigned integer.
Definition MathExtras.h:208
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
bool execMayBeModifiedBeforeUse(const MachineRegisterInfo &MRI, Register VReg, const MachineInstr &DefMI, const MachineInstr &UseMI)
Return false if EXEC is not changed between the def of VReg at DefMI and the use at UseMI.
RegState
Flags to represent properties of register accesses.
@ Implicit
Not emitted register (e.g. carry, or temporary result).
@ Dead
Unused definition.
@ Kill
The last use of a register.
@ Undef
Value of the register doesn't matter.
@ Define
Register definition.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
constexpr RegState getKillRegState(bool B)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:649
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
Op::Description Desc
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
Definition bit.h:156
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
TargetInstrInfo::RegSubRegPair getRegSequenceSubReg(MachineInstr &MI, unsigned SubReg)
Return the SubReg component from REG_SEQUENCE.
static const MachineMemOperand::Flags MONoClobber
Mark the MMO of a uniform load if there are no potentially clobbering stores on any path from the sta...
Definition SIInstrInfo.h:45
constexpr bool has_single_bit(T Value) noexcept
Definition bit.h:149
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
MachineInstr * getVRegSubRegDef(const TargetInstrInfo::RegSubRegPair &P, const MachineRegisterInfo &MRI)
Return the defining instruction for a given reg:subreg pair skipping copy like instructions and subre...
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth, bool MustPreserveProvenance=false)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI VirtRegInfo AnalyzeVirtRegInBundle(MachineInstr &MI, Register Reg, SmallVectorImpl< std::pair< MachineInstr *, unsigned > > *Ops=nullptr)
AnalyzeVirtRegInBundle - Analyze how the current instruction or bundle uses a virtual register.
static const MachineMemOperand::Flags MOCooperative
Mark the MMO of cooperative load/store atomics.
Definition SIInstrInfo.h:53
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
Definition ModRef.h:74
@ Xor
Bitwise or logical XOR of integers.
@ Sub
Subtraction of integers.
@ Add
Sum of integers.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
bool isTargetSpecificOpcode(unsigned Opcode)
Check whether the given Opcode is a target-specific opcode.
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
constexpr unsigned DefaultMemoryClusterDWordsLimit
Definition SIInstrInfo.h:41
constexpr unsigned BitWidth
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
constexpr bool isIntN(unsigned N, int64_t x)
Checks if an signed integer fits into the given (dynamic) bit width.
Definition MathExtras.h:249
static const MachineMemOperand::Flags MOLastUse
Mark the MMO of a load as the last use.
Definition SIInstrInfo.h:49
constexpr T reverseBits(T Val)
Reverse the bits in Val.
Definition MathExtras.h:119
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
Definition MathExtras.h:567
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:78
constexpr RegState getUndefRegState(bool B)
ValueUniformity
Enum describing how values behave with respect to uniformity and divergence, to answer the question: ...
Definition Uniformity.h:18
@ AlwaysUniform
The result value is always uniform.
Definition Uniformity.h:23
@ NeverUniform
The result value can never be assumed to be uniform.
Definition Uniformity.h:26
@ Default
The result value is uniform if and only if all operands are uniform.
Definition Uniformity.h:20
static const MachineMemOperand::Flags MOThreadPrivate
Mark the MMO of accesses to memory locations that are never written to by other threads.
Definition SIInstrInfo.h:64
bool execMayBeModifiedBeforeAnyUse(const MachineRegisterInfo &MRI, Register VReg, const MachineInstr &DefMI)
Return false if EXEC is not changed between the def of VReg at DefMI and all its uses.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
Helper struct for the implementation of 3-address conversion to communicate updates made to instructi...
MachineInstr * RemoveMIUse
Other instruction whose def is no longer used by the converted instruction.
static constexpr uint64_t encode(Fields... Values)
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
constexpr bool all() const
Definition LaneBitmask.h:54
Summarize the scheduling resources required for an instruction of a particular scheduling class.
Definition MCSchedule.h:129
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
Utility to store machine instructions worklist.
Definition SIInstrInfo.h:68
MachineInstr * top() const
Definition SIInstrInfo.h:73
bool isDeferred(MachineInstr *MI)
SetVector< MachineInstr * > & getDeferredList()
Definition SIInstrInfo.h:91
void insert(MachineInstr *MI)
A pair composed of a register and a sub-register index.
VirtRegInfo - Information about a virtual register used by a set of operands.
bool Reads
Reads - One of the operands read the virtual register.
bool Writes
Writes - One of the operands writes the virtual register.