LLVM 24.0.0git
SIFoldOperands.cpp
Go to the documentation of this file.
1//===-- SIFoldOperands.cpp - Fold operands --- ----------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7/// \file
8//===----------------------------------------------------------------------===//
9//
10
11#include "SIFoldOperands.h"
12#include "AMDGPU.h"
13#include "AMDGPULaneMaskUtils.h"
14#include "GCNSubtarget.h"
15#include "SIInstrInfo.h"
17#include "SIRegisterInfo.h"
24
25#define DEBUG_TYPE "si-fold-operands"
26using namespace llvm;
27
28namespace {
29
30/// Track a value we may want to fold into downstream users, applying
31/// subregister extracts along the way.
32struct FoldableDef {
33 union {
34 MachineOperand *OpToFold = nullptr;
35 uint64_t ImmToFold;
36 int FrameIndexToFold;
37 };
38
39 /// Register class of the originally defined value.
40 const TargetRegisterClass *DefRC = nullptr;
41
42 /// Track the original defining instruction for the value.
43 const MachineInstr *DefMI = nullptr;
44
45 /// Subregister to apply to the value at the use point.
46 unsigned DefSubReg = AMDGPU::NoSubRegister;
47
48 /// Kind of value stored in the union.
50
51 FoldableDef() = delete;
52 FoldableDef(MachineOperand &FoldOp, const TargetRegisterClass *DefRC,
53 unsigned DefSubReg = AMDGPU::NoSubRegister)
54 : DefRC(DefRC), DefSubReg(DefSubReg), Kind(FoldOp.getType()) {
55
56 if (FoldOp.isImm()) {
57 ImmToFold = FoldOp.getImm();
58 } else if (FoldOp.isFI()) {
59 FrameIndexToFold = FoldOp.getIndex();
60 } else {
61 assert(FoldOp.isReg() || FoldOp.isGlobal());
62 OpToFold = &FoldOp;
63 }
64
65 DefMI = FoldOp.getParent();
66 }
67
68 FoldableDef(int64_t FoldImm, const TargetRegisterClass *DefRC,
69 unsigned DefSubReg = AMDGPU::NoSubRegister)
70 : ImmToFold(FoldImm), DefRC(DefRC), DefSubReg(DefSubReg),
72
73 /// Copy the current def and apply \p SubReg to the value.
74 FoldableDef getWithSubReg(const SIRegisterInfo &TRI, unsigned SubReg) const {
75 FoldableDef Copy(*this);
76 Copy.DefSubReg = TRI.composeSubRegIndices(DefSubReg, SubReg);
77 return Copy;
78 }
79
80 bool isReg() const { return Kind == MachineOperand::MO_Register; }
81
82 Register getReg() const {
83 assert(isReg());
84 return OpToFold->getReg();
85 }
86
87 unsigned getSubReg() const {
88 assert(isReg());
89 return OpToFold->getSubReg();
90 }
91
92 bool isImm() const { return Kind == MachineOperand::MO_Immediate; }
93
94 bool isFI() const {
95 return Kind == MachineOperand::MO_FrameIndex;
96 }
97
98 int getFI() const {
99 assert(isFI());
100 return FrameIndexToFold;
101 }
102
103 bool isGlobal() const { return Kind == MachineOperand::MO_GlobalAddress; }
104
105 /// Return the effective immediate value defined by this instruction, after
106 /// application of any subregister extracts which may exist between the use
107 /// and def instruction.
108 std::optional<int64_t> getEffectiveImmVal() const {
109 assert(isImm());
110 return SIInstrInfo::extractSubregFromImm(ImmToFold, DefSubReg);
111 }
112
113 /// Check if it is legal to fold this effective value into \p MI's \p OpNo
114 /// operand.
115 bool isOperandLegal(const SIInstrInfo &TII, const MachineInstr &MI,
116 unsigned OpIdx) const {
117 switch (Kind) {
119 std::optional<int64_t> ImmToFold = getEffectiveImmVal();
120 if (!ImmToFold)
121 return false;
122
123 // TODO: Should verify the subregister index is supported by the class
124 // TODO: Avoid the temporary MachineOperand
125 MachineOperand TmpOp = MachineOperand::CreateImm(*ImmToFold);
126 return TII.isOperandLegal(MI, OpIdx, &TmpOp);
127 }
129 if (DefSubReg != AMDGPU::NoSubRegister)
130 return false;
131 MachineOperand TmpOp = MachineOperand::CreateFI(FrameIndexToFold);
132 return TII.isOperandLegal(MI, OpIdx, &TmpOp);
133 }
134 default:
135 // TODO: Try to apply DefSubReg, for global address we can extract
136 // low/high.
137 if (DefSubReg != AMDGPU::NoSubRegister)
138 return false;
139 return TII.isOperandLegal(MI, OpIdx, OpToFold);
140 }
141
142 llvm_unreachable("covered MachineOperand kind switch");
143 }
144};
145
146struct FoldCandidate {
148 FoldableDef Def;
149 int ShrinkOpcode;
150 unsigned UseOpNo;
151 bool Commuted;
152
153 FoldCandidate(MachineInstr *MI, unsigned OpNo, FoldableDef Def,
154 bool Commuted = false, int ShrinkOp = -1)
155 : UseMI(MI), Def(Def), ShrinkOpcode(ShrinkOp), UseOpNo(OpNo),
156 Commuted(Commuted) {}
157
158 bool isFI() const { return Def.isFI(); }
159
160 int getFI() const {
161 assert(isFI());
162 return Def.FrameIndexToFold;
163 }
164
165 bool isImm() const { return Def.isImm(); }
166
167 bool isReg() const { return Def.isReg(); }
168
169 Register getReg() const { return Def.getReg(); }
170
171 bool isGlobal() const { return Def.isGlobal(); }
172
173 bool needsShrink() const { return ShrinkOpcode != -1; }
174};
175
176class SIFoldOperandsImpl {
177public:
178 MachineFunction *MF;
180 const SIInstrInfo *TII;
181 const SIRegisterInfo *TRI;
182 const GCNSubtarget *ST;
183 const SIMachineFunctionInfo *MFI;
184 const MachineLoopInfo *MLI;
185
186 bool frameIndexMayFold(const MachineInstr &UseMI, int OpNo,
187 const FoldableDef &OpToFold) const;
188
189 // TODO: Just use TII::getVALUOp
190 unsigned convertToVALUOp(unsigned Opc, bool UseVOP3 = false) const {
191 switch (Opc) {
192 case AMDGPU::S_ADD_I32: {
193 if (ST->hasAddNoCarryInsts())
194 return UseVOP3 ? AMDGPU::V_ADD_U32_e64 : AMDGPU::V_ADD_U32_e32;
195 return UseVOP3 ? AMDGPU::V_ADD_CO_U32_e64 : AMDGPU::V_ADD_CO_U32_e32;
196 }
197 case AMDGPU::S_OR_B32:
198 return UseVOP3 ? AMDGPU::V_OR_B32_e64 : AMDGPU::V_OR_B32_e32;
199 case AMDGPU::S_AND_B32:
200 return UseVOP3 ? AMDGPU::V_AND_B32_e64 : AMDGPU::V_AND_B32_e32;
201 case AMDGPU::S_MUL_I32:
202 return AMDGPU::V_MUL_LO_U32_e64;
203 default:
204 return AMDGPU::INSTRUCTION_LIST_END;
205 }
206 }
207
208 bool foldCopyToVGPROfScalarAddOfFrameIndex(Register DstReg, Register SrcReg,
209 MachineInstr &MI) const;
210
211 bool updateOperand(FoldCandidate &Fold) const;
212
213 bool canUseImmWithOpSel(const MachineInstr *MI, unsigned UseOpNo,
214 int64_t ImmVal) const;
215
216 /// Try to fold immediate \p ImmVal into \p MI's operand at index \p UseOpNo.
217 bool tryFoldImmWithOpSel(MachineInstr *MI, unsigned UseOpNo,
218 int64_t ImmVal) const;
219
220 bool tryAddToFoldList(SmallVectorImpl<FoldCandidate> &FoldList,
221 MachineInstr *MI, unsigned OpNo,
222 const FoldableDef &OpToFold) const;
223 bool isUseSafeToFold(const MachineInstr &MI,
224 const MachineOperand &UseMO) const;
225 bool isTemporallyDivergentUse(const FoldableDef &OpToFold,
226 const MachineInstr &UseMI) const;
227
228 const TargetRegisterClass *getRegSeqInit(
229 MachineInstr &RegSeq,
230 SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs) const;
231
232 const TargetRegisterClass *
233 getRegSeqInit(SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs,
234 Register UseReg) const;
235
236 std::pair<int64_t, const TargetRegisterClass *>
237 isRegSeqSplat(MachineInstr &RegSeg) const;
238
239 bool tryFoldRegSeqSplat(MachineInstr *UseMI, unsigned UseOpIdx,
240 int64_t SplatVal,
241 const TargetRegisterClass *SplatRC) const;
242
243 bool tryToFoldACImm(const FoldableDef &OpToFold, MachineInstr *UseMI,
244 unsigned UseOpIdx,
245 SmallVectorImpl<FoldCandidate> &FoldList) const;
246 bool foldOperand(FoldableDef OpToFold, MachineInstr *UseMI, int UseOpIdx,
248 SmallVectorImpl<MachineInstr *> &CopiesToReplace) const;
249
250 struct ANDMaskResult {
251 int64_t Mask;
253 unsigned RegIdx;
254 };
255
256 std::optional<ANDMaskResult> getANDMaskRegOperand(MachineInstr &AndMI) const;
257
258 bool tryConstantFoldOp(MachineInstr *MI) const;
259 bool tryFoldCndMask(MachineInstr &MI) const;
260 bool tryFoldRedundantAND(MachineInstr &ChildMI) const;
261 bool tryFoldAndExec(MachineInstr &MI) const;
262 bool foldInstOperand(MachineInstr &MI, const FoldableDef &OpToFold) const;
263
264 bool foldCopyToAGPRRegSequence(MachineInstr *CopyMI) const;
265 bool tryFoldFoldableCopy(MachineInstr &MI,
266 MachineOperand *&CurrentKnownM0Val) const;
267
268 const MachineOperand *isClamp(const MachineInstr &MI) const;
269 bool tryFoldClamp(MachineInstr &MI);
270
271 std::pair<const MachineOperand *, int> isOMod(const MachineInstr &MI) const;
272 bool tryFoldOMod(MachineInstr &MI);
273 bool tryFoldSGPRSplatRegSequence(MachineInstr &MI);
274 bool tryFoldRegSequence(MachineInstr &MI);
275 bool tryFoldPhiAGPR(MachineInstr &MI);
276 bool tryFoldLoad(MachineInstr &MI);
277
278 bool tryOptimizeAGPRPhis(MachineBasicBlock &MBB);
279
280public:
281 SIFoldOperandsImpl() = default;
282
283 bool run(MachineFunction &MF, const MachineLoopInfo *MLI);
284};
285
286class SIFoldOperandsLegacy : public MachineFunctionPass {
287public:
288 static char ID;
289
290 SIFoldOperandsLegacy() : MachineFunctionPass(ID) {}
291
292 bool runOnMachineFunction(MachineFunction &MF) override {
293 if (skipFunction(MF.getFunction()))
294 return false;
295 const MachineLoopInfo *MLI =
296 &getAnalysis<MachineLoopInfoWrapperPass>().getLI();
297 return SIFoldOperandsImpl().run(MF, MLI);
298 }
299
300 StringRef getPassName() const override { return "SI Fold Operands"; }
301
302 void getAnalysisUsage(AnalysisUsage &AU) const override {
303 AU.setPreservesCFG();
307 }
308
309 MachineFunctionProperties getRequiredProperties() const override {
310 return MachineFunctionProperties().setIsSSA();
311 }
312};
313
314} // End anonymous namespace.
315
316INITIALIZE_PASS_BEGIN(SIFoldOperandsLegacy, DEBUG_TYPE, "SI Fold Operands",
317 false, false)
319INITIALIZE_PASS_END(SIFoldOperandsLegacy, DEBUG_TYPE, "SI Fold Operands", false,
320 false)
321
322char SIFoldOperandsLegacy::ID = 0;
323
324char &llvm::SIFoldOperandsLegacyID = SIFoldOperandsLegacy::ID;
325
328 const MachineOperand &MO) {
329 const TargetRegisterClass *RC = MRI.getRegClass(MO.getReg());
330 if (const TargetRegisterClass *SubRC =
331 TRI.getSubRegisterClass(RC, MO.getSubReg()))
332 RC = SubRC;
333 return RC;
334}
335
336// Map multiply-accumulate opcode to corresponding multiply-add opcode if any.
337static unsigned macToMad(unsigned Opc) {
338 switch (Opc) {
339 case AMDGPU::V_MAC_F32_e64:
340 return AMDGPU::V_MAD_F32_e64;
341 case AMDGPU::V_MAC_F16_e64:
342 return AMDGPU::V_MAD_F16_e64;
343 case AMDGPU::V_FMAC_F32_e64:
344 return AMDGPU::V_FMA_F32_e64;
345 case AMDGPU::V_FMAC_F16_e64:
346 return AMDGPU::V_FMA_F16_gfx9_e64;
347 case AMDGPU::V_FMAC_F16_t16_e64:
348 return AMDGPU::V_FMA_F16_gfx9_t16_e64;
349 case AMDGPU::V_FMAC_F16_fake16_e64:
350 return AMDGPU::V_FMA_F16_gfx9_fake16_e64;
351 case AMDGPU::V_FMAC_LEGACY_F32_e64:
352 return AMDGPU::V_FMA_LEGACY_F32_e64;
353 case AMDGPU::V_FMAC_F64_e64:
354 return AMDGPU::V_FMA_F64_e64;
355 }
356 return AMDGPU::INSTRUCTION_LIST_END;
357}
358
359// TODO: Add heuristic that the frame index might not fit in the addressing mode
360// immediate offset to avoid materializing in loops.
361bool SIFoldOperandsImpl::frameIndexMayFold(const MachineInstr &UseMI, int OpNo,
362 const FoldableDef &OpToFold) const {
363 if (!OpToFold.isFI())
364 return false;
365
366 const unsigned Opc = UseMI.getOpcode();
367 switch (Opc) {
368 case AMDGPU::S_ADD_I32:
369 case AMDGPU::S_ADD_U32:
370 case AMDGPU::V_ADD_U32_e32:
371 case AMDGPU::V_ADD_CO_U32_e32:
372 // TODO: Possibly relax hasOneUse. It matters more for mubuf, since we have
373 // to insert the wave size shift at every point we use the index.
374 // TODO: Fix depending on visit order to fold immediates into the operand
375 return UseMI.getOperand(OpNo == 1 ? 2 : 1).isImm() &&
376 MRI->hasOneNonDBGUse(UseMI.getOperand(OpNo).getReg());
377 case AMDGPU::V_ADD_U32_e64:
378 case AMDGPU::V_ADD_CO_U32_e64:
379 return UseMI.getOperand(OpNo == 2 ? 3 : 2).isImm() &&
380 MRI->hasOneNonDBGUse(UseMI.getOperand(OpNo).getReg());
381 default:
382 break;
383 }
384
385 if (TII->isMUBUF(UseMI))
386 return OpNo == AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vaddr);
387 if (!TII->isFLATScratch(UseMI))
388 return false;
389
390 int SIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::saddr);
391 if (OpNo == SIdx)
392 return true;
393
394 int VIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vaddr);
395 return OpNo == VIdx && SIdx == -1;
396}
397
398/// Fold %vgpr = COPY (S_ADD_I32 x, frameindex)
399///
400/// => %vgpr = V_ADD_U32 x, frameindex
401bool SIFoldOperandsImpl::foldCopyToVGPROfScalarAddOfFrameIndex(
402 Register DstReg, Register SrcReg, MachineInstr &MI) const {
403 if (!SrcReg.isVirtual())
404 return false;
405
406 if (TRI->isVGPR(*MRI, DstReg) && TRI->isSGPRReg(*MRI, SrcReg) &&
407 MRI->hasOneNonDBGUse(SrcReg)) {
408 MachineInstr *Def = MRI->getVRegDef(SrcReg);
409 if (!Def || Def->getNumOperands() != 4)
410 return false;
411
412 MachineOperand *Src0 = &Def->getOperand(1);
413 MachineOperand *Src1 = &Def->getOperand(2);
414
415 // TODO: This is profitable with more operand types, and for more
416 // opcodes. But ultimately this is working around poor / nonexistent
417 // regbankselect.
418 if (!Src0->isFI() && !Src1->isFI())
419 return false;
420
421 if (Src0->isFI())
422 std::swap(Src0, Src1);
423
424 const bool UseVOP3 = !Src0->isImm() || TII->isInlineConstant(*Src0);
425 unsigned NewOp = convertToVALUOp(Def->getOpcode(), UseVOP3);
426 if (NewOp == AMDGPU::INSTRUCTION_LIST_END ||
427 !Def->getOperand(3).isDead()) // Check if scc is dead
428 return false;
429
430 MachineBasicBlock *MBB = Def->getParent();
431 const DebugLoc &DL = Def->getDebugLoc();
432 if (NewOp != AMDGPU::V_ADD_CO_U32_e32) {
433 MachineInstrBuilder Add =
434 BuildMI(*MBB, *Def, DL, TII->get(NewOp), DstReg);
435
436 if (Add->getDesc().getNumDefs() == 2) {
437 Register CarryOutReg = MRI->createVirtualRegister(TRI->getBoolRC());
438 Add.addDef(CarryOutReg, RegState::Dead);
439 MRI->setRegAllocationHint(CarryOutReg, 0, TRI->getVCC());
440 }
441
442 Add.add(*Src0).add(*Src1).setMIFlags(Def->getFlags());
443 if (AMDGPU::hasNamedOperand(NewOp, AMDGPU::OpName::clamp))
444 Add.addImm(0);
445
446 Def->eraseFromParent();
447 MI.eraseFromParent();
448 return true;
449 }
450
451 assert(NewOp == AMDGPU::V_ADD_CO_U32_e32);
452
454 MBB->computeRegisterLiveness(TRI, AMDGPU::VCC, *Def, 16);
455 if (Liveness == MachineBasicBlock::LQR_Dead) {
456 // TODO: If src1 satisfies operand constraints, use vop3 version.
457 BuildMI(*MBB, *Def, DL, TII->get(NewOp), DstReg)
458 .add(*Src0)
459 .add(*Src1)
460 .setOperandDead(3) // implicit-def $vcc
461 .setMIFlags(Def->getFlags());
462 Def->eraseFromParent();
463 MI.eraseFromParent();
464 return true;
465 }
466 }
467
468 return false;
469}
470
472 return new SIFoldOperandsLegacy();
473}
474
475bool SIFoldOperandsImpl::canUseImmWithOpSel(const MachineInstr *MI,
476 unsigned UseOpNo,
477 int64_t ImmVal) const {
481 return false;
482
483 const MachineOperand &Old = MI->getOperand(UseOpNo);
484 int OpNo = MI->getOperandNo(&Old);
485
486 unsigned Opcode = MI->getOpcode();
487 uint8_t OpType = TII->get(Opcode).operands()[OpNo].OperandType;
488 switch (OpType) {
489 default:
490 return false;
498 // VOP3 packed instructions ignore op_sel source modifiers, we cannot encode
499 // two different constants.
501 static_cast<uint16_t>(ImmVal) != static_cast<uint16_t>(ImmVal >> 16))
502 return false;
503 break;
504 }
505
506 return true;
507}
508
509bool SIFoldOperandsImpl::tryFoldImmWithOpSel(MachineInstr *MI, unsigned UseOpNo,
510 int64_t ImmVal) const {
511 MachineOperand &Old = MI->getOperand(UseOpNo);
512 unsigned Opcode = MI->getOpcode();
513 int OpNo = MI->getOperandNo(&Old);
514 uint8_t OpType = TII->get(Opcode).operands()[OpNo].OperandType;
515
516 bool BF16FromUpperFP32 = ST->hasBF16InlineConstFromUpperFP32() &&
519
520 // If the literal can be inlined as-is, apply it and short-circuit the
521 // tests below. The main motivation for this is to avoid unintuitive
522 // uses of opsel.
523 if (!BF16FromUpperFP32 && AMDGPU::isInlinableLiteralV216(ImmVal, OpType)) {
524 Old.ChangeToImmediate(ImmVal);
525 return true;
526 }
527
528 // Refer to op_sel/op_sel_hi and check if we can change the immediate and
529 // op_sel in a way that allows an inline constant.
530 AMDGPU::OpName ModName = AMDGPU::OpName::NUM_OPERAND_NAMES;
531 unsigned SrcIdx = ~0;
532 if (OpNo == AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0)) {
533 ModName = AMDGPU::OpName::src0_modifiers;
534 SrcIdx = 0;
535 } else if (OpNo == AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src1)) {
536 ModName = AMDGPU::OpName::src1_modifiers;
537 SrcIdx = 1;
538 } else if (OpNo == AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src2)) {
539 ModName = AMDGPU::OpName::src2_modifiers;
540 SrcIdx = 2;
541 }
542 assert(ModName != AMDGPU::OpName::NUM_OPERAND_NAMES);
543 int ModIdx = AMDGPU::getNamedOperandIdx(Opcode, ModName);
544 MachineOperand &Mod = MI->getOperand(ModIdx);
545 unsigned ModVal = Mod.getImm();
546
547 uint16_t ImmLo =
548 static_cast<uint16_t>(ImmVal >> (ModVal & SISrcMods::OP_SEL_0 ? 16 : 0));
549 uint16_t ImmHi =
550 static_cast<uint16_t>(ImmVal >> (ModVal & SISrcMods::OP_SEL_1 ? 16 : 0));
551 uint32_t Imm = (static_cast<uint32_t>(ImmHi) << 16) | ImmLo;
552 unsigned NewModVal = ModVal & ~(SISrcMods::OP_SEL_0 | SISrcMods::OP_SEL_1);
553
554 // Helper function that attempts to inline the given value with a newly
555 // chosen opsel pattern.
556 auto tryFoldToInline = [&](uint32_t Imm) -> bool {
557 if (!BF16FromUpperFP32 && AMDGPU::isInlinableLiteralV216(Imm, OpType)) {
558 Mod.setImm(NewModVal | SISrcMods::OP_SEL_1);
560 return true;
561 }
562
563 // Try to shuffle the halves around and leverage opsel to get an inline
564 // constant.
565 uint16_t Lo = static_cast<uint16_t>(Imm);
566 uint16_t Hi = static_cast<uint16_t>(Imm >> 16);
567 if (Lo == Hi) {
568 if (AMDGPU::isInlinableLiteralV216(Lo, OpType)) {
569 // If the target has feature 'BF16InlineConstFromUpperFP32', packed BF16
570 // instructions using inline constant must use OPSEL to select the upper
571 // 16-bits from FP32.
572 if (BF16FromUpperFP32)
574 Mod.setImm(NewModVal);
576 return true;
577 }
578
579 if (!BF16FromUpperFP32 && static_cast<int16_t>(Lo) < 0) {
580 int32_t SExt = static_cast<int16_t>(Lo);
581 if (AMDGPU::isInlinableLiteralV216(SExt, OpType)) {
582 Mod.setImm(NewModVal);
583 Old.ChangeToImmediate(SExt);
584 return true;
585 }
586 }
587
588 // This check is only useful for integer instructions
589 if (OpType == AMDGPU::OPERAND_REG_IMM_V2INT16) {
590 if (AMDGPU::isInlinableLiteralV216(Lo << 16, OpType)) {
591 Mod.setImm(NewModVal | SISrcMods::OP_SEL_0 | SISrcMods::OP_SEL_1);
592 Old.ChangeToImmediate(static_cast<uint32_t>(Lo) << 16);
593 return true;
594 }
595 }
596 } else {
597 uint32_t Swapped = (static_cast<uint32_t>(Lo) << 16) | Hi;
598 if (!BF16FromUpperFP32 &&
599 AMDGPU::isInlinableLiteralV216(Swapped, OpType)) {
600 Mod.setImm(NewModVal | SISrcMods::OP_SEL_0);
601 Old.ChangeToImmediate(Swapped);
602 return true;
603 }
604 }
605
606 return false;
607 };
608
609 if (tryFoldToInline(Imm))
610 return true;
611
612 // Replace integer addition by subtraction and vice versa if it allows
613 // folding the immediate to an inline constant.
614 //
615 // We should only ever get here for SrcIdx == 1 due to canonicalization
616 // earlier in the pipeline, but we double-check here to be safe / fully
617 // general.
618 bool IsUAdd = Opcode == AMDGPU::V_PK_ADD_U16;
619 bool IsUSub = Opcode == AMDGPU::V_PK_SUB_U16;
620 if (SrcIdx == 1 && (IsUAdd || IsUSub)) {
621 unsigned ClampIdx =
622 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::clamp);
623 bool Clamp = MI->getOperand(ClampIdx).getImm() != 0;
624
625 if (!Clamp) {
626 uint16_t NegLo = -static_cast<uint16_t>(Imm);
627 uint16_t NegHi = -static_cast<uint16_t>(Imm >> 16);
628 uint32_t NegImm = (static_cast<uint32_t>(NegHi) << 16) | NegLo;
629
630 if (tryFoldToInline(NegImm)) {
631 unsigned NegOpcode =
632 IsUAdd ? AMDGPU::V_PK_SUB_U16 : AMDGPU::V_PK_ADD_U16;
633 MI->setDesc(TII->get(NegOpcode));
634 return true;
635 }
636 }
637 }
638
639 return false;
640}
641
642bool SIFoldOperandsImpl::updateOperand(FoldCandidate &Fold) const {
643 MachineInstr *MI = Fold.UseMI;
644 MachineOperand &Old = MI->getOperand(Fold.UseOpNo);
645 assert(Old.isReg());
646
647 std::optional<int64_t> ImmVal;
648 if (Fold.isImm())
649 ImmVal = Fold.Def.getEffectiveImmVal();
650
651 if (ImmVal && canUseImmWithOpSel(Fold.UseMI, Fold.UseOpNo, *ImmVal)) {
652 if (tryFoldImmWithOpSel(Fold.UseMI, Fold.UseOpNo, *ImmVal))
653 return true;
654
655 // We can't represent the candidate as an inline constant. Try as a literal
656 // with the original opsel, checking constant bus limitations.
657 MachineOperand New = MachineOperand::CreateImm(*ImmVal);
658 int OpNo = MI->getOperandNo(&Old);
659 if (!TII->isOperandLegal(*MI, OpNo, &New))
660 return false;
661
662 Old.ChangeToImmediate(*ImmVal);
663 return true;
664 }
665
666 if ((Fold.isImm() || Fold.isFI() || Fold.isGlobal()) && Fold.needsShrink()) {
667 MachineBasicBlock *MBB = MI->getParent();
668 auto Liveness = MBB->computeRegisterLiveness(TRI, AMDGPU::VCC, MI, 16);
669 if (Liveness != MachineBasicBlock::LQR_Dead) {
670 LLVM_DEBUG(dbgs() << "Not shrinking due to live vcc: " << *MI);
671 return false;
672 }
673
674 int Op32 = Fold.ShrinkOpcode;
675 MachineOperand &Dst0 = MI->getOperand(0);
676 MachineOperand &Dst1 = MI->getOperand(1);
677 assert(Dst0.isDef() && Dst1.isDef());
678
679 bool HaveNonDbgCarryUse = !MRI->use_nodbg_empty(Dst1.getReg());
680
681 const TargetRegisterClass *Dst0RC = MRI->getRegClass(Dst0.getReg());
682 Register NewReg0 = MRI->createVirtualRegister(Dst0RC);
683
684 MachineInstr *Inst32 = TII->buildShrunkInst(*MI, Op32);
685
686 if (HaveNonDbgCarryUse) {
687 BuildMI(*MBB, MI, MI->getDebugLoc(), TII->get(AMDGPU::COPY),
688 Dst1.getReg())
689 .addReg(AMDGPU::VCC, RegState::Kill);
690 } else {
691 // We only reach here when the carry-out vcc is dead so propagate the dead
692 // flag.
693 Inst32->getOperand(3).setIsDead();
694 }
695
696 // Keep the old instruction around to avoid breaking iterators, but
697 // replace it with a dummy instruction to remove uses.
698 //
699 // FIXME: We should not invert how this pass looks at operands to avoid
700 // this. Should track set of foldable movs instead of looking for uses
701 // when looking at a use.
702 Dst0.setReg(NewReg0);
703 for (unsigned I = MI->getNumOperands() - 1; I > 0; --I)
704 MI->removeOperand(I);
705 MI->setDesc(TII->get(AMDGPU::IMPLICIT_DEF));
706
707 if (Fold.Commuted)
708 TII->commuteInstruction(*Inst32, false);
709 return true;
710 }
711
712 assert(!Fold.needsShrink() && "not handled");
713
714 if (ImmVal) {
715 if (Old.isTied()) {
716 int NewMFMAOpc = AMDGPU::getMFMAEarlyClobberOp(MI->getOpcode());
717 if (NewMFMAOpc == -1)
718 return false;
719 MI->setDesc(TII->get(NewMFMAOpc));
720 MI->untieRegOperand(0);
721 const MCInstrDesc &MCID = MI->getDesc();
722 for (unsigned I = 0; I < MI->getNumDefs(); ++I)
724 MI->getOperand(I).setIsEarlyClobber(true);
725 }
726
727 // TODO: Should we try to avoid adding this to the candidate list?
728 MachineOperand New = MachineOperand::CreateImm(*ImmVal);
729 int OpNo = MI->getOperandNo(&Old);
730 if (!TII->isOperandLegal(*MI, OpNo, &New))
731 return false;
732
733 if (ST->hasBF16InlineConstFromUpperFP32() &&
734 OpNo ==
735 AMDGPU::getNamedOperandIdx(MI->getOpcode(), AMDGPU::OpName::src0)) {
736 unsigned Opcode = MI->getOpcode();
737 uint8_t OpType = TII->get(Opcode).operands()[OpNo].OperandType;
738 if ((OpType == AMDGPU::OPERAND_REG_IMM_BF16 ||
740 TII->isInlineConstant(*ImmVal, OpType)) {
741 // We can fold it, but we need to set OPSEL
742 int Mod0 =
743 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0_modifiers);
744 if (Mod0 == -1)
745 return false;
746 MachineOperand &ModOp = MI->getOperand(Mod0);
747 if (ModOp.getImm())
748 return false;
750 }
751 }
752
753 Old.ChangeToImmediate(*ImmVal);
754 return true;
755 }
756
757 if (Fold.isGlobal()) {
758 Old.ChangeToGA(Fold.Def.OpToFold->getGlobal(),
759 Fold.Def.OpToFold->getOffset(),
760 Fold.Def.OpToFold->getTargetFlags());
761 return true;
762 }
763
764 if (Fold.isFI()) {
765 Old.ChangeToFrameIndex(Fold.getFI());
766 return true;
767 }
768
769 MachineOperand *New = Fold.Def.OpToFold;
770
771 // Verify the register is compatible with the operand.
772 if (const TargetRegisterClass *OpRC =
773 TII->getRegClass(MI->getDesc(), Fold.UseOpNo)) {
774 const TargetRegisterClass *NewRC =
775 TRI->getRegClassForReg(*MRI, New->getReg());
776
777 const TargetRegisterClass *ConstrainRC = OpRC;
778 if (New->getSubReg()) {
779 ConstrainRC =
780 TRI->getMatchingSuperRegClass(NewRC, OpRC, New->getSubReg());
781
782 if (!ConstrainRC)
783 return false;
784 }
785
786 if (New->getReg().isVirtual() &&
787 !MRI->constrainRegClass(New->getReg(), ConstrainRC)) {
788 LLVM_DEBUG(dbgs() << "Cannot constrain " << printReg(New->getReg(), TRI)
789 << TRI->getRegClassName(ConstrainRC) << '\n');
790 return false;
791 }
792 }
793
794 // Rework once the VS_16 register class is updated to include proper
795 // 16-bit SGPRs instead of 32-bit ones.
796 if (Old.getSubReg() == AMDGPU::lo16 && TRI->isSGPRReg(*MRI, New->getReg()))
797 Old.setSubReg(AMDGPU::NoSubRegister);
798 if (New->getReg().isPhysical()) {
799 Old.substPhysReg(New->getReg(), *TRI);
800 } else {
801 Register OldReg = Old.getReg();
802 Old.substVirtReg(New->getReg(), New->getSubReg(), *TRI);
803 Old.setIsUndef(New->isUndef());
804
805 // If MI is in a BUNDLE, also update header's matching implicit use.
806 if (MI->isBundledWithPred()) {
807 MachineInstr &Header = *getBundleStart(MI->getIterator());
808 for (MachineOperand &MO : Header.operands()) {
809 if (MO.getReg() == OldReg) {
810 MO.setReg(New->getReg());
811 MO.setSubReg(New->getSubReg());
812 }
813 }
814 }
815 }
816 return true;
817}
818
820 FoldCandidate &&Entry) {
821 // Skip additional folding on the same operand.
822 for (FoldCandidate &Fold : FoldList)
823 if (Fold.UseMI == Entry.UseMI && Fold.UseOpNo == Entry.UseOpNo)
824 return;
825 LLVM_DEBUG(dbgs() << "Append " << (Entry.Commuted ? "commuted" : "normal")
826 << " operand " << Entry.UseOpNo << "\n " << *Entry.UseMI);
827 FoldList.push_back(Entry);
828}
829
831 MachineInstr *MI, unsigned OpNo,
832 const FoldableDef &FoldOp,
833 bool Commuted = false, int ShrinkOp = -1) {
834 appendFoldCandidate(FoldList,
835 FoldCandidate(MI, OpNo, FoldOp, Commuted, ShrinkOp));
836}
837
838// Returns true if the instruction is a packed F32 instruction and the
839// corresponding scalar operand reads 32 bits and replicates the bits to both
840// channels.
842 const GCNSubtarget *ST, MachineInstr *MI, unsigned OpNo) {
843 if (!ST->hasPKF32InstsReplicatingLower32BitsOfScalarInput())
844 return false;
845 const MCOperandInfo &OpDesc = MI->getDesc().operands()[OpNo];
847}
848
849// Packed FP32 instructions only read 32 bits from a scalar operand (SGPR or
850// literal) and replicates the bits to both channels. Therefore, if the hi and
851// lo are not same, we can't fold it.
853 const FoldableDef &OpToFold) {
854 assert(OpToFold.isImm() && "Expected immediate operand");
855 uint64_t ImmVal = OpToFold.getEffectiveImmVal().value();
856 uint32_t Lo = Lo_32(ImmVal);
857 uint32_t Hi = Hi_32(ImmVal);
858 return Lo == Hi;
859}
860
861bool SIFoldOperandsImpl::tryAddToFoldList(
862 SmallVectorImpl<FoldCandidate> &FoldList, MachineInstr *MI, unsigned OpNo,
863 const FoldableDef &OpToFold) const {
864 const unsigned Opc = MI->getOpcode();
865
866 auto tryToFoldAsFMAAKorMK = [&]() {
867 if (!OpToFold.isImm())
868 return false;
869
870 const bool TryAK = OpNo == 3;
871 const unsigned NewOpc = TryAK ? AMDGPU::S_FMAAK_F32 : AMDGPU::S_FMAMK_F32;
872 MI->setDesc(TII->get(NewOpc));
873
874 // We have to fold into operand which would be Imm not into OpNo.
875 bool FoldAsFMAAKorMK =
876 tryAddToFoldList(FoldList, MI, TryAK ? 3 : 2, OpToFold);
877 if (FoldAsFMAAKorMK) {
878 // Untie Src2 of fmac.
879 MI->untieRegOperand(3);
880 // For fmamk swap operands 1 and 2 if OpToFold was meant for operand 1.
881 if (OpNo == 1) {
882 MachineOperand &Op1 = MI->getOperand(1);
883 MachineOperand &Op2 = MI->getOperand(2);
884 Register OldReg = Op1.getReg();
885 // Operand 2 might be an inlinable constant
886 if (Op2.isImm()) {
887 Op1.ChangeToImmediate(Op2.getImm());
888 Op2.ChangeToRegister(OldReg, false);
889 } else {
890 Op1.setReg(Op2.getReg());
891 Op2.setReg(OldReg);
892 }
893 }
894 return true;
895 }
896 MI->setDesc(TII->get(Opc));
897 return false;
898 };
899
900 bool IsLegal = OpToFold.isOperandLegal(*TII, *MI, OpNo);
901 if (!IsLegal && OpToFold.isImm()) {
902 if (std::optional<int64_t> ImmVal = OpToFold.getEffectiveImmVal())
903 IsLegal = canUseImmWithOpSel(MI, OpNo, *ImmVal);
904 }
905
906 if (!IsLegal) {
907 // Special case for v_mac_{f16, f32}_e64 if we are trying to fold into src2
908 unsigned NewOpc = macToMad(Opc);
909 if (NewOpc != AMDGPU::INSTRUCTION_LIST_END) {
910 // Check if changing this to a v_mad_{f16, f32} instruction will allow us
911 // to fold the operand.
912 MI->setDesc(TII->get(NewOpc));
913 bool AddOpSel = !AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::op_sel) &&
914 AMDGPU::hasNamedOperand(NewOpc, AMDGPU::OpName::op_sel);
915 if (AddOpSel)
916 MI->addOperand(MachineOperand::CreateImm(0));
917 bool FoldAsMAD = tryAddToFoldList(FoldList, MI, OpNo, OpToFold);
918 if (FoldAsMAD) {
919 MI->untieRegOperand(OpNo);
920 return true;
921 }
922 if (AddOpSel)
923 MI->removeOperand(MI->getNumExplicitOperands() - 1);
924 MI->setDesc(TII->get(Opc));
925 }
926
927 // Special case for s_fmac_f32 if we are trying to fold into Src2.
928 // By transforming into fmaak we can untie Src2 and make folding legal.
929 if (Opc == AMDGPU::S_FMAC_F32 && OpNo == 3) {
930 if (tryToFoldAsFMAAKorMK())
931 return true;
932 }
933
934 // Inlineable constant might have been folded into Imm operand of fmaak or
935 // fmamk and we are trying to fold a non-inlinable constant.
936 if ((Opc == AMDGPU::S_FMAAK_F32 || Opc == AMDGPU::S_FMAMK_F32) &&
937 OpToFold.isImm()) {
938 std::optional<int64_t> ImmVal = OpToFold.getEffectiveImmVal();
939 if (ImmVal && !TII->isInlineConstant(*MI, OpNo, *ImmVal)) {
940 unsigned ImmIdx = Opc == AMDGPU::S_FMAAK_F32 ? 3 : 2;
941 MachineOperand &OpImm = MI->getOperand(ImmIdx);
942 if (!OpImm.isReg() &&
943 TII->isInlineConstant(*MI, MI->getOperand(OpNo), OpImm))
944 return tryToFoldAsFMAAKorMK();
945 }
946 }
947
948 // Special case for s_setreg_b32
949 if (OpToFold.isImm()) {
950 unsigned ImmOpc = 0;
951 if (Opc == AMDGPU::S_SETREG_B32)
952 ImmOpc = AMDGPU::S_SETREG_IMM32_B32;
953 else if (Opc == AMDGPU::S_SETREG_B32_mode)
954 ImmOpc = AMDGPU::S_SETREG_IMM32_B32_mode;
955 if (ImmOpc) {
956 MI->setDesc(TII->get(ImmOpc));
957 appendFoldCandidate(FoldList, MI, OpNo, OpToFold);
958 return true;
959 }
960 }
961
962 // Operand is not legal, so try to commute the instruction to
963 // see if this makes it possible to fold.
964 unsigned CommuteOpNo = TargetInstrInfo::CommuteAnyOperandIndex;
965 bool CanCommute = TII->findCommutedOpIndices(*MI, OpNo, CommuteOpNo);
966 if (!CanCommute)
967 return false;
968
969 MachineOperand &Op = MI->getOperand(OpNo);
970 MachineOperand &CommutedOp = MI->getOperand(CommuteOpNo);
971
972 // One of operands might be an Imm operand, and OpNo may refer to it after
973 // the call of commuteInstruction() below. Such situations are avoided
974 // here explicitly as OpNo must be a register operand to be a candidate
975 // for memory folding.
976 if (!Op.isReg() || !CommutedOp.isReg())
977 return false;
978
979 // The same situation with an immediate could reproduce if both inputs are
980 // the same register.
981 if (Op.isReg() && CommutedOp.isReg() &&
982 (Op.getReg() == CommutedOp.getReg() &&
983 Op.getSubReg() == CommutedOp.getSubReg()))
984 return false;
985
986 if (!TII->commuteInstruction(*MI, false, OpNo, CommuteOpNo))
987 return false;
988
989 int Op32 = -1;
990 if (!OpToFold.isOperandLegal(*TII, *MI, CommuteOpNo)) {
991 if ((Opc != AMDGPU::V_ADD_CO_U32_e64 && Opc != AMDGPU::V_SUB_CO_U32_e64 &&
992 Opc != AMDGPU::V_SUBREV_CO_U32_e64) || // FIXME
993 (!OpToFold.isImm() && !OpToFold.isFI() && !OpToFold.isGlobal())) {
994 TII->commuteInstruction(*MI, false, OpNo, CommuteOpNo);
995 return false;
996 }
997
998 // Verify the other operand is a VGPR, otherwise we would violate the
999 // constant bus restriction.
1000 MachineOperand &OtherOp = MI->getOperand(OpNo);
1001 if (!OtherOp.isReg() ||
1002 !TII->getRegisterInfo().isVGPR(*MRI, OtherOp.getReg()))
1003 return false;
1004
1005 assert(MI->getOperand(1).isDef());
1006
1007 // Make sure to get the 32-bit version of the commuted opcode.
1008 unsigned MaybeCommutedOpc = MI->getOpcode();
1009 Op32 = AMDGPU::getVOPe32(MaybeCommutedOpc);
1010 }
1011
1012 appendFoldCandidate(FoldList, MI, CommuteOpNo, OpToFold, /*Commuted=*/true,
1013 Op32);
1014 return true;
1015 }
1016
1017 // Special case for s_fmac_f32 if we are trying to fold into Src0 or Src1.
1018 // By changing into fmamk we can untie Src2.
1019 // If folding for Src0 happens first and it is identical operand to Src1 we
1020 // should avoid transforming into fmamk which requires commuting as it would
1021 // cause folding into Src1 to fail later on due to wrong OpNo used.
1022 if (Opc == AMDGPU::S_FMAC_F32 &&
1023 (OpNo != 1 || !MI->getOperand(1).isIdenticalTo(MI->getOperand(2)))) {
1024 if (tryToFoldAsFMAAKorMK())
1025 return true;
1026 }
1027
1028 // Special case for PK_F32 instructions if we are trying to fold an imm to
1029 // src0 or src1.
1030 if (OpToFold.isImm() &&
1033 return false;
1034
1035 appendFoldCandidate(FoldList, MI, OpNo, OpToFold);
1036 return true;
1037}
1038
1039bool SIFoldOperandsImpl::isUseSafeToFold(const MachineInstr &MI,
1040 const MachineOperand &UseMO) const {
1041 // Operands of SDWA instructions must be registers.
1042 return !TII->isSDWA(MI);
1043}
1044
1045// Returns true if any instruction in \p L modifies EXEC.
1046static bool loopModifiesExec(const MachineLoop &L, const SIRegisterInfo &TRI) {
1047 for (const MachineBasicBlock *MBB : L.getBlocks())
1048 for (const MachineInstr &MI : *MBB)
1049 if (MI.modifiesRegister(TRI.getExec(), &TRI))
1050 return true;
1051 return false;
1052}
1053
1054// An SGPR->VGPR copy inside a divergent loop latches each lane value as it
1055// exits. Folding its scalar source into a use after the loop would make every
1056// lane read the same reconverged value, so do not fold across the loop exit.
1057bool SIFoldOperandsImpl::isTemporallyDivergentUse(
1058 const FoldableDef &OpToFold, const MachineInstr &UseMI) const {
1059 if (!OpToFold.isReg())
1060 return false;
1061 const MachineInstr *DefMI = OpToFold.DefMI;
1062 if (!DefMI || !DefMI->isCopy() ||
1063 TRI->isSGPRReg(*MRI, DefMI->getOperand(0).getReg()) ||
1064 !TRI->isSGPRReg(*MRI, OpToFold.getReg()))
1065 return false;
1066 const MachineLoop *DefLoop = MLI->getLoopFor(DefMI->getParent());
1067 return DefLoop && !DefLoop->contains(UseMI.getParent()) &&
1068 loopModifiesExec(*DefLoop, *TRI);
1069}
1070
1072 const MachineRegisterInfo &MRI,
1073 Register SrcReg) {
1074 MachineOperand *Sub = nullptr;
1075 for (MachineInstr *SubDef = MRI.getVRegDef(SrcReg);
1076 SubDef && TII.isFoldableCopy(*SubDef);
1077 SubDef = MRI.getVRegDef(Sub->getReg())) {
1078 unsigned SrcIdx = TII.getFoldableCopySrcIdx(*SubDef);
1079 MachineOperand &SrcOp = SubDef->getOperand(SrcIdx);
1080
1081 if (SrcOp.isImm())
1082 return &SrcOp;
1083 if (!SrcOp.isReg() || SrcOp.getReg().isPhysical())
1084 break;
1085 Sub = &SrcOp;
1086 // TODO: Support compose
1087 if (SrcOp.getSubReg())
1088 break;
1089 }
1090
1091 return Sub;
1092}
1093
1094const TargetRegisterClass *SIFoldOperandsImpl::getRegSeqInit(
1095 MachineInstr &RegSeq,
1096 SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs) const {
1097
1098 assert(RegSeq.isRegSequence());
1099
1100 const TargetRegisterClass *RC = nullptr;
1101
1102 for (unsigned I = 1, E = RegSeq.getNumExplicitOperands(); I != E; I += 2) {
1103 MachineOperand &SrcOp = RegSeq.getOperand(I);
1104 if (SrcOp.getReg().isPhysical())
1105 return nullptr;
1106 unsigned SubRegIdx = RegSeq.getOperand(I + 1).getImm();
1107
1108 // Only accept reg_sequence with uniform reg class inputs for simplicity.
1109 const TargetRegisterClass *OpRC = getRegOpRC(*MRI, *TRI, SrcOp);
1110 if (!RC)
1111 RC = OpRC;
1112 else if (!TRI->getCommonSubClass(RC, OpRC))
1113 return nullptr;
1114
1115 if (SrcOp.getSubReg()) {
1116 // TODO: Handle subregister compose
1117 Defs.emplace_back(&SrcOp, SubRegIdx);
1118 continue;
1119 }
1120
1121 MachineOperand *DefSrc = lookUpCopyChain(*TII, *MRI, SrcOp.getReg());
1122 if (DefSrc && (DefSrc->isReg() || DefSrc->isImm())) {
1123 Defs.emplace_back(DefSrc, SubRegIdx);
1124 continue;
1125 }
1126
1127 Defs.emplace_back(&SrcOp, SubRegIdx);
1128 }
1129
1130 return RC;
1131}
1132
1133// Find a def of the UseReg, check if it is a reg_sequence and find initializers
1134// for each subreg, tracking it to an immediate if possible. Returns the
1135// register class of the inputs on success.
1136const TargetRegisterClass *SIFoldOperandsImpl::getRegSeqInit(
1137 SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs,
1138 Register UseReg) const {
1139 MachineInstr *Def = MRI->getVRegDef(UseReg);
1140 if (!Def || !Def->isRegSequence())
1141 return nullptr;
1142
1143 return getRegSeqInit(*Def, Defs);
1144}
1145
1146std::pair<int64_t, const TargetRegisterClass *>
1147SIFoldOperandsImpl::isRegSeqSplat(MachineInstr &RegSeq) const {
1149 const TargetRegisterClass *SrcRC = getRegSeqInit(RegSeq, Defs);
1150 if (!SrcRC)
1151 return {};
1152
1153 bool TryToMatchSplat64 = false;
1154
1155 std::optional<int64_t> Imm;
1156 for (unsigned I = 0, E = Defs.size(); I != E; ++I) {
1157 const MachineOperand *Op = Defs[I].first;
1158 if (!Op->isImm()) {
1159 if (Op->isReg()) {
1160 MachineInstr *Def = MRI->getVRegDef(Op->getReg());
1161 if (!Def || Def->isImplicitDef())
1162 continue;
1163 }
1164 return {};
1165 }
1166
1167 int64_t SubImm = Op->getImm();
1168 if (!Imm) {
1169 Imm = SubImm;
1170 continue;
1171 }
1172
1173 if (Imm != SubImm) {
1174 if (I == 1 && (E & 1) == 0) {
1175 // If we have an even number of inputs, there's a chance this is a
1176 // 64-bit element splat broken into 32-bit pieces.
1177 TryToMatchSplat64 = true;
1178 break;
1179 }
1180
1181 return {}; // Can only fold splat constants
1182 }
1183 }
1184
1185 if (!TryToMatchSplat64) {
1186 if (Imm)
1187 return {*Imm, SrcRC};
1188 return {};
1189 }
1190
1191 // Fallback to recognizing 64-bit splats broken into 32-bit pieces
1192 // (i.e. recognize every other other element is 0 for 64-bit immediates)
1193 int64_t SplatVal64;
1194 for (unsigned I = 0, E = Defs.size(); I != E; I += 2) {
1195 const MachineOperand *Op0 = Defs[I].first;
1196 const MachineOperand *Op1 = Defs[I + 1].first;
1197
1198 if (!Op0->isImm() || !Op1->isImm())
1199 return {};
1200
1201 unsigned SubReg0 = Defs[I].second;
1202 unsigned SubReg1 = Defs[I + 1].second;
1203
1204 // Assume we're going to generally encounter reg_sequences with sorted
1205 // subreg indexes, so reject any that aren't consecutive.
1206 if (TRI->getChannelFromSubReg(SubReg0) + 1 !=
1207 TRI->getChannelFromSubReg(SubReg1))
1208 return {};
1209
1210 if (TRI->getSubRegIdxSize(SubReg0) != 32)
1211 return {};
1212
1213 int64_t MergedVal = Make_64(Op1->getImm(), Op0->getImm());
1214 if (I == 0)
1215 SplatVal64 = MergedVal;
1216 else if (SplatVal64 != MergedVal)
1217 return {};
1218 }
1219
1220 const TargetRegisterClass *RC64 = TRI->getSubRegisterClass(
1221 MRI->getRegClass(RegSeq.getOperand(0).getReg()), AMDGPU::sub0_sub1);
1222
1223 return {SplatVal64, RC64};
1224}
1225
1226bool SIFoldOperandsImpl::tryFoldRegSeqSplat(
1227 MachineInstr *UseMI, unsigned UseOpIdx, int64_t SplatVal,
1228 const TargetRegisterClass *SplatRC) const {
1229 const MCInstrDesc &Desc = UseMI->getDesc();
1230 if (UseOpIdx >= Desc.getNumOperands())
1231 return false;
1232
1233 // Filter out unhandled pseudos.
1234 if (!AMDGPU::isSISrcOperand(Desc, UseOpIdx))
1235 return false;
1236
1237 int16_t RCID = TII->getOpRegClassID(Desc.operands()[UseOpIdx]);
1238 if (RCID == -1)
1239 return false;
1240
1241 const TargetRegisterClass *OpRC = TRI->getRegClass(RCID);
1242
1243 // Special case 0/-1, since when interpreted as a 64-bit element both halves
1244 // have the same bits. These are the only cases where a splat has the same
1245 // interpretation for 32-bit and 64-bit splats.
1246 if (SplatVal != 0 && SplatVal != -1) {
1247 // We need to figure out the scalar type read by the operand. e.g. the MFMA
1248 // operand will be AReg_128, and we want to check if it's compatible with an
1249 // AReg_32 constant.
1250 uint8_t OpTy = Desc.operands()[UseOpIdx].OperandType;
1251 switch (OpTy) {
1257 OpRC = TRI->getSubRegisterClass(OpRC, AMDGPU::sub0);
1258 break;
1264 OpRC = TRI->getSubRegisterClass(OpRC, AMDGPU::sub0_sub1);
1265 break;
1266 default:
1267 return false;
1268 }
1269
1270 if (!TRI->getCommonSubClass(OpRC, SplatRC))
1271 return false;
1272 }
1273
1274 MachineOperand TmpOp = MachineOperand::CreateImm(SplatVal);
1275 if (!TII->isOperandLegal(*UseMI, UseOpIdx, &TmpOp))
1276 return false;
1277
1278 return true;
1279}
1280
1281bool SIFoldOperandsImpl::tryToFoldACImm(
1282 const FoldableDef &OpToFold, MachineInstr *UseMI, unsigned UseOpIdx,
1283 SmallVectorImpl<FoldCandidate> &FoldList) const {
1284 const MCInstrDesc &Desc = UseMI->getDesc();
1285 if (UseOpIdx >= Desc.getNumOperands())
1286 return false;
1287
1288 // Filter out unhandled pseudos.
1289 if (!AMDGPU::isSISrcOperand(Desc, UseOpIdx))
1290 return false;
1291
1292 if (OpToFold.isImm() && OpToFold.isOperandLegal(*TII, *UseMI, UseOpIdx)) {
1295 return false;
1296 appendFoldCandidate(FoldList, UseMI, UseOpIdx, OpToFold);
1297 return true;
1298 }
1299
1300 return false;
1301}
1302
1303bool SIFoldOperandsImpl::foldOperand(
1304 FoldableDef OpToFold, MachineInstr *UseMI, int UseOpIdx,
1305 SmallVectorImpl<FoldCandidate> &FoldList,
1306 SmallVectorImpl<MachineInstr *> &CopiesToReplace) const {
1307 bool Changed = false;
1308 const MachineOperand *UseOp = &UseMI->getOperand(UseOpIdx);
1309
1310 if (!isUseSafeToFold(*UseMI, *UseOp))
1311 return Changed;
1312
1313 if (isTemporallyDivergentUse(OpToFold, *UseMI))
1314 return Changed;
1315
1316 // FIXME: Fold operands with subregs.
1317 if (UseOp->isReg() && OpToFold.isReg()) {
1318 if (UseOp->isImplicit())
1319 return Changed;
1320 // Allow folding from SGPRs to 16-bit VGPRs.
1321 if (UseOp->getSubReg() != AMDGPU::NoSubRegister &&
1322 (UseOp->getSubReg() != AMDGPU::lo16 ||
1323 !TRI->isSGPRReg(*MRI, OpToFold.getReg())))
1324 return Changed;
1325 }
1326
1327 // Special case for REG_SEQUENCE: We can't fold literals into
1328 // REG_SEQUENCE instructions, so we have to fold them into the
1329 // uses of REG_SEQUENCE.
1330 if (UseMI->isRegSequence()) {
1331 Register RegSeqDstReg = UseMI->getOperand(0).getReg();
1332 unsigned RegSeqDstSubReg = UseMI->getOperand(UseOpIdx + 1).getImm();
1333
1334 int64_t SplatVal;
1335 const TargetRegisterClass *SplatRC;
1336 std::tie(SplatVal, SplatRC) = isRegSeqSplat(*UseMI);
1337
1338 // Grab the use operands first
1340 llvm::make_pointer_range(MRI->use_nodbg_operands(RegSeqDstReg)));
1341 for (unsigned I = 0; I != UsesToProcess.size(); ++I) {
1342 MachineOperand *RSUse = UsesToProcess[I];
1343 MachineInstr *RSUseMI = RSUse->getParent();
1344 unsigned OpNo = RSUseMI->getOperandNo(RSUse);
1345
1346 if (SplatRC) {
1347 if (RSUseMI->isCopy()) {
1348 Register DstReg = RSUseMI->getOperand(0).getReg();
1349 append_range(UsesToProcess,
1351 continue;
1352 }
1353 if (tryFoldRegSeqSplat(RSUseMI, OpNo, SplatVal, SplatRC)) {
1354 FoldableDef SplatDef(SplatVal, SplatRC);
1355 appendFoldCandidate(FoldList, RSUseMI, OpNo, SplatDef);
1356 Changed = true;
1357 continue;
1358 }
1359 }
1360
1361 // TODO: Handle general compose
1362 if (RSUse->getSubReg() != RegSeqDstSubReg)
1363 continue;
1364
1365 // FIXME: We should avoid recursing here. There should be a cleaner split
1366 // between the in-place mutations and adding to the fold list.
1367 Changed |= foldOperand(OpToFold, RSUseMI, RSUseMI->getOperandNo(RSUse),
1368 FoldList, CopiesToReplace);
1369 }
1370
1371 return Changed;
1372 }
1373
1374 if (tryToFoldACImm(OpToFold, UseMI, UseOpIdx, FoldList))
1375 return true;
1376
1377 if (frameIndexMayFold(*UseMI, UseOpIdx, OpToFold)) {
1378 // Verify that this is a stack access.
1379 // FIXME: Should probably use stack pseudos before frame lowering.
1380
1381 if (TII->isMUBUF(*UseMI)) {
1382 if (TII->getNamedOperand(*UseMI, AMDGPU::OpName::srsrc)->getReg() !=
1383 MFI->getScratchRSrcReg())
1384 return Changed;
1385
1386 // Ensure this is either relative to the current frame or the current
1387 // wave.
1388 MachineOperand &SOff =
1389 *TII->getNamedOperand(*UseMI, AMDGPU::OpName::soffset);
1390 if (!SOff.isImm() || SOff.getImm() != 0)
1391 return Changed;
1392 }
1393
1394 const unsigned Opc = UseMI->getOpcode();
1395 if (TII->isFLATScratch(*UseMI) &&
1396 AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::vaddr) &&
1397 !AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::saddr)) {
1398 unsigned NewOpc = AMDGPU::getFlatScratchInstSSfromSV(Opc);
1399 unsigned CPol =
1400 TII->getNamedOperand(*UseMI, AMDGPU::OpName::cpol)->getImm();
1401 if ((CPol & AMDGPU::CPol::SCAL) &&
1403 return Changed;
1404
1405 UseMI->setDesc(TII->get(NewOpc));
1406 }
1407
1408 // A frame index will resolve to a positive constant, so it should always be
1409 // safe to fold the addressing mode, even pre-GFX9.
1410 UseMI->getOperand(UseOpIdx).ChangeToFrameIndex(OpToFold.getFI());
1411
1412 return true;
1413 }
1414
1415 bool FoldingImmLike =
1416 OpToFold.isImm() || OpToFold.isFI() || OpToFold.isGlobal();
1417
1418 if (FoldingImmLike && UseMI->isCopy()) {
1419 Register DestReg = UseMI->getOperand(0).getReg();
1420 Register SrcReg = UseMI->getOperand(1).getReg();
1421 unsigned UseSubReg = UseMI->getOperand(1).getSubReg();
1422 assert(SrcReg.isVirtual());
1423
1424 const TargetRegisterClass *SrcRC = MRI->getRegClass(SrcReg);
1425
1426 // Don't fold into a copy to a physical register with the same class. Doing
1427 // so would interfere with the register coalescer's logic which would avoid
1428 // redundant initializations.
1429 if (DestReg.isPhysical() && SrcRC->contains(DestReg))
1430 return Changed;
1431
1432 const TargetRegisterClass *DestRC = TRI->getRegClassForReg(*MRI, DestReg);
1433 // In order to fold immediates into copies, we need to change the copy to a
1434 // MOV. Find a compatible mov instruction with the value.
1435 for (unsigned MovOp :
1436 {AMDGPU::S_MOV_B32, AMDGPU::V_MOV_B32_e32, AMDGPU::S_MOV_B64,
1437 AMDGPU::V_MOV_B64_PSEUDO, AMDGPU::V_MOV_B16_t16_e64,
1438 AMDGPU::V_ACCVGPR_WRITE_B32_e64, AMDGPU::AV_MOV_B32_IMM_PSEUDO,
1439 AMDGPU::AV_MOV_B64_IMM_PSEUDO}) {
1440 const MCInstrDesc &MovDesc = TII->get(MovOp);
1441 const TargetRegisterClass *MovDstRC =
1442 TRI->getRegClass(TII->getOpRegClassID(MovDesc.operands()[0]));
1443
1444 // Fold if the destination register class of the MOV instruction (ResRC)
1445 // is a superclass of (or equal to) the destination register class of the
1446 // COPY (DestRC). If this condition fails, folding would be illegal.
1447 if (!DestRC->hasSuperClassEq(MovDstRC))
1448 continue;
1449
1450 const int SrcIdx = MovOp == AMDGPU::V_MOV_B16_t16_e64 ? 2 : 1;
1451
1452 int16_t RegClassID = TII->getOpRegClassID(MovDesc.operands()[SrcIdx]);
1453 if (RegClassID != -1) {
1454 const TargetRegisterClass *MovSrcRC = TRI->getRegClass(RegClassID);
1455
1456 if (UseSubReg)
1457 MovSrcRC = TRI->getMatchingSuperRegClass(SrcRC, MovSrcRC, UseSubReg);
1458
1459 // FIXME: We should be able to directly check immediate operand legality
1460 // for all cases, but gfx908 hacks break.
1461 if (MovOp == AMDGPU::AV_MOV_B32_IMM_PSEUDO &&
1462 (!OpToFold.isImm() ||
1463 !TII->isImmOperandLegal(MovDesc, SrcIdx,
1464 *OpToFold.getEffectiveImmVal())))
1465 break;
1466
1467 if (!MRI->constrainRegClass(SrcReg, MovSrcRC))
1468 break;
1469
1470 // FIXME: This is mutating the instruction only and deferring the actual
1471 // fold of the immediate
1472 } else {
1473 // For the _IMM_PSEUDO cases, there can be value restrictions on the
1474 // immediate to verify. Technically we should always verify this, but it
1475 // only matters for these concrete cases.
1476 // TODO: Handle non-imm case if it's useful.
1477 if (!OpToFold.isImm() ||
1478 !TII->isImmOperandLegal(MovDesc, 1, *OpToFold.getEffectiveImmVal()))
1479 break;
1480 }
1481
1484 while (ImpOpI != ImpOpE) {
1485 MachineInstr::mop_iterator Tmp = ImpOpI;
1486 ImpOpI++;
1488 }
1489 UseMI->setDesc(MovDesc);
1490
1491 if (MovOp == AMDGPU::V_MOV_B16_t16_e64) {
1492 const auto &SrcOp = UseMI->getOperand(UseOpIdx);
1493 MachineOperand NewSrcOp(SrcOp);
1494 UseMI->removeOperand(1);
1495 UseMI->addOperand(*MF, MachineOperand::CreateImm(0)); // src0_modifiers
1496 UseMI->addOperand(NewSrcOp); // src0
1497 UseMI->addOperand(*MF, MachineOperand::CreateImm(0)); // op_sel
1498 UseOpIdx = SrcIdx;
1499 UseOp = &UseMI->getOperand(UseOpIdx);
1500 }
1501 CopiesToReplace.push_back(UseMI);
1502 Changed = true;
1503 break;
1504 }
1505
1506 // We failed to replace the copy, so give up.
1507 if (UseMI->getOpcode() == AMDGPU::COPY)
1508 return Changed;
1509
1510 } else {
1511 if (UseMI->isCopy() && OpToFold.isReg() &&
1512 UseMI->getOperand(0).getReg().isVirtual() &&
1513 !UseMI->getOperand(1).getSubReg() &&
1514 OpToFold.DefMI->implicit_operands().empty()) {
1515 LLVM_DEBUG(dbgs() << "Folding " << *OpToFold.OpToFold << "\n into "
1516 << *UseMI);
1517 unsigned Size = TII->getOpSize(*UseMI, 1);
1518 Register UseReg = OpToFold.getReg();
1520 unsigned SubRegIdx = OpToFold.getSubReg();
1521 // Hack to allow 32-bit SGPRs to be folded into True16 instructions
1522 // Remove this if 16-bit SGPRs (i.e. SGPR_LO16) are added to the
1523 // VS_16RegClass
1524 if (Size == 2 && TRI->isVGPR(*MRI, UseMI->getOperand(0).getReg()) &&
1525 TRI->isSGPRReg(*MRI, UseReg) && SubRegIdx != AMDGPU::NoSubRegister) {
1526 // SGPRs only have lo16 subregisters, so the value is in the low half
1527 // of a 32-bit SGPR. Use that whole 32-bit SGPR instead.
1528 unsigned Channel = TRI->getChannelFromSubReg(SubRegIdx);
1529 const TargetRegisterClass *UseRC = TRI->getRegClassForReg(*MRI, UseReg);
1530 SubRegIdx = TRI->getRegSizeInBits(*UseRC) == 32
1531 ? AMDGPU::NoSubRegister
1533 }
1534 UseMI->getOperand(1).setSubReg(SubRegIdx);
1535 UseMI->getOperand(1).setIsKill(false);
1536 CopiesToReplace.push_back(UseMI);
1537 OpToFold.OpToFold->setIsKill(false);
1538 Changed = true;
1539
1540 // Remove kill flags as kills may now be out of order with uses.
1541 MRI->clearKillFlags(UseReg);
1542 if (foldCopyToAGPRRegSequence(UseMI))
1543 return true;
1544 }
1545
1546 unsigned UseOpc = UseMI->getOpcode();
1547 if (UseOpc == AMDGPU::V_READFIRSTLANE_B32 ||
1548 (UseOpc == AMDGPU::V_READLANE_B32 &&
1549 (int)UseOpIdx ==
1550 AMDGPU::getNamedOperandIdx(UseOpc, AMDGPU::OpName::src0))) {
1551 // %vgpr = V_MOV_B32 imm
1552 // %sgpr = V_READFIRSTLANE_B32 %vgpr
1553 // =>
1554 // %sgpr = S_MOV_B32 imm
1555 if (FoldingImmLike) {
1557 UseMI->getOperand(UseOpIdx).getReg(),
1558 *OpToFold.DefMI, *UseMI))
1559 return Changed;
1560
1561 UseMI->setDesc(TII->get(AMDGPU::S_MOV_B32));
1563
1564 if (OpToFold.isImm()) {
1566 *OpToFold.getEffectiveImmVal());
1567 } else if (OpToFold.isFI())
1568 UseMI->getOperand(1).ChangeToFrameIndex(OpToFold.getFI());
1569 else {
1570 assert(OpToFold.isGlobal());
1571 UseMI->getOperand(1).ChangeToGA(OpToFold.OpToFold->getGlobal(),
1572 OpToFold.OpToFold->getOffset(),
1573 OpToFold.OpToFold->getTargetFlags());
1574 }
1575 UseMI->removeOperand(2); // Remove exec read (or src1 for readlane)
1576 return true;
1577 }
1578
1579 if (OpToFold.isReg() && TRI->isSGPRReg(*MRI, OpToFold.getReg())) {
1581 UseMI->getOperand(UseOpIdx).getReg(),
1582 *OpToFold.DefMI, *UseMI))
1583 return Changed;
1584
1585 // %vgpr = COPY %sgpr0
1586 // %sgpr1 = V_READFIRSTLANE_B32 %vgpr
1587 // =>
1588 // %sgpr1 = COPY %sgpr0
1589 UseMI->setDesc(TII->get(AMDGPU::COPY));
1590 UseMI->getOperand(1).setReg(OpToFold.getReg());
1591 UseMI->getOperand(1).setSubReg(OpToFold.getSubReg());
1592 UseMI->getOperand(1).setIsKill(false);
1593 UseMI->removeOperand(2); // Remove exec read (or src1 for readlane)
1595 return true;
1596 }
1597 }
1598
1599 const MCInstrDesc &UseDesc = UseMI->getDesc();
1600
1601 // Don't fold into target independent nodes. Target independent opcodes
1602 // don't have defined register classes.
1603 if (UseDesc.isVariadic() || UseOp->isImplicit() ||
1604 UseDesc.operands()[UseOpIdx].RegClass == -1)
1605 return Changed;
1606 }
1607
1608 // FIXME: We could try to change the instruction from 64-bit to 32-bit
1609 // to enable more folding opportunities. The shrink operands pass
1610 // already does this.
1611
1612 Changed |= tryAddToFoldList(FoldList, UseMI, UseOpIdx, OpToFold);
1613 return Changed;
1614}
1615
1616static bool evalBinaryInstruction(unsigned Opcode, int32_t &Result,
1618 switch (Opcode) {
1619 case AMDGPU::S_ADD_I32:
1620 case AMDGPU::S_ADD_U32:
1621 Result = LHS + RHS;
1622 return true;
1623 case AMDGPU::S_SUB_I32:
1624 case AMDGPU::S_SUB_U32:
1625 Result = LHS - RHS;
1626 return true;
1627 case AMDGPU::V_AND_B32_e64:
1628 case AMDGPU::V_AND_B32_e32:
1629 case AMDGPU::S_AND_B32:
1630 Result = LHS & RHS;
1631 return true;
1632 case AMDGPU::V_OR_B32_e64:
1633 case AMDGPU::V_OR_B32_e32:
1634 case AMDGPU::S_OR_B32:
1635 Result = LHS | RHS;
1636 return true;
1637 case AMDGPU::V_XOR_B32_e64:
1638 case AMDGPU::V_XOR_B32_e32:
1639 case AMDGPU::S_XOR_B32:
1640 Result = LHS ^ RHS;
1641 return true;
1642 case AMDGPU::S_XNOR_B32:
1643 Result = ~(LHS ^ RHS);
1644 return true;
1645 case AMDGPU::S_NAND_B32:
1646 Result = ~(LHS & RHS);
1647 return true;
1648 case AMDGPU::S_NOR_B32:
1649 Result = ~(LHS | RHS);
1650 return true;
1651 case AMDGPU::S_ANDN2_B32:
1652 Result = LHS & ~RHS;
1653 return true;
1654 case AMDGPU::S_ORN2_B32:
1655 Result = LHS | ~RHS;
1656 return true;
1657 case AMDGPU::V_LSHL_B32_e64:
1658 case AMDGPU::V_LSHL_B32_e32:
1659 case AMDGPU::S_LSHL_B32:
1660 // The instruction ignores the high bits for out of bounds shifts.
1661 Result = LHS << (RHS & 31);
1662 return true;
1663 case AMDGPU::V_LSHLREV_B32_e64:
1664 case AMDGPU::V_LSHLREV_B32_e32:
1665 Result = RHS << (LHS & 31);
1666 return true;
1667 case AMDGPU::V_LSHR_B32_e64:
1668 case AMDGPU::V_LSHR_B32_e32:
1669 case AMDGPU::S_LSHR_B32:
1670 Result = LHS >> (RHS & 31);
1671 return true;
1672 case AMDGPU::V_LSHRREV_B32_e64:
1673 case AMDGPU::V_LSHRREV_B32_e32:
1674 Result = RHS >> (LHS & 31);
1675 return true;
1676 case AMDGPU::V_ASHR_I32_e64:
1677 case AMDGPU::V_ASHR_I32_e32:
1678 case AMDGPU::S_ASHR_I32:
1679 Result = static_cast<int32_t>(LHS) >> (RHS & 31);
1680 return true;
1681 case AMDGPU::V_ASHRREV_I32_e64:
1682 case AMDGPU::V_ASHRREV_I32_e32:
1683 Result = static_cast<int32_t>(RHS) >> (LHS & 31);
1684 return true;
1685 default:
1686 return false;
1687 }
1688}
1689
1690static unsigned getMovOpc(bool IsScalar) {
1691 return IsScalar ? AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32;
1692}
1693
1694// Try to simplify operations with a constant that may appear after instruction
1695// selection.
1696// TODO: See if a frame index with a fixed offset can fold.
1697bool SIFoldOperandsImpl::tryConstantFoldOp(MachineInstr *MI) const {
1698 if (!MI->allImplicitDefsAreDead())
1699 return false;
1700
1701 unsigned Opc = MI->getOpcode();
1702
1703 int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
1704 if (Src0Idx == -1)
1705 return false;
1706
1707 MachineOperand *Src0 = &MI->getOperand(Src0Idx);
1708 std::optional<int64_t> Src0Imm = TII->getImmOrMaterializedImm(*MRI, *Src0);
1709
1710 if ((Opc == AMDGPU::V_NOT_B32_e64 || Opc == AMDGPU::V_NOT_B32_e32 ||
1711 Opc == AMDGPU::S_NOT_B32) &&
1712 Src0Imm) {
1713 MI->getOperand(1).ChangeToImmediate(~*Src0Imm);
1714 TII->mutateAndCleanupImplicit(
1715 *MI, TII->get(getMovOpc(Opc == AMDGPU::S_NOT_B32)));
1716 return true;
1717 }
1718
1719 int Src1Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1);
1720 if (Src1Idx == -1)
1721 return false;
1722
1723 MachineOperand *Src1 = &MI->getOperand(Src1Idx);
1724 std::optional<int64_t> Src1Imm = TII->getImmOrMaterializedImm(*MRI, *Src1);
1725
1726 if (!Src0Imm && !Src1Imm)
1727 return false;
1728
1729 // and k0, k1 -> v_mov_b32 (k0 & k1)
1730 // or k0, k1 -> v_mov_b32 (k0 | k1)
1731 // xor k0, k1 -> v_mov_b32 (k0 ^ k1)
1732 if (Src0Imm && Src1Imm) {
1733 int32_t NewImm;
1734 if (!evalBinaryInstruction(Opc, NewImm, *Src0Imm, *Src1Imm))
1735 return false;
1736
1737 bool IsSGPR = TRI->isSGPRReg(*MRI, MI->getOperand(0).getReg());
1738
1739 // Be careful to change the right operand, src0 may belong to a different
1740 // instruction.
1741 MI->getOperand(Src0Idx).ChangeToImmediate(NewImm);
1742 MI->removeOperand(Src1Idx);
1743 TII->mutateAndCleanupImplicit(*MI, TII->get(getMovOpc(IsSGPR)));
1744 return true;
1745 }
1746
1747 // S_SUB_* is not commutable, so handle it before the commutability gate.
1748 // Only `x - 0 -> copy x` is valid; `0 - x` is a negation, not a copy.
1749 if (Opc == AMDGPU::S_SUB_I32 || Opc == AMDGPU::S_SUB_U32) {
1750 if (Src1Imm && static_cast<int32_t>(*Src1Imm) == 0) {
1751 // y = sub x, 0 => y = copy x
1752 MI->removeOperand(Src1Idx);
1753 TII->mutateAndCleanupImplicit(*MI, TII->get(AMDGPU::COPY));
1754 return true;
1755 }
1756 return false;
1757 }
1758
1759 if (!MI->isCommutable())
1760 return false;
1761
1762 if (Src0Imm && !Src1Imm) {
1763 std::swap(Src0, Src1);
1764 std::swap(Src0Idx, Src1Idx);
1765 std::swap(Src0Imm, Src1Imm);
1766 }
1767
1768 int32_t Src1Val = static_cast<int32_t>(*Src1Imm);
1769 if (Opc == AMDGPU::S_ADD_I32 || Opc == AMDGPU::S_ADD_U32) {
1770 if (Src1Val == 0) {
1771 // y = add x, 0 => y = copy x
1772 MI->removeOperand(Src1Idx);
1773 TII->mutateAndCleanupImplicit(*MI, TII->get(AMDGPU::COPY));
1774 return true;
1775 }
1776 return false;
1777 }
1778
1779 if (Opc == AMDGPU::V_OR_B32_e64 ||
1780 Opc == AMDGPU::V_OR_B32_e32 ||
1781 Opc == AMDGPU::S_OR_B32) {
1782 if (Src1Val == 0) {
1783 // y = or x, 0 => y = copy x
1784 MI->removeOperand(Src1Idx);
1785 TII->mutateAndCleanupImplicit(*MI, TII->get(AMDGPU::COPY));
1786 } else if (Src1Val == -1) {
1787 // y = or x, -1 => y = v_mov_b32 -1
1788 MI->removeOperand(Src0Idx);
1789 TII->mutateAndCleanupImplicit(
1790 *MI, TII->get(getMovOpc(Opc == AMDGPU::S_OR_B32)));
1791 } else
1792 return false;
1793
1794 return true;
1795 }
1796
1797 if (Opc == AMDGPU::V_AND_B32_e64 || Opc == AMDGPU::V_AND_B32_e32 ||
1798 Opc == AMDGPU::S_AND_B32) {
1799 if (Src1Val == 0) {
1800 // y = and x, 0 => y = v_mov_b32 0
1801 MI->removeOperand(Src0Idx);
1802 TII->mutateAndCleanupImplicit(
1803 *MI, TII->get(getMovOpc(Opc == AMDGPU::S_AND_B32)));
1804 } else if (Src1Val == -1) {
1805 // y = and x, -1 => y = copy x
1806 MI->removeOperand(Src1Idx);
1807 TII->mutateAndCleanupImplicit(*MI, TII->get(AMDGPU::COPY));
1808 } else
1809 return false;
1810
1811 return true;
1812 }
1813
1814 if (Opc == AMDGPU::V_XOR_B32_e64 || Opc == AMDGPU::V_XOR_B32_e32 ||
1815 Opc == AMDGPU::S_XOR_B32) {
1816 if (Src1Val == 0) {
1817 // y = xor x, 0 => y = copy x
1818 MI->removeOperand(Src1Idx);
1819 TII->mutateAndCleanupImplicit(*MI, TII->get(AMDGPU::COPY));
1820 return true;
1821 }
1822 }
1823
1824 return false;
1825}
1826
1827// Try to fold an instruction into a simpler one
1828bool SIFoldOperandsImpl::tryFoldCndMask(MachineInstr &MI) const {
1829 unsigned Opc = MI.getOpcode();
1830 if (Opc != AMDGPU::V_CNDMASK_B32_e32 && Opc != AMDGPU::V_CNDMASK_B32_e64 &&
1831 Opc != AMDGPU::V_CNDMASK_B64_PSEUDO)
1832 return false;
1833
1834 MachineOperand *Src0 = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
1835 MachineOperand *Src1 = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
1836 if (!Src1->isIdenticalTo(*Src0)) {
1837 std::optional<int64_t> Src1Imm = TII->getImmOrMaterializedImm(*MRI, *Src1);
1838 if (!Src1Imm)
1839 return false;
1840
1841 std::optional<int64_t> Src0Imm = TII->getImmOrMaterializedImm(*MRI, *Src0);
1842 if (!Src0Imm || *Src0Imm != *Src1Imm)
1843 return false;
1844 }
1845
1846 int Src1ModIdx =
1847 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1_modifiers);
1848 int Src0ModIdx =
1849 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0_modifiers);
1850 if ((Src1ModIdx != -1 && MI.getOperand(Src1ModIdx).getImm() != 0) ||
1851 (Src0ModIdx != -1 && MI.getOperand(Src0ModIdx).getImm() != 0))
1852 return false;
1853
1854 LLVM_DEBUG(dbgs() << "Folded " << MI << " into ");
1855 auto &NewDesc =
1856 TII->get(Src0->isReg() ? (unsigned)AMDGPU::COPY : getMovOpc(false));
1857 int Src2Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2);
1858 if (Src2Idx != -1)
1859 MI.removeOperand(Src2Idx);
1860 MI.removeOperand(AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1));
1861 if (Src1ModIdx != -1)
1862 MI.removeOperand(Src1ModIdx);
1863 if (Src0ModIdx != -1)
1864 MI.removeOperand(Src0ModIdx);
1865 TII->mutateAndCleanupImplicit(MI, NewDesc);
1866 LLVM_DEBUG(dbgs() << MI);
1867 return true;
1868}
1869
1870// Extract mask, register, and register operand index from an AND instruction.
1871// Immediate can be in operand 1 or 2.
1872std::optional<SIFoldOperandsImpl::ANDMaskResult>
1873SIFoldOperandsImpl::getANDMaskRegOperand(MachineInstr &AndMI) const {
1874 unsigned Opc = AndMI.getOpcode();
1875 if (Opc != AMDGPU::V_AND_B32_e64 && Opc != AMDGPU::V_AND_B32_e32 &&
1876 Opc != AMDGPU::S_AND_B32)
1877 return std::nullopt;
1878
1879 std::optional<int64_t> MaskImm =
1880 TII->getImmOrMaterializedImm(*MRI, AndMI.getOperand(1));
1881 if (MaskImm && AndMI.getOperand(2).isReg())
1882 return ANDMaskResult{*MaskImm, AndMI.getOperand(2).getReg(), 2};
1883
1884 MaskImm = TII->getImmOrMaterializedImm(*MRI, AndMI.getOperand(2));
1885 if (MaskImm && AndMI.getOperand(1).isReg())
1886 return ANDMaskResult{*MaskImm, AndMI.getOperand(1).getReg(), 1};
1887
1888 return std::nullopt;
1889}
1890
1891// Eliminate redundant 32-bit AND operations by detecting when ChildMI's mask
1892// contains ParentMI's mask.
1893//
1894// For example:
1895// ParentMI: %1 = AND %0, 0x7fff
1896// ChildMI: %2 = AND %1, 0xffff
1897//
1898// This also handles cases where ParentMI implicitly zeros high bits (e.g., f16
1899// operations that write 16-bit results into 32-bit registers), making a
1900// subsequent AND with 0xffff redundant.
1901bool SIFoldOperandsImpl::tryFoldRedundantAND(MachineInstr &ChildMI) const {
1902 // Ensure implicit defs (e.g., $scc) are not live.
1903 if (!ChildMI.allImplicitDefsAreDead())
1904 return false;
1905
1906 std::optional<ANDMaskResult> ChildResult = getANDMaskRegOperand(ChildMI);
1907 if (!ChildResult)
1908 return false;
1909
1910 if (!ChildResult->Reg.isVirtual())
1911 return false;
1912
1913 MachineInstr *ParentMI = MRI->getVRegDef(ChildResult->Reg);
1914 if (!ParentMI)
1915 return false;
1916
1917 int64_t ParentMask = 0;
1918 std::optional<ANDMaskResult> ParentResult = getANDMaskRegOperand(*ParentMI);
1919 if (ParentResult) {
1920 // Parent is an AND - extract its mask.
1921 ParentMask = ParentResult->Mask;
1922 } else if (ST->zeroesHigh16BitsOfDest(ParentMI->getOpcode())) {
1923 // Parent instruction implicitly zeros high 16 bits.
1924 ParentMask = 0xffff;
1925 } else {
1926 return false;
1927 }
1928
1929 // Check if ChildMI is not redundant.
1930 if ((ParentMask & ChildResult->Mask) != ParentMask)
1931 return false;
1932
1933 Register Dst = ChildMI.getOperand(0).getReg();
1934 Register Src = ChildResult->Reg;
1935
1936 // Src must be legal in every use of Dst. An S_AND_B32 parent with a
1937 // V_AND_B32 child defines Src in the scalar bank, and a use that requires a
1938 // VGPR does not accept it.
1939 if (!Dst.isVirtual() || !MRI->constrainRegClass(Src, MRI->getRegClass(Dst)))
1940 return false;
1941
1942 MRI->replaceRegWith(Dst, Src);
1943
1944 // Clear kill flags if the register operand is not marked as kill.
1945 if (!ChildMI.getOperand(ChildResult->RegIdx).isKill())
1946 MRI->clearKillFlags(Src);
1947
1948 ChildMI.eraseFromParent();
1949 return true;
1950}
1951
1952/// Remove S_AND of a lane mask with EXEC, when the lane mask is already known
1953/// to have 0 in the bits of all inactive lanes.
1954///
1955/// Instruction selection inserts these unconditionally because it has not
1956/// analysed what produced the lane mask.
1957bool SIFoldOperandsImpl::tryFoldAndExec(MachineInstr &MI) const {
1958 const AMDGPU::LaneMaskConstants &LMC = AMDGPU::LaneMaskConstants::get(*ST);
1959 if (MI.getOpcode() != LMC.AndOpc)
1960 return false;
1961
1962 // The AND is going to be removed, so nothing may use the SCC it defines.
1963 if (!MI.allImplicitDefsAreDead())
1964 return false;
1965
1966 // Find the EXEC operand, and the lane mask it is being ANDed with.
1967 unsigned ExecIdx = 0;
1968 for (unsigned I : {1u, 2u}) {
1969 const MachineOperand &MO = MI.getOperand(I);
1970 if (MO.isReg() && MO.getReg() == LMC.ExecReg)
1971 ExecIdx = I;
1972 }
1973 if (!ExecIdx)
1974 return false;
1975 MachineOperand &Src = MI.getOperand(3 - ExecIdx);
1976 if (!Src.isReg() || !Src.getReg().isVirtual() || Src.getSubReg())
1977 return false;
1978
1979 Register SrcReg = Src.getReg();
1980 if (!TII->isMaskedByExec(SrcReg, MI, *MRI))
1981 return false;
1982
1983 LLVM_DEBUG(dbgs() << "Folding redundant AND with EXEC: " << MI);
1984
1985 Register DstReg = MI.getOperand(0).getReg();
1986 if (DstReg.isVirtual()) {
1987 if (!MRI->constrainRegClass(SrcReg, MRI->getRegClass(DstReg)))
1988 return false;
1989 MRI->replaceRegWith(DstReg, SrcReg);
1990 } else {
1991 // A physical destination, e.g. the $vcc written by moveToVALU. Register
1992 // allocation will usually make this copy an identity copy.
1993 MachineBasicBlock *MBB = MI.getParent();
1994 BuildMI(*MBB, MI, MI.getDebugLoc(), TII->get(AMDGPU::COPY), DstReg)
1995 .addReg(SrcReg);
1996 }
1997
1998 if (!Src.isKill())
1999 MRI->clearKillFlags(SrcReg);
2000 MI.eraseFromParent();
2001 return true;
2002}
2003
2004bool SIFoldOperandsImpl::foldInstOperand(MachineInstr &MI,
2005 const FoldableDef &OpToFold) const {
2006 // We need mutate the operands of new mov instructions to add implicit
2007 // uses of EXEC, but adding them invalidates the use_iterator, so defer
2008 // this.
2009 SmallVector<MachineInstr *, 4> CopiesToReplace;
2011 MachineOperand &Dst = MI.getOperand(0);
2012 bool Changed = false;
2013
2015 llvm::make_pointer_range(MRI->use_nodbg_operands(Dst.getReg())));
2016 for (auto *U : UsesToProcess) {
2017 MachineInstr *UseMI = U->getParent();
2018
2019 FoldableDef SubOpToFold = OpToFold.getWithSubReg(*TRI, U->getSubReg());
2020 Changed |= foldOperand(SubOpToFold, UseMI, UseMI->getOperandNo(U), FoldList,
2021 CopiesToReplace);
2022 }
2023
2024 if (CopiesToReplace.empty() && FoldList.empty())
2025 return Changed;
2026
2027 // Make sure we add EXEC uses to any new v_mov instructions created.
2028 for (MachineInstr *Copy : CopiesToReplace)
2029 Copy->addImplicitDefUseOperands(*MF);
2030
2031 SetVector<MachineInstr *> ConstantFoldCandidates;
2032 for (FoldCandidate &Fold : FoldList) {
2033 assert(!Fold.isReg() || Fold.Def.OpToFold);
2034 if (Fold.isReg() && Fold.getReg().isVirtual()) {
2035 Register Reg = Fold.getReg();
2036 const MachineInstr *DefMI = Fold.Def.DefMI;
2037 if (DefMI->readsRegister(AMDGPU::EXEC, TRI) &&
2038 execMayBeModifiedBeforeUse(*MRI, Reg, *DefMI, *Fold.UseMI))
2039 continue;
2040 }
2041 if (updateOperand(Fold)) {
2042 // Clear kill flags.
2043 if (Fold.isReg()) {
2044 assert(Fold.Def.OpToFold && Fold.isReg());
2045 // FIXME: Probably shouldn't bother trying to fold if not an
2046 // SGPR. PeepholeOptimizer can eliminate redundant VGPR->VGPR
2047 // copies.
2048 MRI->clearKillFlags(Fold.getReg());
2049 }
2050 LLVM_DEBUG(dbgs() << "Folded source from " << MI << " into OpNo "
2051 << static_cast<int>(Fold.UseOpNo) << " of "
2052 << *Fold.UseMI);
2053
2054 if (Fold.isImm())
2055 ConstantFoldCandidates.insert(Fold.UseMI);
2056
2057 } else if (Fold.Commuted) {
2058 // Restoring instruction's original operand order if fold has failed.
2059 TII->commuteInstruction(*Fold.UseMI, false);
2060 }
2061 }
2062
2063 for (MachineInstr *MI : ConstantFoldCandidates) {
2064 if (tryConstantFoldOp(MI)) {
2065 LLVM_DEBUG(dbgs() << "Constant folded " << *MI);
2066 Changed = true;
2067 }
2068 }
2069 return true;
2070}
2071
2072/// Fold %agpr = COPY (REG_SEQUENCE x_MOV_B32, ...) into REG_SEQUENCE
2073/// (V_ACCVGPR_WRITE_B32_e64) ... depending on the reg_sequence input values.
2074bool SIFoldOperandsImpl::foldCopyToAGPRRegSequence(MachineInstr *CopyMI) const {
2075 // It is very tricky to store a value into an AGPR. v_accvgpr_write_b32 can
2076 // only accept VGPR or inline immediate. Recreate a reg_sequence with its
2077 // initializers right here, so we will rematerialize immediates and avoid
2078 // copies via different reg classes.
2079 const TargetRegisterClass *DefRC =
2080 MRI->getRegClass(CopyMI->getOperand(0).getReg());
2081 if (!TRI->isAGPRClass(DefRC))
2082 return false;
2083
2084 Register UseReg = CopyMI->getOperand(1).getReg();
2085 MachineInstr *RegSeq = MRI->getVRegDef(UseReg);
2086 if (!RegSeq || !RegSeq->isRegSequence())
2087 return false;
2088
2089 const DebugLoc &DL = CopyMI->getDebugLoc();
2090 MachineBasicBlock &MBB = *CopyMI->getParent();
2091
2092 MachineInstrBuilder B(*MBB.getParent(), CopyMI);
2093 DenseMap<TargetInstrInfo::RegSubRegPair, Register> VGPRCopies;
2094
2095 const TargetRegisterClass *UseRC =
2096 MRI->getRegClass(CopyMI->getOperand(1).getReg());
2097
2098 // Value, subregindex for new REG_SEQUENCE
2100
2101 unsigned NumRegSeqOperands = RegSeq->getNumOperands();
2102 unsigned NumFoldable = 0;
2103
2104 for (unsigned I = 1; I != NumRegSeqOperands; I += 2) {
2105 MachineOperand &RegOp = RegSeq->getOperand(I);
2106 unsigned SubRegIdx = RegSeq->getOperand(I + 1).getImm();
2107
2108 if (RegOp.getSubReg()) {
2109 // TODO: Handle subregister compose
2110 NewDefs.emplace_back(&RegOp, SubRegIdx);
2111 continue;
2112 }
2113
2114 MachineOperand *Lookup = lookUpCopyChain(*TII, *MRI, RegOp.getReg());
2115 if (!Lookup)
2116 Lookup = &RegOp;
2117
2118 if (Lookup->isImm()) {
2119 // Check if this is an agpr_32 subregister.
2120 const TargetRegisterClass *DestSuperRC = TRI->getMatchingSuperRegClass(
2121 DefRC, &AMDGPU::AGPR_32RegClass, SubRegIdx);
2122 if (DestSuperRC &&
2123 TII->isInlineConstant(*Lookup, AMDGPU::OPERAND_REG_INLINE_C_INT32)) {
2124 ++NumFoldable;
2125 NewDefs.emplace_back(Lookup, SubRegIdx);
2126 continue;
2127 }
2128 }
2129
2130 const TargetRegisterClass *InputRC =
2131 Lookup->isReg() ? MRI->getRegClass(Lookup->getReg())
2132 : MRI->getRegClass(RegOp.getReg());
2133
2134 // TODO: Account for Lookup->getSubReg()
2135
2136 // If we can't find a matching super class, this is an SGPR->AGPR or
2137 // VGPR->AGPR subreg copy (or something constant-like we have to materialize
2138 // in the AGPR). We can't directly copy from SGPR to AGPR on gfx908, so we
2139 // want to rewrite to copy to an intermediate VGPR class.
2140 const TargetRegisterClass *MatchRC =
2141 TRI->getMatchingSuperRegClass(DefRC, InputRC, SubRegIdx);
2142 if (!MatchRC) {
2143 ++NumFoldable;
2144 NewDefs.emplace_back(&RegOp, SubRegIdx);
2145 continue;
2146 }
2147
2148 NewDefs.emplace_back(&RegOp, SubRegIdx);
2149 }
2150
2151 // Do not clone a reg_sequence and merely change the result register class.
2152 if (NumFoldable == 0)
2153 return false;
2154
2155 CopyMI->setDesc(TII->get(AMDGPU::REG_SEQUENCE));
2156 for (unsigned I = CopyMI->getNumOperands() - 1; I > 0; --I)
2157 CopyMI->removeOperand(I);
2158
2159 for (auto [Def, DestSubIdx] : NewDefs) {
2160 if (!Def->isReg()) {
2161 // TODO: Should we use single write for each repeated value like in
2162 // register case?
2163 Register Tmp = MRI->createVirtualRegister(&AMDGPU::AGPR_32RegClass);
2164 BuildMI(MBB, CopyMI, DL, TII->get(AMDGPU::V_ACCVGPR_WRITE_B32_e64), Tmp)
2165 .add(*Def);
2166 B.addReg(Tmp);
2167 } else {
2168 TargetInstrInfo::RegSubRegPair Src = getRegSubRegPair(*Def);
2169 Def->setIsKill(false);
2170
2171 Register &VGPRCopy = VGPRCopies[Src];
2172 if (!VGPRCopy) {
2173 const TargetRegisterClass *VGPRUseSubRC =
2174 TRI->getSubRegisterClass(UseRC, DestSubIdx);
2175
2176 // We cannot build a reg_sequence out of the same registers, they
2177 // must be copied. Better do it here before copyPhysReg() created
2178 // several reads to do the AGPR->VGPR->AGPR copy.
2179
2180 // Direct copy from SGPR to AGPR is not possible on gfx908. To avoid
2181 // creation of exploded copies SGPR->VGPR->AGPR in the copyPhysReg()
2182 // later, create a copy here and track if we already have such a copy.
2183 const TargetRegisterClass *SubRC =
2184 TRI->getSubRegisterClass(MRI->getRegClass(Src.Reg), Src.SubReg);
2185 if (!VGPRUseSubRC->hasSubClassEq(SubRC)) {
2186 // TODO: Try to reconstrain class
2187 VGPRCopy = MRI->createVirtualRegister(VGPRUseSubRC);
2188 BuildMI(MBB, CopyMI, DL, TII->get(AMDGPU::COPY), VGPRCopy).add(*Def);
2189 B.addReg(VGPRCopy);
2190 } else {
2191 // If it is already a VGPR, do not copy the register.
2192 B.add(*Def);
2193 }
2194 } else {
2195 B.addReg(VGPRCopy);
2196 }
2197 }
2198
2199 B.addImm(DestSubIdx);
2200 }
2201
2202 LLVM_DEBUG(dbgs() << "Folded " << *CopyMI);
2203 return true;
2204}
2205
2206bool SIFoldOperandsImpl::tryFoldFoldableCopy(
2207 MachineInstr &MI, MachineOperand *&CurrentKnownM0Val) const {
2208 Register DstReg = MI.getOperand(0).getReg();
2209 // Specially track simple redefs of m0 to the same value in a block, so we
2210 // can erase the later ones.
2211 if (DstReg == AMDGPU::M0) {
2212 MachineOperand &NewM0Val = MI.getOperand(1);
2213 if (CurrentKnownM0Val && CurrentKnownM0Val->isIdenticalTo(NewM0Val)) {
2214 MI.eraseFromParent();
2215 return true;
2216 }
2217
2218 // We aren't tracking other physical registers
2219 CurrentKnownM0Val = (NewM0Val.isReg() && NewM0Val.getReg().isPhysical())
2220 ? nullptr
2221 : &NewM0Val;
2222 return false;
2223 }
2224
2225 MachineOperand *OpToFoldPtr;
2226 if (MI.getOpcode() == AMDGPU::V_MOV_B16_t16_e64) {
2227 // Folding when any src_modifiers are non-zero is unsupported
2228 if (TII->hasAnyModifiersSet(MI))
2229 return false;
2230 OpToFoldPtr = &MI.getOperand(2);
2231 } else
2232 OpToFoldPtr = &MI.getOperand(1);
2233 MachineOperand &OpToFold = *OpToFoldPtr;
2234 bool FoldingImm = OpToFold.isImm() || OpToFold.isFI() || OpToFold.isGlobal();
2235
2236 // FIXME: We could also be folding things like TargetIndexes.
2237 if (!FoldingImm && !OpToFold.isReg())
2238 return false;
2239
2240 // Fold virtual registers and constant physical registers.
2241 if (OpToFold.isReg() && OpToFold.getReg().isPhysical() &&
2242 !TRI->isConstantPhysReg(OpToFold.getReg()))
2243 return false;
2244
2245 // Prevent folding operands backwards in the function. For example,
2246 // the COPY opcode must not be replaced by 1 in this example:
2247 //
2248 // %3 = COPY %vgpr0; VGPR_32:%3
2249 // ...
2250 // %vgpr0 = V_MOV_B32_e32 1, implicit %exec
2251 if (!DstReg.isVirtual())
2252 return false;
2253
2254 const TargetRegisterClass *DstRC =
2255 MRI->getRegClass(MI.getOperand(0).getReg());
2256
2257 // True16: Fix malformed 16-bit sgpr COPY produced by peephole-opt
2258 // Can remove this code if proper 16-bit SGPRs are implemented
2259 // Example: Pre-peephole-opt
2260 // %29:sgpr_lo16 = COPY %16.lo16:sreg_32
2261 // %32:sreg_32 = COPY %29:sgpr_lo16
2262 // %30:sreg_32 = S_PACK_LL_B32_B16 killed %31:sreg_32, killed %32:sreg_32
2263 // Post-peephole-opt and DCE
2264 // %32:sreg_32 = COPY %16.lo16:sreg_32
2265 // %30:sreg_32 = S_PACK_LL_B32_B16 killed %31:sreg_32, killed %32:sreg_32
2266 // After this transform
2267 // %32:sreg_32 = COPY %16:sreg_32
2268 // %30:sreg_32 = S_PACK_LL_B32_B16 killed %31:sreg_32, killed %32:sreg_32
2269 // After the fold operands pass
2270 // %30:sreg_32 = S_PACK_LL_B32_B16 killed %31:sreg_32, killed %16:sreg_32
2271 if (MI.getOpcode() == AMDGPU::COPY && OpToFold.isReg() &&
2272 OpToFold.getSubReg()) {
2273 if (DstRC == &AMDGPU::SReg_32RegClass &&
2274 DstRC == MRI->getRegClass(OpToFold.getReg())) {
2275 if (!TRI->getMatchingSuperRegClass(DstRC, &AMDGPU::SGPR_LO16RegClass,
2276 OpToFold.getSubReg()))
2277 return false;
2278 OpToFold.setSubReg(0);
2279 }
2280 }
2281
2282 // Fold copy to AGPR through reg_sequence
2283 // TODO: Handle with subregister extract
2284 if (OpToFold.isReg() && MI.isCopy() && !MI.getOperand(1).getSubReg()) {
2285 if (foldCopyToAGPRRegSequence(&MI))
2286 return true;
2287 }
2288
2289 FoldableDef Def(OpToFold, DstRC);
2290 bool Changed = foldInstOperand(MI, Def);
2291
2292 // If we managed to fold all uses of this copy then we might as well
2293 // delete it now.
2294 // The only reason we need to follow chains of copies here is that
2295 // tryFoldRegSequence looks forward through copies before folding a
2296 // REG_SEQUENCE into its eventual users.
2297 auto *InstToErase = &MI;
2298 while (MRI->use_nodbg_empty(InstToErase->getOperand(0).getReg())) {
2299 auto &SrcOp = InstToErase->getOperand(1);
2300 auto SrcReg = SrcOp.isReg() ? SrcOp.getReg() : Register();
2301 InstToErase->eraseFromParent();
2302 Changed = true;
2303 InstToErase = nullptr;
2304 if (!SrcReg || SrcReg.isPhysical())
2305 break;
2306 InstToErase = MRI->getVRegDef(SrcReg);
2307 if (!InstToErase || !TII->isFoldableCopy(*InstToErase))
2308 break;
2309 }
2310
2311 if (InstToErase && InstToErase->isRegSequence() &&
2312 MRI->use_nodbg_empty(InstToErase->getOperand(0).getReg())) {
2313 InstToErase->eraseFromParent();
2314 Changed = true;
2315 }
2316
2317 if (Changed)
2318 return true;
2319
2320 // Run this after foldInstOperand to avoid turning scalar additions into
2321 // vector additions when the result scalar result could just be folded into
2322 // the user(s).
2323 return OpToFold.isReg() &&
2324 foldCopyToVGPROfScalarAddOfFrameIndex(DstReg, OpToFold.getReg(), MI);
2325}
2326
2327// Clamp patterns are canonically selected to v_max_* instructions, so only
2328// handle them.
2329const MachineOperand *
2330SIFoldOperandsImpl::isClamp(const MachineInstr &MI) const {
2331 unsigned Op = MI.getOpcode();
2332 switch (Op) {
2333 case AMDGPU::V_MAX_F32_e64:
2334 case AMDGPU::V_MAX_F16_e64:
2335 case AMDGPU::V_MAX_F16_t16_e64:
2336 case AMDGPU::V_MAX_F16_fake16_e64:
2337 case AMDGPU::V_MAX_F64_e64:
2338 case AMDGPU::V_MAX_NUM_F64_e64:
2339 case AMDGPU::V_PK_MAX_F16:
2340 case AMDGPU::V_MAX_BF16_PSEUDO_e64:
2341 case AMDGPU::V_PK_MAX_NUM_BF16: {
2342 if (MI.mayRaiseFPException())
2343 return nullptr;
2344
2345 if (!TII->getNamedOperand(MI, AMDGPU::OpName::clamp)->getImm())
2346 return nullptr;
2347
2348 // Make sure sources are identical.
2349 const MachineOperand *Src0 = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
2350 const MachineOperand *Src1 = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
2351 if (!Src0->isReg() || !Src1->isReg() ||
2352 Src0->getReg() != Src1->getReg() ||
2353 Src0->getSubReg() != Src1->getSubReg() ||
2354 Src0->getSubReg() != AMDGPU::NoSubRegister)
2355 return nullptr;
2356
2357 // Can't fold up if we have modifiers.
2358 if (TII->hasModifiersSet(MI, AMDGPU::OpName::omod))
2359 return nullptr;
2360
2361 unsigned Src0Mods
2362 = TII->getNamedOperand(MI, AMDGPU::OpName::src0_modifiers)->getImm();
2363 unsigned Src1Mods
2364 = TII->getNamedOperand(MI, AMDGPU::OpName::src1_modifiers)->getImm();
2365
2366 // Having a 0 op_sel_hi would require swizzling the output in the source
2367 // instruction, which we can't do.
2368 unsigned UnsetMods =
2369 (Op == AMDGPU::V_PK_MAX_F16 || Op == AMDGPU::V_PK_MAX_NUM_BF16)
2371 : 0u;
2372 if (Src0Mods != UnsetMods || Src1Mods != UnsetMods)
2373 return nullptr;
2374 return Src0;
2375 }
2376 default:
2377 return nullptr;
2378 }
2379}
2380
2381// FIXME: Clamp for v_mad_mixhi_f16 handled during isel.
2382bool SIFoldOperandsImpl::tryFoldClamp(MachineInstr &MI) {
2383 const MachineOperand *ClampSrc = isClamp(MI);
2384 if (!ClampSrc || !MRI->hasOneNonDBGUser(ClampSrc->getReg()))
2385 return false;
2386
2387 if (!ClampSrc->getReg().isVirtual())
2388 return false;
2389
2390 // Look through COPY. COPY only observed with True16.
2391 Register DefSrcReg = TRI->lookThruCopyLike(ClampSrc->getReg(), MRI);
2392 MachineInstr *Def =
2393 MRI->getVRegDef(DefSrcReg.isVirtual() ? DefSrcReg : ClampSrc->getReg());
2394
2395 // The type of clamp must be compatible.
2396 if (!SIInstrInfo::hasSameClamp(*Def, MI))
2397 return false;
2398
2399 if (Def->mayRaiseFPException())
2400 return false;
2401
2402 MachineOperand *DefClamp = TII->getNamedOperand(*Def, AMDGPU::OpName::clamp);
2403 if (!DefClamp)
2404 return false;
2405
2406 LLVM_DEBUG(dbgs() << "Folding clamp " << *DefClamp << " into " << *Def);
2407
2408 // Clamp is applied after omod, so it is OK if omod is set.
2409 DefClamp->setImm(1);
2410
2411 Register DefReg = Def->getOperand(0).getReg();
2412 Register MIDstReg = MI.getOperand(0).getReg();
2413 if (TRI->isSGPRReg(*MRI, DefReg)) {
2414 // Pseudo scalar instructions have a SGPR for dst and clamp is a v_max*
2415 // instruction with a VGPR dst.
2416 BuildMI(*MI.getParent(), MI, MI.getDebugLoc(), TII->get(AMDGPU::COPY),
2417 MIDstReg)
2418 .addReg(DefReg);
2419 } else {
2420 MRI->replaceRegWith(MIDstReg, DefReg);
2421 }
2422 MI.eraseFromParent();
2423
2424 // Use of output modifiers forces VOP3 encoding for a VOP2 mac/fmac
2425 // instruction, so we might as well convert it to the more flexible VOP3-only
2426 // mad/fma form.
2427 if (TII->convertToThreeAddress(*Def, /*LIS=*/nullptr))
2428 Def->eraseFromParent();
2429
2430 return true;
2431}
2432
2433static int getOModValue(unsigned Opc, int64_t Val) {
2434 switch (Opc) {
2435 case AMDGPU::V_MUL_F64_e64:
2436 case AMDGPU::V_MUL_F64_pseudo_e64: {
2437 switch (Val) {
2438 case 0x3fe0000000000000: // 0.5
2439 return SIOutMods::DIV2;
2440 case 0x4000000000000000: // 2.0
2441 return SIOutMods::MUL2;
2442 case 0x4010000000000000: // 4.0
2443 return SIOutMods::MUL4;
2444 default:
2445 return SIOutMods::NONE;
2446 }
2447 }
2448 case AMDGPU::V_MUL_F32_e64: {
2449 switch (static_cast<uint32_t>(Val)) {
2450 case 0x3f000000: // 0.5
2451 return SIOutMods::DIV2;
2452 case 0x40000000: // 2.0
2453 return SIOutMods::MUL2;
2454 case 0x40800000: // 4.0
2455 return SIOutMods::MUL4;
2456 default:
2457 return SIOutMods::NONE;
2458 }
2459 }
2460 case AMDGPU::V_MUL_F16_e64:
2461 case AMDGPU::V_MUL_F16_t16_e64:
2462 case AMDGPU::V_MUL_F16_fake16_e64: {
2463 switch (static_cast<uint16_t>(Val)) {
2464 case 0x3800: // 0.5
2465 return SIOutMods::DIV2;
2466 case 0x4000: // 2.0
2467 return SIOutMods::MUL2;
2468 case 0x4400: // 4.0
2469 return SIOutMods::MUL4;
2470 default:
2471 return SIOutMods::NONE;
2472 }
2473 }
2474 case AMDGPU::V_PK_MUL_BF16: {
2475 switch (static_cast<uint16_t>(Val)) {
2476 case 0x3F00: // 0.5 in BF16
2477 return SIOutMods::DIV2;
2478 case 0x4000: // 2.0 in BF16
2479 return SIOutMods::MUL2;
2480 case 0x4080: // 4.0 in BF16
2481 return SIOutMods::MUL4;
2482 default:
2483 return SIOutMods::NONE;
2484 }
2485 }
2486 default:
2487 llvm_unreachable("invalid mul opcode");
2488 }
2489}
2490
2491// FIXME: Does this really not support denormals with f16?
2492// FIXME: Does this need to check IEEE mode bit? SNaNs are generally not
2493// handled, so will anything other than that break?
2494std::pair<const MachineOperand *, int>
2495SIFoldOperandsImpl::isOMod(const MachineInstr &MI) const {
2496 unsigned Op = MI.getOpcode();
2497 switch (Op) {
2498 case AMDGPU::V_MUL_F64_e64:
2499 case AMDGPU::V_MUL_F64_pseudo_e64:
2500 case AMDGPU::V_MUL_F32_e64:
2501 case AMDGPU::V_MUL_F16_t16_e64:
2502 case AMDGPU::V_MUL_F16_fake16_e64:
2503 case AMDGPU::V_MUL_F16_e64: {
2504 // If output denormals are enabled, omod is ignored.
2505 if ((Op == AMDGPU::V_MUL_F32_e64 &&
2507 ((Op == AMDGPU::V_MUL_F64_e64 || Op == AMDGPU::V_MUL_F64_pseudo_e64 ||
2508 Op == AMDGPU::V_MUL_F16_e64 || Op == AMDGPU::V_MUL_F16_t16_e64 ||
2509 Op == AMDGPU::V_MUL_F16_fake16_e64) &&
2512 MI.mayRaiseFPException())
2513 return {nullptr, SIOutMods::NONE};
2514
2515 const MachineOperand *Src0 = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
2516 const MachineOperand *Src1 = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
2517
2518 // If there is an immediate operand, it must be Src1
2519 std::optional<int64_t> Src1Imm = TII->getImmOrMaterializedImm(*MRI, *Src1);
2520 if (!Src1Imm)
2521 return {nullptr, SIOutMods::NONE};
2522
2523 int OMod = getOModValue(Op, *Src1Imm);
2524 if (OMod == SIOutMods::NONE ||
2525 TII->hasModifiersSet(MI, AMDGPU::OpName::src0_modifiers) ||
2526 TII->hasModifiersSet(MI, AMDGPU::OpName::src1_modifiers) ||
2527 TII->hasModifiersSet(MI, AMDGPU::OpName::omod) ||
2528 TII->hasModifiersSet(MI, AMDGPU::OpName::clamp))
2529 return {nullptr, SIOutMods::NONE};
2530
2531 return {Src0, OMod};
2532 }
2533 case AMDGPU::V_ADD_F64_e64:
2534 case AMDGPU::V_ADD_F64_pseudo_e64:
2535 case AMDGPU::V_ADD_F32_e64:
2536 case AMDGPU::V_ADD_F16_e64:
2537 case AMDGPU::V_ADD_F16_t16_e64:
2538 case AMDGPU::V_ADD_F16_fake16_e64: {
2539 // If output denormals are enabled, omod is ignored.
2540 if ((Op == AMDGPU::V_ADD_F32_e64 &&
2542 ((Op == AMDGPU::V_ADD_F64_e64 || Op == AMDGPU::V_ADD_F64_pseudo_e64 ||
2543 Op == AMDGPU::V_ADD_F16_e64 || Op == AMDGPU::V_ADD_F16_t16_e64 ||
2544 Op == AMDGPU::V_ADD_F16_fake16_e64) &&
2546 return {nullptr, SIOutMods::NONE};
2547
2548 // Look through the DAGCombiner canonicalization fmul x, 2 -> fadd x, x
2549 const MachineOperand *Src0 = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
2550 const MachineOperand *Src1 = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
2551
2552 if (Src0->isReg() && Src1->isReg() && Src0->getReg() == Src1->getReg() &&
2553 Src0->getSubReg() == Src1->getSubReg() &&
2554 !TII->hasModifiersSet(MI, AMDGPU::OpName::src0_modifiers) &&
2555 !TII->hasModifiersSet(MI, AMDGPU::OpName::src1_modifiers) &&
2556 !TII->hasModifiersSet(MI, AMDGPU::OpName::clamp) &&
2557 !TII->hasModifiersSet(MI, AMDGPU::OpName::omod))
2558 return {Src0, SIOutMods::MUL2};
2559
2560 return {nullptr, SIOutMods::NONE};
2561 }
2562 case AMDGPU::V_PK_MUL_BF16: {
2563 // OMOD folding for BF16 packed multiply. bf16 has no denormal mode of its
2564 // own; it follows the default ("denormal-fp-math") mode, which is the same
2565 // field as f64/f16.
2567 MI.mayRaiseFPException())
2568 return {nullptr, SIOutMods::NONE};
2569
2570 const MachineOperand *Src0 = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
2571 const MachineOperand *Src1 = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
2572
2573 // If there is an immediate operand, it must be Src1
2574 std::optional<int64_t> Src1Imm = TII->getImmOrMaterializedImm(*MRI, *Src1);
2575 if (!Src1Imm)
2576 return {nullptr, SIOutMods::NONE};
2577
2578 int OMod = getOModValue(AMDGPU::V_PK_MUL_BF16, *Src1Imm);
2579 if (OMod == SIOutMods::NONE)
2580 return {nullptr, SIOutMods::NONE};
2581
2582 // Modifiers other than op_sel_hi block OMOD folding. Per getOModValue
2583 // above, Src1 is an inline constant (0.5/2.0/4.0), which may carry
2584 // op_sel_lo to read it from the upper FP32 half, so allow that on src1.
2585 const MachineOperand *Src0Mods =
2586 TII->getNamedOperand(MI, AMDGPU::OpName::src0_modifiers);
2587 const MachineOperand *Src1Mods =
2588 TII->getNamedOperand(MI, AMDGPU::OpName::src1_modifiers);
2589 if ((Src0Mods->getImm() & ~SISrcMods::OP_SEL_1) ||
2590 (Src1Mods->getImm() & ~(SISrcMods::OP_SEL_0 | SISrcMods::OP_SEL_1)) ||
2591 TII->hasModifiersSet(MI, AMDGPU::OpName::omod) ||
2592 TII->hasModifiersSet(MI, AMDGPU::OpName::clamp))
2593 return {nullptr, SIOutMods::NONE};
2594
2595 return {Src0, OMod};
2596 }
2597 case AMDGPU::V_PK_ADD_BF16: {
2598 // OMOD folding for BF16 packed add: x + x -> x * 2. See the bf16 denormal
2599 // mode note in the V_PK_MUL_BF16 case above.
2601 return {nullptr, SIOutMods::NONE};
2602
2603 const MachineOperand *Src0 = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
2604 const MachineOperand *Src1 = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
2605
2606 if (!Src0->isReg() || !Src1->isReg() || Src0->getReg() != Src1->getReg() ||
2607 Src0->getSubReg() != Src1->getSubReg())
2608 return {nullptr, SIOutMods::NONE};
2609
2610 // Modifiers other than op_sel_hi block OMOD folding
2611 const MachineOperand *Src0Mods =
2612 TII->getNamedOperand(MI, AMDGPU::OpName::src0_modifiers);
2613 const MachineOperand *Src1Mods =
2614 TII->getNamedOperand(MI, AMDGPU::OpName::src1_modifiers);
2615 if ((Src0Mods->getImm() & ~SISrcMods::OP_SEL_1) ||
2616 (Src1Mods->getImm() & ~SISrcMods::OP_SEL_1) ||
2617 TII->hasModifiersSet(MI, AMDGPU::OpName::omod) ||
2618 TII->hasModifiersSet(MI, AMDGPU::OpName::clamp))
2619 return {nullptr, SIOutMods::NONE};
2620
2621 return {Src0, SIOutMods::MUL2};
2622 }
2623 default:
2624 return {nullptr, SIOutMods::NONE};
2625 }
2626}
2627
2628// FIXME: Does this need to check IEEE bit on function?
2629bool SIFoldOperandsImpl::tryFoldOMod(MachineInstr &MI) {
2630 const MachineOperand *RegOp;
2631 int OMod;
2632 std::tie(RegOp, OMod) = isOMod(MI);
2633 if (OMod == SIOutMods::NONE || !RegOp->isReg() ||
2634 RegOp->getSubReg() != AMDGPU::NoSubRegister ||
2635 !MRI->hasOneNonDBGUser(RegOp->getReg()))
2636 return false;
2637
2638 MachineInstr *Def = MRI->getVRegDef(RegOp->getReg());
2639 Register OModSrcReg = Def->getOperand(0).getReg();
2640
2641 // In real-true16 mode, vgpr_16 results are packed into vgpr_32 via
2642 // REG_SEQUENCE. Look through it to find the actual instruction.
2643 if (Def->isRegSequence() && Def->getNumOperands() == 5 &&
2644 Def->getOperand(2).getImm() == AMDGPU::lo16) {
2645 // Only look through if the high 16 bits are undefined
2646 bool CanLookThrough = true;
2647 MachineInstr *Hi16Def = MRI->getVRegDef(Def->getOperand(3).getReg());
2648 if (!Hi16Def || !Hi16Def->isImplicitDef())
2649 CanLookThrough = false;
2650
2651 if (CanLookThrough) {
2652 Register SrcReg = Def->getOperand(1).getReg();
2653 if (!MRI->hasOneNonDBGUse(SrcReg))
2654 return false;
2655
2656 Def = MRI->getVRegDef(SrcReg);
2657 if (!Def)
2658 return false;
2659 }
2660 }
2661
2662 MachineOperand *DefOMod = TII->getNamedOperand(*Def, AMDGPU::OpName::omod);
2663 if (!DefOMod || DefOMod->getImm() != SIOutMods::NONE)
2664 return false;
2665
2666 if (Def->mayRaiseFPException())
2667 return false;
2668
2669 // Clamp is applied after omod. If the source already has clamp set, don't
2670 // fold it.
2671 if (TII->hasModifiersSet(*Def, AMDGPU::OpName::clamp))
2672 return false;
2673
2674 LLVM_DEBUG(dbgs() << "Folding omod " << MI << " into " << *Def);
2675
2676 DefOMod->setImm(OMod);
2677 MRI->replaceRegWith(MI.getOperand(0).getReg(), OModSrcReg);
2678 // Kill flags can be wrong if we replaced a def inside a loop with a def
2679 // outside the loop.
2680 MRI->clearKillFlags(OModSrcReg);
2681 MI.eraseFromParent();
2682
2683 // Use of output modifiers forces VOP3 encoding for a VOP2 mac/fmac
2684 // instruction, so we might as well convert it to the more flexible VOP3-only
2685 // mad/fma form.
2686 if (TII->convertToThreeAddress(*Def, /*LIS=*/nullptr))
2687 Def->eraseFromParent();
2688
2689 return true;
2690}
2691
2692// Try to optimize SGPR reg sequences that are splat <s, s> or <s, s, s, s>
2693// where all uses are PackedSingleSGPR64BitInst, replacing with <s, undef, ...>
2694bool SIFoldOperandsImpl::tryFoldSGPRSplatRegSequence(MachineInstr &MI) {
2695 assert(MI.isRegSequence());
2696
2697 if (!ST->hasPackedFP64SingleSGPROps() && !ST->hasPackedU64SingleSGPROps())
2698 return false;
2699
2700 Register Reg = MI.getOperand(0).getReg();
2701
2702 // Only optimize 128-bit SGPR register sequences
2703 const TargetRegisterClass *RegClass = MRI->getRegClass(Reg);
2704 if (!TRI->isSGPRClass(RegClass) || TRI->getRegSizeInBits(*RegClass) != 128)
2705 return false;
2706
2708 if (!getRegSeqInit(Defs, Reg))
2709 return false;
2710
2711 // Check if this is a splat pattern
2712 if (Defs.size() <= 1)
2713 return false;
2714
2715 const auto &[FirstOp, _] = Defs.front();
2716 if (!FirstOp->isReg())
2717 return false;
2718
2719 Register FirstReg = FirstOp->getReg();
2720 unsigned FirstSubReg = FirstOp->getSubReg();
2721
2722 const TargetRegisterClass *FirstRegClass = MRI->getRegClass(FirstReg);
2723 if (!TRI->isSGPRClass(FirstRegClass))
2724 return false;
2725
2726 // Check remaining elements match first
2727 if (!llvm::all_of(llvm::drop_begin(Defs), [&](const auto &Def) {
2728 const auto &[Op, _] = Def;
2729 return Op->isReg() && Op->getReg() == FirstReg &&
2730 Op->getSubReg() == FirstSubReg;
2731 }))
2732 return false;
2733
2734 // Check if all uses are isSingleSGPRReadInst
2735 for (MachineInstr &UseMI : MRI->use_nodbg_instructions(Reg)) {
2737 return false;
2738 }
2739
2740 // Create new reg sequence with <s, undef, undef, ...>
2741 Register NewDst = MRI->createVirtualRegister(RegClass);
2742 MachineInstrBuilder RS = BuildMI(*MI.getParent(), MI, MI.getDebugLoc(),
2743 TII->get(AMDGPU::REG_SEQUENCE), NewDst);
2744
2745 // Add the first operand
2746 FirstOp->setIsKill(false);
2747 RS.add(*FirstOp);
2748 RS.addImm(Defs[0].second);
2749
2750 // Add undef for remaining lanes
2751 // Create an undef virtual register for the same register class
2752 Register UndefReg = MRI->createVirtualRegister(FirstRegClass);
2753 for (unsigned i = 1; i < Defs.size(); ++i) {
2754 RS.addReg(UndefReg, RegState::Undef);
2755 RS.addImm(Defs[i].second);
2756 }
2757
2758 // Replace all uses
2759 MRI->replaceRegWith(Reg, NewDst);
2760
2761 LLVM_DEBUG(dbgs() << "Folded splat SGPR reg_sequence: " << MI << " into "
2762 << *RS);
2763
2764 MI.eraseFromParent();
2765 return true;
2766}
2767
2768// Try to fold a reg_sequence with vgpr output and agpr inputs into an
2769// instruction which can take an agpr. So far that means a store.
2770bool SIFoldOperandsImpl::tryFoldRegSequence(MachineInstr &MI) {
2771 assert(MI.isRegSequence());
2772
2773 // Try to optimize SGPR splat sequences first
2774 if (tryFoldSGPRSplatRegSequence(MI))
2775 return true;
2776
2777 auto Reg = MI.getOperand(0).getReg();
2778
2779 if (!ST->hasGFX90AInsts() || !TRI->isVGPR(*MRI, Reg) ||
2780 !MRI->hasOneNonDBGUse(Reg))
2781 return false;
2782
2784 if (!getRegSeqInit(Defs, Reg))
2785 return false;
2786
2787 for (auto &[Op, SubIdx] : Defs) {
2788 if (!Op->isReg())
2789 return false;
2790 if (TRI->isAGPR(*MRI, Op->getReg()))
2791 continue;
2792 // Maybe this is a COPY from AREG
2793 const MachineInstr *SubDef = MRI->getVRegDef(Op->getReg());
2794 if (!SubDef || !SubDef->isCopy() || SubDef->getOperand(1).getSubReg())
2795 return false;
2796 if (!TRI->isAGPR(*MRI, SubDef->getOperand(1).getReg()))
2797 return false;
2798 }
2799
2800 MachineOperand *Op = &*MRI->use_nodbg_begin(Reg);
2801 MachineInstr *UseMI = Op->getParent();
2802 while (UseMI->isCopy() && !Op->getSubReg()) {
2803 Reg = UseMI->getOperand(0).getReg();
2804 if (!TRI->isVGPR(*MRI, Reg) || !MRI->hasOneNonDBGUse(Reg))
2805 return false;
2806 Op = &*MRI->use_nodbg_begin(Reg);
2807 UseMI = Op->getParent();
2808 }
2809
2810 if (Op->getSubReg())
2811 return false;
2812
2813 unsigned OpIdx = Op - &UseMI->getOperand(0);
2814 const MCInstrDesc &InstDesc = UseMI->getDesc();
2815 const TargetRegisterClass *OpRC = TII->getRegClass(InstDesc, OpIdx);
2816 if (!OpRC || !TRI->isVectorSuperClass(OpRC))
2817 return false;
2818
2819 const auto *NewDstRC = TRI->getEquivalentAGPRClass(MRI->getRegClass(Reg));
2820 auto Dst = MRI->createVirtualRegister(NewDstRC);
2821 auto RS = BuildMI(*MI.getParent(), MI, MI.getDebugLoc(),
2822 TII->get(AMDGPU::REG_SEQUENCE), Dst);
2823
2824 for (auto &[Def, SubIdx] : Defs) {
2825 Def->setIsKill(false);
2826 if (TRI->isAGPR(*MRI, Def->getReg())) {
2827 RS.add(*Def);
2828 } else { // This is a copy
2829 MachineInstr *SubDef = MRI->getVRegDef(Def->getReg());
2830 SubDef->getOperand(1).setIsKill(false);
2831 RS.addReg(SubDef->getOperand(1).getReg(), {}, Def->getSubReg());
2832 }
2833 RS.addImm(SubIdx);
2834 }
2835
2836 Op->setReg(Dst);
2837 if (!TII->isOperandLegal(*UseMI, OpIdx, Op)) {
2838 Op->setReg(Reg);
2839 RS->eraseFromParent();
2840 return false;
2841 }
2842
2843 LLVM_DEBUG(dbgs() << "Folded " << *RS << " into " << *UseMI);
2844
2845 // Erase the REG_SEQUENCE eagerly, unless we followed a chain of COPY users,
2846 // in which case we can erase them all later in runOnMachineFunction.
2847 if (MRI->use_nodbg_empty(MI.getOperand(0).getReg()))
2848 MI.eraseFromParent();
2849 return true;
2850}
2851
2852/// Checks whether \p Copy is a AGPR -> VGPR copy. Returns `true` on success and
2853/// stores the AGPR register in \p OutReg and the subreg in \p OutSubReg
2854static bool isAGPRCopy(const SIRegisterInfo &TRI,
2855 const MachineRegisterInfo &MRI, const MachineInstr &Copy,
2856 Register &OutReg, unsigned &OutSubReg) {
2857 assert(Copy.isCopy());
2858
2859 const MachineOperand &CopySrc = Copy.getOperand(1);
2860 Register CopySrcReg = CopySrc.getReg();
2861 if (!CopySrcReg.isVirtual())
2862 return false;
2863
2864 // Common case: copy from AGPR directly, e.g.
2865 // %1:vgpr_32 = COPY %0:agpr_32
2866 if (TRI.isAGPR(MRI, CopySrcReg)) {
2867 OutReg = CopySrcReg;
2868 OutSubReg = CopySrc.getSubReg();
2869 return true;
2870 }
2871
2872 // Sometimes it can also involve two copies, e.g.
2873 // %1:vgpr_256 = COPY %0:agpr_256
2874 // %2:vgpr_32 = COPY %1:vgpr_256.sub0
2875 const MachineInstr *CopySrcDef = MRI.getVRegDef(CopySrcReg);
2876 if (!CopySrcDef || !CopySrcDef->isCopy())
2877 return false;
2878
2879 const MachineOperand &OtherCopySrc = CopySrcDef->getOperand(1);
2880 Register OtherCopySrcReg = OtherCopySrc.getReg();
2881 if (!OtherCopySrcReg.isVirtual() ||
2882 CopySrcDef->getOperand(0).getSubReg() != AMDGPU::NoSubRegister ||
2883 OtherCopySrc.getSubReg() != AMDGPU::NoSubRegister ||
2884 !TRI.isAGPR(MRI, OtherCopySrcReg))
2885 return false;
2886
2887 OutReg = OtherCopySrcReg;
2888 OutSubReg = CopySrc.getSubReg();
2889 return true;
2890}
2891
2892// Try to hoist an AGPR to VGPR copy across a PHI.
2893// This should allow folding of an AGPR into a consumer which may support it.
2894//
2895// Example 1: LCSSA PHI
2896// loop:
2897// %1:vreg = COPY %0:areg
2898// exit:
2899// %2:vreg = PHI %1:vreg, %loop
2900// =>
2901// loop:
2902// exit:
2903// %1:areg = PHI %0:areg, %loop
2904// %2:vreg = COPY %1:areg
2905//
2906// Example 2: PHI with multiple incoming values:
2907// entry:
2908// %1:vreg = GLOBAL_LOAD(..)
2909// loop:
2910// %2:vreg = PHI %1:vreg, %entry, %5:vreg, %loop
2911// %3:areg = COPY %2:vreg
2912// %4:areg = (instr using %3:areg)
2913// %5:vreg = COPY %4:areg
2914// =>
2915// entry:
2916// %1:vreg = GLOBAL_LOAD(..)
2917// %2:areg = COPY %1:vreg
2918// loop:
2919// %3:areg = PHI %2:areg, %entry, %X:areg,
2920// %4:areg = (instr using %3:areg)
2921bool SIFoldOperandsImpl::tryFoldPhiAGPR(MachineInstr &PHI) {
2922 assert(PHI.isPHI());
2923
2924 Register PhiOut = PHI.getOperand(0).getReg();
2925 if (!TRI->isVGPR(*MRI, PhiOut))
2926 return false;
2927
2928 // Iterate once over all incoming values of the PHI to check if this PHI is
2929 // eligible, and determine the exact AGPR RC we'll target.
2930 const TargetRegisterClass *ARC = nullptr;
2931 for (unsigned K = 1; K < PHI.getNumExplicitOperands(); K += 2) {
2932 MachineOperand &MO = PHI.getOperand(K);
2933 MachineInstr *Copy = MRI->getVRegDef(MO.getReg());
2934 if (!Copy || !Copy->isCopy())
2935 continue;
2936
2937 Register AGPRSrc;
2938 unsigned AGPRRegMask = AMDGPU::NoSubRegister;
2939 if (!isAGPRCopy(*TRI, *MRI, *Copy, AGPRSrc, AGPRRegMask))
2940 continue;
2941
2942 const TargetRegisterClass *CopyInRC = MRI->getRegClass(AGPRSrc);
2943 if (const auto *SubRC = TRI->getSubRegisterClass(CopyInRC, AGPRRegMask))
2944 CopyInRC = SubRC;
2945
2946 if (ARC && !ARC->hasSubClassEq(CopyInRC))
2947 return false;
2948 ARC = CopyInRC;
2949 }
2950
2951 if (!ARC)
2952 return false;
2953
2954 bool IsAGPR32 = (ARC == &AMDGPU::AGPR_32RegClass);
2955
2956 // Rewrite the PHI's incoming values to ARC.
2957 LLVM_DEBUG(dbgs() << "Folding AGPR copies into: " << PHI);
2958 for (unsigned K = 1; K < PHI.getNumExplicitOperands(); K += 2) {
2959 MachineOperand &MO = PHI.getOperand(K);
2960 Register Reg = MO.getReg();
2961
2963 MachineBasicBlock *InsertMBB = nullptr;
2964
2965 // Look at the def of Reg, ignoring all copies.
2966 unsigned CopyOpc = AMDGPU::COPY;
2967 if (MachineInstr *Def = MRI->getVRegDef(Reg)) {
2968
2969 // Look at pre-existing COPY instructions from ARC: Steal the operand. If
2970 // the copy was single-use, it will be removed by DCE later.
2971 if (Def->isCopy()) {
2972 Register AGPRSrc;
2973 unsigned AGPRSubReg = AMDGPU::NoSubRegister;
2974 if (isAGPRCopy(*TRI, *MRI, *Def, AGPRSrc, AGPRSubReg)) {
2975 MO.setReg(AGPRSrc);
2976 MO.setSubReg(AGPRSubReg);
2977 continue;
2978 }
2979
2980 // If this is a multi-use SGPR -> VGPR copy, use V_ACCVGPR_WRITE on
2981 // GFX908 directly instead of a COPY. Otherwise, SIFoldOperand may try
2982 // to fold the sgpr -> vgpr -> agpr copy into a sgpr -> agpr copy which
2983 // is unlikely to be profitable.
2984 //
2985 // Note that V_ACCVGPR_WRITE is only used for AGPR_32.
2986 MachineOperand &CopyIn = Def->getOperand(1);
2987 if (IsAGPR32 && !ST->hasGFX90AInsts() && !MRI->hasOneNonDBGUse(Reg) &&
2988 TRI->isSGPRReg(*MRI, CopyIn.getReg()))
2989 CopyOpc = AMDGPU::V_ACCVGPR_WRITE_B32_e64;
2990 }
2991
2992 InsertMBB = Def->getParent();
2993 InsertPt = InsertMBB->SkipPHIsLabelsAndDebug(++Def->getIterator());
2994 } else {
2995 InsertMBB = PHI.getOperand(MO.getOperandNo() + 1).getMBB();
2996 InsertPt = InsertMBB->getFirstTerminator();
2997 }
2998
2999 Register NewReg = MRI->createVirtualRegister(ARC);
3000 MachineInstr *MI = BuildMI(*InsertMBB, InsertPt, PHI.getDebugLoc(),
3001 TII->get(CopyOpc), NewReg)
3002 .addReg(Reg);
3003 MO.setReg(NewReg);
3004
3005 (void)MI;
3006 LLVM_DEBUG(dbgs() << " Created COPY: " << *MI);
3007 }
3008
3009 // Replace the PHI's result with a new register.
3010 Register NewReg = MRI->createVirtualRegister(ARC);
3011 PHI.getOperand(0).setReg(NewReg);
3012
3013 // COPY that new register back to the original PhiOut register. This COPY will
3014 // usually be folded out later.
3015 MachineBasicBlock *MBB = PHI.getParent();
3016 BuildMI(*MBB, MBB->getFirstNonPHI(), PHI.getDebugLoc(),
3017 TII->get(AMDGPU::COPY), PhiOut)
3018 .addReg(NewReg);
3019
3020 LLVM_DEBUG(dbgs() << " Done: Folded " << PHI);
3021 return true;
3022}
3023
3024// Attempt to convert VGPR load to an AGPR load.
3025bool SIFoldOperandsImpl::tryFoldLoad(MachineInstr &MI) {
3026 assert(MI.mayLoad());
3027 if (!ST->hasGFX90AInsts() || MI.getNumExplicitDefs() != 1)
3028 return false;
3029
3030 MachineOperand &Def = MI.getOperand(0);
3031 if (!Def.isDef())
3032 return false;
3033
3034 Register DefReg = Def.getReg();
3035
3036 if (DefReg.isPhysical() || !TRI->isVGPR(*MRI, DefReg))
3037 return false;
3038
3041 SmallVector<Register, 8> MoveRegs;
3042
3043 if (Users.empty())
3044 return false;
3045
3046 // Check that all uses a copy to an agpr or a reg_sequence producing an agpr.
3047 while (!Users.empty()) {
3048 const MachineInstr *I = Users.pop_back_val();
3049 if (!I->isCopy() && !I->isRegSequence())
3050 return false;
3051 Register DstReg = I->getOperand(0).getReg();
3052 // Physical registers may have more than one instruction definitions
3053 if (DstReg.isPhysical())
3054 return false;
3055 if (TRI->isAGPR(*MRI, DstReg))
3056 continue;
3057 MoveRegs.push_back(DstReg);
3058 for (const MachineInstr &U : MRI->use_nodbg_instructions(DstReg))
3059 Users.push_back(&U);
3060 }
3061
3062 const TargetRegisterClass *RC = MRI->getRegClass(DefReg);
3063 MRI->setRegClass(DefReg, TRI->getEquivalentAGPRClass(RC));
3064 if (!TII->isOperandLegal(MI, 0, &Def)) {
3065 MRI->setRegClass(DefReg, RC);
3066 return false;
3067 }
3068
3069 while (!MoveRegs.empty()) {
3070 Register Reg = MoveRegs.pop_back_val();
3071 MRI->setRegClass(Reg, TRI->getEquivalentAGPRClass(MRI->getRegClass(Reg)));
3072 }
3073
3074 LLVM_DEBUG(dbgs() << "Folded " << MI);
3075
3076 return true;
3077}
3078
3079// tryFoldPhiAGPR will aggressively try to create AGPR PHIs.
3080// For GFX90A and later, this is pretty much always a good thing, but for GFX908
3081// there's cases where it can create a lot more AGPR-AGPR copies, which are
3082// expensive on this architecture due to the lack of V_ACCVGPR_MOV.
3083//
3084// This function looks at all AGPR PHIs in a basic block and collects their
3085// operands. Then, it checks for register that are used more than once across
3086// all PHIs and caches them in a VGPR. This prevents ExpandPostRAPseudo from
3087// having to create one VGPR temporary per use, which can get very messy if
3088// these PHIs come from a broken-up large PHI (e.g. 32 AGPR phis, one per vector
3089// element).
3090//
3091// Example
3092// a:
3093// %in:agpr_256 = COPY %foo:vgpr_256
3094// c:
3095// %x:agpr_32 = ..
3096// b:
3097// %0:areg = PHI %in.sub0:agpr_32, %a, %x, %c
3098// %1:areg = PHI %in.sub0:agpr_32, %a, %y, %c
3099// %2:areg = PHI %in.sub0:agpr_32, %a, %z, %c
3100// =>
3101// a:
3102// %in:agpr_256 = COPY %foo:vgpr_256
3103// %tmp:vgpr_32 = V_ACCVGPR_READ_B32_e64 %in.sub0:agpr_32
3104// %tmp_agpr:agpr_32 = COPY %tmp
3105// c:
3106// %x:agpr_32 = ..
3107// b:
3108// %0:areg = PHI %tmp_agpr, %a, %x, %c
3109// %1:areg = PHI %tmp_agpr, %a, %y, %c
3110// %2:areg = PHI %tmp_agpr, %a, %z, %c
3111bool SIFoldOperandsImpl::tryOptimizeAGPRPhis(MachineBasicBlock &MBB) {
3112 // This is only really needed on GFX908 where AGPR-AGPR copies are
3113 // unreasonably difficult.
3114 if (ST->hasGFX90AInsts())
3115 return false;
3116
3117 // Look at all AGPR Phis and collect the register + subregister used.
3118 DenseMap<std::pair<Register, unsigned>, std::vector<MachineOperand *>>
3119 RegToMO;
3120
3121 for (auto &MI : MBB) {
3122 if (!MI.isPHI())
3123 break;
3124
3125 if (!TRI->isAGPR(*MRI, MI.getOperand(0).getReg()))
3126 continue;
3127
3128 for (unsigned K = 1; K < MI.getNumOperands(); K += 2) {
3129 MachineOperand &PhiMO = MI.getOperand(K);
3130 if (!PhiMO.getSubReg())
3131 continue;
3132 RegToMO[{PhiMO.getReg(), PhiMO.getSubReg()}].push_back(&PhiMO);
3133 }
3134 }
3135
3136 // For all (Reg, SubReg) pair that are used more than once, cache the value in
3137 // a VGPR.
3138 bool Changed = false;
3139 for (const auto &[Entry, MOs] : RegToMO) {
3140 if (MOs.size() == 1)
3141 continue;
3142
3143 const auto [Reg, SubReg] = Entry;
3144 MachineInstr *Def = MRI->getVRegDef(Reg);
3145 MachineBasicBlock *DefMBB = Def->getParent();
3146
3147 // Create a copy in a VGPR using V_ACCVGPR_READ_B32_e64 so it's not folded
3148 // out.
3149 const TargetRegisterClass *ARC = getRegOpRC(*MRI, *TRI, *MOs.front());
3150 Register TempVGPR =
3151 MRI->createVirtualRegister(TRI->getEquivalentVGPRClass(ARC));
3152 MachineInstr *VGPRCopy =
3153 BuildMI(*DefMBB, ++Def->getIterator(), Def->getDebugLoc(),
3154 TII->get(AMDGPU::V_ACCVGPR_READ_B32_e64), TempVGPR)
3155 .addReg(Reg, /* flags */ {}, SubReg);
3156
3157 // Copy back to an AGPR and use that instead of the AGPR subreg in all MOs.
3158 Register TempAGPR = MRI->createVirtualRegister(ARC);
3159 BuildMI(*DefMBB, ++VGPRCopy->getIterator(), Def->getDebugLoc(),
3160 TII->get(AMDGPU::COPY), TempAGPR)
3161 .addReg(TempVGPR);
3162
3163 LLVM_DEBUG(dbgs() << "Caching AGPR into VGPR: " << *VGPRCopy);
3164 for (MachineOperand *MO : MOs) {
3165 MO->setReg(TempAGPR);
3166 MO->setSubReg(AMDGPU::NoSubRegister);
3167 LLVM_DEBUG(dbgs() << " Changed PHI Operand: " << *MO << "\n");
3168 }
3169
3170 Changed = true;
3171 }
3172
3173 return Changed;
3174}
3175
3176bool SIFoldOperandsImpl::run(MachineFunction &MF, const MachineLoopInfo *MLI) {
3177 this->MF = &MF;
3178 MRI = &MF.getRegInfo();
3179 ST = &MF.getSubtarget<GCNSubtarget>();
3180 TII = ST->getInstrInfo();
3181 TRI = &TII->getRegisterInfo();
3182 MFI = MF.getInfo<SIMachineFunctionInfo>();
3183 this->MLI = MLI;
3184
3185 // omod is ignored by hardware if IEEE bit is enabled. omod also does not
3186 // correctly handle signed zeros.
3187 //
3188 // FIXME: Also need to check strictfp
3189 bool IsIEEEMode = MFI->getMode().IEEE;
3190
3191 bool Changed = false;
3192 for (MachineBasicBlock *MBB : depth_first(&MF)) {
3193 MachineOperand *CurrentKnownM0Val = nullptr;
3194 for (auto &MI : make_early_inc_range(*MBB)) {
3195 Changed |= tryFoldCndMask(MI);
3196
3197 // PeepholeOptimizer may have folded an inline immediate directly onto an
3198 // instruction operand without materializing it into a register first.
3199 // Such an instruction is never reached through a def->use edge in
3200 // foldInstOperand, so try to constant fold it here.
3201 if (tryConstantFoldOp(&MI)) {
3202 Changed = true;
3203 continue;
3204 }
3205
3206 if (tryFoldRedundantAND(MI)) {
3207 Changed = true;
3208 continue;
3209 }
3210
3211 if (tryFoldAndExec(MI)) {
3212 Changed = true;
3213 continue;
3214 }
3215
3216 if (MI.isRegSequence() && tryFoldRegSequence(MI)) {
3217 Changed = true;
3218 continue;
3219 }
3220
3221 if (MI.isPHI() && tryFoldPhiAGPR(MI)) {
3222 Changed = true;
3223 continue;
3224 }
3225
3226 if (MI.mayLoad() && tryFoldLoad(MI)) {
3227 Changed = true;
3228 continue;
3229 }
3230
3231 if (TII->isFoldableCopy(MI)) {
3232 Changed |= tryFoldFoldableCopy(MI, CurrentKnownM0Val);
3233 continue;
3234 }
3235
3236 // Saw an unknown clobber of m0, so we no longer know what it is.
3237 if (CurrentKnownM0Val && MI.modifiesRegister(AMDGPU::M0, TRI))
3238 CurrentKnownM0Val = nullptr;
3239
3240 // TODO: Omod might be OK if there is NSZ only on the source
3241 // instruction, and not the omod multiply.
3242 if (IsIEEEMode || !MI.getFlag(MachineInstr::FmNsz) || !tryFoldOMod(MI))
3243 Changed |= tryFoldClamp(MI);
3244 }
3245
3246 Changed |= tryOptimizeAGPRPhis(*MBB);
3247 }
3248
3249 return Changed;
3250}
3251
3252PreservedAnalyses
3255 MFPropsModifier _(*this, MF);
3256
3257 const MachineLoopInfo *MLI = &MFAM.getResult<MachineLoopAnalysis>(MF);
3258 bool Changed = SIFoldOperandsImpl().run(MF, MLI);
3259 if (!Changed) {
3260 return PreservedAnalyses::all();
3261 }
3263 PA.preserveSet<CFGAnalyses>();
3264 PA.preserve<MachineLoopAnalysis>();
3265 return PA;
3266}
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
aarch64 promote const
unsigned Imm
unsigned uint64_t
Rewrite undef for PHI
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static bool updateOperand(Instruction *Inst, unsigned Idx, Instruction *Mat)
Updates the operand at Idx in instruction Inst with the result of instruction Mat.
This file builds on the ADT/GraphTraits.h file to build generic depth first graph iterator.
AMD GCN specific subclass of TargetSubtarget.
#define DEBUG_TYPE
static Register UseReg(const MachineOperand &MO)
const HexagonInstrInfo * TII
#define _
IRTranslator LLVM IR MI
iv Induction Variable Users
Definition IVUsers.cpp:48
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
static bool isReg(const MCInst &MI, unsigned OpNo)
if(auto Err=PB.parsePassPipeline(MPM, Passes)) return wrap(std MPM run * Mod
if(PassOpts->AAPipeline)
#define INITIALIZE_PASS_DEPENDENCY(depName)
Definition PassSupport.h:42
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
Definition PassSupport.h:44
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
Definition PassSupport.h:39
static bool loopModifiesExec(const MachineLoop &L, const SIRegisterInfo &TRI)
static unsigned macToMad(unsigned Opc)
static bool isAGPRCopy(const SIRegisterInfo &TRI, const MachineRegisterInfo &MRI, const MachineInstr &Copy, Register &OutReg, unsigned &OutSubReg)
Checks whether Copy is a AGPR -> VGPR copy.
static void appendFoldCandidate(SmallVectorImpl< FoldCandidate > &FoldList, FoldCandidate &&Entry)
static const TargetRegisterClass * getRegOpRC(const MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI, const MachineOperand &MO)
static bool evalBinaryInstruction(unsigned Opcode, int32_t &Result, uint32_t LHS, uint32_t RHS)
static int getOModValue(unsigned Opc, int64_t Val)
static unsigned getMovOpc(bool IsScalar)
static MachineOperand * lookUpCopyChain(const SIInstrInfo &TII, const MachineRegisterInfo &MRI, Register SrcReg)
static bool checkImmOpForPKF32InstrReplicatesLower32BitsOfScalarOperand(const FoldableDef &OpToFold)
static bool isPKF32InstrReplicatesLower32BitsOfScalarOperand(const GCNSubtarget *ST, MachineInstr *MI, unsigned OpNo)
Interface definition for SIInstrInfo.
Interface definition for SIRegisterInfo.
#define LLVM_DEBUG(...)
Definition Debug.h:119
static int Lookup(ArrayRef< TableEntry > Table, unsigned Opcode)
Value * RHS
Value * LHS
static const LaneMaskConstants & get(const GCNSubtarget &ST)
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Definition Pass.cpp:278
Represents analyses that only rely on functions' control flow.
Definition Analysis.h:73
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
const SIInstrInfo * getInstrInfo() const override
bool hasDOTOpSelHazard() const
bool zeroesHigh16BitsOfDest(unsigned Opcode) const
Returns if the result of this instruction with a 16-bit result returned in a 32-bit register implicit...
const HexagonRegisterInfo & getRegisterInfo() const
bool contains(const LoopT *L) const
Return true if the specified loop is contained within this loop.
LoopT * getLoopFor(const BlockT *BB) const
Return the inner most loop that BB lives in.
ArrayRef< MCOperandInfo > operands() const
int getOperandConstraint(unsigned OpNum, MCOI::OperandConstraint Constraint) const
Returns the value of the specified operand constraint if it is present.
bool isVariadic() const
Return true if this instruction can have a variable number of operands.
This holds information about one operand of a machine instruction, indicating the register class for ...
Definition MCInstrDesc.h:88
uint8_t OperandType
Information about the type of the operand.
bool hasSuperClassEq(const MCRegisterClass *RC) const
Returns true if RC is a super-class of or equal to this class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
bool hasSubClassEq(const MCRegisterClass *RC) const
Returns true if RC is a sub-class of or equal to this class.
An RAII based helper class to modify MachineFunctionProperties when running pass.
LLVM_ABI iterator SkipPHIsLabelsAndDebug(iterator I, Register Reg=Register(), bool SkipPseudoOp=true)
Return the first instruction in MBB after I that is not a PHI, label or debug.
LLVM_ABI LivenessQueryResult computeRegisterLiveness(const TargetRegisterInfo *TRI, MCRegister Reg, const_iterator Before, unsigned Neighborhood=10) const
Return whether (physical) register Reg has been defined and not killed as of just before Before.
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
LLVM_ABI iterator getFirstNonPHI()
Returns a pointer to the first instruction in this block that is not a PHINode instruction.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
MachineInstrBundleIterator< MachineInstr > iterator
LivenessQueryResult
Possible outcome of a register liveness query to computeRegisterLiveness()
@ LQR_Dead
Register is known to be fully dead.
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
Properties which a MachineFunction may have at a given point in time.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Register getReg(unsigned Idx) const
Get the register for the operand index.
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
bool isImplicitDef() const
bool isCopy() const
const MachineBasicBlock * getParent() const
bool readsRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr reads the specified register.
LLVM_ABI bool allImplicitDefsAreDead() const
Return true if all the implicit defs of this instruction are dead.
unsigned getNumOperands() const
Retuns the total number of operands.
LLVM_ABI void addOperand(MachineFunction &MF, const MachineOperand &Op)
Add the specified operand to the instruction.
unsigned getOperandNo(const_mop_iterator I) const
Returns the number of the operand iterator I points to.
LLVM_ABI unsigned getNumExplicitOperands() const
Returns the number of non-implicit operands.
mop_range implicit_operands()
const MCInstrDesc & getDesc() const
Returns the target instruction descriptor of this MachineInstr.
void clearFlag(MIFlag Flag)
clearFlag - Clear a MI flag.
bool isRegSequence() const
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
MachineOperand * mop_iterator
iterator/begin/end - Iterate over all operands of a machine instruction.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
LLVM_ABI void removeOperand(unsigned OpNo)
Erase an operand from an instruction, leaving it with one fewer operand than it started with.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
Analysis pass that exposes the MachineLoopInfo for a machine function.
MachineOperand class - Representation of each machine instruction operand.
void setSubReg(unsigned subReg)
unsigned getSubReg() const
LLVM_ABI unsigned getOperandNo() const
Returns the index of this operand in the instruction that it belongs to.
LLVM_ABI void substVirtReg(Register Reg, unsigned SubIdx, const TargetRegisterInfo &)
substVirtReg - Substitute the current register with the virtual subregister Reg:SubReg.
LLVM_ABI void ChangeToFrameIndex(int Idx, unsigned TargetFlags=0)
Replace this operand with a frame index.
void setImm(int64_t immVal)
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
void setIsDead(bool Val=true)
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
LLVM_ABI void ChangeToImmediate(int64_t ImmVal, unsigned TargetFlags=0)
ChangeToImmediate - Replace this operand with a new immediate operand of the specified value.
LLVM_ABI void ChangeToGA(const GlobalValue *GV, int64_t Offset, unsigned TargetFlags=0)
ChangeToGA - Replace this operand with a new global address operand.
void setIsKill(bool Val=true)
LLVM_ABI void ChangeToRegister(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isDebug=false)
ChangeToRegister - Replace this operand with a new register operand of the specified value.
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
LLVM_ABI void substPhysReg(MCRegister Reg, const TargetRegisterInfo &)
substPhysReg - Substitute the current register with the physical register Reg, taking any existing Su...
static MachineOperand CreateImm(int64_t Val)
bool isGlobal() const
isGlobal - Tests if this is a MO_GlobalAddress operand.
MachineOperandType getType() const
getType - Returns the MachineOperandType for this operand.
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
bool isFI() const
isFI - Tests if this is a MO_FrameIndex operand.
LLVM_ABI bool isIdenticalTo(const MachineOperand &Other) const
Returns true if this operand is identical to the specified operand except for liveness related flags ...
@ MO_Immediate
Immediate operand.
@ MO_GlobalAddress
Address of a global value.
@ MO_FrameIndex
Abstract Stack Frame Index.
@ MO_Register
Register operand.
static MachineOperand CreateFI(int Idx)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
use_nodbg_iterator use_nodbg_begin(Register RegNo) const
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
iterator_range< use_nodbg_iterator > use_nodbg_operands(Register Reg) const
bool use_nodbg_empty(Register RegNo) const
use_nodbg_empty - Return true if there are no non-Debug instructions using the specified register.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLVM_ABI bool hasOneNonDBGUser(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug instruction using the specified regis...
iterator_range< use_instr_nodbg_iterator > use_nodbg_instructions(Register Reg) const
void setRegAllocationHint(Register VReg, unsigned Type, Register PrefReg)
setRegAllocationHint - Specify a register allocation hint for the specified virtual register.
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
LLVM_ABI const TargetRegisterClass * constrainRegClass(Register Reg, const TargetRegisterClass *RC, unsigned MinNumRegs=0)
constrainRegClass - Constrain the register class of the specified virtual register to be a common sub...
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
Wrapper class representing virtual and physical registers.
Definition Register.h:20
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
Definition Register.h:79
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Definition Register.h:83
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
static bool hasSameClamp(const MachineInstr &A, const MachineInstr &B)
static std::optional< int64_t > extractSubregFromImm(int64_t ImmVal, unsigned SubRegIndex)
Return the extracted immediate value in a subregister use from a constant materialized in a super reg...
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
Register getScratchRSrcReg() const
Returns the physical register reserved for use as the resource descriptor for scratch accesses.
SIModeRegisterDefaults getMode() const
static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs=1)
bool insert(const value_type &X)
Insert a new element into the SetVector.
Definition SetVector.h:157
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
Register getReg() const
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
static const unsigned CommuteAnyOperandIndex
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
self_iterator getIterator()
Definition ilist_node.h:123
IteratorT end() const
IteratorT begin() const
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
bool isInlinableLiteralV216(uint32_t Literal, uint8_t OpType)
LLVM_READONLY int32_t getMFMAEarlyClobberOp(uint32_t Opcode)
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
bool isPackedSingleSGPR64BitInst(unsigned Opc)
The opcode is a packed 64-bit instruction which only reads low 64 bits of a scalar operand and propag...
LLVM_READONLY int32_t getVOPe32(uint32_t Opcode)
constexpr bool isSISrcOperand(const MCOperandInfo &OpInfo)
Is this an AMDGPU specific source operand?
@ OPERAND_REG_IMM_V2FP64
Definition SIDefines.h:441
@ OPERAND_REG_IMM_V2FP16
Definition SIDefines.h:434
@ OPERAND_REG_INLINE_C_FP64
Definition SIDefines.h:450
@ OPERAND_REG_INLINE_C_BF16
Definition SIDefines.h:447
@ OPERAND_REG_INLINE_C_V2BF16
Definition SIDefines.h:452
@ OPERAND_REG_IMM_V2INT64
Definition SIDefines.h:437
@ OPERAND_REG_IMM_V2INT16
Definition SIDefines.h:436
@ OPERAND_REG_IMM_BF16
Definition SIDefines.h:430
@ OPERAND_REG_IMM_V2BF16
Definition SIDefines.h:433
@ OPERAND_REG_INLINE_C_INT64
Definition SIDefines.h:446
@ OPERAND_REG_IMM_NOINLINE_V2FP16
Definition SIDefines.h:438
@ OPERAND_REG_INLINE_C_V2FP16
Definition SIDefines.h:453
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
Definition SIDefines.h:464
@ OPERAND_REG_INLINE_AC_FP32
Definition SIDefines.h:465
@ OPERAND_REG_INLINE_C_FP32
Definition SIDefines.h:449
@ OPERAND_REG_INLINE_C_INT32
Definition SIDefines.h:445
@ OPERAND_REG_INLINE_C_V2INT16
Definition SIDefines.h:451
@ OPERAND_REG_IMM_V2FP32
Definition SIDefines.h:440
@ OPERAND_REG_INLINE_AC_FP64
Definition SIDefines.h:466
LLVM_READONLY int32_t getFlatScratchInstSSfromSV(uint32_t Opcode)
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
@ Entry
Definition COFF.h:862
constexpr bool isVOP3(const T &...O)
Definition SIDefines.h:236
constexpr bool isMAI(const T &...O)
Definition SIDefines.h:349
constexpr bool isSWMMAC(const T &...O)
Definition SIDefines.h:376
constexpr bool isVOP3P(const T &...O)
Definition SIDefines.h:239
constexpr bool isWMMA(const T &...O)
Definition SIDefines.h:364
constexpr bool isDOT(const T &...O)
Definition SIDefines.h:352
constexpr bool isPacked(const T &...O)
Definition SIDefines.h:337
NodeAddr< DefNode * > Def
Definition RDFGraph.h:384
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
TargetInstrInfo::RegSubRegPair getRegSubRegPair(const MachineOperand &O)
Create RegSubRegPair from a register MachineOperand.
MachineBasicBlock::instr_iterator getBundleStart(MachineBasicBlock::instr_iterator I)
Returns an iterator to the first instruction in the bundle containing I.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
bool execMayBeModifiedBeforeUse(const MachineRegisterInfo &MRI, Register VReg, const MachineInstr &DefMI, const MachineInstr &UseMI)
Return false if EXEC is not changed between the def of VReg at DefMI and the use at UseMI.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
Definition STLExtras.h:2224
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:649
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
Op::Description Desc
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
FunctionPass * createSIFoldOperandsLegacyPass()
char & SIFoldOperandsLegacyID
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
@ Sub
Subtraction of integers.
@ Add
Sum of integers.
DWARFExpression::Operation Op
iterator_range< pointer_iterator< WrappedIteratorT > > make_pointer_range(RangeT &&Range)
Definition iterator.h:368
iterator_range< df_iterator< T > > depth_first(const T &G)
LLVM_ABI Printable printReg(Register Reg, const TargetRegisterInfo *TRI=nullptr, unsigned SubIdx=0, const MachineRegisterInfo *MRI=nullptr)
Prints virtual and physical registers with or without a TRI instance.
constexpr uint64_t Make_64(uint32_t High, uint32_t Low)
Make a 64-bit integer from a high / low pair of 32-bit integers.
Definition MathExtras.h:161
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
@ PreserveSign
The sign of a flushed-to-zero number is preserved in the sign of 0.
DenormalModeKind Output
Denormal flushing mode for floating point instruction results in the default floating point environme...
DenormalMode FP64FP16Denormals
If this is set, neither input or output denormals are flushed for both f64 and f16/v2f16 instructions...
bool IEEE
Floating point opcodes that support exception flag gathering quiet and propagate signaling NaN inputs...
DenormalMode FP32Denormals
If this is set, neither input or output denormals are flushed for most f32 instructions.