LLVM 24.0.0git
AMDGPUDisassembler.cpp
Go to the documentation of this file.
1//===- AMDGPUDisassembler.cpp - Disassembler for AMDGPU ISA ---------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9//===----------------------------------------------------------------------===//
10//
11/// \file
12///
13/// This file contains definition for AMDGPU ISA disassembler
14//
15//===----------------------------------------------------------------------===//
16
17// ToDo: What to do with instruction suffixes (v_mov_b32 vs v_mov_b32_e32)?
18
22#include "SIDefines.h"
23#include "SIRegisterInfo.h"
29#include "llvm/MC/MCAsmInfo.h"
30#include "llvm/MC/MCContext.h"
31#include "llvm/MC/MCDecoder.h"
33#include "llvm/MC/MCExpr.h"
34#include "llvm/MC/MCInstrDesc.h"
40
41using namespace llvm;
42using namespace llvm::MCD;
43
44#define DEBUG_TYPE "amdgpu-disassembler"
45
46#define SGPR_MAX \
47 (isGFX10Plus() ? AMDGPU::EncValues::SGPR_MAX_GFX10 \
48 : AMDGPU::EncValues::SGPR_MAX_SI)
49
51
52static int64_t getInlineImmValF16(unsigned Imm);
53static int64_t getInlineImmValBF16(unsigned Imm);
54static int64_t getInlineImmVal32(unsigned Imm);
55static int64_t getInlineImmVal64(unsigned Imm);
56
58 MCContext &Ctx, MCInstrInfo const *MCII)
59 : MCDisassembler(STI, Ctx), MCII(MCII), MRI(*Ctx.getRegisterInfo()),
60 MAI(Ctx.getAsmInfo()),
61 HwModeRegClass(STI.getHwMode(MCSubtargetInfo::HwMode_RegInfo)),
62 TargetMaxInstBytes(MAI.getMaxInstLength(&STI)),
63 TargetID(AMDGPU::createAMDGPUTargetID(STI, "")),
64 CodeObjectVersion(AMDGPU::getDefaultAMDHSACodeObjectVersion()) {
65 // ToDo: AMDGPUDisassembler supports only VI ISA.
66 if (!STI.hasFeature(AMDGPU::FeatureGCN3Encoding) && !isGFX10Plus())
67 reportFatalUsageError("disassembly not yet supported for subtarget");
68
69 for (auto [Symbol, Code] : AMDGPU::UCVersion::getGFXVersions())
70 createConstantSymbolExpr(Symbol, Code);
71
72 UCVersionW64Expr = createConstantSymbolExpr("UC_VERSION_W64_BIT", 0x2000);
73 UCVersionW32Expr = createConstantSymbolExpr("UC_VERSION_W32_BIT", 0x4000);
74 UCVersionMDPExpr = createConstantSymbolExpr("UC_VERSION_MDP_BIT", 0x8000);
75}
76
80
82 unsigned EFlags) const {
83 OS << "\t.amdgcn_target \""
84 << STI.getTargetTriple().normalize(Triple::CanonicalForm::FOUR_IDENT)
85 << '-';
86
87 // Get CPU name from ELF e_flags MACH field
88 unsigned MACH = EFlags & ELF::EF_AMDGPU_MACH;
89
90#define X(NUM, ENUM, NAME) \
91 case ELF::ENUM: \
92 OS << NAME; \
93 break;
94 switch (MACH) {
96 default:
97 OS << "unknown";
98 break;
99 }
100#undef X
101
102 // Add xnack and sramecc from ELF flags (v4 format)
103 if (CodeObjectVersion >= AMDGPU::AMDHSA_COV4) {
104 // Hardwired-on features are not selectable target-ID modifiers.
105 bool SramEccHardwiredOn = TargetID.isSramEccSupported() &&
106 !STI.hasFeature(AMDGPU::FeatureSRAMECCOnOffModes);
107 unsigned SrameccSetting = EFlags & ELF::EF_AMDGPU_FEATURE_SRAMECC_V4;
108 switch (SrameccSetting) {
110 break;
112 TargetID.setSramEccSetting(AMDGPU::TargetIDSetting::Any);
113 break;
115 TargetID.setSramEccSetting(AMDGPU::TargetIDSetting::Off);
116 if (!SramEccHardwiredOn)
117 OS << ":sramecc-";
118 break;
120 TargetID.setSramEccSetting(AMDGPU::TargetIDSetting::On);
121 if (!SramEccHardwiredOn)
122 OS << ":sramecc+";
123 break;
124 }
125
126 // Targets that hardwire xnack on (e.g. gfx1250) don't expose it as a
127 // selectable modifier, so don't print it.
128 bool XnackHardwiredOn = TargetID.isXnackSupported() &&
129 !STI.hasFeature(AMDGPU::FeatureXNACKOnOffModes);
131 switch (XnackSetting) {
133 break;
135 TargetID.setXnackSetting(AMDGPU::TargetIDSetting::Any);
136 break;
138 TargetID.setXnackSetting(AMDGPU::TargetIDSetting::Off);
139 if (!XnackHardwiredOn)
140 OS << ":xnack-";
141 break;
143 TargetID.setXnackSetting(AMDGPU::TargetIDSetting::On);
144 if (!XnackHardwiredOn)
145 OS << ":xnack+";
146 break;
147 }
148 }
149
150 OS << "\"\n";
151}
152
154addOperand(MCInst &Inst, const MCOperand& Opnd) {
155 Inst.addOperand(Opnd);
156 return Opnd.isValid() ?
159}
160
162 AMDGPU::OpName Name) {
163 int OpIdx = AMDGPU::getNamedOperandIdx(MI.getOpcode(), Name);
164 if (OpIdx != -1) {
165 auto *I = MI.begin();
166 std::advance(I, OpIdx);
167 MI.insert(I, Op);
168 }
169 return OpIdx;
170}
171
173 uint64_t Addr,
174 const MCDisassembler *Decoder) {
175 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
176
177 // Our branches take a simm16.
178 int64_t Offset = SignExtend64<16>(Imm) * 4 + 4 + Addr;
179
180 if (DAsm->tryAddingSymbolicOperand(Inst, Offset, Addr, true, 2, 2, 0))
182 return addOperand(Inst, MCOperand::createImm(Imm));
183}
184
185static DecodeStatus decodeSMEMOffset(MCInst &Inst, unsigned Imm, uint64_t Addr,
186 const MCDisassembler *Decoder) {
187 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
188 int64_t Offset;
189 if (DAsm->isGFX12Plus()) { // GFX12 supports 24-bit signed offsets.
191 } else if (DAsm->isVI()) { // VI supports 20-bit unsigned offsets.
192 Offset = Imm & 0xFFFFF;
193 } else { // GFX9+ supports 21-bit signed offsets.
195 }
197}
198
199static DecodeStatus decodeBoolReg(MCInst &Inst, unsigned Val, uint64_t Addr,
200 const MCDisassembler *Decoder) {
201 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
202 return addOperand(Inst, DAsm->decodeBoolReg(Inst, Val));
203}
204
205static DecodeStatus decodeSplitBarrier(MCInst &Inst, unsigned Val,
206 uint64_t Addr,
207 const MCDisassembler *Decoder) {
208 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
209 return addOperand(Inst, DAsm->decodeSplitBarrier(Inst, Val));
210}
211
212static DecodeStatus decodeDpp8FI(MCInst &Inst, unsigned Val, uint64_t Addr,
213 const MCDisassembler *Decoder) {
214 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
215 return addOperand(Inst, DAsm->decodeDpp8FI(Val));
216}
217
218#define DECODE_OPERAND(StaticDecoderName, DecoderName) \
219 static DecodeStatus StaticDecoderName(MCInst &Inst, unsigned Imm, \
220 uint64_t /*Addr*/, \
221 const MCDisassembler *Decoder) { \
222 auto DAsm = static_cast<const AMDGPUDisassembler *>(Decoder); \
223 return addOperand(Inst, DAsm->DecoderName(Imm)); \
224 }
225
226// Decoder for registers, decode directly using RegClassID. Imm(8-bit) is
227// number of register. Used by VGPR only and AGPR only operands.
228#define DECODE_OPERAND_REG_8(RegClass) \
229 static DecodeStatus Decode##RegClass##RegisterClass( \
230 MCInst &Inst, unsigned Imm, uint64_t /*Addr*/, \
231 const MCDisassembler *Decoder) { \
232 assert(Imm < (1 << 8) && "8-bit encoding"); \
233 auto DAsm = static_cast<const AMDGPUDisassembler *>(Decoder); \
234 return addOperand( \
235 Inst, DAsm->createRegOperand(AMDGPU::RegClass##RegClassID, Imm)); \
236 }
237
238#define DECODE_SrcOp(Name, EncSize, OpWidth, EncImm) \
239 static DecodeStatus Name(MCInst &Inst, unsigned Imm, uint64_t /*Addr*/, \
240 const MCDisassembler *Decoder) { \
241 if (!isUInt<EncSize>(Imm)) \
242 return MCDisassembler::Fail; \
243 auto DAsm = static_cast<const AMDGPUDisassembler *>(Decoder); \
244 return addOperand(Inst, DAsm->decodeSrcOp(Inst, OpWidth, EncImm)); \
245 }
246
247static DecodeStatus decodeSrcOp(MCInst &Inst, unsigned EncSize,
248 unsigned OpWidth, unsigned Imm, unsigned EncImm,
249 const MCDisassembler *Decoder) {
250 assert(Imm < (1U << EncSize) && "Operand doesn't fit encoding!");
251 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
252 return addOperand(Inst, DAsm->decodeSrcOp(Inst, OpWidth, EncImm));
253}
254
255// Decode an indexed-resource (rsrcidx) 9-bit srsrc field into a 32-bit index
256// register. SGPRs are encoded as 128-251, VGPRs have bit 8 set.
257static DecodeStatus decodeRsrcRegOp(MCInst &Inst, unsigned Imm,
258 uint64_t /* Addr */,
259 const MCDisassembler *Decoder,
260 unsigned OpWidth) {
261 // Uniform-indexed resource. SGPR[0..123] encoded as 128-251.
262 if (Imm >= 128 && Imm < 256)
263 Imm -= 128;
264 return decodeSrcOp(Inst, 9, OpWidth, Imm, Imm, Decoder);
265}
266
268 uint64_t /* Addr */,
269 const MCDisassembler *Decoder) {
270 unsigned OpWidth = 32;
271 // 0-127: Uniform-direct resource in SGPRs (SReg_128).
272 if (Imm < 128)
273 OpWidth = 128;
274 return decodeRsrcRegOp(Inst, Imm, 0, Decoder, OpWidth);
275}
276
278 uint64_t /* Addr */,
279 const MCDisassembler *Decoder) {
280 unsigned OpWidth = 32;
281 // 0-127: Uniform-direct resource in SGPRs (SReg_256).
282 if (Imm < 128)
283 OpWidth = 256;
284 return decodeRsrcRegOp(Inst, Imm, 0, Decoder, OpWidth);
285}
286
287// Decoder for registers. Imm(7-bit) is number of register, uses decodeSrcOp to
288// get register class. Used by SGPR only operands.
289#define DECODE_OPERAND_SREG_7(RegClass, OpWidth) \
290 DECODE_SrcOp(Decode##RegClass##RegisterClass, 7, OpWidth, Imm)
291
292#define DECODE_OPERAND_SREG_8(RegClass, OpWidth) \
293 DECODE_SrcOp(Decode##RegClass##RegisterClass, 8, OpWidth, Imm)
294
295#define DECODE_OPERAND_SREG_9(RegClass, OpWidth) \
296 DECODE_SrcOp(Decode##RegClass##RegisterClass, 9, OpWidth, Imm)
297
298// Decoder for registers. Imm(10-bit): Imm{7-0} is number of register,
299// Imm{9} is acc(agpr or vgpr) Imm{8} should be 0 (see VOP3Pe_SMFMAC).
300// Set Imm{8} to 1 (IS_VGPR) to decode using 'enum10' from decodeSrcOp.
301// Used by AV_ register classes (AGPR or VGPR only register operands).
302template <unsigned OpWidth>
303static DecodeStatus decodeAV10(MCInst &Inst, unsigned Imm, uint64_t /* Addr */,
304 const MCDisassembler *Decoder) {
305 return decodeSrcOp(Inst, 10, OpWidth, Imm, Imm | AMDGPU::EncValues::IS_VGPR,
306 Decoder);
307}
308
309// Decoder for Src(9-bit encoding) registers only.
310template <unsigned OpWidth>
311static DecodeStatus decodeSrcReg9(MCInst &Inst, unsigned Imm,
312 uint64_t /* Addr */,
313 const MCDisassembler *Decoder) {
314 return decodeSrcOp(Inst, 9, OpWidth, Imm, Imm, Decoder);
315}
316
317// Decoder for Src(9-bit encoding) AGPR, register number encoded in 9bits, set
318// Imm{9} to 1 (set acc) and decode using 'enum10' from decodeSrcOp, registers
319// only.
320template <unsigned OpWidth>
321static DecodeStatus decodeSrcA9(MCInst &Inst, unsigned Imm, uint64_t /* Addr */,
322 const MCDisassembler *Decoder) {
323 // A clear Imm{8} names an SGPR or an inline constant, which this
324 // register-only operand cannot hold.
327 return decodeSrcOp(Inst, 9, OpWidth, Imm, Imm | 512, Decoder);
328}
329
330// Decoder for 'enum10' from decodeSrcOp, Imm{0-8} is 9-bit Src encoding
331// Imm{9} is acc, registers only.
332template <unsigned OpWidth>
333static DecodeStatus decodeSrcAV10(MCInst &Inst, unsigned Imm,
334 uint64_t /* Addr */,
335 const MCDisassembler *Decoder) {
336 // A clear Imm{8} names an SGPR or an inline constant, which this
337 // register-only operand cannot hold.
340 return decodeSrcOp(Inst, 10, OpWidth, Imm, Imm, Decoder);
341}
342
343// Decoder for RegisterOperands using 9-bit Src encoding. Operand can be
344// register from RegClass or immediate. Registers that don't belong to RegClass
345// will be decoded and InstPrinter will report warning. Immediate will be
346// decoded into constant matching the OperandType (important for floating point
347// types).
348template <unsigned OpWidth>
350 uint64_t /* Addr */,
351 const MCDisassembler *Decoder) {
352 return decodeSrcOp(Inst, 9, OpWidth, Imm, Imm, Decoder);
353}
354
355// Decoder for Src(9-bit encoding) AGPR or immediate. Set Imm{9} to 1 (set acc)
356// and decode using 'enum10' from decodeSrcOp.
357template <unsigned OpWidth>
359 uint64_t /* Addr */,
360 const MCDisassembler *Decoder) {
361 return decodeSrcOp(Inst, 9, OpWidth, Imm, Imm | 512, Decoder);
362}
363
364// Default decoders generated by tablegen: 'Decode<RegClass>RegisterClass'
365// when RegisterClass is used as an operand. Most often used for destination
366// operands.
367
369DECODE_OPERAND_REG_8(VGPR_32_Lo128)
372DECODE_OPERAND_REG_8(VReg_128)
373DECODE_OPERAND_REG_8(VReg_192)
374DECODE_OPERAND_REG_8(VReg_256)
375DECODE_OPERAND_REG_8(VReg_288)
376DECODE_OPERAND_REG_8(VReg_320)
377DECODE_OPERAND_REG_8(VReg_352)
378DECODE_OPERAND_REG_8(VReg_384)
379DECODE_OPERAND_REG_8(VReg_512)
380DECODE_OPERAND_REG_8(VReg_1024)
381
382DECODE_OPERAND_SREG_7(SReg_32, 32)
383DECODE_OPERAND_SREG_7(SReg_32_XM0, 32)
384DECODE_OPERAND_SREG_7(SReg_32_XEXEC, 32)
385DECODE_OPERAND_SREG_7(SReg_32_XM0_XEXEC, 32)
386DECODE_OPERAND_SREG_7(SReg_32_XEXEC_HI, 32)
387DECODE_OPERAND_SREG_7(SReg_64_XEXEC, 64)
388DECODE_OPERAND_SREG_7(SReg_64_XEXEC_XNULL, 64)
389DECODE_OPERAND_SREG_7(SReg_96, 96)
390DECODE_OPERAND_SREG_7(SReg_128, 128)
391DECODE_OPERAND_SREG_7(SReg_256, 256)
392DECODE_OPERAND_SREG_7(SReg_256_XNULL, 256)
393DECODE_OPERAND_SREG_7(SReg_512, 512)
394
395DECODE_OPERAND_SREG_8(SReg_64, 64)
396
397// GFX13 VBUFFER instructions use a 9-bit srsrc field. For the non-indexed form
398// the two extra MSBs are always 0, so the value still decodes to an SReg_128.
399DECODE_OPERAND_SREG_9(SReg_128_XNULL, 128)
400
403DECODE_OPERAND_REG_8(AReg_128)
404DECODE_OPERAND_REG_8(AReg_256)
405DECODE_OPERAND_REG_8(AReg_512)
406DECODE_OPERAND_REG_8(AReg_1024)
407
409 uint64_t /*Addr*/,
411 assert(isUInt<10>(Imm) && "10-bit encoding expected");
412 assert((Imm & (1 << 8)) == 0 && "Imm{8} should not be used");
413
414 bool IsHi = Imm & (1 << 9);
415 unsigned RegIdx = Imm & 0xff;
416 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
417 return addOperand(Inst, DAsm->createVGPR16Operand(RegIdx, IsHi));
418}
419
420static DecodeStatus
422 const MCDisassembler *Decoder) {
423 assert(isUInt<8>(Imm) && "8-bit encoding expected");
424
425 bool IsHi = Imm & (1 << 7);
426 unsigned RegIdx = Imm & 0x7f;
427 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
428 return addOperand(Inst, DAsm->createVGPR16Operand(RegIdx, IsHi));
429}
430
431template <unsigned OpWidth>
433 uint64_t /*Addr*/,
434 const MCDisassembler *Decoder) {
435 assert(isUInt<9>(Imm) && "9-bit encoding expected");
436
437 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
439 bool IsHi = Imm & (1 << 7);
440 unsigned RegIdx = Imm & 0x7f;
441 return addOperand(Inst, DAsm->createVGPR16Operand(RegIdx, IsHi));
442 }
443 return addOperand(Inst, DAsm->decodeNonVGPRSrcOp(Inst, OpWidth, Imm & 0xFF));
444}
445
446template <unsigned OpWidth>
448 uint64_t /*Addr*/,
449 const MCDisassembler *Decoder) {
450 assert(isUInt<10>(Imm) && "10-bit encoding expected");
451
452 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
454 bool IsHi = Imm & (1 << 9);
455 unsigned RegIdx = Imm & 0xff;
456 return addOperand(Inst, DAsm->createVGPR16Operand(RegIdx, IsHi));
457 }
458 return addOperand(Inst, DAsm->decodeNonVGPRSrcOp(Inst, OpWidth, Imm & 0xFF));
459}
460
462 uint64_t /*Addr*/,
463 const MCDisassembler *Decoder) {
464 assert(isUInt<10>(Imm) && "10-bit encoding expected");
467
468 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
469
470 bool IsHi = Imm & (1 << 9);
471 unsigned RegIdx = Imm & 0xff;
472 return addOperand(Inst, DAsm->createVGPR16Operand(RegIdx, IsHi));
473}
474
476 uint64_t Addr,
477 const MCDisassembler *Decoder) {
478 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
479 return addOperand(Inst, DAsm->decodeMandatoryLiteralConstant(Imm));
480}
481
483 uint64_t Addr,
484 const MCDisassembler *Decoder) {
485 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
486 return addOperand(Inst, DAsm->decodeMandatoryLiteral64Constant(Imm));
487}
488
489static DecodeStatus decodeOperandVOPDDstY(MCInst &Inst, unsigned Val,
490 uint64_t Addr, const void *Decoder) {
491 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
492 return addOperand(Inst, DAsm->decodeVOPDDstYOp(Inst, Val));
493}
494
495static DecodeStatus decodeAVLdSt(MCInst &Inst, unsigned Imm, unsigned Opw,
496 const MCDisassembler *Decoder) {
497 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
498 return addOperand(Inst, DAsm->decodeSrcOp(Inst, Opw, Imm | 256));
499}
500
501template <unsigned Opw>
502static DecodeStatus decodeAVLdSt(MCInst &Inst, unsigned Imm,
503 uint64_t /* Addr */,
504 const MCDisassembler *Decoder) {
505 return decodeAVLdSt(Inst, Imm, Opw, Decoder);
506}
507
509 uint64_t Addr,
510 const MCDisassembler *Decoder) {
511 assert(Imm < (1 << 9) && "9-bit encoding");
512 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
513 return addOperand(Inst, DAsm->decodeSrcOp(Inst, 64, Imm));
514}
515
516#define DECODE_SDWA(DecName) \
517DECODE_OPERAND(decodeSDWA##DecName, decodeSDWA##DecName)
518
519DECODE_SDWA(Src32)
520DECODE_SDWA(Src16)
521DECODE_SDWA(VopcDst)
522
523#define DECODE_SDWA_IMM_FIELD(Name, MaxImm) \
524 static DecodeStatus Name(MCInst &Inst, unsigned Imm, uint64_t /* Addr */, \
525 const MCDisassembler * /* Decoder */) { \
526 if (Imm > (MaxImm)) \
527 return MCDisassembler::Fail; \
528 return addOperand(Inst, MCOperand::createImm(Imm)); \
529 }
530
531// The 3-bit SDWA sel fields only define values up to DWORD; 7 is reserved.
533// The 2-bit SDWA dst_unused field only defines values up to UNUSED_PRESERVE;
534// 3 is reserved.
535DECODE_SDWA_IMM_FIELD(decodeSDWADstUnused,
536 AMDGPU::SDWA::DstUnused::UNUSED_PRESERVE)
537#undef DECODE_SDWA_IMM_FIELD
538
539static DecodeStatus decodeVersionImm(MCInst &Inst, unsigned Imm,
540 uint64_t /* Addr */,
542 const auto *DAsm = static_cast<const AMDGPUDisassembler *>(Decoder);
543 return addOperand(Inst, DAsm->decodeVersionImm(Imm));
544}
545
546#include "AMDGPUGenDisassemblerTables.inc"
547
548namespace {
549// Define bitwidths for various types used to instantiate the decoder.
550template <> constexpr uint32_t InsnBitWidth<uint32_t> = 32;
551template <> constexpr uint32_t InsnBitWidth<uint64_t> = 64;
552template <> constexpr uint32_t InsnBitWidth<std::bitset<96>> = 96;
553template <> constexpr uint32_t InsnBitWidth<std::bitset<128>> = 128;
554} // namespace
555
556//===----------------------------------------------------------------------===//
557//
558//===----------------------------------------------------------------------===//
559
560template <typename InsnType>
562 InsnType Inst, uint64_t Address,
563 raw_ostream &Comments) const {
564 assert(MI.getOpcode() == 0);
565 assert(MI.getNumOperands() == 0);
566 MCInst TmpInst;
567 HasLiteral = false;
568 const auto SavedBytes = Bytes;
569
570 SmallString<64> LocalComments;
571 raw_svector_ostream LocalCommentStream(LocalComments);
572 CommentStream = &LocalCommentStream;
573
574 DecodeStatus Res =
575 decodeInstruction(Table, TmpInst, Inst, Address, this, STI);
576 if (Res != MCDisassembler::Fail && !decodeImmOperands(TmpInst, *MCII))
578
579 CommentStream = nullptr;
580
581 if (Res != MCDisassembler::Fail) {
582 MI = TmpInst;
583 Comments << LocalComments;
585 }
586 Bytes = SavedBytes;
588}
589
590template <typename InsnType>
593 MCInst &MI, InsnType Inst, uint64_t Address,
594 raw_ostream &Comments) const {
595 for (const uint8_t *T : {Table1, Table2}) {
596 if (DecodeStatus Res = tryDecodeInst(T, MI, Inst, Address, Comments))
597 return Res;
598 }
600}
601
602template <typename T> static inline T eatBytes(ArrayRef<uint8_t>& Bytes) {
603 assert(Bytes.size() >= sizeof(T));
604 const auto Res =
606 Bytes = Bytes.slice(sizeof(T));
607 return Res;
608}
609
610static inline std::bitset<96> eat12Bytes(ArrayRef<uint8_t> &Bytes) {
611 using namespace llvm::support::endian;
612 assert(Bytes.size() >= 12);
613 std::bitset<96> Lo(read<uint64_t, endianness::little>(Bytes.data()));
614 Bytes = Bytes.slice(8);
615 std::bitset<96> Hi(read<uint32_t, endianness::little>(Bytes.data()));
616 Bytes = Bytes.slice(4);
617 return (Hi << 64) | Lo;
618}
619
620static inline std::bitset<128> eat16Bytes(ArrayRef<uint8_t> &Bytes) {
621 using namespace llvm::support::endian;
622 assert(Bytes.size() >= 16);
623 std::bitset<128> Lo(read<uint64_t, endianness::little>(Bytes.data()));
624 Bytes = Bytes.slice(8);
625 std::bitset<128> Hi(read<uint64_t, endianness::little>(Bytes.data()));
626 Bytes = Bytes.slice(8);
627 return (Hi << 64) | Lo;
628}
629
630bool AMDGPUDisassembler::decodeImmOperands(MCInst &MI,
631 const MCInstrInfo &MCII) const {
632 const MCInstrDesc &Desc = MCII.get(MI.getOpcode());
633 for (auto [OpNo, OpDesc] : enumerate(Desc.operands())) {
634 if (OpNo >= MI.getNumOperands())
635 continue;
636
637 // TODO: Fix V_DUAL_FMAMK_F32_X_FMAAK_F32_gfx12 vsrc operands,
638 // defined to take VGPR_32, but in reality allowing inline constants.
639 bool IsSrc = AMDGPU::OPERAND_SRC_FIRST <= OpDesc.OperandType &&
640 OpDesc.OperandType <= AMDGPU::OPERAND_SRC_LAST;
641 if (!IsSrc && OpDesc.OperandType != MCOI::OPERAND_REGISTER)
642 continue;
643
644 MCOperand &Op = MI.getOperand(OpNo);
645 if (!Op.isImm())
646 continue;
647 int64_t Imm = Op.getImm();
651 continue;
652 }
653
655 Op = decodeLiteralConstant(Desc, OpDesc);
656 if (!Op.isValid())
657 return false;
658 continue;
659 }
660
663 switch (OpDesc.OperandType) {
666 // Inline constant encodings are not allowed for NOINLINE operand types.
667 // Keep the raw encoding value.
668 continue;
674 break;
678 break;
682 break;
684 // V_PK_FMAC_F16 on GFX11+ duplicates the f16 inline constant to both
685 // halves, so we need to produce the duplicated value for correct
686 // round-trip.
687 if (isGFX11Plus()) {
688 int64_t F16Val = getInlineImmValF16(Imm);
689 Imm = (F16Val << 16) | (F16Val & 0xFFFF);
690 } else {
692 }
693 break;
694 }
703 break;
704 default:
706 }
707 Op.setImm(Imm);
708 }
709 }
710 return true;
711}
712
714 ArrayRef<uint8_t> Bytes_,
715 uint64_t Address,
716 raw_ostream &CS) const {
717 unsigned MaxInstBytesNum = std::min((size_t)TargetMaxInstBytes, Bytes_.size());
718 Bytes = Bytes_.slice(0, MaxInstBytesNum);
719
720 // In case the opcode is not recognized we'll assume a Size of 4 bytes (unless
721 // there are fewer bytes left). This will be overridden on success.
722 Size = std::min((size_t)4, Bytes_.size());
723
724 do {
725 // ToDo: better to switch encoding length using some bit predicate
726 // but it is unknown yet, so try all we can
727
728 // Try to decode DPP and SDWA first to solve conflict with VOP1 and VOP2
729 // encodings
730 if (isGFX1250Plus() && Bytes.size() >= 16) {
731 std::bitset<128> DecW = eat16Bytes(Bytes);
732 if (tryDecodeInst(DecoderTableGFX1250128, MI, DecW, Address, CS))
733 break;
734 Bytes = Bytes_.slice(0, MaxInstBytesNum);
735 }
736
737 if (isGFX11Plus() && Bytes.size() >= 12) {
738 std::bitset<96> DecW = eat12Bytes(Bytes);
739
740 if (isGFX1170() &&
741 tryDecodeInst(DecoderTableGFX117096, DecoderTableGFX1170_FAKE1696, MI,
742 DecW, Address, CS))
743 break;
744
745 if (isGFX11() &&
746 tryDecodeInst(DecoderTableGFX1196, DecoderTableGFX11_FAKE1696, MI,
747 DecW, Address, CS))
748 break;
749
750 if (isGFX1250() &&
751 tryDecodeInst(DecoderTableGFX125096, DecoderTableGFX1250_FAKE1696, MI,
752 DecW, Address, CS))
753 break;
754
755 if (isGFX12() &&
756 tryDecodeInst(DecoderTableGFX1296, DecoderTableGFX12_FAKE1696, MI,
757 DecW, Address, CS))
758 break;
759
760 if (isGFX12() &&
761 tryDecodeInst(DecoderTableGFX12W6496, MI, DecW, Address, CS))
762 break;
763
764 if (isGFX13() &&
765 tryDecodeInst(DecoderTableGFX1396, DecoderTableGFX13_FAKE1696, MI,
766 DecW, Address, CS))
767 break;
768
769 if (STI.hasFeature(AMDGPU::Feature64BitLiterals)) {
770 // Return 8 bytes for a potential literal.
771 Bytes = Bytes_.slice(4, MaxInstBytesNum - 4);
772
773 if (isGFX1250() &&
774 tryDecodeInst(DecoderTableGFX125096, MI, DecW, Address, CS))
775 break;
776 }
777
778 // Reinitialize Bytes
779 Bytes = Bytes_.slice(0, MaxInstBytesNum);
780
781 } else if (Bytes.size() >= 16 &&
782 STI.hasFeature(AMDGPU::FeatureGFX950Insts)) {
783 std::bitset<128> DecW = eat16Bytes(Bytes);
784 if (tryDecodeInst(DecoderTableGFX940128, MI, DecW, Address, CS))
785 break;
786
787 // Reinitialize Bytes
788 Bytes = Bytes_.slice(0, MaxInstBytesNum);
789 }
790
791 if (Bytes.size() >= 8) {
792 const uint64_t QW = eatBytes<uint64_t>(Bytes);
793
794 if (STI.hasFeature(AMDGPU::FeatureGFX10_BEncoding) &&
795 tryDecodeInst(DecoderTableGFX10_B64, MI, QW, Address, CS))
796 break;
797
798 if (STI.hasFeature(AMDGPU::FeatureUnpackedD16VMem) &&
799 tryDecodeInst(DecoderTableGFX80_UNPACKED64, MI, QW, Address, CS))
800 break;
801
802 if (STI.hasFeature(AMDGPU::FeatureGFX950Insts) &&
803 tryDecodeInst(DecoderTableGFX95064, MI, QW, Address, CS))
804 break;
805
806 // Some GFX9 subtargets repurposed the v_mad_mix_f32, v_mad_mixlo_f16 and
807 // v_mad_mixhi_f16 for FMA variants. Try to decode using this special
808 // table first so we print the correct name.
809 if (STI.hasFeature(AMDGPU::FeatureFmaMixInsts) &&
810 tryDecodeInst(DecoderTableGFX9_DL64, MI, QW, Address, CS))
811 break;
812
813 if (STI.hasFeature(AMDGPU::FeatureGFX940Insts) &&
814 tryDecodeInst(DecoderTableGFX94064, MI, QW, Address, CS))
815 break;
816
817 if (STI.hasFeature(AMDGPU::FeatureGFX90AInsts) &&
818 tryDecodeInst(DecoderTableGFX90A64, MI, QW, Address, CS))
819 break;
820
821 if ((isVI() || isGFX9()) &&
822 tryDecodeInst(DecoderTableGFX864, MI, QW, Address, CS))
823 break;
824
825 if (isGFX9() && tryDecodeInst(DecoderTableGFX964, MI, QW, Address, CS))
826 break;
827
828 if (isGFX10() && tryDecodeInst(DecoderTableGFX1064, MI, QW, Address, CS))
829 break;
830
831 if (isGFX1250() &&
832 tryDecodeInst(DecoderTableGFX125064, DecoderTableGFX1250_FAKE1664, MI,
833 QW, Address, CS))
834 break;
835
836 if (isGFX12() &&
837 tryDecodeInst(DecoderTableGFX1264, DecoderTableGFX12_FAKE1664, MI, QW,
838 Address, CS))
839 break;
840
841 if (isGFX1170() &&
842 tryDecodeInst(DecoderTableGFX117064, DecoderTableGFX1170_FAKE1664, MI,
843 QW, Address, CS))
844 break;
845
846 if (isGFX11() &&
847 tryDecodeInst(DecoderTableGFX1164, DecoderTableGFX11_FAKE1664, MI, QW,
848 Address, CS))
849 break;
850
851 if (isGFX1170() &&
852 tryDecodeInst(DecoderTableGFX1170W6464, MI, QW, Address, CS))
853 break;
854
855 if (isGFX11() &&
856 tryDecodeInst(DecoderTableGFX11W6464, MI, QW, Address, CS))
857 break;
858
859 if (isGFX12() &&
860 tryDecodeInst(DecoderTableGFX12W6464, MI, QW, Address, CS))
861 break;
862
863 if (isGFX13() &&
864 tryDecodeInst(DecoderTableGFX1364, DecoderTableGFX13_FAKE1664, MI, QW,
865 Address, CS))
866 break;
867
868 // Reinitialize Bytes
869 Bytes = Bytes_.slice(0, MaxInstBytesNum);
870 }
871
872 // Try decode 32-bit instruction
873 if (Bytes.size() >= 4) {
874 const uint32_t DW = eatBytes<uint32_t>(Bytes);
875
876 if ((isVI() || isGFX9()) &&
877 tryDecodeInst(DecoderTableGFX832, MI, DW, Address, CS))
878 break;
879
880 if (tryDecodeInst(DecoderTableAMDGPU32, MI, DW, Address, CS))
881 break;
882
883 if (isGFX9() && tryDecodeInst(DecoderTableGFX932, MI, DW, Address, CS))
884 break;
885
886 if (STI.hasFeature(AMDGPU::FeatureGFX950Insts) &&
887 tryDecodeInst(DecoderTableGFX95032, MI, DW, Address, CS))
888 break;
889
890 if (STI.hasFeature(AMDGPU::FeatureGFX90AInsts) &&
891 tryDecodeInst(DecoderTableGFX90A32, MI, DW, Address, CS))
892 break;
893
894 if (STI.hasFeature(AMDGPU::FeatureGFX10_BEncoding) &&
895 tryDecodeInst(DecoderTableGFX10_B32, MI, DW, Address, CS))
896 break;
897
898 if (isGFX10() && tryDecodeInst(DecoderTableGFX1032, MI, DW, Address, CS))
899 break;
900
901 if (isGFX1170() &&
902 tryDecodeInst(DecoderTableGFX117032, DecoderTableGFX1170_FAKE1632, MI,
903 DW, Address, CS))
904 break;
905
906 if (isGFX11() &&
907 tryDecodeInst(DecoderTableGFX1132, DecoderTableGFX11_FAKE1632, MI, DW,
908 Address, CS))
909 break;
910
911 if (isGFX1250() &&
912 tryDecodeInst(DecoderTableGFX125032, DecoderTableGFX1250_FAKE1632, MI,
913 DW, Address, CS))
914 break;
915
916 if (isGFX12() &&
917 tryDecodeInst(DecoderTableGFX1232, DecoderTableGFX12_FAKE1632, MI, DW,
918 Address, CS))
919 break;
920
921 if (isGFX13() &&
922 tryDecodeInst(DecoderTableGFX1332, DecoderTableGFX13_FAKE1632, MI, DW,
923 Address, CS))
924 break;
925 }
926
928 } while (false);
929
931
932 if (SIInstrFlags::isDPP(*MCII, MI)) {
933 if (isMacDPP(MI))
935
936 if (SIInstrFlags::isVOP3P(*MCII, MI))
938 else if (SIInstrFlags::isVOPC(*MCII, MI))
939 convertVOPCDPPInst(MI); // Special VOP3 case
940 else if (AMDGPU::isVOPC64DPP(MI.getOpcode()))
941 convertVOPC64DPPInst(MI); // Special VOP3 case
942 else if (AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::dpp8) !=
943 -1)
945 else if (SIInstrFlags::isVOP3(*MCII, MI))
946 convertVOP3DPPInst(MI); // Regular VOP3 case
947 }
948
950
951 if (AMDGPU::isMAC(MI.getOpcode())) {
952 // Insert dummy unused src2_modifiers.
954 AMDGPU::OpName::src2_modifiers);
955 }
956
957 if (MI.getOpcode() == AMDGPU::V_CVT_SR_BF8_F32_e64_dpp ||
958 MI.getOpcode() == AMDGPU::V_CVT_SR_FP8_F32_e64_dpp) {
959 // Insert dummy unused src2_modifiers.
961 AMDGPU::OpName::src2_modifiers);
962 }
963
964 if (SIInstrFlags::isDS(*MCII, MI) && !AMDGPU::hasGDS(STI)) {
965 insertNamedMCOperand(MI, MCOperand::createImm(0), AMDGPU::OpName::gds);
966 }
967
968 if (SIInstrFlags::isMUBUF(*MCII, MI) || SIInstrFlags::isFLAT(*MCII, MI) ||
969 SIInstrFlags::isSMRD(*MCII, MI)) {
970 int CPolPos = AMDGPU::getNamedOperandIdx(MI.getOpcode(),
971 AMDGPU::OpName::cpol);
972 if (CPolPos != -1) {
973 unsigned CPol =
975 if (MI.getNumOperands() <= (unsigned)CPolPos) {
977 AMDGPU::OpName::cpol);
978 } else if (CPol) {
979 MI.getOperand(CPolPos).setImm(MI.getOperand(CPolPos).getImm() | CPol);
980 }
981 }
982 }
983
984 if (SIInstrFlags::isBuffer(*MCII, MI) &&
985 (STI.hasFeature(AMDGPU::FeatureGFX90AInsts))) {
986 // GFX90A lost TFE, its place is occupied by ACC.
987 int TFEOpIdx =
988 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::tfe);
989 if (TFEOpIdx != -1) {
990 auto *TFEIter = MI.begin();
991 std::advance(TFEIter, TFEOpIdx);
992 MI.insert(TFEIter, MCOperand::createImm(0));
993 }
994 }
995
996 // Validate buffer instruction offsets for GFX12+ - must not be a negative.
998 int OffsetIdx =
999 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::offset);
1000 if (OffsetIdx != -1) {
1001 uint32_t Imm = MI.getOperand(OffsetIdx).getImm();
1002 int64_t SignedOffset = SignExtend64<24>(Imm);
1003 if (SignedOffset < 0)
1004 return MCDisassembler::Fail;
1005 }
1006 }
1007
1008 if (SIInstrFlags::isBuffer(*MCII, MI)) {
1009 int SWZOpIdx =
1010 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::swz);
1011 if (SWZOpIdx != -1) {
1012 auto *SWZIter = MI.begin();
1013 std::advance(SWZIter, SWZOpIdx);
1014 MI.insert(SWZIter, MCOperand::createImm(0));
1015 }
1016 }
1017
1018 const MCInstrDesc &Desc = MCII->get(MI.getOpcode());
1020 int VAddr0Idx =
1021 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::vaddr0);
1022 int RsrcIdx =
1023 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::srsrc);
1024 unsigned NSAArgs = RsrcIdx - VAddr0Idx - 1;
1025 if (VAddr0Idx >= 0 && NSAArgs > 0) {
1026 unsigned NSAWords = (NSAArgs + 3) / 4;
1027 if (Bytes.size() < 4 * NSAWords)
1028 return MCDisassembler::Fail;
1029 for (unsigned i = 0; i < NSAArgs; ++i) {
1030 const unsigned VAddrIdx = VAddr0Idx + 1 + i;
1031 auto VAddrRCID =
1032 MCII->getOpRegClassID(Desc.operands()[VAddrIdx], HwModeRegClass);
1033 MI.insert(MI.begin() + VAddrIdx, createRegOperand(VAddrRCID, Bytes[i]));
1034 }
1035 Bytes = Bytes.slice(4 * NSAWords);
1036 }
1037
1039 }
1040
1041 if (SIInstrFlags::isVIMAGE(*MCII, MI) || SIInstrFlags::isVSAMPLE(*MCII, MI))
1043
1044 if (SIInstrFlags::isEXP(*MCII, MI))
1046
1047 if (SIInstrFlags::isVINTERP(*MCII, MI))
1049
1050 if (SIInstrFlags::isSDWA(*MCII, MI))
1052
1053 if (SIInstrFlags::isMAI(*MCII, MI) && !convertMAIInst(MI))
1054 return MCDisassembler::Fail;
1055
1056 if (SIInstrFlags::isWMMA(*MCII, MI) && !convertWMMAInst(MI))
1057 return MCDisassembler::Fail;
1058
1059 int VDstIn_Idx = AMDGPU::getNamedOperandIdx(MI.getOpcode(),
1060 AMDGPU::OpName::vdst_in);
1061 if (VDstIn_Idx != -1) {
1062 int Tied = MCII->get(MI.getOpcode()).getOperandConstraint(VDstIn_Idx,
1064 if (Tied != -1 && (MI.getNumOperands() <= (unsigned)VDstIn_Idx ||
1065 !MI.getOperand(VDstIn_Idx).isReg() ||
1066 MI.getOperand(VDstIn_Idx).getReg() != MI.getOperand(Tied).getReg())) {
1067 if (MI.getNumOperands() > (unsigned)VDstIn_Idx)
1068 MI.erase(&MI.getOperand(VDstIn_Idx));
1070 MCOperand::createReg(MI.getOperand(Tied).getReg()),
1071 AMDGPU::OpName::vdst_in);
1072 }
1073 }
1074
1075 bool IsSOPK = SIInstrFlags::isSOPK(*MCII, MI);
1076 if (AMDGPU::hasNamedOperand(MI.getOpcode(), AMDGPU::OpName::imm) && !IsSOPK)
1078
1079 // Some VOPC instructions, e.g., v_cmpx_f_f64, use VOP3 encoding and
1080 // have EXEC as implicit destination. Issue a warning if encoding for
1081 // vdst is not EXEC.
1082 if (SIInstrFlags::isVOP3(*MCII, MI) &&
1083 MCII->get(MI.getOpcode()).getNumDefs() == 0 &&
1084 MCII->get(MI.getOpcode()).hasImplicitDefOfPhysReg(AMDGPU::EXEC)) {
1085 auto ExecEncoding = MRI.getEncodingValue(AMDGPU::EXEC_LO);
1086 if (Bytes_[0] != ExecEncoding)
1088 }
1089
1090 Size = MaxInstBytesNum - Bytes.size();
1091 return Status;
1092}
1093
1095 if (STI.hasFeature(AMDGPU::FeatureGFX11Insts)) {
1096 // The MCInst still has these fields even though they are no longer encoded
1097 // in the GFX11 instruction.
1098 insertNamedMCOperand(MI, MCOperand::createImm(0), AMDGPU::OpName::vm);
1099 insertNamedMCOperand(MI, MCOperand::createImm(0), AMDGPU::OpName::compr);
1100 }
1101}
1102
1105 if (MI.getOpcode() == AMDGPU::V_INTERP_P10_F16_F32_inreg_t16_gfx11 ||
1106 MI.getOpcode() == AMDGPU::V_INTERP_P10_F16_F32_inreg_fake16_gfx11 ||
1107 MI.getOpcode() == AMDGPU::V_INTERP_P10_F16_F32_inreg_t16_gfx12 ||
1108 MI.getOpcode() == AMDGPU::V_INTERP_P10_F16_F32_inreg_fake16_gfx12 ||
1109 MI.getOpcode() == AMDGPU::V_INTERP_P10_F16_F32_inreg_t16_gfx13 ||
1110 MI.getOpcode() == AMDGPU::V_INTERP_P10_F16_F32_inreg_fake16_gfx13 ||
1111 MI.getOpcode() == AMDGPU::V_INTERP_P10_RTZ_F16_F32_inreg_t16_gfx11 ||
1112 MI.getOpcode() == AMDGPU::V_INTERP_P10_RTZ_F16_F32_inreg_fake16_gfx11 ||
1113 MI.getOpcode() == AMDGPU::V_INTERP_P10_RTZ_F16_F32_inreg_t16_gfx12 ||
1114 MI.getOpcode() == AMDGPU::V_INTERP_P10_RTZ_F16_F32_inreg_fake16_gfx12 ||
1115 MI.getOpcode() == AMDGPU::V_INTERP_P10_RTZ_F16_F32_inreg_t16_gfx13 ||
1116 MI.getOpcode() == AMDGPU::V_INTERP_P10_RTZ_F16_F32_inreg_fake16_gfx13 ||
1117 MI.getOpcode() == AMDGPU::V_INTERP_P2_F16_F32_inreg_t16_gfx11 ||
1118 MI.getOpcode() == AMDGPU::V_INTERP_P2_F16_F32_inreg_fake16_gfx11 ||
1119 MI.getOpcode() == AMDGPU::V_INTERP_P2_F16_F32_inreg_t16_gfx12 ||
1120 MI.getOpcode() == AMDGPU::V_INTERP_P2_F16_F32_inreg_fake16_gfx12 ||
1121 MI.getOpcode() == AMDGPU::V_INTERP_P2_F16_F32_inreg_t16_gfx13 ||
1122 MI.getOpcode() == AMDGPU::V_INTERP_P2_F16_F32_inreg_fake16_gfx13 ||
1123 MI.getOpcode() == AMDGPU::V_INTERP_P2_RTZ_F16_F32_inreg_t16_gfx11 ||
1124 MI.getOpcode() == AMDGPU::V_INTERP_P2_RTZ_F16_F32_inreg_fake16_gfx11 ||
1125 MI.getOpcode() == AMDGPU::V_INTERP_P2_RTZ_F16_F32_inreg_t16_gfx12 ||
1126 MI.getOpcode() == AMDGPU::V_INTERP_P2_RTZ_F16_F32_inreg_fake16_gfx12 ||
1127 MI.getOpcode() == AMDGPU::V_INTERP_P2_RTZ_F16_F32_inreg_t16_gfx13 ||
1128 MI.getOpcode() == AMDGPU::V_INTERP_P2_RTZ_F16_F32_inreg_fake16_gfx13) {
1129 // The MCInst has this field that is not directly encoded in the
1130 // instruction.
1131 insertNamedMCOperand(MI, MCOperand::createImm(0), AMDGPU::OpName::op_sel);
1132 }
1133}
1134
1136 if (STI.hasFeature(AMDGPU::FeatureGFX9) ||
1137 STI.hasFeature(AMDGPU::FeatureGFX10)) {
1138 if (AMDGPU::hasNamedOperand(MI.getOpcode(), AMDGPU::OpName::sdst))
1139 // VOPC - insert clamp
1140 insertNamedMCOperand(MI, MCOperand::createImm(0), AMDGPU::OpName::clamp);
1141 } else if (STI.hasFeature(AMDGPU::FeatureVolcanicIslands)) {
1142 int SDst = AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::sdst);
1143 if (SDst != -1) {
1144 // VOPC - insert VCC register as sdst
1146 AMDGPU::OpName::sdst);
1147 } else {
1148 // VOP1/2 - insert omod if present in instruction
1149 insertNamedMCOperand(MI, MCOperand::createImm(0), AMDGPU::OpName::omod);
1150 }
1151 }
1152}
1153
1154/// Adjust the register values used by V_MFMA_F8F6F4_f8_f8 instructions to the
1155/// appropriate subregister for the used format width.
1156///
1157/// \returns false if the operand cannot be narrowed down to \p NumRegs, which
1158/// means the encoding is malformed.
1160 MCOperand &MO, uint8_t NumRegs) {
1161 // A malformed encoding can select an operand that is not a register at all.
1162 if (!MO.isReg())
1163 return false;
1164
1165 MCRegister NewReg;
1166 switch (NumRegs) {
1167 case 4:
1168 NewReg = MRI.getSubReg(MO.getReg(), AMDGPU::sub0_sub1_sub2_sub3);
1169 break;
1170 case 6:
1171 NewReg = MRI.getSubReg(MO.getReg(), AMDGPU::sub0_sub1_sub2_sub3_sub4_sub5);
1172 break;
1173 case 8:
1174 NewReg = MRI.getSubReg(MO.getReg(),
1175 AMDGPU::sub0_sub1_sub2_sub3_sub4_sub5_sub6_sub7);
1176 // For mfma f8/f8 is the widest format, so the operand already has the
1177 // requested width and there is no subregister to select.
1178 if (!NewReg)
1179 return true;
1180 break;
1181 case 12:
1182 // There is no 384-bit subreg index defined.
1183 if (MCRegister BaseReg = MRI.getSubReg(MO.getReg(), AMDGPU::sub0)) {
1184 NewReg = MRI.getMatchingSuperReg(
1185 BaseReg, AMDGPU::sub0, &MRI.getRegClass(AMDGPU::VReg_384RegClassID));
1186 }
1187 break;
1188 case 16:
1189 // No-op in cases where one operand is still f8/bf8.
1190 return true;
1191 default:
1192 llvm_unreachable("Unexpected size for mfma/wmma f8f6f4 operand");
1193 }
1194
1195 if (!NewReg)
1196 return false;
1197
1198 MO.setReg(NewReg);
1199 return true;
1200}
1201
1202/// f8f6f4 instructions have different pseudos depending on the used formats. In
1203/// the disassembler table, we only have the variants with the largest register
1204/// classes which assume using an fp8/bf8 format for both operands. The actual
1205/// register class depends on the format in blgp and cbsz operands. Adjust the
1206/// register classes depending on the used format.
1208 int BlgpIdx =
1209 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::blgp);
1210 if (BlgpIdx == -1)
1211 return true;
1212
1213 int CbszIdx =
1214 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::cbsz);
1215
1216 unsigned CBSZ = MI.getOperand(CbszIdx).getImm();
1217 unsigned BLGP = MI.getOperand(BlgpIdx).getImm();
1218
1219 const AMDGPU::MFMA_F8F6F4_Info *AdjustedRegClassOpcode =
1220 AMDGPU::getMFMA_F8F6F4_WithFormatArgs(CBSZ, BLGP, MI.getOpcode());
1221 if (!AdjustedRegClassOpcode ||
1222 AdjustedRegClassOpcode->Opcode == MI.getOpcode())
1223 return true;
1224
1225 MI.setOpcode(AdjustedRegClassOpcode->Opcode);
1226 int Src0Idx =
1227 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::src0);
1228 int Src1Idx =
1229 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::src1);
1230 return adjustMFMA_F8F6F4OpRegClass(MRI, MI.getOperand(Src0Idx),
1231 AdjustedRegClassOpcode->NumRegsSrcA) &&
1232 adjustMFMA_F8F6F4OpRegClass(MRI, MI.getOperand(Src1Idx),
1233 AdjustedRegClassOpcode->NumRegsSrcB);
1234}
1235
1237 int FmtAIdx =
1238 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::matrix_a_fmt);
1239 if (FmtAIdx == -1)
1240 return true;
1241
1242 int FmtBIdx =
1243 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::matrix_b_fmt);
1244
1245 unsigned FmtA = MI.getOperand(FmtAIdx).getImm();
1246 unsigned FmtB = MI.getOperand(FmtBIdx).getImm();
1247
1248 const AMDGPU::MFMA_F8F6F4_Info *AdjustedRegClassOpcode =
1249 AMDGPU::getWMMA_F8F6F4_WithFormatArgs(FmtA, FmtB, MI.getOpcode());
1250 if (!AdjustedRegClassOpcode ||
1251 AdjustedRegClassOpcode->Opcode == MI.getOpcode())
1252 return true;
1253
1254 MI.setOpcode(AdjustedRegClassOpcode->Opcode);
1255 int Src0Idx =
1256 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::src0);
1257 int Src1Idx =
1258 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::src1);
1259 return adjustMFMA_F8F6F4OpRegClass(MRI, MI.getOperand(Src0Idx),
1260 AdjustedRegClassOpcode->NumRegsSrcA) &&
1261 adjustMFMA_F8F6F4OpRegClass(MRI, MI.getOperand(Src1Idx),
1262 AdjustedRegClassOpcode->NumRegsSrcB);
1263}
1264
1266 unsigned OpSel = 0;
1267 unsigned OpSelHi = 0;
1268 unsigned NegLo = 0;
1269 unsigned NegHi = 0;
1270};
1271
1272// Reconstruct values of VOP3/VOP3P operands such as op_sel.
1273// Note that these values do not affect disassembler output,
1274// so this is only necessary for consistency with src_modifiers.
1276 bool IsVOP3P = false) {
1277 VOPModifiers Modifiers;
1278 unsigned Opc = MI.getOpcode();
1279 const AMDGPU::OpName ModOps[] = {AMDGPU::OpName::src0_modifiers,
1280 AMDGPU::OpName::src1_modifiers,
1281 AMDGPU::OpName::src2_modifiers};
1282 for (int J = 0; J < 3; ++J) {
1283 int OpIdx = AMDGPU::getNamedOperandIdx(Opc, ModOps[J]);
1284 if (OpIdx == -1)
1285 continue;
1286
1287 unsigned Val = MI.getOperand(OpIdx).getImm();
1288
1289 Modifiers.OpSel |= !!(Val & SISrcMods::OP_SEL_0) << J;
1290 if (IsVOP3P) {
1291 Modifiers.OpSelHi |= !!(Val & SISrcMods::OP_SEL_1) << J;
1292 Modifiers.NegLo |= !!(Val & SISrcMods::NEG) << J;
1293 Modifiers.NegHi |= !!(Val & SISrcMods::NEG_HI) << J;
1294 } else if (J == 0) {
1295 Modifiers.OpSel |= !!(Val & SISrcMods::DST_OP_SEL) << 3;
1296 }
1297 }
1298
1299 return Modifiers;
1300}
1301
1302// Instructions decode the op_sel/suffix bits into the src_modifier
1303// operands. Copy those bits into the src operands for true16 VGPRs.
1305 const unsigned Opc = MI.getOpcode();
1306 const MCRegisterClass &ConversionRC =
1307 MRI.getRegClass(AMDGPU::VGPR_16RegClassID);
1308 constexpr std::array<std::tuple<AMDGPU::OpName, AMDGPU::OpName, unsigned>, 4>
1309 OpAndOpMods = {{{AMDGPU::OpName::src0, AMDGPU::OpName::src0_modifiers,
1311 {AMDGPU::OpName::src1, AMDGPU::OpName::src1_modifiers,
1313 {AMDGPU::OpName::src2, AMDGPU::OpName::src2_modifiers,
1315 {AMDGPU::OpName::vdst, AMDGPU::OpName::src0_modifiers,
1317 for (const auto &[OpName, OpModsName, OpSelMask] : OpAndOpMods) {
1318 int OpIdx = AMDGPU::getNamedOperandIdx(Opc, OpName);
1319 int OpModsIdx = AMDGPU::getNamedOperandIdx(Opc, OpModsName);
1320 if (OpIdx == -1 || OpModsIdx == -1)
1321 continue;
1322 MCOperand &Op = MI.getOperand(OpIdx);
1323 if (!Op.isReg())
1324 continue;
1325 if (!ConversionRC.contains(Op.getReg()))
1326 continue;
1327 unsigned OpEnc = MRI.getEncodingValue(Op.getReg());
1328 const MCOperand &OpMods = MI.getOperand(OpModsIdx);
1329 unsigned ModVal = OpMods.getImm();
1330 if (ModVal & OpSelMask) { // isHi
1331 unsigned RegIdx = OpEnc & AMDGPU::HWEncoding::REG_IDX_MASK;
1332 Op.setReg(ConversionRC.getRegister(RegIdx * 2 + 1));
1333 }
1334 }
1335}
1336
1337// MAC opcodes have special old and src2 operands.
1338// src2 is tied to dst, while old is not tied (but assumed to be).
1340 constexpr int DST_IDX = 0;
1341 auto Opcode = MI.getOpcode();
1342 const auto &Desc = MCII->get(Opcode);
1343 auto OldIdx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::old);
1344
1345 if (OldIdx != -1 && Desc.getOperandConstraint(
1346 OldIdx, MCOI::OperandConstraint::TIED_TO) == -1) {
1347 assert(AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::src2));
1348 assert(Desc.getOperandConstraint(
1349 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src2),
1351 (void)DST_IDX;
1352 return true;
1353 }
1354
1355 return false;
1356}
1357
1358// Create dummy old operand and insert dummy unused src2_modifiers
1360 assert(MI.getNumOperands() + 1 < MCII->get(MI.getOpcode()).getNumOperands());
1361 insertNamedMCOperand(MI, MCOperand::createReg(0), AMDGPU::OpName::old);
1363 AMDGPU::OpName::src2_modifiers);
1364}
1365
1367 unsigned Opc = MI.getOpcode();
1368
1369 int VDstInIdx =
1370 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::vdst_in);
1371 if (VDstInIdx != -1)
1372 insertNamedMCOperand(MI, MI.getOperand(0), AMDGPU::OpName::vdst_in);
1373
1374 unsigned DescNumOps = MCII->get(Opc).getNumOperands();
1375 if (MI.getNumOperands() < DescNumOps &&
1376 AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::op_sel)) {
1378 auto Mods = collectVOPModifiers(MI);
1380 AMDGPU::OpName::op_sel);
1381 } else {
1382 // Insert dummy unused src modifiers.
1383 if (MI.getNumOperands() < DescNumOps &&
1384 AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::src0_modifiers))
1386 AMDGPU::OpName::src0_modifiers);
1387
1388 if (MI.getNumOperands() < DescNumOps &&
1389 AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::src1_modifiers))
1391 AMDGPU::OpName::src1_modifiers);
1392 }
1393}
1394
1397
1398 int VDstInIdx =
1399 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::vdst_in);
1400 if (VDstInIdx != -1)
1401 insertNamedMCOperand(MI, MI.getOperand(0), AMDGPU::OpName::vdst_in);
1402
1403 unsigned Opc = MI.getOpcode();
1404 unsigned DescNumOps = MCII->get(Opc).getNumOperands();
1405 if (MI.getNumOperands() < DescNumOps &&
1406 AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::op_sel)) {
1407 auto Mods = collectVOPModifiers(MI);
1409 AMDGPU::OpName::op_sel);
1410 }
1411}
1412
1413// Given a wide tuple \p Reg check if it will overflow 256 registers.
1414// \returns \p Reg on success or NoRegister otherwise.
1416 const MCRegisterInfo &MRI) {
1417 unsigned NumRegs = RC.getSizeInBits() / 32;
1418 MCRegister Sub0 = MRI.getSubReg(Reg, AMDGPU::sub0);
1419 if (!Sub0)
1420 return Reg;
1421
1422 MCRegister BaseReg;
1423 if (MRI.getRegClass(AMDGPU::VGPR_32RegClassID).contains(Sub0))
1424 BaseReg = AMDGPU::VGPR0;
1425 else if (MRI.getRegClass(AMDGPU::AGPR_32RegClassID).contains(Sub0))
1426 BaseReg = AMDGPU::AGPR0;
1427
1428 assert(BaseReg && "Only vector registers expected");
1429
1430 return (Sub0 - BaseReg + NumRegs <= 256) ? Reg : MCRegister();
1431}
1432
1433// Note that before gfx10, the MIMG encoding provided no information about
1434// VADDR size. Consequently, decoded instructions always show address as if it
1435// has 1 dword, which could be not really so.
1437 int VDstIdx = AMDGPU::getNamedOperandIdx(MI.getOpcode(),
1438 AMDGPU::OpName::vdst);
1439
1440 int VDataIdx = AMDGPU::getNamedOperandIdx(MI.getOpcode(),
1441 AMDGPU::OpName::vdata);
1442 int VAddr0Idx =
1443 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::vaddr0);
1444 AMDGPU::OpName RsrcOpName = SIInstrFlags::isMIMG(*MCII, MI)
1445 ? AMDGPU::OpName::srsrc
1446 : AMDGPU::OpName::rsrc;
1447 int RsrcIdx = AMDGPU::getNamedOperandIdx(MI.getOpcode(), RsrcOpName);
1448 int DMaskIdx = AMDGPU::getNamedOperandIdx(MI.getOpcode(),
1449 AMDGPU::OpName::dmask);
1450
1451 int TFEIdx = AMDGPU::getNamedOperandIdx(MI.getOpcode(),
1452 AMDGPU::OpName::tfe);
1453 int D16Idx = AMDGPU::getNamedOperandIdx(MI.getOpcode(),
1454 AMDGPU::OpName::d16);
1455
1456 const AMDGPU::MIMGInfo *Info = AMDGPU::getMIMGInfo(MI.getOpcode());
1457 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode =
1458 AMDGPU::getMIMGBaseOpcodeInfo(Info->BaseOpcode);
1459
1460 assert(VDataIdx != -1);
1461 if (BaseOpcode->BVH) {
1462 // Add A16 operand for intersect_ray instructions
1463 addOperand(MI, MCOperand::createImm(BaseOpcode->A16));
1464 return;
1465 }
1466
1467 bool IsAtomic = (VDstIdx != -1);
1468 bool IsGather4 = SIInstrFlags::isGather4(*MCII, MI);
1469 bool IsVSample = SIInstrFlags::isVSAMPLE(*MCII, MI);
1470 bool IsNSA = false;
1471 bool IsPartialNSA = false;
1472 unsigned AddrSize = Info->VAddrDwords;
1473
1474 if (isGFX10Plus()) {
1475 unsigned DimIdx =
1476 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::dim);
1477 int A16Idx =
1478 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::a16);
1479 const AMDGPU::MIMGDimInfo *Dim =
1480 AMDGPU::getMIMGDimInfoByEncoding(MI.getOperand(DimIdx).getImm());
1481 const bool IsA16 = (A16Idx != -1 && MI.getOperand(A16Idx).getImm());
1482
1483 AddrSize =
1484 AMDGPU::getAddrSizeMIMGOp(BaseOpcode, Dim, IsA16, AMDGPU::hasG16(STI));
1485
1486 // VSAMPLE insts that do not use vaddr3 behave the same as NSA forms.
1487 // VIMAGE insts other than BVH never use vaddr4.
1488 IsNSA = Info->MIMGEncoding == AMDGPU::MIMGEncGfx10NSA ||
1489 Info->MIMGEncoding == AMDGPU::MIMGEncGfx11NSA ||
1490 Info->MIMGEncoding == AMDGPU::MIMGEncGfx12 ||
1491 Info->MIMGEncoding == AMDGPU::MIMGEncGfx13;
1492 if (!IsNSA) {
1493 if (!IsVSample && AddrSize > 12)
1494 AddrSize = 16;
1495 } else {
1496 if (AddrSize > Info->VAddrDwords) {
1497 if (!STI.hasFeature(AMDGPU::FeaturePartialNSAEncoding)) {
1498 // The NSA encoding does not contain enough operands for the
1499 // combination of base opcode / dimension. Should this be an error?
1500 return;
1501 }
1502 IsPartialNSA = true;
1503 }
1504 }
1505 }
1506
1507 unsigned DMask = MI.getOperand(DMaskIdx).getImm() & 0xf;
1508 unsigned DstSize = IsGather4 ? 4 : std::max(llvm::popcount(DMask), 1);
1509
1510 bool D16 = D16Idx >= 0 && MI.getOperand(D16Idx).getImm();
1511 if (D16 && AMDGPU::hasPackedD16(STI)) {
1512 DstSize = (DstSize + 1) / 2;
1513 }
1514
1515 if (TFEIdx != -1 && MI.getOperand(TFEIdx).getImm())
1516 DstSize += 1;
1517
1518 if (DstSize == Info->VDataDwords && AddrSize == Info->VAddrDwords)
1519 return;
1520
1521 int NewOpcode =
1522 AMDGPU::getMIMGOpcode(Info->BaseOpcode, Info->MIMGEncoding, DstSize,
1523 AddrSize, Info->IndexedRsrc, Info->IndexedSamp);
1524 if (NewOpcode == -1)
1525 return;
1526
1527 // Widen the register to the correct number of enabled channels.
1528 MCRegister NewVdata;
1529 if (DstSize != Info->VDataDwords) {
1530 auto DataRCID = MCII->getOpRegClassID(
1531 MCII->get(NewOpcode).operands()[VDataIdx], HwModeRegClass);
1532
1533 // Get first subregister of VData
1534 MCRegister Vdata0 = MI.getOperand(VDataIdx).getReg();
1535 MCRegister VdataSub0 = MRI.getSubReg(Vdata0, AMDGPU::sub0);
1536 Vdata0 = (VdataSub0 != 0)? VdataSub0 : Vdata0;
1537
1538 const MCRegisterClass &NewRC = MRI.getRegClass(DataRCID);
1539 NewVdata = MRI.getMatchingSuperReg(Vdata0, AMDGPU::sub0, &NewRC);
1540 NewVdata = CheckVGPROverflow(NewVdata, NewRC, MRI);
1541 if (!NewVdata) {
1542 // It's possible to encode this such that the low register + enabled
1543 // components exceeds the register count.
1544 return;
1545 }
1546 }
1547
1548 // If not using NSA on GFX10+, widen vaddr0 address register to correct size.
1549 // If using partial NSA on GFX11+ widen last address register.
1550 int VAddrSAIdx = IsPartialNSA ? (RsrcIdx - 1) : VAddr0Idx;
1551 MCRegister NewVAddrSA;
1552 if (STI.hasFeature(AMDGPU::FeatureNSAEncoding) && (!IsNSA || IsPartialNSA) &&
1553 AddrSize != Info->VAddrDwords) {
1554 MCRegister VAddrSA = MI.getOperand(VAddrSAIdx).getReg();
1555 MCRegister VAddrSubSA = MRI.getSubReg(VAddrSA, AMDGPU::sub0);
1556 VAddrSA = VAddrSubSA ? VAddrSubSA : VAddrSA;
1557
1558 auto AddrRCID = MCII->getOpRegClassID(
1559 MCII->get(NewOpcode).operands()[VAddrSAIdx], HwModeRegClass);
1560
1561 const MCRegisterClass &NewRC = MRI.getRegClass(AddrRCID);
1562 NewVAddrSA = MRI.getMatchingSuperReg(VAddrSA, AMDGPU::sub0, &NewRC);
1563 NewVAddrSA = CheckVGPROverflow(NewVAddrSA, NewRC, MRI);
1564 if (!NewVAddrSA)
1565 return;
1566 }
1567
1568 MI.setOpcode(NewOpcode);
1569
1570 if (NewVdata != AMDGPU::NoRegister) {
1571 MI.getOperand(VDataIdx) = MCOperand::createReg(NewVdata);
1572
1573 if (IsAtomic) {
1574 // Atomic operations have an additional operand (a copy of data)
1575 MI.getOperand(VDstIdx) = MCOperand::createReg(NewVdata);
1576 }
1577 }
1578
1579 if (NewVAddrSA) {
1580 MI.getOperand(VAddrSAIdx) = MCOperand::createReg(NewVAddrSA);
1581 } else if (IsNSA) {
1582 assert(AddrSize <= Info->VAddrDwords);
1583 MI.erase(MI.begin() + VAddr0Idx + AddrSize,
1584 MI.begin() + VAddr0Idx + Info->VAddrDwords);
1585 }
1586}
1587
1588// Opsel and neg bits are used in src_modifiers and standalone operands. Autogen
1589// decoder only adds to src_modifiers, so manually add the bits to the other
1590// operands.
1592 unsigned Opc = MI.getOpcode();
1593 unsigned DescNumOps = MCII->get(Opc).getNumOperands();
1594 auto Mods = collectVOPModifiers(MI, true);
1595
1596 if (MI.getNumOperands() < DescNumOps &&
1597 AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::vdst_in))
1598 insertNamedMCOperand(MI, MCOperand::createImm(0), AMDGPU::OpName::vdst_in);
1599
1600 if (MI.getNumOperands() < DescNumOps &&
1601 AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::op_sel))
1603 AMDGPU::OpName::op_sel);
1604 if (MI.getNumOperands() < DescNumOps &&
1605 AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::op_sel_hi))
1607 AMDGPU::OpName::op_sel_hi);
1608 if (MI.getNumOperands() < DescNumOps &&
1609 AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::neg_lo))
1611 AMDGPU::OpName::neg_lo);
1612 if (MI.getNumOperands() < DescNumOps &&
1613 AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::neg_hi))
1615 AMDGPU::OpName::neg_hi);
1616}
1617
1618// Create dummy old operand and insert optional operands
1620 unsigned Opc = MI.getOpcode();
1621 unsigned DescNumOps = MCII->get(Opc).getNumOperands();
1622
1623 if (MI.getNumOperands() < DescNumOps &&
1624 AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::old))
1625 insertNamedMCOperand(MI, MCOperand::createReg(0), AMDGPU::OpName::old);
1626
1627 if (MI.getNumOperands() < DescNumOps &&
1628 AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::src0_modifiers))
1630 AMDGPU::OpName::src0_modifiers);
1631
1632 if (MI.getNumOperands() < DescNumOps &&
1633 AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::src1_modifiers))
1635 AMDGPU::OpName::src1_modifiers);
1636}
1637
1639 unsigned Opc = MI.getOpcode();
1640 unsigned DescNumOps = MCII->get(Opc).getNumOperands();
1641
1643
1644 if (MI.getNumOperands() < DescNumOps &&
1645 AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::op_sel)) {
1648 AMDGPU::OpName::op_sel);
1649 }
1650}
1651
1653 assert(HasLiteral && "Should have decoded a literal");
1654 insertNamedMCOperand(MI, MCOperand::createImm(Literal), AMDGPU::OpName::immX);
1655}
1656
1657const char* AMDGPUDisassembler::getRegClassName(unsigned RegClassID) const {
1659 &getAMDGPUMCRegisterClass(RegClassID));
1660}
1661
1662inline
1664 const Twine& ErrMsg) const {
1665 *CommentStream << "Error: " + ErrMsg;
1666
1667 // ToDo: add support for error operands to MCInst.h
1668 // return MCOperand::createError(V);
1669 return MCOperand();
1670}
1671
1675
1676inline
1678 unsigned Val) const {
1679 const auto &RegCl = getAMDGPUMCRegisterClass(RegClassID);
1680 if (Val >= RegCl.getNumRegs())
1681 return errOperand(Val, Twine(getRegClassName(RegClassID)) +
1682 ": unknown register " + Twine(Val));
1683 return createRegOperand(RegCl.getRegister(Val));
1684}
1685
1686inline
1688 unsigned Val) const {
1689 // ToDo: SI/CI have 104 SGPRs, VI - 102
1690 // Valery: here we accepting as much as we can, let assembler sort it out
1691 int shift = 0;
1692 switch (SRegClassID) {
1693 case AMDGPU::SGPR_32RegClassID:
1694 case AMDGPU::TTMP_32RegClassID:
1695 break;
1696 case AMDGPU::SGPR_64RegClassID:
1697 case AMDGPU::TTMP_64RegClassID:
1698 shift = 1;
1699 break;
1700 case AMDGPU::SGPR_96RegClassID:
1701 case AMDGPU::TTMP_96RegClassID:
1702 case AMDGPU::SGPR_128RegClassID:
1703 case AMDGPU::TTMP_128RegClassID:
1704 // ToDo: unclear if s[100:104] is available on VI. Can we use VCC as SGPR in
1705 // this bundle?
1706 case AMDGPU::SGPR_256RegClassID:
1707 case AMDGPU::TTMP_256RegClassID:
1708 // ToDo: unclear if s[96:104] is available on VI. Can we use VCC as SGPR in
1709 // this bundle?
1710 case AMDGPU::SGPR_288RegClassID:
1711 case AMDGPU::TTMP_288RegClassID:
1712 case AMDGPU::SGPR_320RegClassID:
1713 case AMDGPU::TTMP_320RegClassID:
1714 case AMDGPU::SGPR_352RegClassID:
1715 case AMDGPU::TTMP_352RegClassID:
1716 case AMDGPU::SGPR_384RegClassID:
1717 case AMDGPU::TTMP_384RegClassID:
1718 case AMDGPU::SGPR_512RegClassID:
1719 case AMDGPU::TTMP_512RegClassID:
1720 shift = 2;
1721 break;
1722 // ToDo: unclear if s[88:104] is available on VI. Can we use VCC as SGPR in
1723 // this bundle?
1724 default:
1725 llvm_unreachable("unhandled register class");
1726 }
1727
1728 if (Val % (1 << shift)) {
1729 *CommentStream << "Warning: " << getRegClassName(SRegClassID)
1730 << ": scalar reg isn't aligned " << Val;
1731 }
1732
1733 return createRegOperand(SRegClassID, Val >> shift);
1734}
1735
1737 bool IsHi) const {
1738 unsigned RegIdxInVGPR16 = RegIdx * 2 + (IsHi ? 1 : 0);
1739 return createRegOperand(AMDGPU::VGPR_16RegClassID, RegIdxInVGPR16);
1740}
1741
1742// Decode Literals for insts which always have a literal in the encoding
1745 if (HasLiteral) {
1746 assert(
1748 "Should only decode multiple kimm with VOPD, check VSrc operand types");
1749 if (Literal != Val)
1750 return errOperand(Val, "More than one unique literal is illegal");
1751 }
1752 HasLiteral = true;
1753 Literal = Val;
1754 return MCOperand::createImm(Literal);
1755}
1756
1759 if (HasLiteral) {
1760 if (Literal != Val)
1761 return errOperand(Val, "More than one unique literal is illegal");
1762 }
1763 HasLiteral = true;
1764 Literal = Val;
1765
1766 bool UseLit64 = Hi_32(Literal) == 0;
1768 LitModifier::Lit64, Literal, getContext()))
1769 : MCOperand::createImm(Literal);
1770}
1771
1774 const MCOperandInfo &OpDesc) const {
1775 // For now all literal constants are supposed to be unsigned integer
1776 // ToDo: deal with signed/unsigned 64-bit integer constants
1777 // ToDo: deal with float/double constants
1778 if (!HasLiteral) {
1779 if (Bytes.size() < 4) {
1780 return errOperand(0, "cannot read literal, inst bytes left " +
1781 Twine(Bytes.size()));
1782 }
1783 HasLiteral = true;
1784 Literal = eatBytes<uint32_t>(Bytes);
1785 }
1786
1787 // For disassembling always assume all inline constants are available.
1788 bool HasInv2Pi = true;
1789
1790 // Invalid instruction codes may contain literals for inline-only
1791 // operands, so we support them here as well.
1792 int64_t Val = Literal;
1793 bool UseLit = false;
1794 switch (OpDesc.OperandType) {
1795 default:
1796 llvm_unreachable("Unexpected operand type!");
1800 UseLit = AMDGPU::isInlinableLiteralBF16(Val, HasInv2Pi);
1801 break;
1804 break;
1808 UseLit = AMDGPU::isInlinableLiteralFP16(Val, HasInv2Pi);
1809 break;
1811 UseLit = AMDGPU::isInlinableLiteralV2F16(Val);
1812 break;
1815 break;
1818 break;
1822 UseLit = AMDGPU::isInlinableLiteralI16(Val, HasInv2Pi);
1823 break;
1825 UseLit = AMDGPU::isInlinableLiteralV2I16(Val);
1826 break;
1836 UseLit = AMDGPU::isInlinableLiteral32(Val, HasInv2Pi);
1837 break;
1842 UseLit = AMDGPU::isInlinableLiteral64(Val << 32, HasInv2Pi);
1843 if (!UseLit)
1844 Val <<= 32;
1845 break;
1849 UseLit = AMDGPU::isInlinableLiteral64(Val, HasInv2Pi);
1850 break;
1852 // TODO: Disassembling V_DUAL_FMAMK_F32_X_FMAMK_F32_gfx11 hits
1853 // decoding a literal in a position of a register operand. Give
1854 // it special handling in the caller, decodeImmOperands(), instead
1855 // of quietly allowing it here.
1856 break;
1857 }
1858
1861 : MCOperand::createImm(Val);
1862}
1863
1865 assert(STI.hasFeature(AMDGPU::Feature64BitLiterals));
1866
1867 if (!HasLiteral) {
1868 if (Bytes.size() < 8) {
1869 return errOperand(0, "cannot read literal64, inst bytes left " +
1870 Twine(Bytes.size()));
1871 }
1872 HasLiteral = true;
1873 Literal = eatBytes<uint64_t>(Bytes);
1874 }
1875
1876 bool UseLit64 = Hi_32(Literal) == 0;
1877
1878 UseLit64 |= AMDGPU::isInlinableLiteral64(
1879 Literal, STI.hasFeature(AMDGPU::FeatureInv2PiInlineImm));
1880
1882 LitModifier::Lit64, Literal, getContext()))
1883 : MCOperand::createImm(Literal);
1884}
1885
1887 using namespace AMDGPU::EncValues;
1888
1889 assert(Imm >= INLINE_INTEGER_C_MIN && Imm <= INLINE_INTEGER_C_MAX);
1890 return MCOperand::createImm((Imm <= INLINE_INTEGER_C_POSITIVE_MAX) ?
1891 (static_cast<int64_t>(Imm) - INLINE_INTEGER_C_MIN) :
1892 (INLINE_INTEGER_C_POSITIVE_MAX - static_cast<int64_t>(Imm)));
1893 // Cast prevents negative overflow.
1894}
1895
1896static int64_t getInlineImmVal32(unsigned Imm) {
1897 switch (Imm) {
1898 case 240:
1899 return llvm::bit_cast<uint32_t>(0.5f);
1900 case 241:
1901 return llvm::bit_cast<uint32_t>(-0.5f);
1902 case 242:
1903 return llvm::bit_cast<uint32_t>(1.0f);
1904 case 243:
1905 return llvm::bit_cast<uint32_t>(-1.0f);
1906 case 244:
1907 return llvm::bit_cast<uint32_t>(2.0f);
1908 case 245:
1909 return llvm::bit_cast<uint32_t>(-2.0f);
1910 case 246:
1911 return llvm::bit_cast<uint32_t>(4.0f);
1912 case 247:
1913 return llvm::bit_cast<uint32_t>(-4.0f);
1914 case 248: // 1 / (2 * PI)
1915 return 0x3e22f983;
1916 default:
1917 llvm_unreachable("invalid fp inline imm");
1918 }
1919}
1920
1921static int64_t getInlineImmVal64(unsigned Imm) {
1922 switch (Imm) {
1923 case 240:
1924 return llvm::bit_cast<uint64_t>(0.5);
1925 case 241:
1926 return llvm::bit_cast<uint64_t>(-0.5);
1927 case 242:
1928 return llvm::bit_cast<uint64_t>(1.0);
1929 case 243:
1930 return llvm::bit_cast<uint64_t>(-1.0);
1931 case 244:
1932 return llvm::bit_cast<uint64_t>(2.0);
1933 case 245:
1934 return llvm::bit_cast<uint64_t>(-2.0);
1935 case 246:
1936 return llvm::bit_cast<uint64_t>(4.0);
1937 case 247:
1938 return llvm::bit_cast<uint64_t>(-4.0);
1939 case 248: // 1 / (2 * PI)
1940 return 0x3fc45f306dc9c882;
1941 default:
1942 llvm_unreachable("invalid fp inline imm");
1943 }
1944}
1945
1946static int64_t getInlineImmValF16(unsigned Imm) {
1947 switch (Imm) {
1948 case 240:
1949 return 0x3800;
1950 case 241:
1951 return 0xB800;
1952 case 242:
1953 return 0x3C00;
1954 case 243:
1955 return 0xBC00;
1956 case 244:
1957 return 0x4000;
1958 case 245:
1959 return 0xC000;
1960 case 246:
1961 return 0x4400;
1962 case 247:
1963 return 0xC400;
1964 case 248: // 1 / (2 * PI)
1965 return 0x3118;
1966 default:
1967 llvm_unreachable("invalid fp inline imm");
1968 }
1969}
1970
1971static int64_t getInlineImmValBF16(unsigned Imm) {
1972 switch (Imm) {
1973 case 240:
1974 return 0x3F00;
1975 case 241:
1976 return 0xBF00;
1977 case 242:
1978 return 0x3F80;
1979 case 243:
1980 return 0xBF80;
1981 case 244:
1982 return 0x4000;
1983 case 245:
1984 return 0xC000;
1985 case 246:
1986 return 0x4080;
1987 case 247:
1988 return 0xC080;
1989 case 248: // 1 / (2 * PI)
1990 return 0x3E22;
1991 default:
1992 llvm_unreachable("invalid fp inline imm");
1993 }
1994}
1995
1996unsigned AMDGPUDisassembler::getVgprClassId(unsigned Width) const {
1997 using namespace AMDGPU;
1998
1999 switch (Width) {
2000 case 16:
2001 case 32:
2002 return VGPR_32RegClassID;
2003 case 64:
2004 return VReg_64RegClassID;
2005 case 96:
2006 return VReg_96RegClassID;
2007 case 128:
2008 return VReg_128RegClassID;
2009 case 160:
2010 return VReg_160RegClassID;
2011 case 192:
2012 return VReg_192RegClassID;
2013 case 256:
2014 return VReg_256RegClassID;
2015 case 288:
2016 return VReg_288RegClassID;
2017 case 320:
2018 return VReg_320RegClassID;
2019 case 352:
2020 return VReg_352RegClassID;
2021 case 384:
2022 return VReg_384RegClassID;
2023 case 512:
2024 return VReg_512RegClassID;
2025 case 1024:
2026 return VReg_1024RegClassID;
2027 }
2028 llvm_unreachable("Invalid register width!");
2029}
2030
2031unsigned AMDGPUDisassembler::getAgprClassId(unsigned Width) const {
2032 using namespace AMDGPU;
2033
2034 switch (Width) {
2035 case 16:
2036 case 32:
2037 return AGPR_32RegClassID;
2038 case 64:
2039 return AReg_64RegClassID;
2040 case 96:
2041 return AReg_96RegClassID;
2042 case 128:
2043 return AReg_128RegClassID;
2044 case 160:
2045 return AReg_160RegClassID;
2046 case 256:
2047 return AReg_256RegClassID;
2048 case 288:
2049 return AReg_288RegClassID;
2050 case 320:
2051 return AReg_320RegClassID;
2052 case 352:
2053 return AReg_352RegClassID;
2054 case 384:
2055 return AReg_384RegClassID;
2056 case 512:
2057 return AReg_512RegClassID;
2058 case 1024:
2059 return AReg_1024RegClassID;
2060 }
2061 llvm_unreachable("Invalid register width!");
2062}
2063
2064std::optional<unsigned>
2066 using namespace AMDGPU;
2067
2068 switch (Width) {
2069 case 16:
2070 case 32:
2071 return SGPR_32RegClassID;
2072 case 64:
2073 return SGPR_64RegClassID;
2074 case 96:
2075 return SGPR_96RegClassID;
2076 case 128:
2077 return SGPR_128RegClassID;
2078 case 160:
2079 return SGPR_160RegClassID;
2080 case 256:
2081 return SGPR_256RegClassID;
2082 case 288:
2083 return SGPR_288RegClassID;
2084 case 320:
2085 return SGPR_320RegClassID;
2086 case 352:
2087 return SGPR_352RegClassID;
2088 case 384:
2089 return SGPR_384RegClassID;
2090 case 512:
2091 return SGPR_512RegClassID;
2092 }
2093 return std::nullopt;
2094}
2095
2096std::optional<unsigned>
2098 using namespace AMDGPU;
2099
2100 switch (Width) {
2101 case 16:
2102 case 32:
2103 return TTMP_32RegClassID;
2104 case 64:
2105 return TTMP_64RegClassID;
2106 case 128:
2107 return TTMP_128RegClassID;
2108 case 256:
2109 return TTMP_256RegClassID;
2110 case 288:
2111 return TTMP_288RegClassID;
2112 case 320:
2113 return TTMP_320RegClassID;
2114 case 352:
2115 return TTMP_352RegClassID;
2116 case 384:
2117 return TTMP_384RegClassID;
2118 case 512:
2119 return TTMP_512RegClassID;
2120 }
2121 return std::nullopt;
2122}
2123
2124int AMDGPUDisassembler::getTTmpIdx(unsigned Val) const {
2125 using namespace AMDGPU::EncValues;
2126
2127 unsigned TTmpMin = isGFX9Plus() ? TTMP_GFX9PLUS_MIN : TTMP_VI_MIN;
2128 unsigned TTmpMax = isGFX9Plus() ? TTMP_GFX9PLUS_MAX : TTMP_VI_MAX;
2129
2130 return (TTmpMin <= Val && Val <= TTmpMax)? Val - TTmpMin : -1;
2131}
2132
2134 unsigned Val) const {
2135 using namespace AMDGPU::EncValues;
2136
2137 assert(Val < 1024); // enum10
2138
2139 bool IsAGPR = Val & 512;
2140 Val &= 511;
2141
2142 if (VGPR_MIN <= Val && Val <= VGPR_MAX) {
2143 return createRegOperand(IsAGPR ? getAgprClassId(Width)
2144 : getVgprClassId(Width), Val - VGPR_MIN);
2145 }
2146 return decodeNonVGPRSrcOp(Inst, Width, Val & 0xFF);
2147}
2148
2150 unsigned Width,
2151 unsigned Val) const {
2152 // Cases when Val{8} is 1 (vgpr, agpr or true 16 vgpr) should have been
2153 // decoded earlier.
2154 assert(Val < (1 << 8) && "9-bit Src encoding when Val{8} is 0");
2155 using namespace AMDGPU::EncValues;
2156
2157 // Not every operand width has a supported non-VGPR source encoding.
2158 // Selecting an unsupported SGPR, ttmp, or special register is malformed.
2159 auto UnsupportedWidth = [&]() {
2160 return errOperand(Val, "unsupported " + Twine(Width) +
2161 "-bit non-VGPR operand encoding " + Twine(Val));
2162 };
2163
2164 if (Val <= SGPR_MAX) {
2165 // "SGPR_MIN <= Val" is always true and causes compilation warning.
2166 static_assert(SGPR_MIN == 0);
2167 std::optional<unsigned> ClassId = getSgprClassId(Width);
2168 if (!ClassId)
2169 return UnsupportedWidth();
2170 return createSRegOperand(*ClassId, Val - SGPR_MIN);
2171 }
2172
2173 int TTmpIdx = getTTmpIdx(Val);
2174 if (TTmpIdx >= 0) {
2175 std::optional<unsigned> ClassId = getTtmpClassId(Width);
2176 if (!ClassId)
2177 return UnsupportedWidth();
2178 return createSRegOperand(*ClassId, TTmpIdx);
2179 }
2180
2181 if ((INLINE_INTEGER_C_MIN <= Val && Val <= INLINE_INTEGER_C_MAX) ||
2182 (INLINE_FLOATING_C_MIN <= Val && Val <= INLINE_FLOATING_C_MAX) ||
2183 Val == LITERAL_CONST)
2184 return MCOperand::createImm(Val);
2185
2186 if (Val == LITERAL64_CONST && STI.hasFeature(AMDGPU::Feature64BitLiterals)) {
2187 // Only VOP1, VOP2, VOPC, SOP1, SOP2 and SOPC may encode a 64-bit literal.
2188 // VOP3, VOP3P and VOPD have to use a 32-bit one.
2189 if (SIInstrFlags::isVOP3Like(*MCII, Inst) ||
2190 AMDGPU::isVOPD(Inst.getOpcode())) {
2191 return errOperand(Val,
2192 "64-bit literal is not supported by this instruction");
2193 }
2194 return decodeLiteral64Constant();
2195 }
2196
2197 switch (Width) {
2198 case 32:
2199 case 16:
2200 return decodeSpecialReg32(Val);
2201 case 64:
2202 return decodeSpecialReg64(Val);
2203 case 96:
2204 case 128:
2205 case 256:
2206 case 512:
2207 return decodeSpecialReg96Plus(Val);
2208 default:
2209 return UnsupportedWidth();
2210 }
2211}
2212
2213// Bit 0 of DstY isn't stored in the instruction, because it's always the
2214// opposite of bit 0 of DstX.
2216 unsigned Val) const {
2217 int VDstXInd =
2218 AMDGPU::getNamedOperandIdx(Inst.getOpcode(), AMDGPU::OpName::vdstX);
2219 assert(VDstXInd != -1);
2220 assert(Inst.getOperand(VDstXInd).isReg());
2221 unsigned XDstReg = MRI.getEncodingValue(Inst.getOperand(VDstXInd).getReg());
2222 Val |= ~XDstReg & 1;
2223 return createRegOperand(getVgprClassId(32), Val);
2224}
2225
2227 using namespace AMDGPU;
2228
2229 switch (Val) {
2230 // clang-format off
2231 case 102: return createRegOperand(FLAT_SCR_LO);
2232 case 103: return createRegOperand(FLAT_SCR_HI);
2233 case 104: return createRegOperand(XNACK_MASK_LO);
2234 case 105: return createRegOperand(XNACK_MASK_HI);
2235 case 106: return createRegOperand(VCC_LO);
2236 case 107: return createRegOperand(VCC_HI);
2237 case 108: return createRegOperand(TBA_LO);
2238 case 109: return createRegOperand(TBA_HI);
2239 case 110: return createRegOperand(TMA_LO);
2240 case 111: return createRegOperand(TMA_HI);
2241 case 124:
2242 return isGFX11Plus() ? createRegOperand(SGPR_NULL) : createRegOperand(M0);
2243 case 125:
2244 return isGFX11Plus() ? createRegOperand(M0) : createRegOperand(SGPR_NULL);
2245 case 126: return createRegOperand(EXEC_LO);
2246 case 127: return createRegOperand(EXEC_HI);
2247 case 230: return createRegOperand(SRC_FLAT_SCRATCH_BASE_LO);
2248 case 231: return createRegOperand(SRC_FLAT_SCRATCH_BASE_HI);
2249 case 235: return createRegOperand(SRC_SHARED_BASE_LO);
2250 case 236: return createRegOperand(SRC_SHARED_LIMIT_LO);
2251 case 237:
2253 return createRegOperand(SRC_PRIVATE_BASE_LO);
2254 break;
2255 case 238:
2257 return createRegOperand(SRC_PRIVATE_LIMIT_LO);
2258 break;
2259 case 239:
2261 return createRegOperand(SRC_POPS_EXITING_WAVE_ID);
2262 break;
2263 case 251:
2264 if (!isGFX11Plus())
2265 return createRegOperand(SRC_VCCZ);
2266 break;
2267 case 252:
2268 if (!isGFX11Plus())
2269 return createRegOperand(SRC_EXECZ);
2270 break;
2271 case 253: return createRegOperand(SRC_SCC);
2272 case 254: return createRegOperand(LDS_DIRECT);
2273 default: break;
2274 // clang-format on
2275 }
2276 return errOperand(Val, "unknown operand encoding " + Twine(Val));
2277}
2278
2280 using namespace AMDGPU;
2281
2282 switch (Val) {
2283 case 102: return createRegOperand(FLAT_SCR);
2284 case 104: return createRegOperand(XNACK_MASK);
2285 case 106: return createRegOperand(VCC);
2286 case 108: return createRegOperand(TBA);
2287 case 110: return createRegOperand(TMA);
2288 case 124:
2289 if (isGFX11Plus())
2290 return createRegOperand(SGPR_NULL);
2291 break;
2292 case 125:
2293 if (!isGFX11Plus())
2294 return createRegOperand(SGPR_NULL);
2295 break;
2296 case 126: return createRegOperand(EXEC);
2297 case 230: return createRegOperand(SRC_FLAT_SCRATCH_BASE_LO);
2298 case 235: return createRegOperand(SRC_SHARED_BASE);
2299 case 236: return createRegOperand(SRC_SHARED_LIMIT);
2300 case 237:
2302 return createRegOperand(SRC_PRIVATE_BASE);
2303 break;
2304 case 238:
2306 return createRegOperand(SRC_PRIVATE_LIMIT);
2307 break;
2308 case 239:
2310 return createRegOperand(SRC_POPS_EXITING_WAVE_ID);
2311 break;
2312 case 251:
2313 if (!isGFX11Plus())
2314 return createRegOperand(SRC_VCCZ);
2315 break;
2316 case 252:
2317 if (!isGFX11Plus())
2318 return createRegOperand(SRC_EXECZ);
2319 break;
2320 case 253: return createRegOperand(SRC_SCC);
2321 default: break;
2322 }
2323 return errOperand(Val, "unknown operand encoding " + Twine(Val));
2324}
2325
2327 using namespace AMDGPU;
2328
2329 switch (Val) {
2330 case 124:
2331 if (isGFX11Plus())
2332 return createRegOperand(SGPR_NULL);
2333 break;
2334 case 125:
2335 if (!isGFX11Plus())
2336 return createRegOperand(SGPR_NULL);
2337 break;
2338 default:
2339 break;
2340 }
2341 return errOperand(Val, "unknown operand encoding " + Twine(Val));
2342}
2343
2345 const unsigned Val) const {
2346 using namespace AMDGPU::SDWA;
2347 using namespace AMDGPU::EncValues;
2348
2349 if (STI.hasFeature(AMDGPU::FeatureGFX9) ||
2350 STI.hasFeature(AMDGPU::FeatureGFX10)) {
2351 // XXX: cast to int is needed to avoid stupid warning:
2352 // compare with unsigned is always true
2353 if (int(SDWA9EncValues::SRC_VGPR_MIN) <= int(Val) &&
2354 Val <= SDWA9EncValues::SRC_VGPR_MAX) {
2355 return createRegOperand(getVgprClassId(Width),
2356 Val - SDWA9EncValues::SRC_VGPR_MIN);
2357 }
2358 if (SDWA9EncValues::SRC_SGPR_MIN <= Val &&
2359 Val <= (isGFX10Plus() ? SDWA9EncValues::SRC_SGPR_MAX_GFX10
2360 : SDWA9EncValues::SRC_SGPR_MAX_SI)) {
2361 return createSRegOperand(*getSgprClassId(Width),
2362 Val - SDWA9EncValues::SRC_SGPR_MIN);
2363 }
2364 if (SDWA9EncValues::SRC_TTMP_MIN <= Val &&
2365 Val <= SDWA9EncValues::SRC_TTMP_MAX) {
2366 return createSRegOperand(*getTtmpClassId(Width),
2367 Val - SDWA9EncValues::SRC_TTMP_MIN);
2368 }
2369
2370 const unsigned SVal = Val - SDWA9EncValues::SRC_SGPR_MIN;
2371
2372 if ((INLINE_INTEGER_C_MIN <= SVal && SVal <= INLINE_INTEGER_C_MAX) ||
2373 (INLINE_FLOATING_C_MIN <= SVal && SVal <= INLINE_FLOATING_C_MAX))
2374 return MCOperand::createImm(SVal);
2375
2376 return decodeSpecialReg32(SVal);
2377 }
2378 if (STI.hasFeature(AMDGPU::FeatureVolcanicIslands))
2379 return createRegOperand(getVgprClassId(Width), Val);
2380 llvm_unreachable("unsupported target");
2381}
2382
2384 return decodeSDWASrc(16, Val);
2385}
2386
2388 return decodeSDWASrc(32, Val);
2389}
2390
2392 using namespace AMDGPU::SDWA;
2393
2394 assert((STI.hasFeature(AMDGPU::FeatureGFX9) ||
2395 STI.hasFeature(AMDGPU::FeatureGFX10)) &&
2396 "SDWAVopcDst should be present only on GFX9+");
2397
2398 bool IsWave32 = STI.hasFeature(AMDGPU::FeatureWavefrontSize32);
2399
2400 if (Val & SDWA9EncValues::VOPC_DST_VCC_MASK) {
2401 Val &= SDWA9EncValues::VOPC_DST_SGPR_MASK;
2402
2403 int TTmpIdx = getTTmpIdx(Val);
2404 if (TTmpIdx >= 0)
2405 return createSRegOperand(*getTtmpClassId(IsWave32 ? 32 : 64), TTmpIdx);
2406 if (Val > SGPR_MAX) {
2407 return IsWave32 ? decodeSpecialReg32(Val) : decodeSpecialReg64(Val);
2408 }
2409 return createSRegOperand(*getSgprClassId(IsWave32 ? 32 : 64), Val);
2410 }
2411 return createRegOperand(IsWave32 ? AMDGPU::VCC_LO : AMDGPU::VCC);
2412}
2413
2415 unsigned Val) const {
2416 return STI.hasFeature(AMDGPU::FeatureWavefrontSize32)
2417 ? decodeSrcOp(Inst, 32, Val)
2418 : decodeSrcOp(Inst, 64, Val);
2419}
2420
2422 unsigned Val) const {
2423 using namespace AMDGPU::EncValues;
2424 constexpr unsigned M0Encoding = 125;
2425 bool IsValidBarrier =
2426 Val == M0Encoding ||
2427 (INLINE_INTEGER_C_MIN <= Val && Val < INLINE_INTEGER_C_MIN + 32) ||
2428 (INLINE_INTEGER_C_POSITIVE_MAX < Val &&
2429 Val <= INLINE_INTEGER_C_POSITIVE_MAX + 4);
2430 if (!IsValidBarrier)
2431 return MCOperand();
2432 return decodeSrcOp(Inst, 32, Val);
2433}
2434
2437 return MCOperand();
2438 return MCOperand::createImm(Val);
2439}
2440
2442 using VersionField = AMDGPU::EncodingField<7, 0>;
2443 using W64Bit = AMDGPU::EncodingBit<13>;
2444 using W32Bit = AMDGPU::EncodingBit<14>;
2445 using MDPBit = AMDGPU::EncodingBit<15>;
2447
2448 auto [Version, W64, W32, MDP] = Encoding::decode(Imm);
2449
2450 // Decode into a plain immediate if any unused bits are raised.
2451 if (Encoding::encode(Version, W64, W32, MDP) != Imm)
2452 return MCOperand::createImm(Imm);
2453
2454 const auto &Versions = AMDGPU::UCVersion::getGFXVersions();
2455 const auto *I = find_if(
2456 Versions, [Version = Version](const AMDGPU::UCVersion::GFXVersion &V) {
2457 return V.Code == Version;
2458 });
2459 MCContext &Ctx = getContext();
2460 const MCExpr *E;
2461 if (I == Versions.end())
2463 else
2464 E = MCSymbolRefExpr::create(Ctx.getOrCreateSymbol(I->Symbol), Ctx);
2465
2466 if (W64)
2467 E = MCBinaryExpr::createOr(E, UCVersionW64Expr, Ctx);
2468 if (W32)
2469 E = MCBinaryExpr::createOr(E, UCVersionW32Expr, Ctx);
2470 if (MDP)
2471 E = MCBinaryExpr::createOr(E, UCVersionMDPExpr, Ctx);
2472
2473 return MCOperand::createExpr(E);
2474}
2475
2477 return STI.hasFeature(AMDGPU::FeatureVolcanicIslands);
2478}
2479
2481
2483 return STI.hasFeature(AMDGPU::FeatureGFX90AInsts);
2484}
2485
2487
2489
2493
2495 return STI.hasFeature(AMDGPU::FeatureGFX11);
2496}
2497
2501
2503 return STI.hasFeature(AMDGPU::FeatureGFX11_7Insts);
2504}
2505
2507 return STI.hasFeature(AMDGPU::FeatureGFX12);
2508}
2509
2513
2515
2519
2521
2525
2527 return STI.hasFeature(AMDGPU::FeatureArchitectedFlatScratch);
2528}
2529
2533//===----------------------------------------------------------------------===//
2534// AMDGPU specific symbol handling
2535//===----------------------------------------------------------------------===//
2536
2537/// Print a string describing the reserved bit range specified by Mask with
2538/// offset BaseBytes for use in error comments. Mask is a single continuous
2539/// range of 1s surrounded by zeros. The format here is meant to align with the
2540/// tables that describe these bits in llvm.org/docs/AMDGPUUsage.html.
2541static SmallString<32> getBitRangeFromMask(uint32_t Mask, unsigned BaseBytes) {
2542 SmallString<32> Result;
2543 raw_svector_ostream S(Result);
2544
2545 int TrailingZeros = llvm::countr_zero(Mask);
2546 int PopCount = llvm::popcount(Mask);
2547
2548 if (PopCount == 1) {
2549 S << "bit (" << (TrailingZeros + BaseBytes * CHAR_BIT) << ')';
2550 } else {
2551 S << "bits in range ("
2552 << (TrailingZeros + PopCount - 1 + BaseBytes * CHAR_BIT) << ':'
2553 << (TrailingZeros + BaseBytes * CHAR_BIT) << ')';
2554 }
2555
2556 return Result;
2557}
2558
2559#define GET_FIELD(MASK) (AMDHSA_BITS_GET(FourByteBuffer, MASK))
2560#define PRINT_DIRECTIVE(DIRECTIVE, MASK) \
2561 do { \
2562 KdStream << Indent << DIRECTIVE " " << GET_FIELD(MASK) << '\n'; \
2563 } while (0)
2564#define PRINT_PSEUDO_DIRECTIVE_COMMENT(DIRECTIVE, MASK) \
2565 do { \
2566 KdStream << Indent << MAI.getCommentString() << ' ' << DIRECTIVE " " \
2567 << GET_FIELD(MASK) << '\n'; \
2568 } while (0)
2569
2570#define CHECK_RESERVED_BITS_IMPL(MASK, DESC, MSG) \
2571 do { \
2572 if (FourByteBuffer & (MASK)) { \
2573 return createStringError(std::errc::invalid_argument, \
2574 "kernel descriptor " DESC \
2575 " reserved %s set" MSG, \
2576 getBitRangeFromMask((MASK), 0).c_str()); \
2577 } \
2578 } while (0)
2579
2580#define CHECK_RESERVED_BITS(MASK) CHECK_RESERVED_BITS_IMPL(MASK, #MASK, "")
2581#define CHECK_RESERVED_BITS_MSG(MASK, MSG) \
2582 CHECK_RESERVED_BITS_IMPL(MASK, #MASK, ", " MSG)
2583#define CHECK_RESERVED_BITS_DESC(MASK, DESC) \
2584 CHECK_RESERVED_BITS_IMPL(MASK, DESC, "")
2585#define CHECK_RESERVED_BITS_DESC_MSG(MASK, DESC, MSG) \
2586 CHECK_RESERVED_BITS_IMPL(MASK, DESC, ", " MSG)
2587
2588// NOLINTNEXTLINE(readability-identifier-naming)
2590 uint32_t FourByteBuffer, raw_string_ostream &KdStream) const {
2591 using namespace amdhsa;
2592 StringRef Indent = "\t";
2593
2594 // We cannot accurately backward compute #VGPRs used from
2595 // GRANULATED_WORKITEM_VGPR_COUNT. But we are concerned with getting the same
2596 // value of GRANULATED_WORKITEM_VGPR_COUNT in the reassembled binary. So we
2597 // simply calculate the inverse of what the assembler does.
2598
2599 uint32_t GranulatedWorkitemVGPRCount =
2600 GET_FIELD(COMPUTE_PGM_RSRC1_GRANULATED_WORKITEM_VGPR_COUNT);
2601
2602 uint32_t NextFreeVGPR =
2603 (GranulatedWorkitemVGPRCount + 1) *
2604 AMDGPU::IsaInfo::getVGPREncodingGranule(STI, EnableWavefrontSize32);
2605
2606 KdStream << Indent << ".amdhsa_next_free_vgpr " << NextFreeVGPR << '\n';
2607
2608 // We cannot backward compute values used to calculate
2609 // GRANULATED_WAVEFRONT_SGPR_COUNT. Hence the original values for following
2610 // directives can't be computed:
2611 // .amdhsa_reserve_vcc
2612 // .amdhsa_reserve_flat_scratch
2613 // .amdhsa_reserve_xnack_mask
2614 // They take their respective default values if not specified in the assembly.
2615 //
2616 // GRANULATED_WAVEFRONT_SGPR_COUNT
2617 // = f(NEXT_FREE_SGPR + VCC + FLAT_SCRATCH + XNACK_MASK)
2618 //
2619 // We compute the inverse as though all directives apart from NEXT_FREE_SGPR
2620 // are set to 0. So while disassembling we consider that:
2621 //
2622 // GRANULATED_WAVEFRONT_SGPR_COUNT
2623 // = f(NEXT_FREE_SGPR + 0 + 0 + 0)
2624 //
2625 // The disassembler cannot recover the original values of those 3 directives.
2626
2627 uint32_t GranulatedWavefrontSGPRCount =
2628 GET_FIELD(COMPUTE_PGM_RSRC1_GRANULATED_WAVEFRONT_SGPR_COUNT);
2629
2630 if (isGFX10Plus())
2631 CHECK_RESERVED_BITS_MSG(COMPUTE_PGM_RSRC1_GRANULATED_WAVEFRONT_SGPR_COUNT,
2632 "must be zero on gfx10+");
2633
2634 uint32_t NextFreeSGPR = (GranulatedWavefrontSGPRCount + 1) *
2636
2637 KdStream << Indent << ".amdhsa_reserve_vcc " << 0 << '\n';
2639 KdStream << Indent << ".amdhsa_reserve_flat_scratch " << 0 << '\n';
2640 // Only print the directive on xnack-supporting targets (matching the
2641 // asmprinter), unless the binary erronously set xnack on an unsupported
2642 // target
2643 bool ReservedXnackMask = TargetID.isXnackOnOrAny();
2644 if (STI.hasFeature(AMDGPU::FeatureSupportsXNACK) || ReservedXnackMask) {
2645 KdStream << Indent << ".amdhsa_reserve_xnack_mask " << ReservedXnackMask
2646 << '\n';
2647 }
2648 KdStream << Indent << ".amdhsa_next_free_sgpr " << NextFreeSGPR << "\n";
2649
2650 CHECK_RESERVED_BITS(COMPUTE_PGM_RSRC1_PRIORITY);
2651
2652 PRINT_DIRECTIVE(".amdhsa_float_round_mode_32",
2653 COMPUTE_PGM_RSRC1_FLOAT_ROUND_MODE_32);
2654 PRINT_DIRECTIVE(".amdhsa_float_round_mode_16_64",
2655 COMPUTE_PGM_RSRC1_FLOAT_ROUND_MODE_16_64);
2656 PRINT_DIRECTIVE(".amdhsa_float_denorm_mode_32",
2657 COMPUTE_PGM_RSRC1_FLOAT_DENORM_MODE_32);
2658 PRINT_DIRECTIVE(".amdhsa_float_denorm_mode_16_64",
2659 COMPUTE_PGM_RSRC1_FLOAT_DENORM_MODE_16_64);
2660
2661 CHECK_RESERVED_BITS(COMPUTE_PGM_RSRC1_PRIV);
2662
2663 if (STI.hasFeature(AMDGPU::FeatureDX10ClampAndIEEEMode))
2664 PRINT_DIRECTIVE(".amdhsa_dx10_clamp",
2665 COMPUTE_PGM_RSRC1_GFX6_GFX11_ENABLE_DX10_CLAMP);
2666
2667 CHECK_RESERVED_BITS(COMPUTE_PGM_RSRC1_DEBUG_MODE);
2668
2669 if (STI.hasFeature(AMDGPU::FeatureDX10ClampAndIEEEMode))
2670 PRINT_DIRECTIVE(".amdhsa_ieee_mode",
2671 COMPUTE_PGM_RSRC1_GFX6_GFX11_ENABLE_IEEE_MODE);
2672
2673 CHECK_RESERVED_BITS(COMPUTE_PGM_RSRC1_BULKY);
2674 CHECK_RESERVED_BITS(COMPUTE_PGM_RSRC1_CDBG_USER);
2675
2676 // Bits [26].
2677 if (isGFX9Plus()) {
2678 PRINT_DIRECTIVE(".amdhsa_fp16_overflow", COMPUTE_PGM_RSRC1_GFX9_PLUS_FP16_OVFL);
2679 } else {
2680 CHECK_RESERVED_BITS_DESC_MSG(COMPUTE_PGM_RSRC1_GFX6_GFX8_RESERVED0,
2681 "COMPUTE_PGM_RSRC1", "must be zero pre-gfx9");
2682 }
2683
2684 // Bits [27].
2685 if (isGFX1250Plus()) {
2686 PRINT_PSEUDO_DIRECTIVE_COMMENT("FLAT_SCRATCH_IS_NV",
2687 COMPUTE_PGM_RSRC1_GFX125_FLAT_SCRATCH_IS_NV);
2688 } else {
2689 CHECK_RESERVED_BITS_DESC(COMPUTE_PGM_RSRC1_GFX6_GFX120_RESERVED1,
2690 "COMPUTE_PGM_RSRC1");
2691 }
2692
2693 // Bits [28].
2694 CHECK_RESERVED_BITS_DESC(COMPUTE_PGM_RSRC1_RESERVED2, "COMPUTE_PGM_RSRC1");
2695
2696 // Bits [29-31].
2697 if (isGFX10Plus()) {
2698 // WGP_MODE is not available on GFX1250.
2699 if (!isGFX1250Plus()) {
2700 PRINT_DIRECTIVE(".amdhsa_workgroup_processor_mode",
2701 COMPUTE_PGM_RSRC1_GFX10_PLUS_WGP_MODE);
2702 }
2703 PRINT_DIRECTIVE(".amdhsa_memory_ordered", COMPUTE_PGM_RSRC1_GFX10_PLUS_MEM_ORDERED);
2704 PRINT_DIRECTIVE(".amdhsa_forward_progress", COMPUTE_PGM_RSRC1_GFX10_PLUS_FWD_PROGRESS);
2705 } else {
2706 CHECK_RESERVED_BITS_DESC(COMPUTE_PGM_RSRC1_GFX6_GFX9_RESERVED3,
2707 "COMPUTE_PGM_RSRC1");
2708 }
2709
2710 if (isGFX12Plus())
2711 PRINT_DIRECTIVE(".amdhsa_round_robin_scheduling",
2712 COMPUTE_PGM_RSRC1_GFX12_PLUS_ENABLE_WG_RR_EN);
2713
2714 return true;
2715}
2716
2717// NOLINTNEXTLINE(readability-identifier-naming)
2719 uint32_t FourByteBuffer, raw_string_ostream &KdStream) const {
2720 using namespace amdhsa;
2721 StringRef Indent = "\t";
2723 PRINT_DIRECTIVE(".amdhsa_enable_private_segment",
2724 COMPUTE_PGM_RSRC2_ENABLE_PRIVATE_SEGMENT);
2725 else
2726 PRINT_DIRECTIVE(".amdhsa_system_sgpr_private_segment_wavefront_offset",
2727 COMPUTE_PGM_RSRC2_ENABLE_PRIVATE_SEGMENT);
2728 PRINT_DIRECTIVE(".amdhsa_system_sgpr_workgroup_id_x",
2729 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_ID_X);
2730 PRINT_DIRECTIVE(".amdhsa_system_sgpr_workgroup_id_y",
2731 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_ID_Y);
2732 PRINT_DIRECTIVE(".amdhsa_system_sgpr_workgroup_id_z",
2733 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_ID_Z);
2734 PRINT_DIRECTIVE(".amdhsa_system_sgpr_workgroup_info",
2735 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_INFO);
2736 PRINT_DIRECTIVE(".amdhsa_system_vgpr_workitem_id",
2737 COMPUTE_PGM_RSRC2_ENABLE_VGPR_WORKITEM_ID);
2738
2739 CHECK_RESERVED_BITS(COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_ADDRESS_WATCH);
2740 CHECK_RESERVED_BITS(COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_MEMORY);
2741 CHECK_RESERVED_BITS(COMPUTE_PGM_RSRC2_GRANULATED_LDS_SIZE);
2742
2744 ".amdhsa_exception_fp_ieee_invalid_op",
2745 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_INVALID_OPERATION);
2746 PRINT_DIRECTIVE(".amdhsa_exception_fp_denorm_src",
2747 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_FP_DENORMAL_SOURCE);
2749 ".amdhsa_exception_fp_ieee_div_zero",
2750 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_DIVISION_BY_ZERO);
2751 PRINT_DIRECTIVE(".amdhsa_exception_fp_ieee_overflow",
2752 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_OVERFLOW);
2753 PRINT_DIRECTIVE(".amdhsa_exception_fp_ieee_underflow",
2754 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_UNDERFLOW);
2755 PRINT_DIRECTIVE(".amdhsa_exception_fp_ieee_inexact",
2756 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_INEXACT);
2757 PRINT_DIRECTIVE(".amdhsa_exception_int_div_zero",
2758 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_INT_DIVIDE_BY_ZERO);
2759
2760 CHECK_RESERVED_BITS_DESC(COMPUTE_PGM_RSRC2_RESERVED0, "COMPUTE_PGM_RSRC2");
2761
2762 return true;
2763}
2764
2765// NOLINTNEXTLINE(readability-identifier-naming)
2767 uint32_t FourByteBuffer, raw_string_ostream &KdStream) const {
2768 using namespace amdhsa;
2769 StringRef Indent = "\t";
2770 if (isGFX90A()) {
2771 KdStream << Indent << ".amdhsa_accum_offset "
2772 << (GET_FIELD(COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET) + 1) * 4
2773 << '\n';
2774
2775 PRINT_DIRECTIVE(".amdhsa_tg_split", COMPUTE_PGM_RSRC3_GFX90A_TG_SPLIT);
2776
2777 CHECK_RESERVED_BITS_DESC_MSG(COMPUTE_PGM_RSRC3_GFX90A_RESERVED0,
2778 "COMPUTE_PGM_RSRC3", "must be zero on gfx90a");
2779 CHECK_RESERVED_BITS_DESC_MSG(COMPUTE_PGM_RSRC3_GFX90A_RESERVED1,
2780 "COMPUTE_PGM_RSRC3", "must be zero on gfx90a");
2781 } else if (isGFX10Plus()) {
2782 // Bits [0-3].
2783 if (!isGFX12Plus()) {
2784 if (!EnableWavefrontSize32 || !*EnableWavefrontSize32) {
2785 PRINT_DIRECTIVE(".amdhsa_shared_vgpr_count",
2786 COMPUTE_PGM_RSRC3_GFX10_GFX11_SHARED_VGPR_COUNT);
2787 } else {
2789 "SHARED_VGPR_COUNT",
2790 COMPUTE_PGM_RSRC3_GFX10_GFX11_SHARED_VGPR_COUNT);
2791 }
2792 } else {
2793 CHECK_RESERVED_BITS_DESC_MSG(COMPUTE_PGM_RSRC3_GFX12_PLUS_RESERVED0,
2794 "COMPUTE_PGM_RSRC3",
2795 "must be zero on gfx12+");
2796 }
2797
2798 // Bits [4-11].
2799 if (isGFX11()) {
2800 PRINT_DIRECTIVE(".amdhsa_inst_pref_size",
2801 COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE);
2802 PRINT_PSEUDO_DIRECTIVE_COMMENT("TRAP_ON_START",
2803 COMPUTE_PGM_RSRC3_GFX11_TRAP_ON_START);
2804 PRINT_PSEUDO_DIRECTIVE_COMMENT("TRAP_ON_END",
2805 COMPUTE_PGM_RSRC3_GFX11_TRAP_ON_END);
2806 } else if (isGFX12Plus()) {
2807 PRINT_DIRECTIVE(".amdhsa_inst_pref_size",
2808 COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE);
2809 } else {
2810 CHECK_RESERVED_BITS_DESC_MSG(COMPUTE_PGM_RSRC3_GFX10_RESERVED1,
2811 "COMPUTE_PGM_RSRC3",
2812 "must be zero on gfx10");
2813 }
2814
2815 // Bits [12].
2816 CHECK_RESERVED_BITS_DESC_MSG(COMPUTE_PGM_RSRC3_GFX10_PLUS_RESERVED2,
2817 "COMPUTE_PGM_RSRC3", "must be zero on gfx10+");
2818
2819 // Bits [13].
2820 if (isGFX12Plus()) {
2822 COMPUTE_PGM_RSRC3_GFX12_PLUS_GLG_EN);
2823 } else {
2824 CHECK_RESERVED_BITS_DESC_MSG(COMPUTE_PGM_RSRC3_GFX10_GFX11_RESERVED3,
2825 "COMPUTE_PGM_RSRC3",
2826 "must be zero on gfx10 or gfx11");
2827 }
2828
2829 // Bits [14-21].
2830 if (isGFX1250Plus()) {
2831 PRINT_DIRECTIVE(".amdhsa_named_barrier_count",
2832 COMPUTE_PGM_RSRC3_GFX125_NAMED_BAR_CNT);
2834 "ENABLE_DYNAMIC_VGPR", COMPUTE_PGM_RSRC3_GFX125_ENABLE_DYNAMIC_VGPR);
2836 COMPUTE_PGM_RSRC3_GFX125_TCP_SPLIT);
2838 "ENABLE_DIDT_THROTTLE",
2839 COMPUTE_PGM_RSRC3_GFX125_ENABLE_DIDT_THROTTLE);
2840 } else {
2841 CHECK_RESERVED_BITS_DESC_MSG(COMPUTE_PGM_RSRC3_GFX10_GFX120_RESERVED4,
2842 "COMPUTE_PGM_RSRC3",
2843 "must be zero on gfx10+");
2844 }
2845
2846 // Bits [22-30].
2847 CHECK_RESERVED_BITS_DESC_MSG(COMPUTE_PGM_RSRC3_GFX10_PLUS_RESERVED5,
2848 "COMPUTE_PGM_RSRC3", "must be zero on gfx10+");
2849
2850 // Bits [31].
2851 if (isGFX11Plus()) {
2853 COMPUTE_PGM_RSRC3_GFX11_PLUS_IMAGE_OP);
2854 } else {
2855 CHECK_RESERVED_BITS_DESC_MSG(COMPUTE_PGM_RSRC3_GFX10_RESERVED6,
2856 "COMPUTE_PGM_RSRC3",
2857 "must be zero on gfx10");
2858 }
2859 } else if (FourByteBuffer) {
2860 return createStringError(
2861 std::errc::invalid_argument,
2862 "kernel descriptor COMPUTE_PGM_RSRC3 must be all zero before gfx9");
2863 }
2864 return true;
2865}
2866#undef PRINT_PSEUDO_DIRECTIVE_COMMENT
2867#undef PRINT_DIRECTIVE
2868#undef GET_FIELD
2869#undef CHECK_RESERVED_BITS_IMPL
2870#undef CHECK_RESERVED_BITS
2871#undef CHECK_RESERVED_BITS_MSG
2872#undef CHECK_RESERVED_BITS_DESC
2873#undef CHECK_RESERVED_BITS_DESC_MSG
2874
2875/// Create an error object to return from onSymbolStart for reserved kernel
2876/// descriptor bits being set.
2877static Error createReservedKDBitsError(uint32_t Mask, unsigned BaseBytes,
2878 const char *Msg = "") {
2879 return createStringError(
2880 std::errc::invalid_argument, "kernel descriptor reserved %s set%s%s",
2881 getBitRangeFromMask(Mask, BaseBytes).c_str(), *Msg ? ", " : "", Msg);
2882}
2883
2884/// Create an error object to return from onSymbolStart for reserved kernel
2885/// descriptor bytes being set.
2886static Error createReservedKDBytesError(unsigned BaseInBytes,
2887 unsigned WidthInBytes) {
2888 // Create an error comment in the same format as the "Kernel Descriptor"
2889 // table here: https://llvm.org/docs/AMDGPUUsage.html#kernel-descriptor .
2890 return createStringError(
2891 std::errc::invalid_argument,
2892 "kernel descriptor reserved bits in range (%u:%u) set",
2893 (BaseInBytes + WidthInBytes) * CHAR_BIT - 1, BaseInBytes * CHAR_BIT);
2894}
2895
2898 raw_string_ostream &KdStream) const {
2899#define PRINT_DIRECTIVE(DIRECTIVE, MASK) \
2900 do { \
2901 KdStream << Indent << DIRECTIVE " " \
2902 << ((TwoByteBuffer & MASK) >> (MASK##_SHIFT)) << '\n'; \
2903 } while (0)
2904
2905 uint16_t TwoByteBuffer = 0;
2906 uint32_t FourByteBuffer = 0;
2907
2908 StringRef ReservedBytes;
2909 StringRef Indent = "\t";
2910
2911 assert(Bytes.size() == 64);
2912 DataExtractor DE(Bytes, /*IsLittleEndian=*/true);
2913
2914 switch (Cursor.tell()) {
2916 FourByteBuffer = DE.getU32(Cursor);
2917 KdStream << Indent << ".amdhsa_group_segment_fixed_size " << FourByteBuffer
2918 << '\n';
2919 return true;
2920
2922 FourByteBuffer = DE.getU32(Cursor);
2923 KdStream << Indent << ".amdhsa_private_segment_fixed_size "
2924 << FourByteBuffer << '\n';
2925 return true;
2926
2928 FourByteBuffer = DE.getU32(Cursor);
2929 KdStream << Indent << ".amdhsa_kernarg_size "
2930 << FourByteBuffer << '\n';
2931 return true;
2932
2934 // 4 reserved bytes, must be 0.
2935 ReservedBytes = DE.getBytes(Cursor, 4);
2936 for (char B : ReservedBytes) {
2937 if (B != 0)
2939 }
2940 return true;
2941
2943 // KERNEL_CODE_ENTRY_BYTE_OFFSET
2944 // So far no directive controls this for Code Object V3, so simply skip for
2945 // disassembly.
2946 DE.skip(Cursor, 8);
2947 return true;
2948
2950 // 20 reserved bytes, must be 0.
2951 ReservedBytes = DE.getBytes(Cursor, 20);
2952 for (char B : ReservedBytes) {
2953 if (B != 0)
2955 }
2956 return true;
2957
2959 FourByteBuffer = DE.getU32(Cursor);
2960 return decodeCOMPUTE_PGM_RSRC3(FourByteBuffer, KdStream);
2961
2963 FourByteBuffer = DE.getU32(Cursor);
2964 return decodeCOMPUTE_PGM_RSRC1(FourByteBuffer, KdStream);
2965
2967 FourByteBuffer = DE.getU32(Cursor);
2968 return decodeCOMPUTE_PGM_RSRC2(FourByteBuffer, KdStream);
2969
2971 using namespace amdhsa;
2972 TwoByteBuffer = DE.getU16(Cursor);
2973
2975 PRINT_DIRECTIVE(".amdhsa_user_sgpr_private_segment_buffer",
2976 KERNEL_CODE_PROPERTY_ENABLE_SGPR_PRIVATE_SEGMENT_BUFFER);
2977 PRINT_DIRECTIVE(".amdhsa_user_sgpr_dispatch_ptr",
2978 KERNEL_CODE_PROPERTY_ENABLE_SGPR_DISPATCH_PTR);
2979 PRINT_DIRECTIVE(".amdhsa_user_sgpr_queue_ptr",
2980 KERNEL_CODE_PROPERTY_ENABLE_SGPR_QUEUE_PTR);
2981 PRINT_DIRECTIVE(".amdhsa_user_sgpr_kernarg_segment_ptr",
2982 KERNEL_CODE_PROPERTY_ENABLE_SGPR_KERNARG_SEGMENT_PTR);
2983 PRINT_DIRECTIVE(".amdhsa_user_sgpr_dispatch_id",
2984 KERNEL_CODE_PROPERTY_ENABLE_SGPR_DISPATCH_ID);
2986 PRINT_DIRECTIVE(".amdhsa_user_sgpr_flat_scratch_init",
2987 KERNEL_CODE_PROPERTY_ENABLE_SGPR_FLAT_SCRATCH_INIT);
2988 PRINT_DIRECTIVE(".amdhsa_user_sgpr_private_segment_size",
2989 KERNEL_CODE_PROPERTY_ENABLE_SGPR_PRIVATE_SEGMENT_SIZE);
2990
2991 if (TwoByteBuffer & KERNEL_CODE_PROPERTY_RESERVED0)
2992 return createReservedKDBitsError(KERNEL_CODE_PROPERTY_RESERVED0,
2994
2995 // Reserved for GFX9
2996 if (isGFX9() &&
2997 (TwoByteBuffer & KERNEL_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32)) {
2999 KERNEL_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32,
3000 amdhsa::KERNEL_CODE_PROPERTIES_OFFSET, "must be zero on gfx9");
3001 }
3002 if (isGFX10Plus()) {
3003 PRINT_DIRECTIVE(".amdhsa_wavefront_size32",
3004 KERNEL_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32);
3005 }
3006
3007 if (CodeObjectVersion >= AMDGPU::AMDHSA_COV5)
3008 PRINT_DIRECTIVE(".amdhsa_uses_dynamic_stack",
3009 KERNEL_CODE_PROPERTY_USES_DYNAMIC_STACK);
3010
3011 if (TwoByteBuffer & KERNEL_CODE_PROPERTY_RESERVED1) {
3012 return createReservedKDBitsError(KERNEL_CODE_PROPERTY_RESERVED1,
3014 }
3015
3016 return true;
3017
3019 using namespace amdhsa;
3020 TwoByteBuffer = DE.getU16(Cursor);
3021 if (TwoByteBuffer & KERNARG_PRELOAD_SPEC_LENGTH) {
3022 PRINT_DIRECTIVE(".amdhsa_user_sgpr_kernarg_preload_length",
3023 KERNARG_PRELOAD_SPEC_LENGTH);
3024 }
3025
3026 if (TwoByteBuffer & KERNARG_PRELOAD_SPEC_OFFSET) {
3027 PRINT_DIRECTIVE(".amdhsa_user_sgpr_kernarg_preload_offset",
3028 KERNARG_PRELOAD_SPEC_OFFSET);
3029 }
3030 return true;
3031
3033 // 4 bytes from here are reserved, must be 0.
3034 ReservedBytes = DE.getBytes(Cursor, 4);
3035 for (char B : ReservedBytes) {
3036 if (B != 0)
3038 }
3039 return true;
3040
3041 default:
3042 llvm_unreachable("Unhandled index. Case statements cover everything.");
3043 return true;
3044 }
3045#undef PRINT_DIRECTIVE
3046}
3047
3049 StringRef KdName, ArrayRef<uint8_t> Bytes, uint64_t KdAddress) const {
3050
3051 // CP microcode requires the kernel descriptor to be 64 aligned.
3052 if (Bytes.size() != 64 || KdAddress % 64 != 0)
3053 return createStringError(std::errc::invalid_argument,
3054 "kernel descriptor must be 64-byte aligned");
3055
3056 // FIXME: We can't actually decode "in order" as is done below, as e.g. GFX10
3057 // requires us to know the setting of .amdhsa_wavefront_size32 in order to
3058 // accurately produce .amdhsa_next_free_vgpr, and they appear in the wrong
3059 // order. Workaround this by first looking up .amdhsa_wavefront_size32 here
3060 // when required.
3061 if (isGFX10Plus()) {
3062 uint16_t KernelCodeProperties =
3065 EnableWavefrontSize32 =
3066 AMDHSA_BITS_GET(KernelCodeProperties,
3067 amdhsa::KERNEL_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32);
3068 }
3069
3070 std::string Kd;
3071 raw_string_ostream KdStream(Kd);
3072 KdStream << ".amdhsa_kernel " << KdName << '\n';
3073
3075 while (C && C.tell() < Bytes.size()) {
3076 Expected<bool> Res = decodeKernelDescriptorDirective(C, Bytes, KdStream);
3077
3078 cantFail(C.takeError());
3079
3080 if (!Res)
3081 return Res;
3082 }
3083 KdStream << ".end_amdhsa_kernel\n";
3084 outs() << KdStream.str();
3085 return true;
3086}
3087
3089 uint64_t &Size,
3090 ArrayRef<uint8_t> Bytes,
3091 uint64_t Address) const {
3092 // Right now only kernel descriptor needs to be handled.
3093 // We ignore all other symbols for target specific handling.
3094 // TODO:
3095 // Fix the spurious symbol issue for AMDGPU kernels. Exists for both Code
3096 // Object V2 and V3 when symbols are marked protected.
3097
3098 // amd_kernel_code_t for Code Object V2.
3099 if (Symbol.Type == ELF::STT_AMDGPU_HSA_KERNEL) {
3100 Size = 256;
3101 return createStringError(std::errc::invalid_argument,
3102 "code object v2 is not supported");
3103 }
3104
3105 // Code Object V3 kernel descriptors.
3106 StringRef Name = Symbol.Name;
3107 if (Symbol.Type == ELF::STT_OBJECT && Name.ends_with(StringRef(".kd"))) {
3108 Size = 64; // Size = 64 regardless of success or failure.
3109 return decodeKernelDescriptor(Name.drop_back(3), Bytes, Address);
3110 }
3111
3112 return false;
3113}
3114
3115const MCExpr *AMDGPUDisassembler::createConstantSymbolExpr(StringRef Id,
3116 int64_t Val) {
3117 MCContext &Ctx = getContext();
3118 MCSymbol *Sym = Ctx.getOrCreateSymbol(Id);
3119 // Note: only set value to Val on a new symbol in case an dissassembler
3120 // has already been initialized in this context.
3121 if (!Sym->isVariable()) {
3123 } else {
3124 int64_t Res = ~Val;
3125 bool Valid = Sym->getVariableValue()->evaluateAsAbsolute(Res);
3126 if (!Valid || Res != Val)
3127 Ctx.reportWarning(SMLoc(), "unsupported redefinition of " + Id);
3128 }
3129 return MCSymbolRefExpr::create(Sym, Ctx);
3130}
3131
3133 // Check for MUBUF and MTBUF instructions
3134 if (SIInstrFlags::isBuffer(*MCII, MI))
3135 return true;
3136
3137 // Check for SMEM buffer instructions (S_BUFFER_* instructions)
3138 if (SIInstrFlags::isSMRD(*MCII, MI) &&
3139 AMDGPU::getSMEMIsBuffer(MI.getOpcode()))
3140 return true;
3141
3142 return false;
3143}
3144
3145//===----------------------------------------------------------------------===//
3146// AMDGPUSymbolizer
3147//===----------------------------------------------------------------------===//
3148
3149// Try to find symbol name for specified label
3151 MCInst &Inst, raw_ostream & /*cStream*/, int64_t Value,
3152 uint64_t /*Address*/, bool IsBranch, uint64_t /*Offset*/,
3153 uint64_t /*OpSize*/, uint64_t /*InstSize*/) {
3154
3155 if (!IsBranch) {
3156 return false;
3157 }
3158
3159 auto *Symbols = static_cast<SectionSymbolsTy *>(DisInfo);
3160 if (!Symbols)
3161 return false;
3162
3163 auto Result = llvm::find_if(*Symbols, [Value](const SymbolInfoTy &Val) {
3164 return Val.Addr == static_cast<uint64_t>(Value) &&
3165 Val.Type == ELF::STT_NOTYPE;
3166 });
3167 if (Result != Symbols->end()) {
3168 auto *Sym = Ctx.getOrCreateSymbol(Result->Name);
3169 const auto *Add = MCSymbolRefExpr::create(Sym, Ctx);
3171 return true;
3172 }
3173 // Add to list of referenced addresses, so caller can synthesize a label.
3174 ReferencedAddresses.push_back(static_cast<uint64_t>(Value));
3175 return false;
3176}
3177
3179 int64_t Value,
3180 uint64_t Address) {
3181 llvm_unreachable("unimplemented");
3182}
3183
3184//===----------------------------------------------------------------------===//
3185// Initialization
3186//===----------------------------------------------------------------------===//
3187
3189 LLVMOpInfoCallback /*GetOpInfo*/,
3190 LLVMSymbolLookupCallback /*SymbolLookUp*/,
3191 void *DisInfo,
3192 MCContext *Ctx,
3193 std::unique_ptr<MCRelocationInfo> &&RelInfo) {
3194 return new AMDGPUSymbolizer(*Ctx, std::move(RelInfo), DisInfo);
3195}
3196
3198 const MCSubtargetInfo &STI,
3199 MCContext &Ctx) {
3200 return new AMDGPUDisassembler(STI, Ctx, T.createMCInstrInfo());
3201}
3202
3203extern "C" LLVM_ABI LLVM_EXTERNAL_VISIBILITY void
MCDisassembler::DecodeStatus DecodeStatus
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
aarch64 promote const
#define CHECK_RESERVED_BITS_DESC(MASK, DESC)
static DecodeStatus decodeRsrcRegOp(MCInst &Inst, unsigned Imm, uint64_t, const MCDisassembler *Decoder, unsigned OpWidth)
static VOPModifiers collectVOPModifiers(const MCInst &MI, bool IsVOP3P=false)
static int insertNamedMCOperand(MCInst &MI, const MCOperand &Op, AMDGPU::OpName Name)
#define DECODE_OPERAND_SREG_9(RegClass, OpWidth)
LLVM_ABI LLVM_EXTERNAL_VISIBILITY void LLVMInitializeAMDGPUDisassembler()
static DecodeStatus decodeOperand_VSrcT16_Lo128(MCInst &Inst, unsigned Imm, uint64_t, const MCDisassembler *Decoder)
static DecodeStatus decodeOperand_KImmFP64(MCInst &Inst, uint64_t Imm, uint64_t Addr, const MCDisassembler *Decoder)
static SmallString< 32 > getBitRangeFromMask(uint32_t Mask, unsigned BaseBytes)
Print a string describing the reserved bit range specified by Mask with offset BaseBytes for use in e...
#define DECODE_OPERAND_SREG_8(RegClass, OpWidth)
static DecodeStatus decodeSMEMOffset(MCInst &Inst, unsigned Imm, uint64_t Addr, const MCDisassembler *Decoder)
static std::bitset< 128 > eat16Bytes(ArrayRef< uint8_t > &Bytes)
#define DECODE_OPERAND_SREG_7(RegClass, OpWidth)
static DecodeStatus decodeSrcA9(MCInst &Inst, unsigned Imm, uint64_t, const MCDisassembler *Decoder)
static DecodeStatus decodeOperand_VGPR_16(MCInst &Inst, unsigned Imm, uint64_t, const MCDisassembler *Decoder)
#define PRINT_PSEUDO_DIRECTIVE_COMMENT(DIRECTIVE, MASK)
static DecodeStatus decodeSrcOp(MCInst &Inst, unsigned EncSize, unsigned OpWidth, unsigned Imm, unsigned EncImm, const MCDisassembler *Decoder)
unsigned Imm
static DecodeStatus decodeDpp8FI(MCInst &Inst, unsigned Val, uint64_t Addr, const MCDisassembler *Decoder)
static DecodeStatus decodeOperand_VSrc_f64(MCInst &Inst, unsigned Imm, uint64_t Addr, const MCDisassembler *Decoder)
static MCRegister CheckVGPROverflow(MCRegister Reg, const MCRegisterClass &RC, const MCRegisterInfo &MRI)
static int64_t getInlineImmValBF16(unsigned Imm)
#define DECODE_SDWA(DecName)
static DecodeStatus decodeSOPPBrTarget(MCInst &Inst, unsigned Imm, uint64_t Addr, const MCDisassembler *Decoder)
#define DECODE_OPERAND_REG_8(RegClass)
#define PRINT_DIRECTIVE(DIRECTIVE, MASK)
static DecodeStatus decodeSrcRegOrImm9(MCInst &Inst, unsigned Imm, uint64_t, const MCDisassembler *Decoder)
static DecodeStatus DecodeVGPR_16RegisterClass(MCInst &Inst, unsigned Imm, uint64_t, const MCDisassembler *Decoder)
static DecodeStatus decodeSrcReg9(MCInst &Inst, unsigned Imm, uint64_t, const MCDisassembler *Decoder)
static DecodeStatus decodeRsrcReg256(MCInst &Inst, unsigned Imm, uint64_t, const MCDisassembler *Decoder)
static int64_t getInlineImmVal32(unsigned Imm)
unsigned uint64_t
static MCDisassembler::DecodeStatus addOperand(MCInst &Inst, const MCOperand &Opnd)
#define CHECK_RESERVED_BITS(MASK)
static DecodeStatus decodeSrcAV10(MCInst &Inst, unsigned Imm, uint64_t, const MCDisassembler *Decoder)
#define SGPR_MAX
static int64_t getInlineImmVal64(unsigned Imm)
static T eatBytes(ArrayRef< uint8_t > &Bytes)
static DecodeStatus decodeOperand_KImmFP(MCInst &Inst, unsigned Imm, uint64_t Addr, const MCDisassembler *Decoder)
static DecodeStatus decodeAVLdSt(MCInst &Inst, unsigned Imm, unsigned Opw, const MCDisassembler *Decoder)
#define DECODE_SDWA_IMM_FIELD(Name, MaxImm)
static MCDisassembler * createAMDGPUDisassembler(const Target &T, const MCSubtargetInfo &STI, MCContext &Ctx)
static DecodeStatus decodeSrcRegOrImmA9(MCInst &Inst, unsigned Imm, uint64_t, const MCDisassembler *Decoder)
static DecodeStatus DecodeVGPR_16_Lo128RegisterClass(MCInst &Inst, unsigned Imm, uint64_t, const MCDisassembler *Decoder)
#define CHECK_RESERVED_BITS_MSG(MASK, MSG)
static DecodeStatus decodeOperandVOPDDstY(MCInst &Inst, unsigned Val, uint64_t Addr, const void *Decoder)
static MCSymbolizer * createAMDGPUSymbolizer(const Triple &, LLVMOpInfoCallback, LLVMSymbolLookupCallback, void *DisInfo, MCContext *Ctx, std::unique_ptr< MCRelocationInfo > &&RelInfo)
static DecodeStatus decodeBoolReg(MCInst &Inst, unsigned Val, uint64_t Addr, const MCDisassembler *Decoder)
static int64_t getInlineImmValF16(unsigned Imm)
unsigned const MCDisassembler * Decoder
#define GET_FIELD(MASK)
static std::bitset< 96 > eat12Bytes(ArrayRef< uint8_t > &Bytes)
static DecodeStatus decodeRsrcReg128(MCInst &Inst, unsigned Imm, uint64_t, const MCDisassembler *Decoder)
static DecodeStatus decodeOperand_VSrcT16(MCInst &Inst, unsigned Imm, uint64_t, const MCDisassembler *Decoder)
static Error createReservedKDBytesError(unsigned BaseInBytes, unsigned WidthInBytes)
Create an error object to return from onSymbolStart for reserved kernel descriptor bytes being set.
static DecodeStatus decodeSplitBarrier(MCInst &Inst, unsigned Val, uint64_t Addr, const MCDisassembler *Decoder)
static DecodeStatus decodeAV10(MCInst &Inst, unsigned Imm, uint64_t, const MCDisassembler *Decoder)
static bool adjustMFMA_F8F6F4OpRegClass(const MCRegisterInfo &MRI, MCOperand &MO, uint8_t NumRegs)
Adjust the register values used by V_MFMA_F8F6F4_f8_f8 instructions to the appropriate subregister fo...
#define CHECK_RESERVED_BITS_DESC_MSG(MASK, DESC, MSG)
static Error createReservedKDBitsError(uint32_t Mask, unsigned BaseBytes, const char *Msg="")
Create an error object to return from onSymbolStart for reserved kernel descriptor bits being set.
This file contains declaration for AMDGPU ISA disassembler.
Provides AMDGPU specific target descriptions.
static cl::opt< bool > XnackSetting("amdgpu-xnack", cl::desc("Force amdgpu.xnack value for testing"), cl::ReallyHidden)
AMDHSA kernel descriptor definitions.
#define AMDHSA_BITS_GET(SRC, MSK)
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
#define AMDGPU_MACH_LIST(X)
Definition ELF.h:768
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
#define LLVM_ABI
Definition Compiler.h:215
#define LLVM_EXTERNAL_VISIBILITY
Definition Compiler.h:132
IRTranslator LLVM IR MI
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
#define T
Interface definition for SIRegisterInfo.
const char * Msg
std::optional< unsigned > getSgprClassId(unsigned Width) const
Return the SGPR/TTMP register class accepted by source decoding for Width, or std::nullopt if that wi...
MCOperand decodeNonVGPRSrcOp(const MCInst &Inst, unsigned Width, unsigned Val) const
MCOperand decodeLiteral64Constant() const
void convertVOPC64DPPInst(MCInst &MI) const
bool isBufferInstruction(const MCInst &MI) const
Check if the instruction is a buffer operation (MUBUF, MTBUF, or S_BUFFER)
void convertEXPInst(MCInst &MI) const
MCOperand decodeSpecialReg64(unsigned Val) const
const char * getRegClassName(unsigned RegClassID) const
Expected< bool > decodeCOMPUTE_PGM_RSRC1(uint32_t FourByteBuffer, raw_string_ostream &KdStream) const
Decode as directives that handle COMPUTE_PGM_RSRC1.
MCOperand decodeSplitBarrier(const MCInst &Inst, unsigned Val) const
Expected< bool > decodeKernelDescriptorDirective(DataExtractor::Cursor &Cursor, ArrayRef< uint8_t > Bytes, raw_string_ostream &KdStream) const
void convertVOPCDPPInst(MCInst &MI) const
MCOperand decodeSpecialReg96Plus(unsigned Val) const
MCOperand decodeSDWASrc32(unsigned Val) const
void setABIVersion(unsigned Version) override
ELF-specific, set the ABI version from the object header.
Expected< bool > decodeCOMPUTE_PGM_RSRC2(uint32_t FourByteBuffer, raw_string_ostream &KdStream) const
Decode as directives that handle COMPUTE_PGM_RSRC2.
unsigned getAgprClassId(unsigned Width) const
MCOperand decodeDpp8FI(unsigned Val) const
MCOperand decodeSDWASrc(unsigned Width, unsigned Val) const
void convertFMAanyK(MCInst &MI) const
DecodeStatus tryDecodeInst(const uint8_t *Table, MCInst &MI, InsnType Inst, uint64_t Address, raw_ostream &Comments) const
void convertMacDPPInst(MCInst &MI) const
MCOperand decodeVOPDDstYOp(MCInst &Inst, unsigned Val) const
void convertDPP8Inst(MCInst &MI) const
MCOperand createVGPR16Operand(unsigned RegIdx, bool IsHi) const
MCOperand errOperand(unsigned V, const Twine &ErrMsg) const
MCOperand decodeVersionImm(unsigned Imm) const
Expected< bool > decodeKernelDescriptor(StringRef KdName, ArrayRef< uint8_t > Bytes, uint64_t KdAddress) const
void convertVOP3DPPInst(MCInst &MI) const
void convertTrue16OpSel(MCInst &MI) const
MCOperand decodeSrcOp(const MCInst &Inst, unsigned Width, unsigned Val) const
bool convertMAIInst(MCInst &MI) const
f8f6f4 instructions have different pseudos depending on the used formats.
MCOperand decodeMandatoryLiteralConstant(unsigned Imm) const
MCOperand decodeLiteralConstant(const MCInstrDesc &Desc, const MCOperandInfo &OpDesc) const
Expected< bool > decodeCOMPUTE_PGM_RSRC3(uint32_t FourByteBuffer, raw_string_ostream &KdStream) const
Decode as directives that handle COMPUTE_PGM_RSRC3.
AMDGPUDisassembler(const MCSubtargetInfo &STI, MCContext &Ctx, MCInstrInfo const *MCII)
MCOperand decodeSpecialReg32(unsigned Val) const
MCOperand createRegOperand(MCRegister Reg) const
MCOperand decodeSDWAVopcDst(unsigned Val) const
void convertVINTERPInst(MCInst &MI) const
void convertSDWAInst(MCInst &MI) const
static MCOperand decodeIntImmed(unsigned Imm)
MCOperand decodeBoolReg(const MCInst &Inst, unsigned Val) const
void emitTargetIDIfSupported(raw_ostream &OS, unsigned EFlags) const override
Emit something based on ELF's e_flags if the target needs to.
unsigned getVgprClassId(unsigned Width) const
DecodeStatus getInstruction(MCInst &MI, uint64_t &Size, ArrayRef< uint8_t > Bytes, uint64_t Address, raw_ostream &CS) const override
Returns the disassembly of a single instruction.
std::optional< unsigned > getTtmpClassId(unsigned Width) const
MCOperand decodeMandatoryLiteral64Constant(uint64_t Imm) const
void convertMIMGInst(MCInst &MI) const
bool isMacDPP(MCInst &MI) const
int getTTmpIdx(unsigned Val) const
void convertVOP3PDPPInst(MCInst &MI) const
bool convertWMMAInst(MCInst &MI) const
MCOperand createSRegOperand(unsigned SRegClassID, unsigned Val) const
MCOperand decodeSDWASrc16(unsigned Val) const
Expected< bool > onSymbolStart(SymbolInfoTy &Symbol, uint64_t &Size, ArrayRef< uint8_t > Bytes, uint64_t Address) const override
Used to perform separate target specific disassembly for a particular symbol.
static const AMDGPUMCExpr * createLit(LitModifier Lit, int64_t Value, MCContext &Ctx)
bool tryAddingSymbolicOperand(MCInst &Inst, raw_ostream &cStream, int64_t Value, uint64_t Address, bool IsBranch, uint64_t Offset, uint64_t OpSize, uint64_t InstSize) override
Try to add a symbolic operand instead of Value to the MCInst.
void tryAddingPcLoadReferenceComment(raw_ostream &cStream, int64_t Value, uint64_t Address) override
Try to add a comment on the PC-relative load.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
const T * data() const
Definition ArrayRef.h:138
ArrayRef< T > slice(size_t N, size_t M) const
slice(n, m) - Chop off the first N elements of the array, and keep M elements in the array.
Definition ArrayRef.h:185
A class representing a position in a DataExtractor, as well as any error encountered during extractio...
LLVM_ABI uint32_t getU32(uint64_t *offset_ptr, Error *Err=nullptr) const
Extract a uint32_t value from *offset_ptr.
LLVM_ABI uint16_t getU16(uint64_t *offset_ptr, Error *Err=nullptr) const
Extract a uint16_t value from *offset_ptr.
LLVM_ABI void skip(Cursor &C, uint64_t Length) const
Advance the Cursor position by the given number of bytes.
LLVM_ABI StringRef getBytes(uint64_t *OffsetPtr, uint64_t Length, Error *Err=nullptr) const
Extract a fixed number of bytes from the specified offset.
Lightweight error class with error context and mandatory checking.
Definition Error.h:159
Tagged union holding either a T or a Error.
Definition Error.h:485
static const MCBinaryExpr * createOr(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:407
static LLVM_ABI const MCConstantExpr * create(int64_t Value, MCContext &Ctx, bool PrintInHex=false, unsigned SizeInBytes=0)
Definition MCExpr.cpp:212
Context object for machine code objects.
Definition MCContext.h:83
const MCRegisterInfo * getRegisterInfo() const
Definition MCContext.h:411
Superclass for all disassemblers.
MCDisassembler(const MCSubtargetInfo &STI, MCContext &Ctx)
MCContext & getContext() const
const MCSubtargetInfo & STI
raw_ostream * CommentStream
DecodeStatus
Ternary decode status.
Base class for the full range of assembler expressions which are needed for parsing.
Definition MCExpr.h:34
Instances of this class represent a single low-level machine instruction.
Definition MCInst.h:188
unsigned getOpcode() const
Definition MCInst.h:202
void addOperand(const MCOperand Op)
Definition MCInst.h:215
const MCOperand & getOperand(unsigned i) const
Definition MCInst.h:210
Describe properties that are true of each instruction in the target description file.
Interface to description of machine instruction set.
Definition MCInstrInfo.h:27
This holds information about one operand of a machine instruction, indicating the register class for ...
Definition MCInstrDesc.h:88
uint8_t OperandType
Information about the type of the operand.
Instances of this class represent operands of the MCInst class.
Definition MCInst.h:40
static MCOperand createExpr(const MCExpr *Val)
Definition MCInst.h:166
int64_t getImm() const
Definition MCInst.h:84
static MCOperand createReg(MCRegister Reg)
Definition MCInst.h:138
static MCOperand createImm(int64_t Val)
Definition MCInst.h:145
void setReg(MCRegister Reg)
Set the register number.
Definition MCInst.h:79
bool isReg() const
Definition MCInst.h:65
MCRegister getReg() const
Returns the register number.
Definition MCInst.h:73
bool isValid() const
Definition MCInst.h:64
MCRegisterClass - Base class of TargetRegisterClass.
MCRegister getRegister(unsigned i) const
getRegister - Return the specified register in the class.
unsigned getSizeInBits() const
Return the size of the physical register in bits if we are able to determine it.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
MCRegisterInfo base class - We assume that the target defines a static array of MCRegisterDesc object...
MCRegister getMatchingSuperReg(MCRegister Reg, unsigned SubIdx, const MCRegisterClass *RC) const
Return a super-register of the specified register Reg so its sub-register of index SubIdx is Reg.
const char * getRegClassName(const MCRegisterClass *Class) const
const MCRegisterClass & getRegClass(unsigned i) const
Returns the register class associated with the enumeration value.
MCRegister getSubReg(MCRegister Reg, unsigned Idx) const
Returns the physical register number of sub-register "Index" for physical register RegNo.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
Generic base class for all target subtargets.
static const MCSymbolRefExpr * create(const MCSymbol *Symbol, MCContext &Ctx, SMLoc Loc=SMLoc())
Definition MCExpr.h:213
MCSymbol - Instances of this class represent a symbol name in the MC file, and MCSymbols are created ...
Definition MCSymbol.h:42
bool isVariable() const
isVariable - Check if this is a variable symbol.
Definition MCSymbol.h:267
LLVM_ABI void setVariableValue(const MCExpr *Value)
Definition MCSymbol.cpp:50
const MCExpr * getVariableValue() const
Get the expression of the variable symbol.
Definition MCSymbol.h:270
Symbolize and annotate disassembled instructions.
Represents a location in source code.
Definition SMLoc.h:22
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Target - Wrapper for Target specific information.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
LLVM Value Representation.
Definition Value.h:75
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
A raw_ostream that writes to an std::string.
std::string & str()
Returns the string's reference.
A raw_ostream that writes to an SmallVector or SmallString.
const char *(* LLVMSymbolLookupCallback)(void *DisInfo, uint64_t ReferenceValue, uint64_t *ReferenceType, uint64_t ReferencePC, const char **ReferenceName)
The type for the symbol lookup function.
int(* LLVMOpInfoCallback)(void *DisInfo, uint64_t PC, uint64_t Offset, uint64_t OpSize, uint64_t InstSize, int TagType, void *TagBuf)
The type for the operand information call back function.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI, std::optional< bool > EnableWavefrontSize32)
unsigned getSGPREncodingGranule(const MCSubtargetInfo &STI)
ArrayRef< GFXVersion > getGFXVersions()
bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi)
EncodingField< Bit, Bit, D > EncodingBit
bool isPKFMACF16InlineConstant(uint32_t Literal, bool IsGFX11Plus)
LLVM_READONLY const MIMGInfo * getMIMGInfo(unsigned Opc)
bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi)
MCRegister getMCReg(MCRegister Reg, const MCSubtargetInfo &STI)
If Reg is a pseudo reg, return the correct hardware register given STI otherwise return Reg.
bool isInlinableLiteralV2I16(uint32_t Literal)
bool isGFX10(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2BF16(uint32_t Literal)
bool isGFX12Plus(const MCSubtargetInfo &STI)
bool hasPackedD16(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2F16(uint32_t Literal)
bool getSMEMIsBuffer(unsigned Opc)
bool isGFX13(const MCSubtargetInfo &STI)
bool isVOPC64DPP(unsigned Opc)
bool hasPrivateApertureRegs(const MCSubtargetInfo &STI)
unsigned getAMDHSACodeObjectVersion(const Module &M)
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
bool isGFX9(const MCSubtargetInfo &STI)
int getMIMGOpcode(unsigned BaseOpcode, unsigned MIMGEncoding, unsigned VDataDwords, unsigned VAddrDwords, bool IndexedRsrc, bool IndexedSamp)
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfoByEncoding(uint8_t DimEnc)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
const MFMA_F8F6F4_Info * getWMMA_F8F6F4_WithFormatArgs(unsigned FmtA, unsigned FmtB, unsigned F8F8Opcode)
bool hasG16(const MCSubtargetInfo &STI)
unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode, const MIMGDimInfo *Dim, bool IsA16, bool IsG16Supported)
bool isGFX13Plus(const MCSubtargetInfo &STI)
bool isGFX11Plus(const MCSubtargetInfo &STI)
bool isGFX10Plus(const MCSubtargetInfo &STI)
@ OPERAND_REG_IMM_V2FP64
Definition SIDefines.h:441
@ OPERAND_KIMM32
Operand with 32-bit immediate that uses the constant bus.
Definition SIDefines.h:459
@ OPERAND_REG_IMM_INT64
Definition SIDefines.h:426
@ OPERAND_REG_IMM_V2FP16
Definition SIDefines.h:434
@ OPERAND_REG_INLINE_C_FP64
Definition SIDefines.h:450
@ OPERAND_REG_IMM_NOINLINE_FP16
Definition SIDefines.h:432
@ OPERAND_REG_INLINE_C_BF16
Definition SIDefines.h:447
@ OPERAND_REG_INLINE_C_V2BF16
Definition SIDefines.h:452
@ OPERAND_REG_IMM_V2INT64
Definition SIDefines.h:437
@ OPERAND_REG_IMM_V2INT16
Definition SIDefines.h:436
@ OPERAND_REG_IMM_BF16
Definition SIDefines.h:430
@ OPERAND_REG_IMM_INT32
Operands with register, 32-bit, or 64-bit immediate.
Definition SIDefines.h:425
@ OPERAND_REG_IMM_V2BF16
Definition SIDefines.h:433
@ OPERAND_REG_IMM_FP16
Definition SIDefines.h:431
@ OPERAND_REG_IMM_V2FP16_SPLAT
Definition SIDefines.h:435
@ OPERAND_REG_INLINE_C_INT64
Definition SIDefines.h:446
@ OPERAND_REG_INLINE_C_INT16
Operands with register or inline constant.
Definition SIDefines.h:444
@ OPERAND_REG_IMM_NOINLINE_V2FP16
Definition SIDefines.h:438
@ OPERAND_REG_IMM_FP64
Definition SIDefines.h:429
@ OPERAND_REG_INLINE_C_V2FP16
Definition SIDefines.h:453
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
Definition SIDefines.h:464
@ OPERAND_REG_INLINE_AC_FP32
Definition SIDefines.h:465
@ OPERAND_REG_IMM_V2INT32
Definition SIDefines.h:439
@ OPERAND_REG_IMM_FP32
Definition SIDefines.h:428
@ OPERAND_REG_INLINE_C_FP32
Definition SIDefines.h:449
@ OPERAND_REG_INLINE_C_INT32
Definition SIDefines.h:445
@ OPERAND_REG_INLINE_C_V2INT16
Definition SIDefines.h:451
@ OPERAND_REG_IMM_V2FP32
Definition SIDefines.h:440
@ OPERAND_REG_INLINE_AC_FP64
Definition SIDefines.h:466
@ OPERAND_REG_INLINE_C_FP16
Definition SIDefines.h:448
@ OPERAND_REG_IMM_INT16
Definition SIDefines.h:427
bool hasGDS(const MCSubtargetInfo &STI)
bool isGFX9Plus(const MCSubtargetInfo &STI)
bool isVOPD(unsigned Opc)
bool isGFX1250(const MCSubtargetInfo &STI)
unsigned hasKernargPreload(const MCSubtargetInfo &STI)
bool isMAC(unsigned Opc)
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
bool isGFX1250Plus(const MCSubtargetInfo &STI)
bool hasPopsExitingWaveID(const MCSubtargetInfo &STI)
bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi)
bool hasVOPD(const MCSubtargetInfo &STI)
bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi)
Is this literal inlinable.
const MFMA_F8F6F4_Info * getMFMA_F8F6F4_WithFormatArgs(unsigned CBSZ, unsigned BLGP, unsigned F8F8Opcode)
@ STT_NOTYPE
Definition ELF.h:1433
@ STT_AMDGPU_HSA_KERNEL
Definition ELF.h:1447
@ STT_OBJECT
Definition ELF.h:1434
@ EF_AMDGPU_FEATURE_XNACK_ANY_V4
Definition ELF.h:910
@ EF_AMDGPU_FEATURE_SRAMECC_UNSUPPORTED_V4
Definition ELF.h:921
@ EF_AMDGPU_FEATURE_SRAMECC_OFF_V4
Definition ELF.h:925
@ EF_AMDGPU_FEATURE_XNACK_UNSUPPORTED_V4
Definition ELF.h:908
@ EF_AMDGPU_FEATURE_XNACK_OFF_V4
Definition ELF.h:912
@ EF_AMDGPU_FEATURE_XNACK_V4
Definition ELF.h:906
@ EF_AMDGPU_FEATURE_SRAMECC_V4
Definition ELF.h:919
@ EF_AMDGPU_FEATURE_XNACK_ON_V4
Definition ELF.h:914
@ EF_AMDGPU_MACH
Definition ELF.h:852
@ EF_AMDGPU_FEATURE_SRAMECC_ANY_V4
Definition ELF.h:923
@ EF_AMDGPU_FEATURE_SRAMECC_ON_V4
Definition ELF.h:927
constexpr bool isAtomicRet(const T &...O)
Definition SIDefines.h:361
constexpr bool isVOPC(const T &...O)
Definition SIDefines.h:233
constexpr bool isVOP3(const T &...O)
Definition SIDefines.h:236
constexpr bool isMAI(const T &...O)
Definition SIDefines.h:349
constexpr bool isFLAT(const T &...O)
Definition SIDefines.h:283
constexpr bool isVOP3P(const T &...O)
Definition SIDefines.h:239
constexpr bool isBuffer(const T &...O)
Definition SIDefines.h:264
constexpr bool isVIMAGE(const T &...O)
Definition SIDefines.h:274
constexpr bool isSMRD(const T &...O)
Definition SIDefines.h:268
constexpr bool isVOP3Like(const T &...O)
Definition SIDefines.h:242
constexpr bool isMIMG(const T &...O)
Definition SIDefines.h:271
constexpr bool isWMMA(const T &...O)
Definition SIDefines.h:364
constexpr bool isMUBUF(const T &...O)
Definition SIDefines.h:258
constexpr bool isSDWA(const T &...O)
Definition SIDefines.h:249
constexpr bool isEXP(const T &...O)
Definition SIDefines.h:280
constexpr bool isSOPK(const T &...O)
Definition SIDefines.h:221
constexpr bool isVINTERP(const T &...O)
Definition SIDefines.h:295
constexpr bool isVSAMPLE(const T &...O)
Definition SIDefines.h:277
constexpr bool isDS(const T &...O)
Definition SIDefines.h:286
constexpr bool isGather4(const T &...O)
Definition SIDefines.h:304
constexpr bool isDPP(const T &...O)
Definition SIDefines.h:252
value_type read(const void *memory, endianness endian)
Read a value of a particular endianness from memory.
Definition Endian.h:53
uint16_t read16(const void *P, endianness E)
Definition Endian.h:389
This is an optimization pass for GlobalISel generic memory operations.
@ Offset
Definition DWP.cpp:577
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
LLVM_ABI raw_fd_ostream & outs()
This returns a reference to a raw_fd_ostream for standard output.
SmallVectorImpl< T >::const_pointer c_str(SmallVectorImpl< T > &str)
Error createStringError(std::error_code EC, char const *Fmt, const Ts &... Vals)
Create formatted StringError object.
Definition Error.h:1321
Op::Description Desc
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
Definition bit.h:156
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
void cantFail(Error Err, const char *Msg=nullptr)
Report a fatal error if Err is a failure value.
Definition Error.h:769
Target & getTheGCNTarget()
The target for GCN GPUs.
To bit_cast(const From &from) noexcept
Definition bit.h:90
@ Add
Sum of integers.
DWARFExpression::Operation Op
unsigned M0(unsigned Val)
Definition VE.h:376
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
Target & getTheGCNLegacyTarget()
The target for GCN GPUs, registered under the legacy "amdgcn" architecture name for use with -march.
std::vector< SymbolInfoTy > SectionSymbolsTy
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
Definition MathExtras.h:567
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
static void RegisterMCSymbolizer(Target &T, Target::MCSymbolizerCtorTy Fn)
RegisterMCSymbolizer - Register an MCSymbolizer implementation for the given target.
static void RegisterMCDisassembler(Target &T, Target::MCDisassemblerCtorTy Fn)
RegisterMCDisassembler - Register a MCDisassembler implementation for the given target.