LLVM 24.0.0git
AMDGPUAsmParser.cpp
Go to the documentation of this file.
1//===- AMDGPUAsmParser.cpp - Parse SI asm to MCInst instructions ----------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
9#include "AMDKernelCodeT.h"
16#include "SIDefines.h"
17#include "SIInstrInfo.h"
22#include "llvm/ADT/APFloat.h"
24#include "llvm/ADT/Twine.h"
27#include "llvm/MC/MCAsmInfo.h"
28#include "llvm/MC/MCContext.h"
29#include "llvm/MC/MCExpr.h"
30#include "llvm/MC/MCInst.h"
31#include "llvm/MC/MCInstrDesc.h"
37#include "llvm/MC/MCSymbol.h"
46#include <optional>
47
48using namespace llvm;
49using namespace llvm::AMDGPU;
50using namespace llvm::amdhsa;
51
52namespace {
53
54class AMDGPUAsmParser;
55
56enum RegisterKind {
57 IS_UNKNOWN,
58 IS_VGPR,
59 IS_SGPR,
60 IS_AGPR,
61 IS_TTMP,
62 IS_SPECIAL
63};
64
65//===----------------------------------------------------------------------===//
66// Operand
67//===----------------------------------------------------------------------===//
68
69class AMDGPUOperand : public MCParsedAsmOperand {
70 enum KindTy { Token, Immediate, Register, Expression } Kind;
71
72 SMLoc StartLoc, EndLoc;
73 const AMDGPUAsmParser *AsmParser;
74
75public:
76 AMDGPUOperand(KindTy Kind_, const AMDGPUAsmParser *AsmParser_)
77 : Kind(Kind_), AsmParser(AsmParser_) {}
78
79 using Ptr = std::unique_ptr<AMDGPUOperand>;
80
81 struct Modifiers {
82 bool Abs = false;
83 bool Neg = false;
84 bool Sext = false;
85 LitModifier Lit = LitModifier::None;
86
87 bool hasFPModifiers() const { return Abs || Neg; }
88 bool hasIntModifiers() const { return Sext; }
89 bool hasModifiers() const { return hasFPModifiers() || hasIntModifiers(); }
90 bool isForcedLit() const { return Lit == LitModifier::Lit; }
91 bool isForcedLit64() const { return Lit == LitModifier::Lit64; }
92
93 int64_t getFPModifiersOperand() const {
94 int64_t Operand = 0;
95 Operand |= Abs ? SISrcMods::ABS : 0u;
96 Operand |= Neg ? SISrcMods::NEG : 0u;
97 return Operand;
98 }
99
100 int64_t getIntModifiersOperand() const {
101 int64_t Operand = 0;
102 Operand |= Sext ? SISrcMods::SEXT : 0u;
103 return Operand;
104 }
105
106 int64_t getModifiersOperand() const {
107 assert(!(hasFPModifiers() && hasIntModifiers()) &&
108 "fp and int modifiers should not be used simultaneously");
109 if (hasFPModifiers())
110 return getFPModifiersOperand();
111 if (hasIntModifiers())
112 return getIntModifiersOperand();
113 return 0;
114 }
115
116 friend raw_ostream &operator<<(raw_ostream &OS,
117 AMDGPUOperand::Modifiers Mods);
118 };
119
120 enum ImmTy {
121 ImmTyNone,
122 ImmTyGDS,
123 ImmTyLDS,
124 ImmTyOffen,
125 ImmTyIdxen,
126 ImmTyAddr64,
127 ImmTyOffset,
128 ImmTyInstOffset,
129 ImmTyOffset0,
130 ImmTyOffset1,
131 ImmTySMEMOffsetMod,
132 ImmTyCPol,
133 ImmTyTFE,
134 ImmTyIsAsync,
135 ImmTyD16,
136 ImmTyClamp,
137 ImmTyOModSI,
138 ImmTySDWADstSel,
139 ImmTySDWASrc0Sel,
140 ImmTySDWASrc1Sel,
141 ImmTySDWADstUnused,
142 ImmTyDMask,
143 ImmTyDim,
144 ImmTyUNorm,
145 ImmTyDA,
146 ImmTyR128A16,
147 ImmTyA16,
148 ImmTyLWE,
149 ImmTyExpTgt,
150 ImmTyExpCompr,
151 ImmTyExpVM,
152 ImmTyDone,
153 ImmTyRowEn,
154 ImmTyFORMAT,
155 ImmTyHwreg,
156 ImmTyOff,
157 ImmTySendMsg,
158 ImmTyWaitEvent,
159 ImmTyInterpSlot,
160 ImmTyInterpAttr,
161 ImmTyInterpAttrChan,
162 ImmTyOpSel,
163 ImmTyOpSelHi,
164 ImmTyNegLo,
165 ImmTyNegHi,
166 ImmTyIndexKey8bit,
167 ImmTyIndexKey16bit,
168 ImmTyIndexKey32bit,
169 ImmTyDPP8,
170 ImmTyDppCtrl,
171 ImmTyDppRowMask,
172 ImmTyDppBankMask,
173 ImmTyDppBoundCtrl,
174 ImmTyDppFI,
175 ImmTySwizzle,
176 ImmTyGprIdxMode,
177 ImmTyHigh,
178 ImmTyBLGP,
179 ImmTyCBSZ,
180 ImmTyABID,
181 ImmTyEndpgm,
182 ImmTyWaitVDST,
183 ImmTyWaitEXP,
184 ImmTyWaitVAVDst,
185 ImmTyWaitVMVSrc,
186 ImmTyBitOp3,
187 ImmTyMatrixAFMT,
188 ImmTyMatrixBFMT,
189 ImmTyMatrixAScale,
190 ImmTyMatrixBScale,
191 ImmTyMatrixAScaleFmt,
192 ImmTyMatrixBScaleFmt,
193 ImmTyMatrixAReuse,
194 ImmTyMatrixBReuse,
195 ImmTyScaleSel,
196 ImmTyByteSel,
197 };
198
199private:
200 struct TokOp {
201 const char *Data;
202 unsigned Length;
203 };
204
205 struct ImmOp {
206 int64_t Val;
207 ImmTy Type;
208 bool IsFPImm;
209 Modifiers Mods;
210 };
211
212 struct RegOp {
213 MCRegister RegNo;
214 Modifiers Mods;
215 };
216
217 union {
218 TokOp Tok;
219 ImmOp Imm;
220 RegOp Reg;
221 const MCExpr *Expr;
222 };
223
224 // The index of the associated MCInst operand.
225 mutable int MCOpIdx = -1;
226
227public:
228 bool isToken() const override { return Kind == Token; }
229
230 bool isSymbolRefExpr() const {
231 return isExpr() && Expr && isa<MCSymbolRefExpr>(Expr);
232 }
233
234 bool isImm() const override { return Kind == Immediate; }
235
236 bool isInlinableImm(MVT type) const;
237 bool isLiteralImm(MVT type) const;
238
239 bool isRegKind() const { return Kind == Register; }
240
241 bool isReg() const override { return isRegKind() && !hasModifiers(); }
242
243 bool isRegOrInline(unsigned RCID, MVT type) const {
244 return isRegClass(RCID) || isInlinableImm(type);
245 }
246
247 bool isRegOrInlineTarget(unsigned TargetRCIdx, MVT type) const {
248 return isRegClassTarget(TargetRCIdx) || isInlinableImm(type);
249 }
250
251 bool isRegOrImmWithInputMods(unsigned RCID, MVT type) const {
252 return isRegOrInline(RCID, type) || isLiteralImm(type);
253 }
254
255 bool isRegOrImmWithInputModsTarget(unsigned TargetRCIdx, MVT type) const {
256 return isRegOrInlineTarget(TargetRCIdx, type) || isLiteralImm(type);
257 }
258
259 bool isRegOrImmWithInt16InputMods() const {
260 return isRegOrImmWithInputMods(AMDGPU::VS_32RegClassID, MVT::i16);
261 }
262
263 template <bool IsFake16> bool isRegOrImmWithIntT16InputMods() const {
265 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::i16);
266 }
267
268 bool isRegOrImmWithInt32InputMods() const {
269 return isRegOrImmWithInputMods(AMDGPU::VS_32RegClassID, MVT::i32);
270 }
271
272 bool isRegOrInlineImmWithInt16InputMods() const {
273 return isRegOrInline(AMDGPU::VS_32RegClassID, MVT::i16);
274 }
275
276 template <bool IsFake16> bool isRegOrInlineImmWithIntT16InputMods() const {
277 return isRegOrInline(
278 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::i16);
279 }
280
281 bool isRegOrInlineImmWithInt32InputMods() const {
282 return isRegOrInline(AMDGPU::VS_32RegClassID, MVT::i32);
283 }
284
285 bool isRegOrImmWithInt64InputMods() const {
286 return isRegOrImmWithInputModsTarget(AMDGPU::VS_64_AlignTarget, MVT::i64);
287 }
288
289 bool isRegOrImmWithFP16InputMods() const {
290 return isRegOrImmWithInputMods(AMDGPU::VS_32RegClassID, MVT::f16);
291 }
292
293 template <bool IsFake16> bool isRegOrImmWithFPT16InputMods() const {
295 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::f16);
296 }
297
298 bool isRegOrImmWithFP32InputMods() const {
299 return isRegOrImmWithInputMods(AMDGPU::VS_32RegClassID, MVT::f32);
300 }
301
302 bool isRegOrImmWithFP64InputMods() const {
303 return isRegOrImmWithInputModsTarget(AMDGPU::VS_64_AlignTarget, MVT::f64);
304 }
305
306 template <bool IsFake16> bool isRegOrInlineImmWithFP16InputMods() const {
307 return isRegOrInline(
308 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::f16);
309 }
310
311 bool isRegOrInlineImmWithFP32InputMods() const {
312 return isRegOrInline(AMDGPU::VS_32RegClassID, MVT::f32);
313 }
314
315 bool isRegOrInlineImmWithFP64InputMods() const {
316 return isRegOrInlineTarget(AMDGPU::VS_64_AlignTarget, MVT::f64);
317 }
318
319 bool isVRegWithInputMods(unsigned RCID) const { return isRegClass(RCID); }
320
321 bool isVRegWithFP32InputMods() const {
322 return isVRegWithInputMods(AMDGPU::VGPR_32RegClassID);
323 }
324
325 bool isVRegWithFP64InputMods() const {
326 return isRegClassTarget(AMDGPU::VReg_64_AlignTarget);
327 }
328
329 bool isPackedFP16InputMods() const {
330 return isRegOrImmWithInputMods(AMDGPU::VS_32RegClassID, MVT::v2f16);
331 }
332
333 bool isPackedVGPRFP32InputMods() const {
334 return isRegOrImmWithInputMods(AMDGPU::VReg_64RegClassID, MVT::v2f32);
335 }
336
337 bool isVReg() const {
338 return isRegClass(AMDGPU::VGPR_32RegClassID) ||
339 isRegClass(AMDGPU::VReg_64RegClassID) ||
340 isRegClass(AMDGPU::VReg_96RegClassID) ||
341 isRegClass(AMDGPU::VReg_128RegClassID) ||
342 isRegClass(AMDGPU::VReg_160RegClassID) ||
343 isRegClass(AMDGPU::VReg_192RegClassID) ||
344 isRegClass(AMDGPU::VReg_256RegClassID) ||
345 isRegClass(AMDGPU::VReg_512RegClassID) ||
346 isRegClass(AMDGPU::VReg_1024RegClassID);
347 }
348
349 bool isVReg32() const { return isRegClass(AMDGPU::VGPR_32RegClassID); }
350
351 bool isVReg32OrOff() const { return isOff() || isVReg32(); }
352
353 bool isRsrcReg32() const { return isRegClass(AMDGPU::RsrcReg32RegClassID); }
354
355 bool isNull() const { return isRegKind() && getReg() == AMDGPU::SGPR_NULL; }
356
357 bool isAV_LdSt_32_Align2_RegOp() const {
358 return isRegClass(AMDGPU::VGPR_32RegClassID) ||
359 isRegClass(AMDGPU::AGPR_32RegClassID);
360 }
361
362 bool isVRegWithInputMods() const;
363 template <bool IsFake16> bool isT16_Lo128VRegWithInputMods() const;
364 template <bool IsFake16> bool isT16VRegWithInputMods() const;
365
366 bool isSDWAOperand(MVT type) const;
367 bool isSDWAFP16Operand() const;
368 bool isSDWAFP32Operand() const;
369 bool isSDWAInt16Operand() const;
370 bool isSDWAInt32Operand() const;
371
372 bool isImmTy(ImmTy ImmT) const { return isImm() && Imm.Type == ImmT; }
373
374 template <ImmTy Ty> bool isImmTy() const { return isImmTy(Ty); }
375
376 bool isImmLiteral() const { return isImmTy(ImmTyNone); }
377
378 bool isImmModifier() const { return isImm() && Imm.Type != ImmTyNone; }
379
380 bool isOModSI() const { return isImmTy(ImmTyOModSI); }
381 bool isDim() const { return isImmTy(ImmTyDim); }
382 bool isR128A16() const { return isImmTy(ImmTyR128A16); }
383 bool isOff() const { return isImmTy(ImmTyOff); }
384 bool isExpTgt() const { return isImmTy(ImmTyExpTgt); }
385 bool isOffen() const { return isImmTy(ImmTyOffen); }
386 bool isIdxen() const { return isImmTy(ImmTyIdxen); }
387 bool isAddr64() const { return isImmTy(ImmTyAddr64); }
388 bool isSMEMOffsetMod() const { return isImmTy(ImmTySMEMOffsetMod); }
389 bool isFlatOffset() const {
390 return isImmTy(ImmTyOffset) || isImmTy(ImmTyInstOffset);
391 }
392 bool isGDS() const { return isImmTy(ImmTyGDS); }
393 bool isLDS() const { return isImmTy(ImmTyLDS); }
394 bool isCPol() const { return isImmTy(ImmTyCPol); }
395 bool isIndexKey8bit() const { return isImmTy(ImmTyIndexKey8bit); }
396 bool isIndexKey16bit() const { return isImmTy(ImmTyIndexKey16bit); }
397 bool isIndexKey32bit() const { return isImmTy(ImmTyIndexKey32bit); }
398 bool isMatrixAFMT() const { return isImmTy(ImmTyMatrixAFMT); }
399 bool isMatrixBFMT() const { return isImmTy(ImmTyMatrixBFMT); }
400 bool isMatrixAScale() const { return isImmTy(ImmTyMatrixAScale); }
401 bool isMatrixBScale() const { return isImmTy(ImmTyMatrixBScale); }
402 bool isMatrixAScaleFmt() const { return isImmTy(ImmTyMatrixAScaleFmt); }
403 bool isMatrixBScaleFmt() const { return isImmTy(ImmTyMatrixBScaleFmt); }
404 bool isMatrixAReuse() const { return isImmTy(ImmTyMatrixAReuse); }
405 bool isMatrixBReuse() const { return isImmTy(ImmTyMatrixBReuse); }
406 bool isTFE() const { return isImmTy(ImmTyTFE); }
407 bool isFORMAT() const { return isImmTy(ImmTyFORMAT) && isUInt<7>(getImm()); }
408 bool isDppFI() const { return isImmTy(ImmTyDppFI); }
409 bool isSDWADstSel() const { return isImmTy(ImmTySDWADstSel); }
410 bool isSDWASrc0Sel() const { return isImmTy(ImmTySDWASrc0Sel); }
411 bool isSDWASrc1Sel() const { return isImmTy(ImmTySDWASrc1Sel); }
412 bool isSDWADstUnused() const { return isImmTy(ImmTySDWADstUnused); }
413 bool isInterpSlot() const { return isImmTy(ImmTyInterpSlot); }
414 bool isInterpAttr() const { return isImmTy(ImmTyInterpAttr); }
415 bool isInterpAttrChan() const { return isImmTy(ImmTyInterpAttrChan); }
416 bool isOpSel() const { return isImmTy(ImmTyOpSel); }
417 bool isOpSelHi() const { return isImmTy(ImmTyOpSelHi); }
418 bool isNegLo() const { return isImmTy(ImmTyNegLo); }
419 bool isNegHi() const { return isImmTy(ImmTyNegHi); }
420 bool isBitOp3() const { return isImmTy(ImmTyBitOp3) && isUInt<8>(getImm()); }
421 bool isDone() const { return isImmTy(ImmTyDone); }
422 bool isRowEn() const { return isImmTy(ImmTyRowEn); }
423
424 bool isRegOrImm() const { return isReg() || isImm(); }
425
426 bool isRegClass(unsigned RCID) const;
427
428 // Check the register against the HwMode-resolved operand class.
429 bool isRegClassTarget(unsigned TargetRCIdx) const;
430
431 bool isInlineValue() const;
432
433 bool isRegOrInlineNoMods(unsigned RCID, MVT type) const {
434 return isRegOrInline(RCID, type) && !hasModifiers();
435 }
436
437 bool isRegOrInlineNoModsTarget(unsigned TargetRCIdx, MVT type) const {
438 return isRegOrInlineTarget(TargetRCIdx, type) && !hasModifiers();
439 }
440
441 bool isSCSrcB16() const {
442 return isRegOrInlineNoMods(AMDGPU::SReg_32RegClassID, MVT::i16);
443 }
444
445 bool isSCSrcV2B16() const { return isSCSrcB16(); }
446
447 bool isSCSrc_b32() const {
448 return isRegOrInlineNoMods(AMDGPU::SReg_32RegClassID, MVT::i32);
449 }
450
451 bool isSCSrc_b64() const {
452 return isRegOrInlineNoMods(AMDGPU::SReg_64RegClassID, MVT::i64);
453 }
454
455 bool isBoolReg() const;
456
457 bool isSCSrcF16() const {
458 return isRegOrInlineNoMods(AMDGPU::SReg_32RegClassID, MVT::f16);
459 }
460
461 bool isSCSrcV2F16() const { return isSCSrcF16(); }
462
463 bool isSCSrcF32() const {
464 return isRegOrInlineNoMods(AMDGPU::SReg_32RegClassID, MVT::f32);
465 }
466
467 bool isSCSrcF64() const {
468 return isRegOrInlineNoMods(AMDGPU::SReg_64RegClassID, MVT::f64);
469 }
470
471 bool isSSrc_b32() const {
472 return isSCSrc_b32() || isLiteralImm(MVT::i32) || isExpr();
473 }
474
475 bool isSSrc_b16() const { return isSCSrcB16() || isLiteralImm(MVT::i16); }
476
477 bool isSSrcV2B16() const {
478 llvm_unreachable("cannot happen");
479 return isSSrc_b16();
480 }
481
482 bool isSSrc_b64() const {
483 // TODO: Find out how SALU supports extension of 32-bit literals to 64 bits.
484 // See isVSrc64().
485 return isSCSrc_b64() || isLiteralImm(MVT::i64) ||
486 (((const MCTargetAsmParser *)AsmParser)
487 ->getAvailableFeatures()[AMDGPU::Feature64BitLiterals] &&
488 isExpr());
489 }
490
491 bool isSSrc_f32() const {
492 return isSCSrc_b32() || isLiteralImm(MVT::f32) || isExpr();
493 }
494
495 bool isSSrcF64() const { return isSCSrc_b64() || isLiteralImm(MVT::f64); }
496
497 bool isSSrc_bf16() const { return isSCSrcB16() || isLiteralImm(MVT::bf16); }
498
499 bool isSSrc_f16() const { return isSCSrcB16() || isLiteralImm(MVT::f16); }
500
501 bool isSSrc_NoInline_f16() const { return isSSrc_f16(); }
502
503 bool isSSrcV2F16() const {
504 llvm_unreachable("cannot happen");
505 return isSSrc_f16();
506 }
507
508 bool isSSrcV2FP32() const {
509 llvm_unreachable("cannot happen");
510 return isSSrc_f32();
511 }
512
513 bool isSCSrcV2FP32() const {
514 llvm_unreachable("cannot happen");
515 return isSCSrcF32();
516 }
517
518 bool isSSrcV2INT32() const {
519 llvm_unreachable("cannot happen");
520 return isSSrc_b32();
521 }
522
523 bool isSCSrcV2INT32() const {
524 llvm_unreachable("cannot happen");
525 return isSCSrc_b32();
526 }
527
528 bool isSSrcOrLds_b32() const {
529 return isRegOrInlineNoMods(AMDGPU::SRegOrLds_32RegClassID, MVT::i32) ||
530 isLiteralImm(MVT::i32) || isExpr();
531 }
532
533 bool isVCSrc_b32() const {
534 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::i32);
535 }
536
537 bool isVCSrc_b32_Lo256() const {
538 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo256RegClassID, MVT::i32);
539 }
540
541 bool isVCSrc_b64_Lo256() const {
542 return isRegOrInlineNoMods(AMDGPU::VS_64_Lo256RegClassID, MVT::i64);
543 }
544
545 bool isVCSrc_b64() const {
546 return isRegOrInlineNoModsTarget(AMDGPU::VS_64_AlignTarget, MVT::i64);
547 }
548
549 bool isVCSrcT_b16() const {
550 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::i16);
551 }
552
553 bool isVCSrcTB16_Lo128() const {
554 return isRegOrInlineNoMods(AMDGPU::VS_16_Lo128RegClassID, MVT::i16);
555 }
556
557 bool isVCSrcFake16B16_Lo128() const {
558 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo128RegClassID, MVT::i16);
559 }
560
561 bool isVCSrc_b16() const {
562 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::i16);
563 }
564
565 bool isVCSrc_v2b16() const { return isVCSrc_b16(); }
566
567 bool isVCSrc_f32() const {
568 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::f32);
569 }
570
571 bool isVCSrc_f64() const {
572 return isRegOrInlineNoModsTarget(AMDGPU::VS_64_AlignTarget, MVT::f64);
573 }
574
575 bool isVCSrcTBF16() const {
576 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::bf16);
577 }
578
579 bool isVCSrcT_f16() const {
580 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::f16);
581 }
582
583 bool isVCSrcT_bf16() const {
584 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::f16);
585 }
586
587 bool isVCSrcTBF16_Lo128() const {
588 return isRegOrInlineNoMods(AMDGPU::VS_16_Lo128RegClassID, MVT::bf16);
589 }
590
591 bool isVCSrcTF16_Lo128() const {
592 return isRegOrInlineNoMods(AMDGPU::VS_16_Lo128RegClassID, MVT::f16);
593 }
594
595 bool isVCSrcFake16BF16_Lo128() const {
596 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo128RegClassID, MVT::bf16);
597 }
598
599 bool isVCSrcFake16F16_Lo128() const {
600 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo128RegClassID, MVT::f16);
601 }
602
603 bool isVCSrc_bf16() const {
604 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::bf16);
605 }
606
607 bool isVCSrc_f16() const {
608 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::f16);
609 }
610
611 bool isVCSrc_v2bf16() const { return isVCSrc_bf16(); }
612
613 bool isVCSrc_v2f16() const { return isVCSrc_f16(); }
614
615 bool isVSrc_b32() const {
616 return isVCSrc_f32() || isLiteralImm(MVT::i32) || isExpr();
617 }
618
619 bool isVSrc_b64() const { return isVCSrc_f64() || isLiteralImm(MVT::i64); }
620
621 bool isVSrc_v2b64() const {
622 return isRegOrInlineNoMods(AMDGPU::VS_128RegClassID, MVT::i64) ||
623 isLiteralImm(MVT::i64);
624 }
625
626 bool isVSrc_v2f64() const {
627 return isRegOrInlineNoMods(AMDGPU::VS_128RegClassID, MVT::f64) ||
628 isLiteralImm(MVT::f64);
629 }
630
631 bool isVSrcT_b16() const { return isVCSrcT_b16() || isLiteralImm(MVT::i16); }
632
633 bool isVSrcT_b16_Lo128() const {
634 return isVCSrcTB16_Lo128() || isLiteralImm(MVT::i16);
635 }
636
637 bool isVSrcFake16_b16_Lo128() const {
638 return isVCSrcFake16B16_Lo128() || isLiteralImm(MVT::i16);
639 }
640
641 bool isVSrc_b16() const { return isVCSrc_b16() || isLiteralImm(MVT::i16); }
642
643 bool isVSrc_v2b16() const { return isVSrc_b16() || isLiteralImm(MVT::v2i16); }
644
645 bool isVCSrcV2FP32() const { return isVCSrc_f64(); }
646
647 bool isVSrc_v2f32() const { return isVSrc_f64() || isLiteralImm(MVT::v2f32); }
648
649 bool isVCSrc_v2b32() const { return isVCSrc_b64(); }
650
651 bool isVSrc_v2b32() const { return isVSrc_b64() || isLiteralImm(MVT::v2i32); }
652
653 bool isVSrc_f32() const {
654 return isVCSrc_f32() || isLiteralImm(MVT::f32) || isExpr();
655 }
656
657 bool isVSrc_f64() const {
658 return isRegOrInlineNoModsTarget(AMDGPU::VS_64_AlignTarget, MVT::f64) ||
659 isLiteralImm(MVT::f64);
660 }
661
662 bool isVSrcT_bf16() const {
663 return isVCSrcTBF16() || isLiteralImm(MVT::bf16);
664 }
665
666 bool isVSrcT_f16() const { return isVCSrcT_f16() || isLiteralImm(MVT::f16); }
667
668 bool isVSrcT_bf16_Lo128() const {
669 return isVCSrcTBF16_Lo128() || isLiteralImm(MVT::bf16);
670 }
671
672 bool isVSrcT_f16_Lo128() const {
673 return isVCSrcTF16_Lo128() || isLiteralImm(MVT::f16);
674 }
675
676 bool isVSrcFake16_bf16_Lo128() const {
677 return isVCSrcFake16BF16_Lo128() || isLiteralImm(MVT::bf16);
678 }
679
680 bool isVSrcFake16_f16_Lo128() const {
681 return isVCSrcFake16F16_Lo128() || isLiteralImm(MVT::f16);
682 }
683
684 bool isVSrc_bf16() const { return isVCSrc_bf16() || isLiteralImm(MVT::bf16); }
685
686 bool isVSrc_f16() const { return isVCSrc_f16() || isLiteralImm(MVT::f16); }
687
688 bool isVSrc_v2bf16() const {
689 return isVSrc_bf16() || isLiteralImm(MVT::v2bf16);
690 }
691
692 bool isVSrc_v2f16() const { return isVSrc_f16() || isLiteralImm(MVT::v2f16); }
693
694 bool isVSrc_v2f16_splat() const { return isVSrc_v2f16(); }
695
696 bool isVSrc_NoInline_v2f16() const { return isVSrc_v2f16(); }
697
698 bool isVISrcB32() const {
699 return isRegOrInlineNoMods(AMDGPU::VGPR_32RegClassID, MVT::i32);
700 }
701
702 bool isVISrcB16() const {
703 return isRegOrInlineNoMods(AMDGPU::VGPR_32RegClassID, MVT::i16);
704 }
705
706 bool isVISrcV2B16() const { return isVISrcB16(); }
707
708 bool isVISrcF32() const {
709 return isRegOrInlineNoMods(AMDGPU::VGPR_32RegClassID, MVT::f32);
710 }
711
712 bool isVISrcF16() const {
713 return isRegOrInlineNoMods(AMDGPU::VGPR_32RegClassID, MVT::f16);
714 }
715
716 bool isVISrcV2F16() const { return isVISrcF16() || isVISrcB32(); }
717
718 bool isVISrc_64_bf16() const {
719 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::bf16);
720 }
721
722 bool isVISrc_64_f16() const {
723 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::f16);
724 }
725
726 bool isVISrc_64_b32() const {
727 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::i32);
728 }
729
730 bool isVISrc_64B64() const {
731 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::i64);
732 }
733
734 bool isVISrc_64_f64() const {
735 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::f64);
736 }
737
738 bool isVISrc_64V2FP32() const {
739 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::f32);
740 }
741
742 bool isVISrc_64V2INT32() const {
743 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::i32);
744 }
745
746 bool isVISrc_256_b32() const {
747 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::i32);
748 }
749
750 bool isVISrc_256_f32() const {
751 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::f32);
752 }
753
754 bool isVISrc_256B64() const {
755 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::i64);
756 }
757
758 bool isVISrc_256_f64() const {
759 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::f64);
760 }
761
762 bool isVISrc_512_f64() const {
763 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::f64);
764 }
765
766 bool isVISrc_128B16() const {
767 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::i16);
768 }
769
770 bool isVISrc_128V2B16() const { return isVISrc_128B16(); }
771
772 bool isVISrc_128_b32() const {
773 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::i32);
774 }
775
776 bool isVISrc_128_f32() const {
777 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::f32);
778 }
779
780 bool isVISrc_256V2FP32() const {
781 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::f32);
782 }
783
784 bool isVISrc_256V2INT32() const {
785 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::i32);
786 }
787
788 bool isVISrc_512_b32() const {
789 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::i32);
790 }
791
792 bool isVISrc_512B16() const {
793 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::i16);
794 }
795
796 bool isVISrc_512V2B16() const { return isVISrc_512B16(); }
797
798 bool isVISrc_512_f32() const {
799 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::f32);
800 }
801
802 bool isVISrc_512F16() const {
803 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::f16);
804 }
805
806 bool isVISrc_512V2F16() const {
807 return isVISrc_512F16() || isVISrc_512_b32();
808 }
809
810 bool isVISrc_1024_b32() const {
811 return isRegOrInlineNoModsTarget(AMDGPU::VReg_1024_AlignTarget, MVT::i32);
812 }
813
814 bool isVISrc_1024B16() const {
815 return isRegOrInlineNoModsTarget(AMDGPU::VReg_1024_AlignTarget, MVT::i16);
816 }
817
818 bool isVISrc_1024V2B16() const { return isVISrc_1024B16(); }
819
820 bool isVISrc_1024_f32() const {
821 return isRegOrInlineNoModsTarget(AMDGPU::VReg_1024_AlignTarget, MVT::f32);
822 }
823
824 bool isVISrc_1024F16() const {
825 return isRegOrInlineNoModsTarget(AMDGPU::VReg_1024_AlignTarget, MVT::f16);
826 }
827
828 bool isVISrc_1024V2F16() const {
829 return isVISrc_1024F16() || isVISrc_1024_b32();
830 }
831
832 bool isAISrcB32() const {
833 return isRegOrInlineNoMods(AMDGPU::AGPR_32RegClassID, MVT::i32);
834 }
835
836 bool isAISrcB16() const {
837 return isRegOrInlineNoMods(AMDGPU::AGPR_32RegClassID, MVT::i16);
838 }
839
840 bool isAISrcV2B16() const { return isAISrcB16(); }
841
842 bool isAISrcF32() const {
843 return isRegOrInlineNoMods(AMDGPU::AGPR_32RegClassID, MVT::f32);
844 }
845
846 bool isAISrcF16() const {
847 return isRegOrInlineNoMods(AMDGPU::AGPR_32RegClassID, MVT::f16);
848 }
849
850 bool isAISrcV2F16() const { return isAISrcF16() || isAISrcB32(); }
851
852 bool isAISrc_64B64() const {
853 return isRegOrInlineNoModsTarget(AMDGPU::AReg_64_AlignTarget, MVT::i64);
854 }
855
856 bool isAISrc_64_f64() const {
857 return isRegOrInlineNoModsTarget(AMDGPU::AReg_64_AlignTarget, MVT::f64);
858 }
859
860 bool isAISrc_128_b32() const {
861 return isRegOrInlineNoModsTarget(AMDGPU::AReg_128_AlignTarget, MVT::i32);
862 }
863
864 bool isAISrc_128B16() const {
865 return isRegOrInlineNoModsTarget(AMDGPU::AReg_128_AlignTarget, MVT::i16);
866 }
867
868 bool isAISrc_128V2B16() const { return isAISrc_128B16(); }
869
870 bool isAISrc_128_f32() const {
871 return isRegOrInlineNoModsTarget(AMDGPU::AReg_128_AlignTarget, MVT::f32);
872 }
873
874 bool isAISrc_128F16() const {
875 return isRegOrInlineNoModsTarget(AMDGPU::AReg_128_AlignTarget, MVT::f16);
876 }
877
878 bool isAISrc_128V2F16() const {
879 return isAISrc_128F16() || isAISrc_128_b32();
880 }
881
882 bool isVISrc_128_bf16() const {
883 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::bf16);
884 }
885
886 bool isVISrc_128_f16() const {
887 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::f16);
888 }
889
890 bool isVISrc_128V2F16() const {
891 return isVISrc_128_f16() || isVISrc_128_b32();
892 }
893
894 bool isAISrc_256B64() const {
895 return isRegOrInlineNoModsTarget(AMDGPU::AReg_256_AlignTarget, MVT::i64);
896 }
897
898 bool isAISrc_256_f64() const {
899 return isRegOrInlineNoModsTarget(AMDGPU::AReg_256_AlignTarget, MVT::f64);
900 }
901
902 bool isAISrc_512_b32() const {
903 return isRegOrInlineNoModsTarget(AMDGPU::AReg_512_AlignTarget, MVT::i32);
904 }
905
906 bool isAISrc_512B16() const {
907 return isRegOrInlineNoModsTarget(AMDGPU::AReg_512_AlignTarget, MVT::i16);
908 }
909
910 bool isAISrc_512V2B16() const { return isAISrc_512B16(); }
911
912 bool isAISrc_512_f32() const {
913 return isRegOrInlineNoModsTarget(AMDGPU::AReg_512_AlignTarget, MVT::f32);
914 }
915
916 bool isAISrc_512F16() const {
917 return isRegOrInlineNoModsTarget(AMDGPU::AReg_512_AlignTarget, MVT::f16);
918 }
919
920 bool isAISrc_512V2F16() const {
921 return isAISrc_512F16() || isAISrc_512_b32();
922 }
923
924 bool isAISrc_1024_b32() const {
925 return isRegOrInlineNoModsTarget(AMDGPU::AReg_1024_AlignTarget, MVT::i32);
926 }
927
928 bool isAISrc_1024B16() const {
929 return isRegOrInlineNoModsTarget(AMDGPU::AReg_1024_AlignTarget, MVT::i16);
930 }
931
932 bool isAISrc_1024V2B16() const { return isAISrc_1024B16(); }
933
934 bool isAISrc_1024_f32() const {
935 return isRegOrInlineNoModsTarget(AMDGPU::AReg_1024_AlignTarget, MVT::f32);
936 }
937
938 bool isAISrc_1024F16() const {
939 return isRegOrInlineNoModsTarget(AMDGPU::AReg_1024_AlignTarget, MVT::f16);
940 }
941
942 bool isAISrc_1024V2F16() const {
943 return isAISrc_1024F16() || isAISrc_1024_b32();
944 }
945
946 bool isKImmFP32() const { return isLiteralImm(MVT::f32); }
947
948 bool isKImmFP16() const { return isLiteralImm(MVT::f16); }
949
950 bool isKImmFP64() const { return isLiteralImm(MVT::f64); }
951
952 bool isMem() const override { return false; }
953
954 bool isExpr() const { return Kind == Expression; }
955
956 bool isSOPPBrTarget() const { return isExpr() || isImm(); }
957
958 bool isSWaitCnt() const;
959 bool isDepCtr() const;
960 bool isSDelayALU() const;
961 bool isHwreg() const;
962 bool isSendMsg() const;
963 bool isWaitEvent() const;
964 bool isSplitBarrier() const;
965 bool isSwizzle() const;
966 bool isSMRDOffset8() const;
967 bool isSMEMOffset() const;
968 bool isSMRDLiteralOffset() const;
969 bool isDPP8() const;
970 bool isDPPCtrl() const;
971 bool isBLGP() const;
972 bool isGPRIdxMode() const;
973 bool isS16Imm() const;
974 bool isU16Imm() const;
975 bool isEndpgm() const;
976
977 auto getPredicate(std::function<bool(const AMDGPUOperand &Op)> P) const {
978 return [this, P]() { return P(*this); };
979 }
980
981 StringRef getToken() const {
982 assert(isToken());
983 return StringRef(Tok.Data, Tok.Length);
984 }
985
986 int64_t getImm() const {
987 assert(isImm());
988 return Imm.Val;
989 }
990
991 void setImm(int64_t Val) {
992 assert(isImm());
993 Imm.Val = Val;
994 }
995
996 ImmTy getImmTy() const {
997 assert(isImm());
998 return Imm.Type;
999 }
1000
1001 MCRegister getReg() const override {
1002 assert(isRegKind());
1003 return Reg.RegNo;
1004 }
1005
1006 SMLoc getStartLoc() const override { return StartLoc; }
1007
1008 SMLoc getEndLoc() const override { return EndLoc; }
1009
1010 SMRange getLocRange() const { return SMRange(StartLoc, EndLoc); }
1011
1012 int getMCOpIdx() const { return MCOpIdx; }
1013
1014 Modifiers getModifiers() const {
1015 assert(isRegKind() || isImmTy(ImmTyNone));
1016 return isRegKind() ? Reg.Mods : Imm.Mods;
1017 }
1018
1019 void setModifiers(Modifiers Mods) {
1020 assert(isRegKind() || isImmTy(ImmTyNone));
1021 if (isRegKind())
1022 Reg.Mods = Mods;
1023 else
1024 Imm.Mods = Mods;
1025 }
1026
1027 bool hasModifiers() const { return getModifiers().hasModifiers(); }
1028
1029 bool hasFPModifiers() const { return getModifiers().hasFPModifiers(); }
1030
1031 bool hasIntModifiers() const { return getModifiers().hasIntModifiers(); }
1032
1033 bool isForcedLit() const {
1034 return isImmLiteral() && getModifiers().isForcedLit();
1035 }
1036
1037 bool isForcedLit64() const {
1038 return isImmLiteral() && getModifiers().isForcedLit64();
1039 }
1040
1041 uint64_t applyInputFPModifiers(uint64_t Val, unsigned Size) const;
1042
1043 void addImmOperands(MCInst &Inst, unsigned N,
1044 bool ApplyModifiers = true) const;
1045
1046 void addLiteralImmOperand(MCInst &Inst, int64_t Val,
1047 bool ApplyModifiers) const;
1048
1049 void addRegOperands(MCInst &Inst, unsigned N) const;
1050
1051 void addRegOrImmOperands(MCInst &Inst, unsigned N) const {
1052 if (isRegKind())
1053 addRegOperands(Inst, N);
1054 else
1055 addImmOperands(Inst, N);
1056 }
1057
1058 void addRegOrImmWithInputModsOperands(MCInst &Inst, unsigned N) const {
1059 Modifiers Mods = getModifiers();
1060 Inst.addOperand(MCOperand::createImm(Mods.getModifiersOperand()));
1061 if (isRegKind()) {
1062 addRegOperands(Inst, N);
1063 } else {
1064 addImmOperands(Inst, N, false);
1065 }
1066 }
1067
1068 void addRegOrImmWithFPInputModsOperands(MCInst &Inst, unsigned N) const {
1069 assert(!hasIntModifiers());
1070 addRegOrImmWithInputModsOperands(Inst, N);
1071 }
1072
1073 void addRegOrImmWithIntInputModsOperands(MCInst &Inst, unsigned N) const {
1074 assert(!hasFPModifiers());
1075 addRegOrImmWithInputModsOperands(Inst, N);
1076 }
1077
1078 void addRegWithInputModsOperands(MCInst &Inst, unsigned N) const {
1079 Modifiers Mods = getModifiers();
1080 Inst.addOperand(MCOperand::createImm(Mods.getModifiersOperand()));
1081 assert(isRegKind());
1082 addRegOperands(Inst, N);
1083 }
1084
1085 void addRegWithFPInputModsOperands(MCInst &Inst, unsigned N) const {
1086 assert(!hasIntModifiers());
1087 addRegWithInputModsOperands(Inst, N);
1088 }
1089
1090 void addRegWithIntInputModsOperands(MCInst &Inst, unsigned N) const {
1091 assert(!hasFPModifiers());
1092 addRegWithInputModsOperands(Inst, N);
1093 }
1094
1095 static void printImmTy(raw_ostream &OS, ImmTy Type) {
1096 // clang-format off
1097 switch (Type) {
1098 case ImmTyNone: OS << "None"; break;
1099 case ImmTyGDS: OS << "GDS"; break;
1100 case ImmTyLDS: OS << "LDS"; break;
1101 case ImmTyOffen: OS << "Offen"; break;
1102 case ImmTyIdxen: OS << "Idxen"; break;
1103 case ImmTyAddr64: OS << "Addr64"; break;
1104 case ImmTyOffset: OS << "Offset"; break;
1105 case ImmTyInstOffset: OS << "InstOffset"; break;
1106 case ImmTyOffset0: OS << "Offset0"; break;
1107 case ImmTyOffset1: OS << "Offset1"; break;
1108 case ImmTySMEMOffsetMod: OS << "SMEMOffsetMod"; break;
1109 case ImmTyCPol: OS << "CPol"; break;
1110 case ImmTyIndexKey8bit: OS << "index_key"; break;
1111 case ImmTyIndexKey16bit: OS << "index_key"; break;
1112 case ImmTyIndexKey32bit: OS << "index_key"; break;
1113 case ImmTyTFE: OS << "TFE"; break;
1114 case ImmTyIsAsync: OS << "IsAsync"; break;
1115 case ImmTyD16: OS << "D16"; break;
1116 case ImmTyFORMAT: OS << "FORMAT"; break;
1117 case ImmTyClamp: OS << "Clamp"; break;
1118 case ImmTyOModSI: OS << "OModSI"; break;
1119 case ImmTyDPP8: OS << "DPP8"; break;
1120 case ImmTyDppCtrl: OS << "DppCtrl"; break;
1121 case ImmTyDppRowMask: OS << "DppRowMask"; break;
1122 case ImmTyDppBankMask: OS << "DppBankMask"; break;
1123 case ImmTyDppBoundCtrl: OS << "DppBoundCtrl"; break;
1124 case ImmTyDppFI: OS << "DppFI"; break;
1125 case ImmTySDWADstSel: OS << "SDWADstSel"; break;
1126 case ImmTySDWASrc0Sel: OS << "SDWASrc0Sel"; break;
1127 case ImmTySDWASrc1Sel: OS << "SDWASrc1Sel"; break;
1128 case ImmTySDWADstUnused: OS << "SDWADstUnused"; break;
1129 case ImmTyDMask: OS << "DMask"; break;
1130 case ImmTyDim: OS << "Dim"; break;
1131 case ImmTyUNorm: OS << "UNorm"; break;
1132 case ImmTyDA: OS << "DA"; break;
1133 case ImmTyR128A16: OS << "R128A16"; break;
1134 case ImmTyA16: OS << "A16"; break;
1135 case ImmTyLWE: OS << "LWE"; break;
1136 case ImmTyOff: OS << "Off"; break;
1137 case ImmTyExpTgt: OS << "ExpTgt"; break;
1138 case ImmTyExpCompr: OS << "ExpCompr"; break;
1139 case ImmTyExpVM: OS << "ExpVM"; break;
1140 case ImmTyDone: OS << "Done"; break;
1141 case ImmTyRowEn: OS << "RowEn"; break;
1142 case ImmTyHwreg: OS << "Hwreg"; break;
1143 case ImmTySendMsg: OS << "SendMsg"; break;
1144 case ImmTyWaitEvent: OS << "WaitEvent"; break;
1145 case ImmTyInterpSlot: OS << "InterpSlot"; break;
1146 case ImmTyInterpAttr: OS << "InterpAttr"; break;
1147 case ImmTyInterpAttrChan: OS << "InterpAttrChan"; break;
1148 case ImmTyOpSel: OS << "OpSel"; break;
1149 case ImmTyOpSelHi: OS << "OpSelHi"; break;
1150 case ImmTyNegLo: OS << "NegLo"; break;
1151 case ImmTyNegHi: OS << "NegHi"; break;
1152 case ImmTySwizzle: OS << "Swizzle"; break;
1153 case ImmTyGprIdxMode: OS << "GprIdxMode"; break;
1154 case ImmTyHigh: OS << "High"; break;
1155 case ImmTyBLGP: OS << "BLGP"; break;
1156 case ImmTyCBSZ: OS << "CBSZ"; break;
1157 case ImmTyABID: OS << "ABID"; break;
1158 case ImmTyEndpgm: OS << "Endpgm"; break;
1159 case ImmTyWaitVDST: OS << "WaitVDST"; break;
1160 case ImmTyWaitEXP: OS << "WaitEXP"; break;
1161 case ImmTyWaitVAVDst: OS << "WaitVAVDst"; break;
1162 case ImmTyWaitVMVSrc: OS << "WaitVMVSrc"; break;
1163 case ImmTyBitOp3: OS << "BitOp3"; break;
1164 case ImmTyMatrixAFMT: OS << "ImmTyMatrixAFMT"; break;
1165 case ImmTyMatrixBFMT: OS << "ImmTyMatrixBFMT"; break;
1166 case ImmTyMatrixAScale: OS << "ImmTyMatrixAScale"; break;
1167 case ImmTyMatrixBScale: OS << "ImmTyMatrixBScale"; break;
1168 case ImmTyMatrixAScaleFmt: OS << "ImmTyMatrixAScaleFmt"; break;
1169 case ImmTyMatrixBScaleFmt: OS << "ImmTyMatrixBScaleFmt"; break;
1170 case ImmTyMatrixAReuse: OS << "ImmTyMatrixAReuse"; break;
1171 case ImmTyMatrixBReuse: OS << "ImmTyMatrixBReuse"; break;
1172 case ImmTyScaleSel: OS << "ScaleSel" ; break;
1173 case ImmTyByteSel: OS << "ByteSel" ; break;
1174 }
1175 // clang-format on
1176 }
1177
1178 void print(raw_ostream &OS, const MCAsmInfo &MAI) const override {
1179 switch (Kind) {
1180 case Register:
1181 OS << "<register " << AMDGPUInstPrinter::getRegisterName(getReg())
1182 << " mods: " << Reg.Mods << '>';
1183 break;
1184 case Immediate:
1185 OS << '<' << getImm();
1186 if (getImmTy() != ImmTyNone) {
1187 OS << " type: ";
1188 printImmTy(OS, getImmTy());
1189 }
1190 OS << " mods: " << Imm.Mods << '>';
1191 break;
1192 case Token:
1193 OS << '\'' << getToken() << '\'';
1194 break;
1195 case Expression:
1196 OS << "<expr ";
1197 MAI.printExpr(OS, *Expr);
1198 OS << '>';
1199 break;
1200 }
1201 }
1202
1203 static AMDGPUOperand::Ptr CreateImm(const AMDGPUAsmParser *AsmParser,
1204 int64_t Val, SMLoc Loc,
1205 ImmTy Type = ImmTyNone,
1206 bool IsFPImm = false) {
1207 auto Op = std::make_unique<AMDGPUOperand>(Immediate, AsmParser);
1208 Op->Imm.Val = Val;
1209 Op->Imm.IsFPImm = IsFPImm;
1210 Op->Imm.Type = Type;
1211 Op->Imm.Mods = Modifiers();
1212 Op->StartLoc = Loc;
1213 Op->EndLoc = Loc;
1214 return Op;
1215 }
1216
1217 static AMDGPUOperand::Ptr CreateToken(const AMDGPUAsmParser *AsmParser,
1218 StringRef Str, SMLoc Loc,
1219 bool HasExplicitEncodingSize = true) {
1220 auto Res = std::make_unique<AMDGPUOperand>(Token, AsmParser);
1221 Res->Tok.Data = Str.data();
1222 Res->Tok.Length = Str.size();
1223 Res->StartLoc = Loc;
1224 Res->EndLoc = Loc;
1225 return Res;
1226 }
1227
1228 static AMDGPUOperand::Ptr CreateReg(const AMDGPUAsmParser *AsmParser,
1229 MCRegister Reg, SMLoc S, SMLoc E) {
1230 auto Op = std::make_unique<AMDGPUOperand>(Register, AsmParser);
1231 Op->Reg.RegNo = Reg;
1232 Op->Reg.Mods = Modifiers();
1233 Op->StartLoc = S;
1234 Op->EndLoc = E;
1235 return Op;
1236 }
1237
1238 static AMDGPUOperand::Ptr CreateExpr(const AMDGPUAsmParser *AsmParser,
1239 const class MCExpr *Expr, SMLoc S) {
1240 auto Op = std::make_unique<AMDGPUOperand>(Expression, AsmParser);
1241 Op->Expr = Expr;
1242 Op->StartLoc = S;
1243 Op->EndLoc = S;
1244 return Op;
1245 }
1246};
1247
1248raw_ostream &operator<<(raw_ostream &OS, AMDGPUOperand::Modifiers Mods) {
1249 OS << "abs:" << Mods.Abs << " neg: " << Mods.Neg << " sext:" << Mods.Sext;
1250 return OS;
1251}
1252
1253//===----------------------------------------------------------------------===//
1254// AsmParser
1255//===----------------------------------------------------------------------===//
1256
1257// TODO: define GET_SUBTARGET_FEATURE_NAME
1258#define GET_REGISTER_MATCHER
1259#include "AMDGPUGenAsmMatcher.inc"
1260#undef GET_REGISTER_MATCHER
1261#undef GET_SUBTARGET_FEATURE_NAME
1262
1263// Holds info related to the current kernel, e.g. count of SGPRs used.
1264// Kernel scope begins at .amdgpu_hsa_kernel directive, ends at next
1265// .amdgpu_hsa_kernel or at EOF.
1266class KernelScopeInfo {
1267 int SgprIndexUnusedMin = -1;
1268 int VgprIndexUnusedMin = -1;
1269 int AgprIndexUnusedMin = -1;
1270 MCContext *Ctx = nullptr;
1271 MCSubtargetInfo const *MSTI = nullptr;
1272
1273 void usesSgprAt(int i) {
1274 if (i >= SgprIndexUnusedMin) {
1275 SgprIndexUnusedMin = ++i;
1276 if (Ctx) {
1277 MCSymbol *const Sym =
1278 Ctx->getOrCreateSymbol(Twine(".kernel.sgpr_count"));
1279 Sym->setVariableValue(MCConstantExpr::create(SgprIndexUnusedMin, *Ctx));
1280 }
1281 }
1282 }
1283
1284 void usesVgprAt(int i) {
1285 if (i >= VgprIndexUnusedMin) {
1286 VgprIndexUnusedMin = ++i;
1287 if (Ctx) {
1288 MCSymbol *const Sym =
1289 Ctx->getOrCreateSymbol(Twine(".kernel.vgpr_count"));
1290 int totalVGPR = getTotalNumVGPRs(isGFX90A(*MSTI), AgprIndexUnusedMin,
1291 VgprIndexUnusedMin);
1292 Sym->setVariableValue(MCConstantExpr::create(totalVGPR, *Ctx));
1293 }
1294 }
1295 }
1296
1297 void usesAgprAt(int i) {
1298 // Instruction will error in AMDGPUAsmParser::matchAndEmitInstruction
1299 if (!hasMAIInsts(*MSTI))
1300 return;
1301
1302 if (i >= AgprIndexUnusedMin) {
1303 AgprIndexUnusedMin = ++i;
1304 if (Ctx) {
1305 MCSymbol *const Sym =
1306 Ctx->getOrCreateSymbol(Twine(".kernel.agpr_count"));
1307 Sym->setVariableValue(MCConstantExpr::create(AgprIndexUnusedMin, *Ctx));
1308
1309 // Also update vgpr_count (dependent on agpr_count for gfx908/gfx90a)
1310 MCSymbol *const vSym =
1311 Ctx->getOrCreateSymbol(Twine(".kernel.vgpr_count"));
1312 int totalVGPR = getTotalNumVGPRs(isGFX90A(*MSTI), AgprIndexUnusedMin,
1313 VgprIndexUnusedMin);
1314 vSym->setVariableValue(MCConstantExpr::create(totalVGPR, *Ctx));
1315 }
1316 }
1317 }
1318
1319public:
1320 KernelScopeInfo() = default;
1321
1322 void initialize(MCContext &Context) {
1323 Ctx = &Context;
1324 MSTI = Ctx->getSubtargetInfo();
1325
1326 usesSgprAt(SgprIndexUnusedMin = -1);
1327 usesVgprAt(VgprIndexUnusedMin = -1);
1328 if (hasMAIInsts(*MSTI)) {
1329 usesAgprAt(AgprIndexUnusedMin = -1);
1330 }
1331 }
1332
1333 void usesRegister(RegisterKind RegKind, unsigned DwordRegIndex,
1334 unsigned RegWidth) {
1335 switch (RegKind) {
1336 case IS_SGPR:
1337 usesSgprAt(DwordRegIndex + divideCeil(RegWidth, 32) - 1);
1338 break;
1339 case IS_AGPR:
1340 usesAgprAt(DwordRegIndex + divideCeil(RegWidth, 32) - 1);
1341 break;
1342 case IS_VGPR:
1343 usesVgprAt(DwordRegIndex + divideCeil(RegWidth, 32) - 1);
1344 break;
1345 default:
1346 break;
1347 }
1348 }
1349};
1350
1351class AMDGPUAsmParser : public MCTargetAsmParser {
1352 MCAsmParser &Parser;
1353
1354 unsigned ForcedEncodingSize = 0;
1355 bool ForcedDPP = false;
1356 bool ForcedSDWA = false;
1357 KernelScopeInfo KernelScope;
1358 const unsigned HwMode;
1359 const AMDGPU::GPUKind Gfx;
1360 const AMDGPU::IsaVersion ISA;
1361
1362 /// @name Auto-generated Match Functions
1363 /// {
1364
1365#define GET_ASSEMBLER_HEADER
1366#include "AMDGPUGenAsmMatcher.inc"
1367
1368 /// }
1369
1370 /// Get size of register operand
1371 unsigned getRegOperandSize(const MCInstrDesc &Desc, unsigned OpNo) const {
1372 assert(OpNo < Desc.NumOperands);
1373 int16_t RCID = MII.getOpRegClassID(Desc.operands()[OpNo], HwMode);
1374 return getRegBitWidth(RCID) / 8;
1375 }
1376
1377 std::optional<AMDGPU::InfoSectionData> InfoData;
1378
1379 /// Whether the leading .amdgcn_target directive has been emitted to the
1380 /// output streamer yet. The emission is deferred until the first piece of
1381 /// content (instruction or kernel descriptor) so that any leading
1382 /// .amdgcn_target/.amd_amdgpu_isa directive in the source has had a chance to
1383 /// update the target ID first.
1384 bool TargetDirectiveEmitted = false;
1385
1386 /// State for checking that every kernel named in a .amdhsa_kernel directive
1387 /// begins with the required prologue instruction sequence. Because the
1388 /// directive may appear either before or after the kernel's label (it is
1389 /// normally emitted after the function body, in .rodata), validation is
1390 /// deferred to onEndOfFile(). We record an order-independent timeline of
1391 /// parsed labels and emitted instruction opcodes, plus the set of symbols
1392 /// named by .amdhsa_kernel directives, and match them up at end of file.
1393 SmallVector<unsigned> OpcodeStream;
1395 OpcodeStreamSymbols;
1396 SmallPtrSet<const MCSymbol *, 8> AMDHSAKernelSymbols;
1397
1398 /// Verify recorded kernel prologues.
1399 void checkKernelPrologues();
1400
1401private:
1402 void createConstantSymbol(StringRef Id, int64_t Val);
1403
1404 bool ParseAsAbsoluteExpression(uint32_t &Ret);
1405 bool OutOfRangeError(SMRange Range);
1406 /// Calculate VGPR/SGPR blocks required for given target, reserved
1407 /// registers, and user-specified NextFreeXGPR values.
1408 ///
1409 /// \param Features [in] Target features, used for bug corrections.
1410 /// \param VCCUsed [in] Whether VCC special SGPR is reserved.
1411 /// \param FlatScrUsed [in] Whether FLAT_SCRATCH special SGPR is reserved.
1412 /// \param XNACKUsed [in] Whether XNACK_MASK special SGPR is reserved.
1413 /// \param EnableWavefrontSize32 [in] Value of ENABLE_WAVEFRONT_SIZE32 kernel
1414 /// descriptor field, if valid.
1415 /// \param NextFreeVGPR [in] Max VGPR number referenced, plus one.
1416 /// \param VGPRRange [in] Token range, used for VGPR diagnostics.
1417 /// \param NextFreeSGPR [in] Max SGPR number referenced, plus one.
1418 /// \param SGPRRange [in] Token range, used for SGPR diagnostics.
1419 /// \param VGPRBlocks [out] Result VGPR block count.
1420 /// \param SGPRBlocks [out] Result SGPR block count.
1421 bool calculateGPRBlocks(const FeatureBitset &Features, const MCExpr *VCCUsed,
1422 const MCExpr *FlatScrUsed, bool XNACKUsed,
1423 std::optional<bool> EnableWavefrontSize32,
1424 const MCExpr *NextFreeVGPR, SMRange VGPRRange,
1425 const MCExpr *NextFreeSGPR, SMRange SGPRRange,
1426 const MCExpr *&VGPRBlocks, const MCExpr *&SGPRBlocks);
1427 bool ParseDirectiveAMDGCNTarget();
1428 bool ParseDirectiveAMDHSACodeObjectVersion();
1429 bool ParseDirectiveAMDHSAKernel();
1430 bool ParseAMDKernelCodeTValue(StringRef ID, AMDGPUMCKernelCodeT &Header);
1431 bool ParseDirectiveAMDKernelCodeT();
1432 // TODO: Possibly make subtargetHasRegister const.
1433 bool subtargetHasRegister(const MCRegisterInfo &MRI, MCRegister Reg);
1434 bool ParseDirectiveAMDGPUHsaKernel();
1435
1436 bool ParseDirectiveISAVersion();
1437 bool ParseDirectiveHSAMetadata();
1438 bool ParseDirectivePALMetadataBegin();
1439 bool ParseDirectivePALMetadata();
1440 bool ParseDirectiveAMDGPULDS();
1441 bool ParseDirectiveAMDGPUInfo();
1442
1443 /// Common code to parse out a block of text (typically YAML) between start
1444 /// and end directives.
1445 bool ParseToEndDirective(const char *AssemblerDirectiveBegin,
1446 const char *AssemblerDirectiveEnd,
1447 std::string &CollectString);
1448
1449 bool AddNextRegisterToList(MCRegister &Reg, unsigned &RegWidth,
1450 RegisterKind RegKind, MCRegister Reg1,
1451 RegisterKind RegKind1, SMLoc Loc);
1452 bool ParseAMDGPURegister(RegisterKind &RegKind, MCRegister &Reg,
1453 unsigned &RegNum, unsigned &RegWidth,
1454 bool RestoreOnFailure = false);
1455 bool ParseAMDGPURegister(RegisterKind &RegKind, MCRegister &Reg,
1456 unsigned &RegNum, unsigned &RegWidth,
1457 SmallVectorImpl<AsmToken> &Tokens);
1458 MCRegister ParseRegularReg(RegisterKind &RegKind, unsigned &RegNum,
1459 unsigned &RegWidth,
1460 SmallVectorImpl<AsmToken> &Tokens);
1461 MCRegister ParseSpecialReg(RegisterKind &RegKind, unsigned &RegNum,
1462 unsigned &RegWidth,
1463 SmallVectorImpl<AsmToken> &Tokens);
1464 MCRegister ParseRegList(RegisterKind &RegKind, unsigned &RegNum,
1465 unsigned &RegWidth,
1466 SmallVectorImpl<AsmToken> &Tokens);
1467 bool ParseRegRange(unsigned &Num, unsigned &Width, unsigned &SubReg);
1468 MCRegister getRegularReg(RegisterKind RegKind, unsigned RegNum,
1469 unsigned SubReg, unsigned RegWidth, SMLoc Loc);
1470
1471 bool isRegister();
1472 bool isRegister(const AsmToken &Token, const AsmToken &NextToken) const;
1473 std::optional<StringRef> getGprCountSymbolName(RegisterKind RegKind);
1474 void initializeGprCountSymbol(RegisterKind RegKind);
1475 bool updateGprCountSymbols(RegisterKind RegKind, unsigned DwordRegIndex,
1476 unsigned RegWidth);
1477 void cvtMubufImpl(MCInst &Inst, const OperandVector &Operands, bool IsAtomic);
1478
1479public:
1480 enum OperandMode {
1481 OperandMode_Default,
1482 OperandMode_NSA,
1483 };
1484
1485 using OptionalImmIndexMap = std::map<AMDGPUOperand::ImmTy, unsigned>;
1486
1487 AMDGPUAsmParser(const MCSubtargetInfo &STI, MCAsmParser &_Parser,
1488 const MCInstrInfo &MII)
1489 : MCTargetAsmParser(STI, MII), Parser(_Parser),
1490 HwMode(STI.getHwMode(MCSubtargetInfo::HwMode_RegInfo)),
1491 Gfx(AMDGPU::parseArchAMDGCN(STI.getCPU())),
1492 ISA(AMDGPU::getIsaVersion(STI.getCPU())) {
1494
1495 setAvailableFeatures(ComputeAvailableFeatures(getFeatureBits()));
1496
1497 if (ISA.Major >= 6 && isHsaAbi(getSTI())) {
1498 createConstantSymbol(".amdgcn.gfx_generation_number", ISA.Major);
1499 createConstantSymbol(".amdgcn.gfx_generation_minor", ISA.Minor);
1500 createConstantSymbol(".amdgcn.gfx_generation_stepping", ISA.Stepping);
1501 } else {
1502 createConstantSymbol(".option.machine_version_major", ISA.Major);
1503 createConstantSymbol(".option.machine_version_minor", ISA.Minor);
1504 createConstantSymbol(".option.machine_version_stepping", ISA.Stepping);
1505 }
1506 if (ISA.Major >= 6 && isHsaAbi(getSTI())) {
1507 initializeGprCountSymbol(IS_VGPR);
1508 initializeGprCountSymbol(IS_SGPR);
1509 } else
1510 KernelScope.initialize(getContext());
1511
1512 for (auto [Symbol, Code] : AMDGPU::UCVersion::getGFXVersions())
1513 createConstantSymbol(Symbol, Code);
1514
1515 createConstantSymbol("UC_VERSION_W64_BIT", 0x2000);
1516 createConstantSymbol("UC_VERSION_W32_BIT", 0x4000);
1517 createConstantSymbol("UC_VERSION_MDP_BIT", 0x8000);
1518 }
1519
1520 bool hasMIMG_R128() const { return AMDGPU::hasMIMG_R128(getSTI()); }
1521
1522 bool hasPackedD16() const { return AMDGPU::hasPackedD16(getSTI()); }
1523
1524 bool hasA16() const { return AMDGPU::hasA16(getSTI()); }
1525
1526 bool hasG16() const { return AMDGPU::hasG16(getSTI()); }
1527
1528 bool hasGDS() const { return AMDGPU::hasGDS(getSTI()); }
1529
1530 bool isSI() const { return AMDGPU::isSI(getSTI()); }
1531
1532 bool isCI() const { return AMDGPU::isCI(getSTI()); }
1533
1534 bool isVI() const { return AMDGPU::isVI(getSTI()); }
1535
1536 bool isGFX9() const { return AMDGPU::isGFX9(getSTI()); }
1537
1538 // TODO: isGFX90A is also true for GFX940. We need to clean it.
1539 bool isGFX90A() const { return AMDGPU::isGFX90A(getSTI()); }
1540
1541 bool isGFX940() const { return AMDGPU::isGFX940(getSTI()); }
1542
1543 bool isGFX9Plus() const { return AMDGPU::isGFX9Plus(getSTI()); }
1544
1545 bool isGFX10() const { return AMDGPU::isGFX10(getSTI()); }
1546
1547 bool isGFX10Plus() const { return AMDGPU::isGFX10Plus(getSTI()); }
1548
1549 bool isGFX11() const { return AMDGPU::isGFX11(getSTI()); }
1550
1551 bool isGFX11Plus() const { return AMDGPU::isGFX11Plus(getSTI()); }
1552
1553 bool isGFX12() const { return AMDGPU::isGFX12(getSTI()); }
1554
1555 bool isGFX12Plus() const { return AMDGPU::isGFX12Plus(getSTI()); }
1556
1557 bool isGFX1250() const { return AMDGPU::isGFX1250(getSTI()); }
1558
1559 bool isGFX1250Plus() const { return AMDGPU::isGFX1250Plus(getSTI()); }
1560
1561 bool isGFX13() const { return AMDGPU::isGFX13(getSTI()); }
1562
1563 bool isGFX13Plus() const { return AMDGPU::isGFX13Plus(getSTI()); }
1564
1565 bool hasBVHRayTracingInsts() const {
1566 return getFeatureBits()[AMDGPU::FeatureBVHRayTracingInsts];
1567 }
1568
1569 bool isGFX10_BEncoding() const { return AMDGPU::isGFX10_BEncoding(getSTI()); }
1570
1571 bool isWave32() const { return getAvailableFeatures()[Feature_isWave32Bit]; }
1572
1573 bool isWave64() const { return getAvailableFeatures()[Feature_isWave64Bit]; }
1574
1575 bool hasInv2PiInlineImm() const {
1576 return getFeatureBits()[AMDGPU::FeatureInv2PiInlineImm];
1577 }
1578
1579 bool has64BitLiterals() const {
1580 return getFeatureBits()[AMDGPU::Feature64BitLiterals];
1581 }
1582
1583 bool hasFlatOffsets() const {
1584 return getFeatureBits()[AMDGPU::FeatureFlatInstOffsets];
1585 }
1586
1587 bool hasTrue16Insts() const {
1588 return getFeatureBits()[AMDGPU::FeatureTrue16BitInsts];
1589 }
1590
1591 bool hasArchitectedFlatScratch() const {
1592 return getFeatureBits()[AMDGPU::FeatureArchitectedFlatScratch];
1593 }
1594
1595 bool hasSGPR102_SGPR103() const { return !isVI() && !isGFX9(); }
1596
1597 bool hasSGPR104_SGPR105() const { return isGFX10Plus(); }
1598
1599 bool hasIntClamp() const { return getFeatureBits()[AMDGPU::FeatureIntClamp]; }
1600
1601 bool hasPartialNSAEncoding() const {
1602 return getFeatureBits()[AMDGPU::FeaturePartialNSAEncoding];
1603 }
1604
1605 bool hasGloballyAddressableScratch() const {
1606 return getFeatureBits()[AMDGPU::FeatureGloballyAddressableScratch];
1607 }
1608
1609 unsigned getNSAMaxSize(bool HasSampler = false) const {
1610 return AMDGPU::getNSAMaxSize(getSTI(), HasSampler);
1611 }
1612
1613 unsigned getMaxNumUserSGPRs() const {
1614 return AMDGPU::getMaxNumUserSGPRs(getSTI());
1615 }
1616
1617 bool hasKernargPreload() const { return AMDGPU::hasKernargPreload(getSTI()); }
1618
1619 AMDGPUTargetStreamer &getTargetStreamer() {
1620 MCTargetStreamer &TS = *getParser().getStreamer().getTargetStreamer();
1621 return static_cast<AMDGPUTargetStreamer &>(TS);
1622 }
1623
1624 MCContext &getContext() const {
1625 // We need this const_cast because for some reason getContext() is not const
1626 // in MCAsmParser.
1627 return const_cast<AMDGPUAsmParser *>(this)->MCTargetAsmParser::getContext();
1628 }
1629
1630 const MCRegisterInfo *getMRI() const {
1631 return getContext().getRegisterInfo();
1632 }
1633
1634 const MCInstrInfo *getMII() const { return &MII; }
1635
1636 // Resolve a RegClassByHwModeUses index to a register class id for the active
1637 // HwMode; -1 if the mode has no entry.
1638 int16_t getTargetRegClass(unsigned TargetRCIdx) const {
1639 return MII.getRegClassByHwModeTable(HwMode)[TargetRCIdx];
1640 }
1641
1642 // FIXME: This should not be used. Instead, should use queries derived from
1643 // getAvailableFeatures().
1644 const FeatureBitset &getFeatureBits() const {
1645 return getSTI().getFeatureBits();
1646 }
1647
1648 void setForcedEncodingSize(unsigned Size) { ForcedEncodingSize = Size; }
1649 void setForcedDPP(bool ForceDPP_) { ForcedDPP = ForceDPP_; }
1650 void setForcedSDWA(bool ForceSDWA_) { ForcedSDWA = ForceSDWA_; }
1651
1652 unsigned getForcedEncodingSize() const { return ForcedEncodingSize; }
1653 bool isForcedVOP3() const { return ForcedEncodingSize == 64; }
1654 bool isForcedDPP() const { return ForcedDPP; }
1655 bool isForcedSDWA() const { return ForcedSDWA; }
1656 ArrayRef<unsigned> getMatchedVariants() const;
1657 StringRef getMatchedVariantName() const;
1658
1659 std::unique_ptr<AMDGPUOperand> parseRegister(bool RestoreOnFailure = false);
1660 bool ParseRegister(MCRegister &RegNo, SMLoc &StartLoc, SMLoc &EndLoc,
1661 bool RestoreOnFailure);
1662 bool parseRegister(MCRegister &Reg, SMLoc &StartLoc, SMLoc &EndLoc) override;
1663 ParseStatus tryParseRegister(MCRegister &Reg, SMLoc &StartLoc,
1664 SMLoc &EndLoc) override;
1665 unsigned checkTargetMatchPredicate(MCInst &Inst) override;
1666 unsigned validateTargetOperandClass(MCParsedAsmOperand &Op,
1667 unsigned Kind) override;
1668 bool matchAndEmitInstruction(SMLoc IDLoc, unsigned &Opcode,
1669 OperandVector &Operands, MCStreamer &Out,
1670 uint64_t &ErrorInfo,
1671 bool MatchingInlineAsm) override;
1672 bool ParseDirective(AsmToken DirectiveID) override;
1673 void doBeforeLabelEmit(MCSymbol *Symbol, SMLoc IDLoc) override;
1674 void onEndOfFile() override;
1675 ParseStatus parseOperand(OperandVector &Operands, StringRef Mnemonic,
1676 OperandMode Mode = OperandMode_Default);
1677 StringRef parseMnemonicSuffix(StringRef Name);
1678 bool parseInstruction(ParseInstructionInfo &Info, StringRef Name,
1679 SMLoc NameLoc, OperandVector &Operands) override;
1680 // bool ProcessInstruction(MCInst &Inst);
1681
1682 ParseStatus parseTokenOp(StringRef Name, OperandVector &Operands);
1683
1684 ParseStatus parseIntWithPrefix(const char *Prefix, int64_t &Int);
1685
1686 ParseStatus
1687 parseIntWithPrefix(const char *Prefix, OperandVector &Operands,
1688 AMDGPUOperand::ImmTy ImmTy = AMDGPUOperand::ImmTyNone,
1689 std::function<bool(int64_t &)> ConvertResult = nullptr);
1690
1691 ParseStatus parseOperandArrayWithPrefix(
1692 const char *Prefix, OperandVector &Operands,
1693 AMDGPUOperand::ImmTy ImmTy = AMDGPUOperand::ImmTyNone,
1694 bool (*ConvertResult)(int64_t &) = nullptr);
1695
1696 ParseStatus
1697 parseNamedBit(StringRef Name, OperandVector &Operands,
1698 AMDGPUOperand::ImmTy ImmTy = AMDGPUOperand::ImmTyNone,
1699 bool IgnoreNegative = false);
1700 unsigned getCPolKind(StringRef Id, StringRef Mnemo, bool &Disabling) const;
1701 ParseStatus parseCPol(OperandVector &Operands);
1702 ParseStatus parseScope(OperandVector &Operands, int64_t &Scope);
1703 ParseStatus parseTH(OperandVector &Operands, int64_t &TH);
1704 ParseStatus parseStringWithPrefix(StringRef Prefix, StringRef &Value,
1705 SMLoc &StringLoc);
1706 ParseStatus parseStringOrIntWithPrefix(OperandVector &Operands,
1707 StringRef Name,
1708 ArrayRef<const char *> Ids,
1709 int64_t &IntVal);
1710 ParseStatus parseStringOrIntWithPrefix(OperandVector &Operands,
1711 StringRef Name,
1712 ArrayRef<const char *> Ids,
1713 AMDGPUOperand::ImmTy Type);
1714
1715 bool isModifier();
1716 bool isOperandModifier(const AsmToken &Token,
1717 const AsmToken &NextToken) const;
1718 bool isRegOrOperandModifier(const AsmToken &Token,
1719 const AsmToken &NextToken) const;
1720 bool isNamedOperandModifier(const AsmToken &Token,
1721 const AsmToken &NextToken) const;
1722 bool isOpcodeModifierWithVal(const AsmToken &Token,
1723 const AsmToken &NextToken) const;
1724 bool parseSP3NegModifier();
1725 ParseStatus parseImm(OperandVector &Operands, bool HasSP3AbsModifier = false,
1726 LitModifier Lit = LitModifier::None);
1727 ParseStatus parseReg(OperandVector &Operands);
1728 ParseStatus parseRegOrImm(OperandVector &Operands, bool HasSP3AbsMod = false,
1729 LitModifier Lit = LitModifier::None);
1730 ParseStatus parseRegOrImmWithFPInputMods(OperandVector &Operands,
1731 bool AllowImm = true);
1732 ParseStatus parseRegOrImmWithIntInputMods(OperandVector &Operands,
1733 bool AllowImm = true);
1734 ParseStatus parseRegWithFPInputMods(OperandVector &Operands);
1735 ParseStatus parseRegWithIntInputMods(OperandVector &Operands);
1736 ParseStatus parseRsrcReg(OperandVector &Operands);
1737 ParseStatus parseVReg32OrOff(OperandVector &Operands);
1738 ParseStatus tryParseIndexKey(OperandVector &Operands,
1739 AMDGPUOperand::ImmTy ImmTy);
1740 ParseStatus parseIndexKey8bit(OperandVector &Operands);
1741 ParseStatus parseIndexKey16bit(OperandVector &Operands);
1742 ParseStatus parseIndexKey32bit(OperandVector &Operands);
1743 ParseStatus tryParseMatrixFMT(OperandVector &Operands, StringRef Name,
1744 AMDGPUOperand::ImmTy Type);
1745 ParseStatus parseMatrixAFMT(OperandVector &Operands);
1746 ParseStatus parseMatrixBFMT(OperandVector &Operands);
1747 ParseStatus tryParseMatrixScale(OperandVector &Operands, StringRef Name,
1748 AMDGPUOperand::ImmTy Type);
1749 ParseStatus parseMatrixAScale(OperandVector &Operands);
1750 ParseStatus parseMatrixBScale(OperandVector &Operands);
1751 ParseStatus tryParseMatrixScaleFmt(OperandVector &Operands, StringRef Name,
1752 AMDGPUOperand::ImmTy Type);
1753 ParseStatus parseMatrixAScaleFmt(OperandVector &Operands);
1754 ParseStatus parseMatrixBScaleFmt(OperandVector &Operands);
1755
1756 ParseStatus parseDfmtNfmt(int64_t &Format);
1757 ParseStatus parseUfmt(int64_t &Format);
1758 ParseStatus parseSymbolicSplitFormat(StringRef FormatStr, SMLoc Loc,
1759 int64_t &Format);
1760 ParseStatus parseSymbolicUnifiedFormat(StringRef FormatStr, SMLoc Loc,
1761 int64_t &Format);
1762 ParseStatus parseFORMAT(OperandVector &Operands);
1763 ParseStatus parseSymbolicOrNumericFormat(int64_t &Format);
1764 ParseStatus parseNumericFormat(int64_t &Format);
1765 ParseStatus parseFlatOffset(OperandVector &Operands);
1766 ParseStatus parseR128A16(OperandVector &Operands);
1767 ParseStatus parseBLGP(OperandVector &Operands);
1768 bool tryParseFmt(const char *Pref, int64_t MaxVal, int64_t &Val);
1769 bool matchDfmtNfmt(int64_t &Dfmt, int64_t &Nfmt, StringRef FormatStr,
1770 SMLoc Loc);
1771
1772 void cvtExp(MCInst &Inst, const OperandVector &Operands);
1773
1774 bool parseCnt(int64_t &IntVal);
1775 ParseStatus parseSWaitCnt(OperandVector &Operands);
1776
1777 bool parseDepCtr(int64_t &IntVal, unsigned &Mask);
1778 void depCtrError(SMLoc Loc, int ErrorId, StringRef DepCtrName);
1779 ParseStatus parseDepCtr(OperandVector &Operands);
1780
1781 bool parseDelay(int64_t &Delay);
1782 ParseStatus parseSDelayALU(OperandVector &Operands);
1783
1784 ParseStatus parseHwreg(OperandVector &Operands);
1785
1786private:
1787 struct OperandInfoTy {
1788 SMLoc Loc;
1789 int64_t Val;
1790 bool IsSymbolic = false;
1791 bool IsDefined = false;
1792
1793 constexpr OperandInfoTy(int64_t Val) : Val(Val) {}
1794 };
1795
1796 struct StructuredOpField : OperandInfoTy {
1797 StringLiteral Id;
1798 StringLiteral Desc;
1799 unsigned Width;
1800 bool IsDefined = false;
1801
1802 constexpr StructuredOpField(StringLiteral Id, StringLiteral Desc,
1803 unsigned Width, int64_t Default)
1804 : OperandInfoTy(Default), Id(Id), Desc(Desc), Width(Width) {}
1805 virtual ~StructuredOpField() = default;
1806
1807 bool Error(AMDGPUAsmParser &Parser, const Twine &Err) const {
1808 Parser.Error(Loc, "invalid " + Desc + ": " + Err);
1809 return false;
1810 }
1811
1812 virtual bool validate(AMDGPUAsmParser &Parser) const {
1813 if (IsSymbolic && Val == OPR_ID_UNSUPPORTED)
1814 return Error(Parser, "not supported on this GPU");
1815 if (!isUIntN(Width, Val))
1816 return Error(Parser, "only " + Twine(Width) + "-bit values are legal");
1817 return true;
1818 }
1819 };
1820
1821 ParseStatus parseStructuredOpFields(ArrayRef<StructuredOpField *> Fields);
1822 bool validateStructuredOpFields(ArrayRef<const StructuredOpField *> Fields);
1823
1824 bool parseSendMsgBody(OperandInfoTy &Msg, OperandInfoTy &Op,
1825 OperandInfoTy &Stream);
1826 bool validateSendMsg(const OperandInfoTy &Msg, const OperandInfoTy &Op,
1827 const OperandInfoTy &Stream);
1828
1829 ParseStatus parseHwregFunc(OperandInfoTy &HwReg, OperandInfoTy &Offset,
1830 OperandInfoTy &Width);
1831
1832 const AMDGPUOperand &findMCOperand(const OperandVector &Operands,
1833 int MCOpIdx) const;
1834
1835 static SMLoc getLaterLoc(SMLoc a, SMLoc b);
1836
1837 SMLoc getFlatOffsetLoc(const OperandVector &Operands) const;
1838 SMLoc getSMEMOffsetLoc(const OperandVector &Operands) const;
1839 SMLoc getBLGPLoc(const OperandVector &Operands) const;
1840
1841 SMLoc getOperandLoc(const OperandVector &Operands, int MCOpIdx) const;
1842 SMLoc getOperandLoc(std::function<bool(const AMDGPUOperand &)> Test,
1843 const OperandVector &Operands) const;
1844 SMLoc getImmLoc(AMDGPUOperand::ImmTy Type,
1845 const OperandVector &Operands) const;
1846 SMLoc getInstLoc(const OperandVector &Operands) const;
1847
1848 bool validateInstruction(const MCInst &Inst, SMLoc IDLoc,
1849 const OperandVector &Operands);
1850 bool validateOffset(const MCInst &Inst, const OperandVector &Operands);
1851 bool validateFlatOffset(const MCInst &Inst, const OperandVector &Operands);
1852 bool validateSMEMOffset(const MCInst &Inst, const OperandVector &Operands);
1853 bool validateBF16InlineConst(const MCInst &Inst,
1854 const OperandVector &Operands);
1855 bool validateSOPLiteral(const MCInst &Inst, const OperandVector &Operands);
1856 bool validateConstantBusLimitations(const MCInst &Inst,
1857 const OperandVector &Operands);
1858 std::optional<unsigned> checkVOPDRegBankConstraints(const MCInst &Inst,
1859 bool AsVOPD3);
1860 bool validateVOPD(const MCInst &Inst, const OperandVector &Operands);
1861 bool tryVOPD(const MCInst &Inst);
1862 bool tryVOPD3(const MCInst &Inst);
1863 bool tryAnotherVOPDEncoding(const MCInst &Inst);
1864
1865 bool validateIntClampSupported(const MCInst &Inst);
1866 bool validateMIMGAtomicDMask(const MCInst &Inst);
1867 bool validateMIMGGatherDMask(const MCInst &Inst);
1868 bool validateMovrels(const MCInst &Inst, const OperandVector &Operands);
1869 bool validateMIMGDataSize(const MCInst &Inst, SMLoc IDLoc);
1870 bool validateMIMGAddrSize(const MCInst &Inst, SMLoc IDLoc);
1871 bool validateMIMGD16(const MCInst &Inst);
1872 bool validateMIMGDim(const MCInst &Inst, const OperandVector &Operands);
1873 bool validateTensorR128(const MCInst &Inst);
1874 bool validateMIMGMSAA(const MCInst &Inst);
1875 bool validateOpSel(const MCInst &Inst);
1876 bool validateTrue16OpSel(const MCInst &Inst);
1877 bool validateNeg(const MCInst &Inst, AMDGPU::OpName OpName);
1878 bool validateDPP(const MCInst &Inst, const OperandVector &Operands);
1879 bool validateVccOperand(MCRegister Reg) const;
1880 bool validateVOPLiteral(const MCInst &Inst, const OperandVector &Operands);
1881 bool validateMAIAccWrite(const MCInst &Inst, const OperandVector &Operands);
1882 bool validateMAISrc2(const MCInst &Inst, const OperandVector &Operands);
1883 bool validateMFMA(const MCInst &Inst, const OperandVector &Operands);
1884 bool validateAGPRLdSt(const MCInst &Inst) const;
1885 bool validateVGPRAlign(const MCInst &Inst) const;
1886 bool validateBLGP(const MCInst &Inst, const OperandVector &Operands);
1887 bool validateDS(const MCInst &Inst, const OperandVector &Operands);
1888 bool validateGWS(const MCInst &Inst, const OperandVector &Operands);
1889 bool validateDivScale(const MCInst &Inst);
1890 bool validateWaitCnt(const MCInst &Inst, const OperandVector &Operands);
1891 bool validateCoherencyBits(const MCInst &Inst, const OperandVector &Operands,
1892 SMLoc IDLoc);
1893 bool validateTHAndScopeBits(const MCInst &Inst, const OperandVector &Operands,
1894 const unsigned CPol);
1895 bool validateTFE(const MCInst &Inst, const OperandVector &Operands);
1896 bool validateLdsDirect(const MCInst &Inst, const OperandVector &Operands);
1897 bool validateWMMA(const MCInst &Inst, const OperandVector &Operands);
1898 bool validateMonitorSleep(const MCInst &Inst, const OperandVector &Operands);
1899 bool validateClusterBarrierIsFirst(const MCInst &Inst,
1900 const OperandVector &Operands);
1901 bool validateScaleSel(const MCInst &Inst, const OperandVector &Operands);
1902 unsigned getConstantBusLimit(unsigned Opcode) const;
1903 bool usesConstantBus(const MCInst &Inst, unsigned OpIdx);
1904 bool isInlineConstant(const MCInst &Inst, unsigned OpIdx) const;
1905 MCRegister findImplicitSGPRReadInVOP(const MCInst &Inst) const;
1906
1907 bool isSupportedMnemo(StringRef Mnemo, const FeatureBitset &FBS);
1908 bool isSupportedMnemo(StringRef Mnemo, const FeatureBitset &FBS,
1909 ArrayRef<unsigned> Variants);
1910 bool checkUnsupportedInstruction(StringRef Name, SMLoc IDLoc);
1911
1912 bool isId(const StringRef Id) const;
1913 bool isId(const AsmToken &Token, const StringRef Id) const;
1914 bool isToken(const AsmToken::TokenKind Kind) const;
1915 StringRef getId() const;
1916 bool trySkipId(const StringRef Id);
1917 bool trySkipId(const StringRef Pref, const StringRef Id);
1918 bool trySkipId(const StringRef Id, const AsmToken::TokenKind Kind);
1919 bool trySkipToken(const AsmToken::TokenKind Kind);
1920 bool skipToken(const AsmToken::TokenKind Kind, const StringRef ErrMsg);
1921 bool parseString(StringRef &Val,
1922 const StringRef ErrMsg = "expected a string");
1923 bool parseId(StringRef &Val, const StringRef ErrMsg = "");
1924
1925 void peekTokens(MutableArrayRef<AsmToken> Tokens);
1926 AsmToken::TokenKind getTokenKind() const;
1927 bool parseExpr(int64_t &Imm, StringRef Expected = "");
1929 StringRef getTokenStr() const;
1930 AsmToken peekToken(bool ShouldSkipSpace = true);
1931 AsmToken getToken() const;
1932 SMLoc getLoc() const;
1933 void lex();
1934
1935public:
1936 void onBeginOfFile() override;
1937 /// Emit the deferred leading .amdgcn_target directive if it has not been
1938 /// emitted yet. Called before emitting the first instruction or kernel
1939 /// descriptor.
1940 void emitTargetDirective();
1941 bool parsePrimaryExpr(const MCExpr *&Res, SMLoc &EndLoc) override;
1942
1943 ParseStatus parseCustomOperand(OperandVector &Operands, unsigned MCK);
1944
1945 ParseStatus parseExpTgt(OperandVector &Operands);
1946 ParseStatus parseSendMsg(OperandVector &Operands);
1947 ParseStatus parseWaitEvent(OperandVector &Operands);
1948 ParseStatus parseInterpSlot(OperandVector &Operands);
1949 ParseStatus parseInterpAttr(OperandVector &Operands);
1950 ParseStatus parseSOPPBrTarget(OperandVector &Operands);
1951 ParseStatus parseBoolReg(OperandVector &Operands);
1952
1953 bool parseSwizzleOperand(int64_t &Op, const unsigned MinVal,
1954 const unsigned MaxVal, const Twine &ErrMsg,
1955 SMLoc &Loc);
1956 bool parseSwizzleOperands(const unsigned OpNum, int64_t *Op,
1957 const unsigned MinVal, const unsigned MaxVal,
1958 const StringRef ErrMsg);
1959 ParseStatus parseSwizzle(OperandVector &Operands);
1960 bool parseSwizzleOffset(int64_t &Imm);
1961 bool parseSwizzleMacro(int64_t &Imm);
1962 bool parseSwizzleQuadPerm(int64_t &Imm);
1963 bool parseSwizzleBitmaskPerm(int64_t &Imm);
1964 bool parseSwizzleBroadcast(int64_t &Imm);
1965 bool parseSwizzleSwap(int64_t &Imm);
1966 bool parseSwizzleReverse(int64_t &Imm);
1967 bool parseSwizzleFFT(int64_t &Imm);
1968 bool parseSwizzleRotate(int64_t &Imm);
1969
1970 ParseStatus parseGPRIdxMode(OperandVector &Operands);
1971 int64_t parseGPRIdxMacro();
1972
1973 void cvtMubuf(MCInst &Inst, const OperandVector &Operands) {
1974 cvtMubufImpl(Inst, Operands, false);
1975 }
1976 void cvtMubufAtomic(MCInst &Inst, const OperandVector &Operands) {
1977 cvtMubufImpl(Inst, Operands, true);
1978 }
1979
1980 ParseStatus parseOModSI(OperandVector &Operands);
1981
1982 void cvtVOP3(MCInst &Inst, const OperandVector &Operands,
1983 OptionalImmIndexMap &OptionalIdx);
1984 void cvtScaledMFMA(MCInst &Inst, const OperandVector &Operands);
1985 void cvtVOP3OpSel(MCInst &Inst, const OperandVector &Operands);
1986 void cvtVOP3(MCInst &Inst, const OperandVector &Operands);
1987 void cvtVOP3P(MCInst &Inst, const OperandVector &Operands);
1988 void cvtSWMMAC(MCInst &Inst, const OperandVector &Operands);
1989
1990 void cvtVOPD(MCInst &Inst, const OperandVector &Operands);
1991 void cvtVOP3OpSel(MCInst &Inst, const OperandVector &Operands,
1992 OptionalImmIndexMap &OptionalIdx);
1993 void cvtVOP3P(MCInst &Inst, const OperandVector &Operands,
1994 OptionalImmIndexMap &OptionalIdx);
1995
1996 void cvtVOP3Interp(MCInst &Inst, const OperandVector &Operands);
1997 void cvtVINTERP(MCInst &Inst, const OperandVector &Operands);
1998 void cvtOpSelHelper(MCInst &Inst, unsigned OpSel);
1999
2000 bool parseDimId(unsigned &Encoding);
2001 ParseStatus parseDim(OperandVector &Operands);
2002 bool convertDppBoundCtrl(int64_t &BoundCtrl);
2003 ParseStatus parseDPP8(OperandVector &Operands);
2004 ParseStatus parseDPPCtrl(OperandVector &Operands);
2005 bool isSupportedDPPCtrl(StringRef Ctrl, const OperandVector &Operands);
2006 int64_t parseDPPCtrlSel(StringRef Ctrl);
2007 int64_t parseDPPCtrlPerm();
2008 void cvtDPP(MCInst &Inst, const OperandVector &Operands, bool IsDPP8 = false);
2009 void cvtDPP8(MCInst &Inst, const OperandVector &Operands) {
2010 cvtDPP(Inst, Operands, true);
2011 }
2012 void cvtVOP3DPP(MCInst &Inst, const OperandVector &Operands,
2013 bool IsDPP8 = false);
2014 void cvtVOP3DPP8(MCInst &Inst, const OperandVector &Operands) {
2015 cvtVOP3DPP(Inst, Operands, true);
2016 }
2017
2018 ParseStatus parseSDWASel(OperandVector &Operands, StringRef Prefix,
2019 AMDGPUOperand::ImmTy Type);
2020 ParseStatus parseSDWADstUnused(OperandVector &Operands);
2021 void cvtSdwaVOP1(MCInst &Inst, const OperandVector &Operands);
2022 void cvtSdwaVOP2(MCInst &Inst, const OperandVector &Operands);
2023 void cvtSdwaVOP2b(MCInst &Inst, const OperandVector &Operands);
2024 void cvtSdwaVOP2e(MCInst &Inst, const OperandVector &Operands);
2025 void cvtSdwaVOPC(MCInst &Inst, const OperandVector &Operands);
2026
2027 enum class SDWAInstType : unsigned { VOP1 = 0, VOP2 = 1, VOPC = 2 };
2028
2029 void cvtSDWA(MCInst &Inst, const OperandVector &Operands,
2030 SDWAInstType BasicInstType, bool SkipDstVcc = false,
2031 bool SkipSrcVcc = false);
2032
2033 ParseStatus parseEndpgm(OperandVector &Operands);
2034
2035 ParseStatus parseVOPD(OperandVector &Operands);
2036};
2037
2038} // end anonymous namespace
2039
2040// May be called with integer type with equivalent bitwidth.
2041static const fltSemantics *getFltSemantics(unsigned Size) {
2042 switch (Size) {
2043 case 4:
2044 return &APFloat::IEEEsingle();
2045 case 8:
2046 return &APFloat::IEEEdouble();
2047 case 2:
2048 return &APFloat::IEEEhalf();
2049 default:
2050 llvm_unreachable("unsupported fp type");
2051 }
2052}
2053
2055 return getFltSemantics(VT.getScalarSizeInBits() / 8);
2056}
2057
2059 switch (OperandType) {
2060 // When floating-point immediate is used as operand of type i16, the 32-bit
2061 // representation of the constant truncated to the 16 LSBs should be used.
2076 return &APFloat::IEEEsingle();
2085 return &APFloat::IEEEdouble();
2094 return &APFloat::IEEEhalf();
2099 return &APFloat::BFloat();
2100 default:
2101 llvm_unreachable("unsupported fp type");
2102 }
2103}
2104
2105//===----------------------------------------------------------------------===//
2106// Operand
2107//===----------------------------------------------------------------------===//
2108
2109static bool canLosslesslyConvertToFPType(APFloat &FPLiteral, MVT VT) {
2110 bool Lost;
2111
2112 // Convert literal to single precision
2113 APFloat::opStatus Status = FPLiteral.convert(
2115 // We allow precision lost but not overflow or underflow
2116 if (Status != APFloat::opOK && Lost &&
2117 ((Status & APFloat::opOverflow) != 0 ||
2118 (Status & APFloat::opUnderflow) != 0)) {
2119 return false;
2120 }
2121
2122 return true;
2123}
2124
2125static bool isSafeTruncation(int64_t Val, unsigned Size) {
2126 return isUIntN(Size, Val) || isIntN(Size, Val);
2127}
2128
2129static bool isInlineableLiteralOp16(int64_t Val, MVT VT, bool HasInv2Pi) {
2130 if (VT.getScalarType() == MVT::i16)
2131 return isInlinableLiteral32(Val, HasInv2Pi);
2132
2133 if (VT.getScalarType() == MVT::f16)
2134 return AMDGPU::isInlinableLiteralFP16(Val, HasInv2Pi);
2135
2136 assert(VT.getScalarType() == MVT::bf16);
2137
2138 return AMDGPU::isInlinableLiteralBF16(Val, HasInv2Pi);
2139}
2140
2141bool AMDGPUOperand::isInlinableImm(MVT type) const {
2142
2143 // This is a hack to enable named inline values like
2144 // shared_base with both 32-bit and 64-bit operands.
2145 // Note that these values are defined as
2146 // 32-bit operands only.
2147 if (isInlineValue()) {
2148 return true;
2149 }
2150
2151 if (!isImmTy(ImmTyNone)) {
2152 // Only plain immediates are inlinable (e.g. "clamp" attribute is not)
2153 return false;
2154 }
2155
2156 if (getModifiers().Lit != LitModifier::None)
2157 return false;
2158
2159 // TODO: We should avoid using host float here. It would be better to
2160 // check the float bit values which is what a few other places do.
2161 // We've had bot failures before due to weird NaN support on mips hosts.
2162
2163 APInt Literal(64, Imm.Val);
2164
2165 if (Imm.IsFPImm) { // We got fp literal token
2166 if (type == MVT::f64 || type == MVT::i64) { // Expected 64-bit operand
2168 AsmParser->hasInv2PiInlineImm());
2169 }
2170
2171 APFloat FPLiteral(APFloat::IEEEdouble(), APInt(64, Imm.Val));
2172 if (!canLosslesslyConvertToFPType(FPLiteral, type))
2173 return false;
2174
2175 if (type.getScalarSizeInBits() == 16) {
2176 bool Lost = false;
2177 switch (type.getScalarType().SimpleTy) {
2178 default:
2179 llvm_unreachable("unknown 16-bit type");
2180 case MVT::bf16:
2181 FPLiteral.convert(APFloatBase::BFloat(), APFloat::rmNearestTiesToEven,
2182 &Lost);
2183 break;
2184 case MVT::f16:
2185 FPLiteral.convert(APFloatBase::IEEEhalf(), APFloat::rmNearestTiesToEven,
2186 &Lost);
2187 break;
2188 case MVT::i16:
2189 FPLiteral.convert(APFloatBase::IEEEsingle(),
2190 APFloat::rmNearestTiesToEven, &Lost);
2191 break;
2192 }
2193 // We need to use 32-bit representation here because when a floating-point
2194 // inline constant is used as an i16 operand, its 32-bit representation
2195 // representation will be used. We will need the 32-bit value to check if
2196 // it is FP inline constant.
2197 uint32_t ImmVal = FPLiteral.bitcastToAPInt().getZExtValue();
2198 return isInlineableLiteralOp16(ImmVal, type,
2199 AsmParser->hasInv2PiInlineImm());
2200 }
2201
2202 // Check if single precision literal is inlinable
2204 static_cast<int32_t>(FPLiteral.bitcastToAPInt().getZExtValue()),
2205 AsmParser->hasInv2PiInlineImm());
2206 }
2207
2208 // We got int literal token.
2209 if (type == MVT::f64 || type == MVT::i64) { // Expected 64-bit operand
2211 AsmParser->hasInv2PiInlineImm());
2212 }
2213
2214 if (!isSafeTruncation(Imm.Val, type.getScalarSizeInBits())) {
2215 return false;
2216 }
2217
2218 if (type.getScalarSizeInBits() == 16) {
2220 static_cast<int16_t>(Literal.getLoBits(16).getSExtValue()), type,
2221 AsmParser->hasInv2PiInlineImm());
2222 }
2223
2225 static_cast<int32_t>(Literal.getLoBits(32).getZExtValue()),
2226 AsmParser->hasInv2PiInlineImm());
2227}
2228
2229bool AMDGPUOperand::isLiteralImm(MVT type) const {
2230 // Check that this immediate can be added as literal
2231 if (!isImmTy(ImmTyNone)) {
2232 return false;
2233 }
2234
2235 bool Allow64Bit =
2236 (type == MVT::i64 || type == MVT::f64) && AsmParser->has64BitLiterals();
2237
2238 if (!Imm.IsFPImm) {
2239 // We got int literal token.
2240
2241 if (type == MVT::f64 && hasFPModifiers()) {
2242 // Cannot apply fp modifiers to int literals preserving the same semantics
2243 // for VOP1/2/C and VOP3 because of integer truncation. To avoid
2244 // ambiguity, disable these cases.
2245 return false;
2246 }
2247
2248 unsigned Size = type.getSizeInBits();
2249 if (Size == 64) {
2250 if (Allow64Bit && !AMDGPU::isValid32BitLiteral(Imm.Val, false))
2251 return true;
2252 Size = 32;
2253 }
2254
2255 // FIXME: 64-bit operands can zero extend, sign extend, or pad zeroes for FP
2256 // types.
2257 return isSafeTruncation(Imm.Val, Size);
2258 }
2259
2260 // We got fp literal token
2261 if (type == MVT::f64) { // Expected 64-bit fp operand
2262 // We would set low 64-bits of literal to zeroes but we accept this literals
2263 return true;
2264 }
2265
2266 if (type == MVT::i64) { // Expected 64-bit int operand
2267 // We don't allow fp literals in 64-bit integer instructions. It is
2268 // unclear how we should encode them.
2269 return false;
2270 }
2271
2272 // We allow fp literals with f16x2 operands assuming that the specified
2273 // literal goes into the lower half and the upper half is zero. We also
2274 // require that the literal may be losslessly converted to f16.
2275 //
2276 // For i16x2 operands, we assume that the specified literal is encoded as a
2277 // single-precision float. This is pretty odd, but it matches SP3 and what
2278 // happens in hardware.
2279 MVT ExpectedType = (type == MVT::v2f16) ? MVT::f16
2280 : (type == MVT::v2i16) ? MVT::f32
2281 : (type == MVT::v2f32) ? MVT::f32
2282 : type;
2283
2284 APFloat FPLiteral(APFloat::IEEEdouble(), APInt(64, Imm.Val));
2285 return canLosslesslyConvertToFPType(FPLiteral, ExpectedType);
2286}
2287
2288bool AMDGPUOperand::isRegClassTarget(unsigned TargetRCIdx) const {
2289 if (!isRegKind())
2290 return false;
2291 int16_t RCID = AsmParser->getTargetRegClass(TargetRCIdx);
2292 return RCID >= 0 && isRegClass(RCID);
2293}
2294
2295bool AMDGPUOperand::isRegClass(unsigned RCID) const {
2296 return isRegKind() &&
2297 AsmParser->getMRI()->getRegClass(RCID).contains(getReg());
2298}
2299
2300bool AMDGPUOperand::isVRegWithInputMods() const {
2301 return isRegClass(AMDGPU::VGPR_32RegClassID) ||
2302 // GFX90A allows DPP on 64-bit operands.
2303 (AsmParser->getFeatureBits()[AMDGPU::FeatureDPALU_DPP] &&
2304 isRegClassTarget(AMDGPU::VReg_64_AlignTarget));
2305}
2306
2307template <bool IsFake16>
2308bool AMDGPUOperand::isT16_Lo128VRegWithInputMods() const {
2309 return isRegClass(IsFake16 ? AMDGPU::VGPR_32_Lo128RegClassID
2310 : AMDGPU::VGPR_16_Lo128RegClassID);
2311}
2312
2313template <bool IsFake16> bool AMDGPUOperand::isT16VRegWithInputMods() const {
2314 return isRegClass(IsFake16 ? AMDGPU::VGPR_32RegClassID
2315 : AMDGPU::VGPR_16RegClassID);
2316}
2317
2318bool AMDGPUOperand::isSDWAOperand(MVT type) const {
2319 if (AsmParser->isVI())
2320 return isVReg32();
2321 if (AsmParser->isGFX9Plus())
2322 return isRegClass(AMDGPU::VS_32RegClassID) || isInlinableImm(type);
2323 return false;
2324}
2325
2326bool AMDGPUOperand::isSDWAFP16Operand() const {
2327 return isSDWAOperand(MVT::f16);
2328}
2329
2330bool AMDGPUOperand::isSDWAFP32Operand() const {
2331 return isSDWAOperand(MVT::f32);
2332}
2333
2334bool AMDGPUOperand::isSDWAInt16Operand() const {
2335 return isSDWAOperand(MVT::i16);
2336}
2337
2338bool AMDGPUOperand::isSDWAInt32Operand() const {
2339 return isSDWAOperand(MVT::i32);
2340}
2341
2342bool AMDGPUOperand::isBoolReg() const {
2343 return isReg() && ((AsmParser->isWave64() && isSCSrc_b64()) ||
2344 (AsmParser->isWave32() && isSCSrc_b32()));
2345}
2346
2347uint64_t AMDGPUOperand::applyInputFPModifiers(uint64_t Val,
2348 unsigned Size) const {
2349 assert(isImmTy(ImmTyNone) && Imm.Mods.hasFPModifiers());
2350 assert(Size == 2 || Size == 4 || Size == 8);
2351
2352 const uint64_t FpSignMask = (1ULL << (Size * 8 - 1));
2353
2354 if (Imm.Mods.Abs) {
2355 Val &= ~FpSignMask;
2356 }
2357 if (Imm.Mods.Neg) {
2358 Val ^= FpSignMask;
2359 }
2360
2361 return Val;
2362}
2363
2364void AMDGPUOperand::addImmOperands(MCInst &Inst, unsigned N,
2365 bool ApplyModifiers) const {
2366 MCOpIdx = Inst.getNumOperands();
2367
2368 if (isExpr()) {
2370 return;
2371 }
2372
2373 if (AMDGPU::isSISrcOperand(AsmParser->getMII()->get(Inst.getOpcode()),
2374 Inst.getNumOperands())) {
2375 addLiteralImmOperand(Inst, Imm.Val,
2376 ApplyModifiers & isImmTy(ImmTyNone) &&
2377 Imm.Mods.hasFPModifiers());
2378 } else {
2379 assert(!isImmTy(ImmTyNone) || !hasModifiers());
2381 }
2382}
2383
2384void AMDGPUOperand::addLiteralImmOperand(MCInst &Inst, int64_t Val,
2385 bool ApplyModifiers) const {
2386 const auto &InstDesc = AsmParser->getMII()->get(Inst.getOpcode());
2387 auto OpNum = Inst.getNumOperands();
2388 // Check that this operand accepts literals
2389 assert(AMDGPU::isSISrcOperand(InstDesc, OpNum));
2390
2391 if (ApplyModifiers) {
2392 assert(AMDGPU::isSISrcFPOperand(InstDesc, OpNum));
2393 const unsigned Size =
2394 Imm.IsFPImm ? sizeof(double) : getOperandSize(InstDesc, OpNum);
2395 Val = applyInputFPModifiers(Val, Size);
2396 }
2397
2398 APInt Literal(64, Val);
2399 uint8_t OpTy = InstDesc.operands()[OpNum].OperandType;
2400
2401 bool CanUse64BitLiterals =
2402 AsmParser->has64BitLiterals() && !SIInstrFlags::isVOP3Like(InstDesc);
2403 LitModifier Lit = getModifiers().Lit;
2404 MCContext &Ctx = AsmParser->getContext();
2405
2406 if (Imm.IsFPImm) { // We got fp literal token
2407 switch (OpTy) {
2415 if (Lit == LitModifier::None &&
2417 AsmParser->hasInv2PiInlineImm())) {
2418 Inst.addOperand(MCOperand::createImm(Literal.getZExtValue()));
2419 return;
2420 }
2421
2422 // Non-inlineable
2423 if (AMDGPU::isSISrcFPOperand(InstDesc,
2424 OpNum)) { // Expected 64-bit fp operand
2425 bool HasMandatoryLiteral =
2426 AMDGPU::hasNamedOperand(Inst.getOpcode(), AMDGPU::OpName::imm);
2427 // For fp operands we check if low 32 bits are zeros
2428 if (Literal.getLoBits(32) != 0 &&
2429 (InstDesc.getSize() != 4 || !AsmParser->has64BitLiterals()) &&
2430 !HasMandatoryLiteral) {
2431 const_cast<AMDGPUAsmParser *>(AsmParser)->Warning(
2432 Inst.getLoc(),
2433 "Can't encode literal as exact 64-bit floating-point operand. "
2434 "Low 32-bits will be set to zero");
2435 Val &= 0xffffffff00000000u;
2436 }
2437
2438 if ((OpTy == AMDGPU::OPERAND_REG_IMM_FP64 ||
2441 if (CanUse64BitLiterals && Lit == LitModifier::None &&
2442 (isInt<32>(Val) || isUInt<32>(Val))) {
2443 // The floating-point operand will be verbalized as an
2444 // integer one. If that integer happens to fit 32 bits, on
2445 // re-assembling it will be intepreted as the high half of
2446 // the actual value, so we have to wrap it into lit64().
2447 Lit = LitModifier::Lit64;
2448 } else if (Lit == LitModifier::Lit) {
2449 // For FP64 operands lit() specifies the high half of the value.
2450 Val = Hi_32(Val);
2451 }
2452 }
2453 break;
2454 }
2455
2456 // We don't allow fp literals in 64-bit integer instructions. It is
2457 // unclear how we should encode them. This case should be checked earlier
2458 // in predicate methods (isLiteralImm())
2459 llvm_unreachable("fp literal in 64-bit integer instruction.");
2460
2462 if (CanUse64BitLiterals && Lit == LitModifier::None &&
2463 (isInt<32>(Val) || isUInt<32>(Val)))
2464 Lit = LitModifier::Lit64;
2465 break;
2466
2471 if (Lit == LitModifier::None && AsmParser->hasInv2PiInlineImm() &&
2472 Literal == 0x3fc45f306725feed) {
2473 // This is the 1/(2*pi) which is going to be truncated to bf16 with the
2474 // loss of precision. The constant represents ideomatic fp32 value of
2475 // 1/(2*pi) = 0.15915494 since bf16 is in fact fp32 with cleared low 16
2476 // bits. Prevent rounding below.
2477 Inst.addOperand(MCOperand::createImm(0x3e22));
2478 return;
2479 }
2480 [[fallthrough]];
2481
2504 bool lost;
2505 APFloat FPLiteral(APFloat::IEEEdouble(), Literal);
2506 // Convert literal to single precision
2507 FPLiteral.convert(*getOpFltSemantics(OpTy), APFloat::rmNearestTiesToEven,
2508 &lost);
2509 // We allow precision lost but not overflow or underflow. This should be
2510 // checked earlier in isLiteralImm()
2511
2512 Val = FPLiteral.bitcastToAPInt().getZExtValue();
2513 break;
2514 }
2515 default:
2516 llvm_unreachable("invalid operand size");
2517 }
2518
2519 if (Lit != LitModifier::None) {
2520 Inst.addOperand(
2522 } else {
2524 }
2525 return;
2526 }
2527
2528 // We got int literal token.
2529 // Only sign extend inline immediates.
2530 switch (OpTy) {
2545 break;
2546
2550 if (Lit == LitModifier::None &&
2551 AMDGPU::isInlinableLiteral64(Val, AsmParser->hasInv2PiInlineImm())) {
2553 return;
2554 }
2555
2556 // When the 32 MSBs are not zero (effectively means it can't be safely
2557 // truncated to uint32_t), if the target doesn't support 64-bit literals, or
2558 // the lit modifier is explicitly used, we need to truncate it to the 32
2559 // LSBs.
2560 if (!AsmParser->has64BitLiterals() || Lit == LitModifier::Lit)
2561 Val = Lo_32(Val);
2562 break;
2563
2568 if (Lit == LitModifier::None &&
2569 AMDGPU::isInlinableLiteral64(Val, AsmParser->hasInv2PiInlineImm())) {
2571 return;
2572 }
2573
2574 // If the target doesn't support 64-bit literals, we need to use the
2575 // constant as the high 32 MSBs of a double-precision floating point value.
2576 if (!AsmParser->has64BitLiterals()) {
2577 Val = static_cast<uint64_t>(Val) << 32;
2578 } else {
2579 // Now the target does support 64-bit literals, there are two cases
2580 // where we still want to use src_literal encoding:
2581 // 1) explicitly forced by using lit modifier;
2582 // 2) the value is a valid 32-bit representation (signed or unsigned),
2583 // meanwhile not forced by lit64 modifier.
2584 if (Lit == LitModifier::Lit ||
2585 (Lit != LitModifier::Lit64 && (isInt<32>(Val) || isUInt<32>(Val))))
2586 Val = static_cast<uint64_t>(Val) << 32;
2587 }
2588
2589 // For FP64 operands lit() specifies the high half of the value.
2590 if (Lit == LitModifier::Lit)
2591 Val = Hi_32(Val);
2592 break;
2593
2606 break;
2607
2609 if ((isInt<32>(Val) || isUInt<32>(Val)) && Lit != LitModifier::Lit64)
2610 Val <<= 32;
2611 break;
2612
2613 default:
2614 llvm_unreachable("invalid operand type");
2615 }
2616
2617 if (Lit != LitModifier::None) {
2618 Inst.addOperand(
2620 } else {
2622 }
2623}
2624
2625void AMDGPUOperand::addRegOperands(MCInst &Inst, unsigned N) const {
2626 MCOpIdx = Inst.getNumOperands();
2627 Inst.addOperand(
2628 MCOperand::createReg(AMDGPU::getMCReg(getReg(), AsmParser->getSTI())));
2629}
2630
2631bool AMDGPUOperand::isInlineValue() const {
2632 return isRegKind() && ::isInlineValue(getReg());
2633}
2634
2635//===----------------------------------------------------------------------===//
2636// AsmParser
2637//===----------------------------------------------------------------------===//
2638
2639void AMDGPUAsmParser::createConstantSymbol(StringRef Id, int64_t Val) {
2640 // TODO: make those pre-defined variables read-only.
2641 // Currently there is none suitable machinery in the core llvm-mc for this.
2642 // MCSymbol::isRedefinable is intended for another purpose, and
2643 // AsmParser::parseDirectiveSet() cannot be specialized for specific target.
2644 MCContext &Ctx = getContext();
2645 MCSymbol *Sym = Ctx.getOrCreateSymbol(Id);
2647}
2648
2649static int getRegClass(RegisterKind Is, unsigned RegWidth) {
2650 if (Is == IS_VGPR) {
2651 switch (RegWidth) {
2652 default:
2653 return -1;
2654 case 32:
2655 return AMDGPU::VGPR_32RegClassID;
2656 case 64:
2657 return AMDGPU::VReg_64RegClassID;
2658 case 96:
2659 return AMDGPU::VReg_96RegClassID;
2660 case 128:
2661 return AMDGPU::VReg_128RegClassID;
2662 case 160:
2663 return AMDGPU::VReg_160RegClassID;
2664 case 192:
2665 return AMDGPU::VReg_192RegClassID;
2666 case 224:
2667 return AMDGPU::VReg_224RegClassID;
2668 case 256:
2669 return AMDGPU::VReg_256RegClassID;
2670 case 288:
2671 return AMDGPU::VReg_288RegClassID;
2672 case 320:
2673 return AMDGPU::VReg_320RegClassID;
2674 case 352:
2675 return AMDGPU::VReg_352RegClassID;
2676 case 384:
2677 return AMDGPU::VReg_384RegClassID;
2678 case 512:
2679 return AMDGPU::VReg_512RegClassID;
2680 case 1024:
2681 return AMDGPU::VReg_1024RegClassID;
2682 }
2683 } else if (Is == IS_TTMP) {
2684 switch (RegWidth) {
2685 default:
2686 return -1;
2687 case 32:
2688 return AMDGPU::TTMP_32RegClassID;
2689 case 64:
2690 return AMDGPU::TTMP_64RegClassID;
2691 case 128:
2692 return AMDGPU::TTMP_128RegClassID;
2693 case 256:
2694 return AMDGPU::TTMP_256RegClassID;
2695 case 512:
2696 return AMDGPU::TTMP_512RegClassID;
2697 }
2698 } else if (Is == IS_SGPR) {
2699 switch (RegWidth) {
2700 default:
2701 return -1;
2702 case 32:
2703 return AMDGPU::SGPR_32RegClassID;
2704 case 64:
2705 return AMDGPU::SGPR_64RegClassID;
2706 case 96:
2707 return AMDGPU::SGPR_96RegClassID;
2708 case 128:
2709 return AMDGPU::SGPR_128RegClassID;
2710 case 160:
2711 return AMDGPU::SGPR_160RegClassID;
2712 case 192:
2713 return AMDGPU::SGPR_192RegClassID;
2714 case 224:
2715 return AMDGPU::SGPR_224RegClassID;
2716 case 256:
2717 return AMDGPU::SGPR_256RegClassID;
2718 case 288:
2719 return AMDGPU::SGPR_288RegClassID;
2720 case 320:
2721 return AMDGPU::SGPR_320RegClassID;
2722 case 352:
2723 return AMDGPU::SGPR_352RegClassID;
2724 case 384:
2725 return AMDGPU::SGPR_384RegClassID;
2726 case 512:
2727 return AMDGPU::SGPR_512RegClassID;
2728 }
2729 } else if (Is == IS_AGPR) {
2730 switch (RegWidth) {
2731 default:
2732 return -1;
2733 case 32:
2734 return AMDGPU::AGPR_32RegClassID;
2735 case 64:
2736 return AMDGPU::AReg_64RegClassID;
2737 case 96:
2738 return AMDGPU::AReg_96RegClassID;
2739 case 128:
2740 return AMDGPU::AReg_128RegClassID;
2741 case 160:
2742 return AMDGPU::AReg_160RegClassID;
2743 case 192:
2744 return AMDGPU::AReg_192RegClassID;
2745 case 224:
2746 return AMDGPU::AReg_224RegClassID;
2747 case 256:
2748 return AMDGPU::AReg_256RegClassID;
2749 case 288:
2750 return AMDGPU::AReg_288RegClassID;
2751 case 320:
2752 return AMDGPU::AReg_320RegClassID;
2753 case 352:
2754 return AMDGPU::AReg_352RegClassID;
2755 case 384:
2756 return AMDGPU::AReg_384RegClassID;
2757 case 512:
2758 return AMDGPU::AReg_512RegClassID;
2759 case 1024:
2760 return AMDGPU::AReg_1024RegClassID;
2761 }
2762 }
2763 return -1;
2764}
2765
2768 .Case("exec", AMDGPU::EXEC)
2769 .Case("vcc", AMDGPU::VCC)
2770 .Case("flat_scratch", AMDGPU::FLAT_SCR)
2771 .Case("xnack_mask", AMDGPU::XNACK_MASK)
2772 .Case("shared_base", AMDGPU::SRC_SHARED_BASE)
2773 .Case("src_shared_base", AMDGPU::SRC_SHARED_BASE)
2774 .Case("shared_limit", AMDGPU::SRC_SHARED_LIMIT)
2775 .Case("src_shared_limit", AMDGPU::SRC_SHARED_LIMIT)
2776 .Case("private_base", AMDGPU::SRC_PRIVATE_BASE)
2777 .Case("src_private_base", AMDGPU::SRC_PRIVATE_BASE)
2778 .Case("private_limit", AMDGPU::SRC_PRIVATE_LIMIT)
2779 .Case("src_private_limit", AMDGPU::SRC_PRIVATE_LIMIT)
2780 .Case("src_flat_scratch_base_lo", AMDGPU::SRC_FLAT_SCRATCH_BASE_LO)
2781 .Case("src_flat_scratch_base_hi", AMDGPU::SRC_FLAT_SCRATCH_BASE_HI)
2782 .Case("pops_exiting_wave_id", AMDGPU::SRC_POPS_EXITING_WAVE_ID)
2783 .Case("src_pops_exiting_wave_id", AMDGPU::SRC_POPS_EXITING_WAVE_ID)
2784 .Case("lds_direct", AMDGPU::LDS_DIRECT)
2785 .Case("src_lds_direct", AMDGPU::LDS_DIRECT)
2786 .Case("m0", AMDGPU::M0)
2787 .Case("vccz", AMDGPU::SRC_VCCZ)
2788 .Case("src_vccz", AMDGPU::SRC_VCCZ)
2789 .Case("execz", AMDGPU::SRC_EXECZ)
2790 .Case("src_execz", AMDGPU::SRC_EXECZ)
2791 .Case("scc", AMDGPU::SRC_SCC)
2792 .Case("src_scc", AMDGPU::SRC_SCC)
2793 .Case("tba", AMDGPU::TBA)
2794 .Case("tma", AMDGPU::TMA)
2795 .Case("flat_scratch_lo", AMDGPU::FLAT_SCR_LO)
2796 .Case("flat_scratch_hi", AMDGPU::FLAT_SCR_HI)
2797 .Case("xnack_mask_lo", AMDGPU::XNACK_MASK_LO)
2798 .Case("xnack_mask_hi", AMDGPU::XNACK_MASK_HI)
2799 .Case("vcc_lo", AMDGPU::VCC_LO)
2800 .Case("vcc_hi", AMDGPU::VCC_HI)
2801 .Case("exec_lo", AMDGPU::EXEC_LO)
2802 .Case("exec_hi", AMDGPU::EXEC_HI)
2803 .Case("tma_lo", AMDGPU::TMA_LO)
2804 .Case("tma_hi", AMDGPU::TMA_HI)
2805 .Case("tba_lo", AMDGPU::TBA_LO)
2806 .Case("tba_hi", AMDGPU::TBA_HI)
2807 .Case("pc", AMDGPU::PC_REG)
2808 .Case("null", AMDGPU::SGPR_NULL)
2809 .Default(AMDGPU::NoRegister);
2810}
2811
2812bool AMDGPUAsmParser::ParseRegister(MCRegister &RegNo, SMLoc &StartLoc,
2813 SMLoc &EndLoc, bool RestoreOnFailure) {
2814 auto R = parseRegister();
2815 if (!R)
2816 return true;
2817 assert(R->isReg());
2818 RegNo = R->getReg();
2819 StartLoc = R->getStartLoc();
2820 EndLoc = R->getEndLoc();
2821 return false;
2822}
2823
2824bool AMDGPUAsmParser::parseRegister(MCRegister &Reg, SMLoc &StartLoc,
2825 SMLoc &EndLoc) {
2826 return ParseRegister(Reg, StartLoc, EndLoc, /*RestoreOnFailure=*/false);
2827}
2828
2829ParseStatus AMDGPUAsmParser::tryParseRegister(MCRegister &Reg, SMLoc &StartLoc,
2830 SMLoc &EndLoc) {
2831 bool Result = ParseRegister(Reg, StartLoc, EndLoc, /*RestoreOnFailure=*/true);
2832 bool PendingErrors = getParser().hasPendingError();
2833 getParser().clearPendingErrors();
2834 if (PendingErrors)
2835 return ParseStatus::Failure;
2836 if (Result)
2837 return ParseStatus::NoMatch;
2838 return ParseStatus::Success;
2839}
2840
2841bool AMDGPUAsmParser::AddNextRegisterToList(MCRegister &Reg, unsigned &RegWidth,
2842 RegisterKind RegKind,
2843 MCRegister Reg1,
2844 RegisterKind RegKind1, SMLoc Loc) {
2845 // Allow VCC_LO/HI at the end of SGPR lists.
2846 if (RegKind == IS_SGPR) {
2847 unsigned RegIdx = (Reg - AMDGPU::SGPR0) + RegWidth / 32;
2848 if ((RegIdx == 106 && Reg1 == AMDGPU::VCC_LO) ||
2849 (RegIdx == 107 && Reg1 == AMDGPU::VCC_HI)) {
2850 RegWidth += 32;
2851 return true;
2852 }
2853 }
2854
2855 if (RegKind != RegKind1) {
2856 Error(Loc, "registers in a list must be of the same kind");
2857 return false;
2858 }
2859
2860 switch (RegKind) {
2861 case IS_SPECIAL:
2862 if (Reg == AMDGPU::EXEC_LO && Reg1 == AMDGPU::EXEC_HI) {
2863 Reg = AMDGPU::EXEC;
2864 RegWidth = 64;
2865 return true;
2866 }
2867 if (Reg == AMDGPU::FLAT_SCR_LO && Reg1 == AMDGPU::FLAT_SCR_HI) {
2868 Reg = AMDGPU::FLAT_SCR;
2869 RegWidth = 64;
2870 return true;
2871 }
2872 if (Reg == AMDGPU::XNACK_MASK_LO && Reg1 == AMDGPU::XNACK_MASK_HI) {
2873 Reg = AMDGPU::XNACK_MASK;
2874 RegWidth = 64;
2875 return true;
2876 }
2877 if (Reg == AMDGPU::VCC_LO && Reg1 == AMDGPU::VCC_HI) {
2878 Reg = AMDGPU::VCC;
2879 RegWidth = 64;
2880 return true;
2881 }
2882 if (Reg == AMDGPU::TBA_LO && Reg1 == AMDGPU::TBA_HI) {
2883 Reg = AMDGPU::TBA;
2884 RegWidth = 64;
2885 return true;
2886 }
2887 if (Reg == AMDGPU::TMA_LO && Reg1 == AMDGPU::TMA_HI) {
2888 Reg = AMDGPU::TMA;
2889 RegWidth = 64;
2890 return true;
2891 }
2892 Error(Loc, "register does not fit in the list");
2893 return false;
2894 case IS_VGPR:
2895 case IS_SGPR:
2896 case IS_AGPR:
2897 case IS_TTMP:
2898 if (Reg1 != Reg + RegWidth / 32) {
2899 Error(Loc, "registers in a list must have consecutive indices");
2900 return false;
2901 }
2902 RegWidth += 32;
2903 return true;
2904 default:
2905 llvm_unreachable("unexpected register kind");
2906 }
2907}
2908
2909struct RegInfo {
2911 RegisterKind Kind;
2912};
2913
2914static constexpr RegInfo RegularRegisters[] = {
2915 {{"v"}, IS_VGPR}, {{"s"}, IS_SGPR}, {{"ttmp"}, IS_TTMP},
2916 {{"acc"}, IS_AGPR}, {{"a"}, IS_AGPR},
2917};
2918
2919static bool isRegularReg(RegisterKind Kind) {
2920 return Kind == IS_VGPR || Kind == IS_SGPR || Kind == IS_TTMP ||
2921 Kind == IS_AGPR;
2922}
2923
2925 for (const RegInfo &Reg : RegularRegisters)
2926 if (Str.starts_with(Reg.Name))
2927 return &Reg;
2928 return nullptr;
2929}
2930
2931static bool getRegNum(StringRef Str, unsigned &Num) {
2932 return !Str.getAsInteger(10, Num);
2933}
2934
2935bool AMDGPUAsmParser::isRegister(const AsmToken &Token,
2936 const AsmToken &NextToken) const {
2937
2938 // A list of consecutive registers: [s0,s1,s2,s3]
2939 if (Token.is(AsmToken::LBrac))
2940 return true;
2941
2942 if (!Token.is(AsmToken::Identifier))
2943 return false;
2944
2945 // A single register like s0 or a range of registers like s[0:1]
2946
2947 StringRef Str = Token.getString();
2948 const RegInfo *Reg = getRegularRegInfo(Str);
2949 if (Reg) {
2950 StringRef RegName = Reg->Name;
2951 StringRef RegSuffix = Str.substr(RegName.size());
2952 if (!RegSuffix.empty()) {
2953 RegSuffix.consume_back(".l");
2954 RegSuffix.consume_back(".h");
2955 unsigned Num;
2956 // A single register with an index: rXX
2957 if (getRegNum(RegSuffix, Num))
2958 return true;
2959 } else {
2960 // A range of registers: r[XX:YY].
2961 if (NextToken.is(AsmToken::LBrac))
2962 return true;
2963 }
2964 }
2965
2966 return getSpecialRegForName(Str).isValid();
2967}
2968
2969bool AMDGPUAsmParser::isRegister() {
2970 return isRegister(getToken(), peekToken());
2971}
2972
2973MCRegister AMDGPUAsmParser::getRegularReg(RegisterKind RegKind, unsigned RegNum,
2974 unsigned SubReg, unsigned RegWidth,
2975 SMLoc Loc) {
2976 assert(isRegularReg(RegKind));
2977
2978 unsigned AlignSize = 1;
2979 if (RegKind == IS_SGPR || RegKind == IS_TTMP) {
2980 // SGPR and TTMP registers must be aligned.
2981 // Max required alignment is 4 dwords.
2982 AlignSize = std::min(llvm::bit_ceil(RegWidth / 32), 4u);
2983 }
2984
2985 if (RegNum % AlignSize != 0) {
2986 Error(Loc, "invalid register alignment");
2987 return MCRegister();
2988 }
2989
2990 unsigned RegIdx = RegNum / AlignSize;
2991 int RCID = getRegClass(RegKind, RegWidth);
2992 if (RCID == -1) {
2993 Error(Loc, "invalid or unsupported register size");
2994 return MCRegister();
2995 }
2996
2997 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
2998 const MCRegisterClass &RC = TRI->getRegClass(RCID);
2999 if (RegIdx >= RC.getNumRegs() || (RegKind == IS_VGPR && RegIdx > 255)) {
3000 Error(Loc, "register index is out of range");
3001 return AMDGPU::NoRegister;
3002 }
3003
3004 if (RegKind == IS_VGPR && !isGFX1250Plus() && RegIdx + RegWidth / 32 > 256) {
3005 Error(Loc, "register index is out of range");
3006 return MCRegister();
3007 }
3008
3009 MCRegister Reg = RC.getRegister(RegIdx);
3010
3011 if (SubReg) {
3012 Reg = TRI->getSubReg(Reg, SubReg);
3013
3014 if (!Reg)
3015 Error(Loc, "invalid subregister");
3016 }
3017
3018 return Reg;
3019}
3020
3021bool AMDGPUAsmParser::ParseRegRange(unsigned &Num, unsigned &RegWidth,
3022 unsigned &SubReg) {
3023 int64_t RegLo, RegHi;
3024 if (!skipToken(AsmToken::LBrac, "missing register index"))
3025 return false;
3026
3027 SMLoc FirstIdxLoc = getLoc();
3028 SMLoc SecondIdxLoc;
3029
3030 if (!parseExpr(RegLo))
3031 return false;
3032
3033 if (trySkipToken(AsmToken::Colon)) {
3034 SecondIdxLoc = getLoc();
3035 if (!parseExpr(RegHi))
3036 return false;
3037 } else {
3038 RegHi = RegLo;
3039 }
3040
3041 if (!skipToken(AsmToken::RBrac, "expected a closing square bracket"))
3042 return false;
3043
3044 if (!isUInt<32>(RegLo)) {
3045 Error(FirstIdxLoc, "invalid register index");
3046 return false;
3047 }
3048
3049 if (!isUInt<32>(RegHi)) {
3050 Error(SecondIdxLoc, "invalid register index");
3051 return false;
3052 }
3053
3054 if (RegLo > RegHi) {
3055 Error(FirstIdxLoc, "first register index should not exceed second index");
3056 return false;
3057 }
3058
3059 if (RegHi == RegLo) {
3060 StringRef RegSuffix = getTokenStr();
3061 if (RegSuffix == ".l") {
3062 SubReg = AMDGPU::lo16;
3063 lex();
3064 } else if (RegSuffix == ".h") {
3065 SubReg = AMDGPU::hi16;
3066 lex();
3067 }
3068 }
3069
3070 Num = static_cast<unsigned>(RegLo);
3071 RegWidth = 32 * ((RegHi - RegLo) + 1);
3072
3073 return true;
3074}
3075
3076MCRegister AMDGPUAsmParser::ParseSpecialReg(RegisterKind &RegKind,
3077 unsigned &RegNum,
3078 unsigned &RegWidth,
3079 SmallVectorImpl<AsmToken> &Tokens) {
3080 assert(isToken(AsmToken::Identifier));
3081 MCRegister Reg = getSpecialRegForName(getTokenStr());
3082 if (Reg) {
3083 RegNum = 0;
3084 RegWidth = 32;
3085 RegKind = IS_SPECIAL;
3086 Tokens.push_back(getToken());
3087 lex(); // skip register name
3088 }
3089 return Reg;
3090}
3091
3092MCRegister AMDGPUAsmParser::ParseRegularReg(RegisterKind &RegKind,
3093 unsigned &RegNum,
3094 unsigned &RegWidth,
3095 SmallVectorImpl<AsmToken> &Tokens) {
3096 assert(isToken(AsmToken::Identifier));
3097 StringRef RegName = getTokenStr();
3098 auto Loc = getLoc();
3099
3100 const RegInfo *RI = getRegularRegInfo(RegName);
3101 if (!RI) {
3102 Error(Loc, "invalid register name");
3103 return MCRegister();
3104 }
3105
3106 Tokens.push_back(getToken());
3107 lex(); // skip register name
3108
3109 RegKind = RI->Kind;
3110 StringRef RegSuffix = RegName.substr(RI->Name.size());
3111 unsigned SubReg = NoSubRegister;
3112 bool IsRange = false;
3113 if (!RegSuffix.empty()) {
3114 if (RegSuffix.consume_back(".l"))
3115 SubReg = AMDGPU::lo16;
3116 else if (RegSuffix.consume_back(".h"))
3117 SubReg = AMDGPU::hi16;
3118
3119 // Single 32-bit register: vXX.
3120 if (!getRegNum(RegSuffix, RegNum)) {
3121 Error(Loc, "invalid register index");
3122 return MCRegister();
3123 }
3124 RegWidth = 32;
3125 } else {
3126 // Range of registers: v[XX:YY]. ":YY" is optional.
3127 IsRange = true;
3128 if (!ParseRegRange(RegNum, RegWidth, SubReg))
3129 return MCRegister();
3130 }
3131
3132 // Do not allow vcc_lo/hi be referred as s106/107.
3133 MCRegister Reg = getRegularReg(RegKind, RegNum, SubReg, RegWidth, Loc);
3134 const MCRegisterInfo &TRI = *getContext().getRegisterInfo();
3135 if (RegKind == IS_SGPR && IsRange
3136 ? (TRI.isSubRegister(Reg, VCC_LO) || TRI.isSubRegister(Reg, VCC_HI))
3137 : (Reg == VCC_LO || Reg == VCC_HI)) {
3138 Error(Loc, "register index is out of range");
3139 return MCRegister();
3140 }
3141
3142 return Reg;
3143}
3144
3145MCRegister AMDGPUAsmParser::ParseRegList(RegisterKind &RegKind,
3146 unsigned &RegNum, unsigned &RegWidth,
3147 SmallVectorImpl<AsmToken> &Tokens) {
3148 MCRegister Reg;
3149 auto ListLoc = getLoc();
3150
3151 if (!skipToken(AsmToken::LBrac,
3152 "expected a register or a list of registers")) {
3153 return MCRegister();
3154 }
3155
3156 // List of consecutive registers, e.g.: [s0,s1,s2,s3]
3157
3158 auto Loc = getLoc();
3159 if (!ParseAMDGPURegister(RegKind, Reg, RegNum, RegWidth))
3160 return MCRegister();
3161 if (RegWidth != 32) {
3162 Error(Loc, "expected a single 32-bit register");
3163 return MCRegister();
3164 }
3165
3166 for (; trySkipToken(AsmToken::Comma);) {
3167 RegisterKind NextRegKind;
3168 MCRegister NextReg;
3169 unsigned NextRegNum, NextRegWidth;
3170 Loc = getLoc();
3171
3172 if (!ParseAMDGPURegister(NextRegKind, NextReg, NextRegNum, NextRegWidth,
3173 Tokens)) {
3174 return MCRegister();
3175 }
3176 if (NextRegWidth != 32) {
3177 Error(Loc, "expected a single 32-bit register");
3178 return MCRegister();
3179 }
3180 if (!AddNextRegisterToList(Reg, RegWidth, RegKind, NextReg, NextRegKind,
3181 Loc))
3182 return MCRegister();
3183 }
3184
3185 if (!skipToken(AsmToken::RBrac,
3186 "expected a comma or a closing square bracket")) {
3187 return MCRegister();
3188 }
3189
3190 if (isRegularReg(RegKind))
3191 Reg = getRegularReg(RegKind, RegNum, NoSubRegister, RegWidth, ListLoc);
3192
3193 return Reg;
3194}
3195
3196bool AMDGPUAsmParser::ParseAMDGPURegister(RegisterKind &RegKind,
3197 MCRegister &Reg, unsigned &RegNum,
3198 unsigned &RegWidth,
3199 SmallVectorImpl<AsmToken> &Tokens) {
3200 auto Loc = getLoc();
3201 Reg = MCRegister();
3202
3203 if (isToken(AsmToken::Identifier)) {
3204 Reg = ParseSpecialReg(RegKind, RegNum, RegWidth, Tokens);
3205 if (!Reg)
3206 Reg = ParseRegularReg(RegKind, RegNum, RegWidth, Tokens);
3207 } else {
3208 Reg = ParseRegList(RegKind, RegNum, RegWidth, Tokens);
3209 }
3210
3211 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
3212 if (!Reg) {
3213 assert(Parser.hasPendingError());
3214 return false;
3215 }
3216
3217 if (!subtargetHasRegister(*TRI, Reg)) {
3218 if (Reg == AMDGPU::SGPR_NULL) {
3219 Error(Loc, "'null' operand is not supported on this GPU");
3220 } else {
3222 " register not available on this GPU");
3223 }
3224 return false;
3225 }
3226
3227 return true;
3228}
3229
3230bool AMDGPUAsmParser::ParseAMDGPURegister(RegisterKind &RegKind,
3231 MCRegister &Reg, unsigned &RegNum,
3232 unsigned &RegWidth,
3233 bool RestoreOnFailure /*=false*/) {
3234 Reg = MCRegister();
3235
3237 if (ParseAMDGPURegister(RegKind, Reg, RegNum, RegWidth, Tokens)) {
3238 if (RestoreOnFailure) {
3239 while (!Tokens.empty()) {
3240 getLexer().UnLex(Tokens.pop_back_val());
3241 }
3242 }
3243 return true;
3244 }
3245 return false;
3246}
3247
3248std::optional<StringRef>
3249AMDGPUAsmParser::getGprCountSymbolName(RegisterKind RegKind) {
3250 switch (RegKind) {
3251 case IS_VGPR:
3252 return StringRef(".amdgcn.next_free_vgpr");
3253 case IS_SGPR:
3254 return StringRef(".amdgcn.next_free_sgpr");
3255 default:
3256 return std::nullopt;
3257 }
3258}
3259
3260void AMDGPUAsmParser::initializeGprCountSymbol(RegisterKind RegKind) {
3261 auto SymbolName = getGprCountSymbolName(RegKind);
3262 assert(SymbolName && "initializing invalid register kind");
3263 MCSymbol *Sym = getContext().getOrCreateSymbol(*SymbolName);
3265 Sym->setRedefinable(true);
3266}
3267
3268bool AMDGPUAsmParser::updateGprCountSymbols(RegisterKind RegKind,
3269 unsigned DwordRegIndex,
3270 unsigned RegWidth) {
3271 // Symbols are only defined for GCN targets
3272 if (ISA.Major < 6)
3273 return true;
3274
3275 auto SymbolName = getGprCountSymbolName(RegKind);
3276 if (!SymbolName)
3277 return true;
3278 MCSymbol *Sym = getContext().getOrCreateSymbol(*SymbolName);
3279
3280 int64_t NewMax = DwordRegIndex + divideCeil(RegWidth, 32) - 1;
3281 int64_t OldCount;
3282
3283 if (!Sym->isVariable())
3284 return !Error(getLoc(),
3285 ".amdgcn.next_free_{v,s}gpr symbols must be variable");
3286 if (!Sym->getVariableValue()->evaluateAsAbsolute(OldCount))
3287 return !Error(
3288 getLoc(),
3289 ".amdgcn.next_free_{v,s}gpr symbols must be absolute expressions");
3290
3291 if (OldCount <= NewMax)
3293
3294 return true;
3295}
3296
3297std::unique_ptr<AMDGPUOperand>
3298AMDGPUAsmParser::parseRegister(bool RestoreOnFailure) {
3299 const auto &Tok = getToken();
3300 SMLoc StartLoc = Tok.getLoc();
3301 SMLoc EndLoc = Tok.getEndLoc();
3302 RegisterKind RegKind;
3303 MCRegister Reg;
3304 unsigned RegNum, RegWidth;
3305
3306 if (!ParseAMDGPURegister(RegKind, Reg, RegNum, RegWidth)) {
3307 return nullptr;
3308 }
3309 if (isHsaAbi(getSTI())) {
3310 if (!updateGprCountSymbols(RegKind, RegNum, RegWidth))
3311 return nullptr;
3312 } else
3313 KernelScope.usesRegister(RegKind, RegNum, RegWidth);
3314 return AMDGPUOperand::CreateReg(this, Reg, StartLoc, EndLoc);
3315}
3316
3317ParseStatus AMDGPUAsmParser::parseImm(OperandVector &Operands,
3318 bool HasSP3AbsModifier, LitModifier Lit) {
3319 // TODO: add syntactic sugar for 1/(2*PI)
3320
3321 if (isRegister() || isModifier())
3322 return ParseStatus::NoMatch;
3323
3324 if (Lit == LitModifier::None) {
3325 if (trySkipId("lit"))
3326 Lit = LitModifier::Lit;
3327 else if (trySkipId("lit64"))
3328 Lit = LitModifier::Lit64;
3329
3330 if (Lit != LitModifier::None) {
3331 if (!skipToken(AsmToken::LParen, "expected left paren after lit"))
3332 return ParseStatus::Failure;
3333 ParseStatus S = parseImm(Operands, HasSP3AbsModifier, Lit);
3334 if (S.isSuccess() &&
3335 !skipToken(AsmToken::RParen, "expected closing parentheses"))
3336 return ParseStatus::Failure;
3337 return S;
3338 }
3339 }
3340
3341 const auto &Tok = getToken();
3342 const auto &NextTok = peekToken();
3343 bool IsReal = Tok.is(AsmToken::Real);
3344 SMLoc S = getLoc();
3345 bool Negate = false;
3346
3347 if (!IsReal && Tok.is(AsmToken::Minus) && NextTok.is(AsmToken::Real)) {
3348 lex();
3349 IsReal = true;
3350 Negate = true;
3351 }
3352
3353 AMDGPUOperand::Modifiers Mods;
3354 Mods.Lit = Lit;
3355
3356 if (IsReal) {
3357 // Floating-point expressions are not supported.
3358 // Can only allow floating-point literals with an
3359 // optional sign.
3360
3361 StringRef Num = getTokenStr();
3362 lex();
3363
3364 APFloat RealVal(APFloat::IEEEdouble());
3365 auto roundMode = APFloat::rmNearestTiesToEven;
3366 if (errorToBool(RealVal.convertFromString(Num, roundMode).takeError()))
3367 return ParseStatus::Failure;
3368 if (Negate)
3369 RealVal.changeSign();
3370
3371 Operands.push_back(
3372 AMDGPUOperand::CreateImm(this, RealVal.bitcastToAPInt().getZExtValue(),
3373 S, AMDGPUOperand::ImmTyNone, true));
3374 AMDGPUOperand &Op = static_cast<AMDGPUOperand &>(*Operands.back());
3375 Op.setModifiers(Mods);
3376
3377 return ParseStatus::Success;
3378
3379 } else {
3380 int64_t IntVal;
3381 const MCExpr *Expr;
3382 SMLoc S = getLoc();
3383
3384 if (HasSP3AbsModifier) {
3385 // This is a workaround for handling expressions
3386 // as arguments of SP3 'abs' modifier, for example:
3387 // |1.0|
3388 // |-1|
3389 // |1+x|
3390 // This syntax is not compatible with syntax of standard
3391 // MC expressions (due to the trailing '|').
3392 SMLoc EndLoc;
3393 if (getParser().parsePrimaryExpr(Expr, EndLoc, nullptr))
3394 return ParseStatus::Failure;
3395 } else {
3396 if (Parser.parseExpression(Expr))
3397 return ParseStatus::Failure;
3398 }
3399
3400 if (Expr->evaluateAsAbsolute(IntVal)) {
3401 if (Lit == LitModifier::Lit && !isInt<32>(IntVal) && !isUInt<32>(IntVal))
3402 return Error(S, "literal value out of range");
3403 Operands.push_back(AMDGPUOperand::CreateImm(this, IntVal, S));
3404 AMDGPUOperand &Op = static_cast<AMDGPUOperand &>(*Operands.back());
3405 Op.setModifiers(Mods);
3406 } else {
3407 if (Lit != LitModifier::None)
3408 return ParseStatus::NoMatch;
3409 Operands.push_back(AMDGPUOperand::CreateExpr(this, Expr, S));
3410 }
3411
3412 return ParseStatus::Success;
3413 }
3414
3415 return ParseStatus::NoMatch;
3416}
3417
3418ParseStatus AMDGPUAsmParser::parseReg(OperandVector &Operands) {
3419 if (!isRegister())
3420 return ParseStatus::NoMatch;
3421
3422 if (auto R = parseRegister()) {
3423 assert(R->isReg());
3424 Operands.push_back(std::move(R));
3425 return ParseStatus::Success;
3426 }
3427 return ParseStatus::Failure;
3428}
3429
3430ParseStatus AMDGPUAsmParser::parseRegOrImm(OperandVector &Operands,
3431 bool HasSP3AbsMod, LitModifier Lit) {
3432 ParseStatus Res = parseReg(Operands);
3433 if (!Res.isNoMatch())
3434 return Res;
3435 if (isModifier())
3436 return ParseStatus::NoMatch;
3437 return parseImm(Operands, HasSP3AbsMod, Lit);
3438}
3439
3440bool AMDGPUAsmParser::isNamedOperandModifier(const AsmToken &Token,
3441 const AsmToken &NextToken) const {
3442 if (Token.is(AsmToken::Identifier) && NextToken.is(AsmToken::LParen)) {
3443 const auto &str = Token.getString();
3444 return str == "abs" || str == "neg" || str == "sext";
3445 }
3446 return false;
3447}
3448
3449bool AMDGPUAsmParser::isOpcodeModifierWithVal(const AsmToken &Token,
3450 const AsmToken &NextToken) const {
3451 return Token.is(AsmToken::Identifier) && NextToken.is(AsmToken::Colon);
3452}
3453
3454bool AMDGPUAsmParser::isOperandModifier(const AsmToken &Token,
3455 const AsmToken &NextToken) const {
3456 return isNamedOperandModifier(Token, NextToken) || Token.is(AsmToken::Pipe);
3457}
3458
3459bool AMDGPUAsmParser::isRegOrOperandModifier(const AsmToken &Token,
3460 const AsmToken &NextToken) const {
3461 return isRegister(Token, NextToken) || isOperandModifier(Token, NextToken);
3462}
3463
3464// Check if this is an operand modifier or an opcode modifier
3465// which may look like an expression but it is not. We should
3466// avoid parsing these modifiers as expressions. Currently
3467// recognized sequences are:
3468// |...|
3469// abs(...)
3470// neg(...)
3471// sext(...)
3472// -reg
3473// -|...|
3474// -abs(...)
3475// name:...
3476//
3477bool AMDGPUAsmParser::isModifier() {
3478
3479 AsmToken Tok = getToken();
3480 AsmToken NextToken[2];
3481 peekTokens(NextToken);
3482
3483 return isOperandModifier(Tok, NextToken[0]) ||
3484 (Tok.is(AsmToken::Minus) &&
3485 isRegOrOperandModifier(NextToken[0], NextToken[1])) ||
3486 isOpcodeModifierWithVal(Tok, NextToken[0]);
3487}
3488
3489// Check if the current token is an SP3 'neg' modifier.
3490// Currently this modifier is allowed in the following context:
3491//
3492// 1. Before a register, e.g. "-v0", "-v[...]" or "-[v0,v1]".
3493// 2. Before an 'abs' modifier: -abs(...)
3494// 3. Before an SP3 'abs' modifier: -|...|
3495//
3496// In all other cases "-" is handled as a part
3497// of an expression that follows the sign.
3498//
3499// Note: When "-" is followed by an integer literal,
3500// this is interpreted as integer negation rather
3501// than a floating-point NEG modifier applied to N.
3502// Beside being contr-intuitive, such use of floating-point
3503// NEG modifier would have resulted in different meaning
3504// of integer literals used with VOP1/2/C and VOP3,
3505// for example:
3506// v_exp_f32_e32 v5, -1 // VOP1: src0 = 0xFFFFFFFF
3507// v_exp_f32_e64 v5, -1 // VOP3: src0 = 0x80000001
3508// Negative fp literals with preceding "-" are
3509// handled likewise for uniformity
3510//
3511bool AMDGPUAsmParser::parseSP3NegModifier() {
3512
3513 AsmToken NextToken[2];
3514 peekTokens(NextToken);
3515
3516 if (isToken(AsmToken::Minus) &&
3517 (isRegister(NextToken[0], NextToken[1]) ||
3518 NextToken[0].is(AsmToken::Pipe) || isId(NextToken[0], "abs"))) {
3519 lex();
3520 return true;
3521 }
3522
3523 return false;
3524}
3525
3526ParseStatus
3527AMDGPUAsmParser::parseRegOrImmWithFPInputMods(OperandVector &Operands,
3528 bool AllowImm) {
3529 bool Neg, SP3Neg;
3530 bool Abs, SP3Abs;
3531 SMLoc Loc;
3532
3533 // Disable ambiguous constructs like '--1' etc. Should use neg(-1) instead.
3534 if (isToken(AsmToken::Minus) && peekToken().is(AsmToken::Minus))
3535 return Error(getLoc(), "invalid syntax, expected 'neg' modifier");
3536
3537 SP3Neg = parseSP3NegModifier();
3538
3539 Loc = getLoc();
3540 Neg = trySkipId("neg");
3541 if (Neg && SP3Neg)
3542 return Error(Loc, "expected register or immediate");
3543 if (Neg && !skipToken(AsmToken::LParen, "expected left paren after neg"))
3544 return ParseStatus::Failure;
3545
3546 Abs = trySkipId("abs");
3547 if (Abs && !skipToken(AsmToken::LParen, "expected left paren after abs"))
3548 return ParseStatus::Failure;
3549
3550 LitModifier Lit = LitModifier::None;
3551 if (trySkipId("lit")) {
3552 Lit = LitModifier::Lit;
3553 if (!skipToken(AsmToken::LParen, "expected left paren after lit"))
3554 return ParseStatus::Failure;
3555 } else if (trySkipId("lit64")) {
3556 Lit = LitModifier::Lit64;
3557 if (!skipToken(AsmToken::LParen, "expected left paren after lit64"))
3558 return ParseStatus::Failure;
3559 if (!has64BitLiterals())
3560 return Error(Loc, "lit64 is not supported on this GPU");
3561 }
3562
3563 Loc = getLoc();
3564 SP3Abs = trySkipToken(AsmToken::Pipe);
3565 if (Abs && SP3Abs)
3566 return Error(Loc, "expected register or immediate");
3567
3568 ParseStatus Res;
3569 if (AllowImm) {
3570 Res = parseRegOrImm(Operands, SP3Abs, Lit);
3571 } else {
3572 Res = parseReg(Operands);
3573 }
3574 if (!Res.isSuccess())
3575 return (SP3Neg || Neg || SP3Abs || Abs || Lit != LitModifier::None)
3577 : Res;
3578
3579 if (Lit != LitModifier::None && !Operands.back()->isImm())
3580 Error(Loc, "expected immediate with lit modifier");
3581
3582 if (SP3Abs && !skipToken(AsmToken::Pipe, "expected vertical bar"))
3583 return ParseStatus::Failure;
3584 if (Abs && !skipToken(AsmToken::RParen, "expected closing parentheses"))
3585 return ParseStatus::Failure;
3586 if (Neg && !skipToken(AsmToken::RParen, "expected closing parentheses"))
3587 return ParseStatus::Failure;
3588 if (Lit != LitModifier::None &&
3589 !skipToken(AsmToken::RParen, "expected closing parentheses"))
3590 return ParseStatus::Failure;
3591
3592 AMDGPUOperand::Modifiers Mods;
3593 Mods.Abs = Abs || SP3Abs;
3594 Mods.Neg = Neg || SP3Neg;
3595 Mods.Lit = Lit;
3596
3597 if (Mods.hasFPModifiers() || Lit != LitModifier::None) {
3598 AMDGPUOperand &Op = static_cast<AMDGPUOperand &>(*Operands.back());
3599 if (Op.isExpr())
3600 return Error(Op.getStartLoc(), "expected an absolute expression");
3601 Op.setModifiers(Mods);
3602 }
3603 return ParseStatus::Success;
3604}
3605
3606ParseStatus
3607AMDGPUAsmParser::parseRegOrImmWithIntInputMods(OperandVector &Operands,
3608 bool AllowImm) {
3609 bool Sext = trySkipId("sext");
3610 if (Sext && !skipToken(AsmToken::LParen, "expected left paren after sext"))
3611 return ParseStatus::Failure;
3612
3613 ParseStatus Res;
3614 if (AllowImm) {
3615 Res = parseRegOrImm(Operands);
3616 } else {
3617 Res = parseReg(Operands);
3618 }
3619 if (!Res.isSuccess())
3620 return Sext ? ParseStatus::Failure : Res;
3621
3622 if (Sext && !skipToken(AsmToken::RParen, "expected closing parentheses"))
3623 return ParseStatus::Failure;
3624
3625 AMDGPUOperand::Modifiers Mods;
3626 Mods.Sext = Sext;
3627
3628 if (Mods.hasIntModifiers()) {
3629 AMDGPUOperand &Op = static_cast<AMDGPUOperand &>(*Operands.back());
3630 if (Op.isExpr())
3631 return Error(Op.getStartLoc(), "expected an absolute expression");
3632 Op.setModifiers(Mods);
3633 }
3634
3635 return ParseStatus::Success;
3636}
3637
3638ParseStatus AMDGPUAsmParser::parseRegWithFPInputMods(OperandVector &Operands) {
3639 return parseRegOrImmWithFPInputMods(Operands, false);
3640}
3641
3642ParseStatus AMDGPUAsmParser::parseRegWithIntInputMods(OperandVector &Operands) {
3643 return parseRegOrImmWithIntInputMods(Operands, false);
3644}
3645
3646ParseStatus AMDGPUAsmParser::parseRsrcReg(OperandVector &Operands) {
3647 // Without the marker, fall back to plain register parsing so the legacy
3648 // bare-register form (e.g. `s8`, `v8`) still assembles for indexed
3649 // buffer/image instructions.
3650 if (!trySkipId("rsrcidx"))
3651 return parseReg(Operands);
3652
3653 if (!skipToken(AsmToken::LParen, "expected left paren after rsrcidx"))
3654 return ParseStatus::Failure;
3655
3656 SMLoc RegLoc = getLoc();
3657 std::unique_ptr<AMDGPUOperand> Reg = parseRegister();
3658 if (!Reg)
3659 return ParseStatus::Failure;
3660
3661 // Enforce that the inner register is a valid index register. The matcher
3662 // predicate alone is not sufficient: if it fails, the matcher will fall back
3663 // to a non-indexed instruction variant whose resource operand happens to
3664 // accept the same register, silently dropping the `rsrcidx` intent.
3665 if (!Reg->isRsrcReg32())
3666 return Error(RegLoc, "rsrcidx operand must be a 32-bit SGPR or VGPR");
3667
3668 if (!skipToken(AsmToken::RParen, "expected closing parenthesis"))
3669 return ParseStatus::Failure;
3670
3671 Operands.push_back(std::move(Reg));
3672 return ParseStatus::Success;
3673}
3674
3675ParseStatus AMDGPUAsmParser::parseVReg32OrOff(OperandVector &Operands) {
3676 auto Loc = getLoc();
3677 if (trySkipId("off")) {
3678 Operands.push_back(
3679 AMDGPUOperand::CreateImm(this, 0, Loc, AMDGPUOperand::ImmTyOff, false));
3680 return ParseStatus::Success;
3681 }
3682
3683 if (!isRegister())
3684 return ParseStatus::NoMatch;
3685
3686 std::unique_ptr<AMDGPUOperand> Reg = parseRegister();
3687 if (Reg) {
3688 Operands.push_back(std::move(Reg));
3689 return ParseStatus::Success;
3690 }
3691
3692 return ParseStatus::Failure;
3693}
3694
3695unsigned AMDGPUAsmParser::checkTargetMatchPredicate(MCInst &Inst) {
3696 if ((getForcedEncodingSize() == 32 && SIInstrFlags::isVOP3(MII, Inst)) ||
3697 (getForcedEncodingSize() == 64 && !SIInstrFlags::isVOP3(MII, Inst)) ||
3698 (isForcedDPP() && !SIInstrFlags::isDPP(MII, Inst)) ||
3699 (isForcedSDWA() && !SIInstrFlags::isSDWA(MII, Inst)))
3700 return Match_InvalidOperand;
3701
3702 if (Inst.getOpcode() == AMDGPU::V_MAC_F32_sdwa_vi ||
3703 Inst.getOpcode() == AMDGPU::V_MAC_F16_sdwa_vi) {
3704 // v_mac_f32/16 allow only dst_sel == DWORD;
3705 auto OpNum =
3706 AMDGPU::getNamedOperandIdx(Inst.getOpcode(), AMDGPU::OpName::dst_sel);
3707 const auto &Op = Inst.getOperand(OpNum);
3708 if (!Op.isImm() || Op.getImm() != AMDGPU::SDWA::SdwaSel::DWORD) {
3709 return Match_InvalidOperand;
3710 }
3711 }
3712
3713 // Asm can first try to match VOPD or VOPD3. By failing early here with
3714 // Match_InvalidOperand, the parser will retry parsing as VOPD3 or VOPD.
3715 // Checking later during validateInstruction does not give a chance to retry
3716 // parsing as a different encoding.
3717 if (tryAnotherVOPDEncoding(Inst))
3718 return Match_InvalidOperand;
3719
3720 return Match_Success;
3721}
3722
3731
3732// What asm variants we should check
3733ArrayRef<unsigned> AMDGPUAsmParser::getMatchedVariants() const {
3734 if (isForcedDPP() && isForcedVOP3()) {
3735 static const unsigned Variants[] = {AMDGPUAsmVariants::VOP3_DPP};
3736 return ArrayRef(Variants);
3737 }
3738 if (getForcedEncodingSize() == 32) {
3739 static const unsigned Variants[] = {AMDGPUAsmVariants::DEFAULT};
3740 return ArrayRef(Variants);
3741 }
3742
3743 if (isForcedVOP3()) {
3744 static const unsigned Variants[] = {AMDGPUAsmVariants::VOP3};
3745 return ArrayRef(Variants);
3746 }
3747
3748 if (isForcedSDWA()) {
3749 static const unsigned Variants[] = {AMDGPUAsmVariants::SDWA,
3751 return ArrayRef(Variants);
3752 }
3753
3754 if (isForcedDPP()) {
3755 static const unsigned Variants[] = {AMDGPUAsmVariants::DPP};
3756 return ArrayRef(Variants);
3757 }
3758
3759 return getAllVariants();
3760}
3761
3762StringRef AMDGPUAsmParser::getMatchedVariantName() const {
3763 if (isForcedDPP() && isForcedVOP3())
3764 return "e64_dpp";
3765
3766 if (getForcedEncodingSize() == 32)
3767 return "e32";
3768
3769 if (isForcedVOP3())
3770 return "e64";
3771
3772 if (isForcedSDWA())
3773 return "sdwa";
3774
3775 if (isForcedDPP())
3776 return "dpp";
3777
3778 return "";
3779}
3780
3781MCRegister
3782AMDGPUAsmParser::findImplicitSGPRReadInVOP(const MCInst &Inst) const {
3783 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
3784 for (MCPhysReg Reg : Desc.implicit_uses()) {
3785 switch (Reg) {
3786 case AMDGPU::FLAT_SCR:
3787 case AMDGPU::VCC:
3788 case AMDGPU::VCC_LO:
3789 case AMDGPU::VCC_HI:
3790 case AMDGPU::M0:
3791 return Reg;
3792 default:
3793 break;
3794 }
3795 }
3796 return MCRegister();
3797}
3798
3799// NB: This code is correct only when used to check constant
3800// bus limitations because GFX7 support no f16 inline constants.
3801// Note that there are no cases when a GFX7 opcode violates
3802// constant bus limitations due to the use of an f16 constant.
3803bool AMDGPUAsmParser::isInlineConstant(const MCInst &Inst,
3804 unsigned OpIdx) const {
3805 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
3806
3807 if (!AMDGPU::isSISrcOperand(Desc, OpIdx) ||
3808 AMDGPU::isKImmOperand(Desc, OpIdx)) {
3809 return false;
3810 }
3811
3812 const MCOperand &MO = Inst.getOperand(OpIdx);
3813
3814 int64_t Val = MO.isImm() ? MO.getImm() : getLitValue(MO.getExpr());
3815 auto OpSize = AMDGPU::getOperandSize(Desc, OpIdx);
3816
3817 switch (OpSize) { // expected operand size
3818 case 8:
3819 return AMDGPU::isInlinableLiteral64(Val, hasInv2PiInlineImm());
3820 case 4:
3821 return AMDGPU::isInlinableLiteral32(Val, hasInv2PiInlineImm());
3822 case 2: {
3823 const unsigned OperandType = Desc.operands()[OpIdx].OperandType;
3826 return AMDGPU::isInlinableLiteralI16(Val, hasInv2PiInlineImm());
3827
3831
3835
3838
3842
3845 return AMDGPU::isInlinableLiteralFP16(Val, hasInv2PiInlineImm());
3846
3849 return AMDGPU::isInlinableLiteralBF16(Val, hasInv2PiInlineImm());
3850
3853 return false;
3854
3855 llvm_unreachable("invalid operand type");
3856 }
3857 default:
3858 llvm_unreachable("invalid operand size");
3859 }
3860}
3861
3862unsigned AMDGPUAsmParser::getConstantBusLimit(unsigned Opcode) const {
3863 if (!isGFX10Plus())
3864 return 1;
3865
3866 switch (Opcode) {
3867 // 64-bit shift instructions can use only one scalar value input
3868 case AMDGPU::V_LSHLREV_B64_e64:
3869 case AMDGPU::V_LSHLREV_B64_gfx10:
3870 case AMDGPU::V_LSHLREV_B64_e64_gfx11:
3871 case AMDGPU::V_LSHLREV_B64_e32_gfx12:
3872 case AMDGPU::V_LSHLREV_B64_e64_gfx12:
3873 case AMDGPU::V_LSHRREV_B64_e64:
3874 case AMDGPU::V_LSHRREV_B64_gfx10:
3875 case AMDGPU::V_LSHRREV_B64_e64_gfx11:
3876 case AMDGPU::V_LSHRREV_B64_e64_gfx12:
3877 case AMDGPU::V_ASHRREV_I64_e64:
3878 case AMDGPU::V_ASHRREV_I64_gfx10:
3879 case AMDGPU::V_ASHRREV_I64_e64_gfx11:
3880 case AMDGPU::V_ASHRREV_I64_e64_gfx12:
3881 case AMDGPU::V_LSHL_B64_e64:
3882 case AMDGPU::V_LSHR_B64_e64:
3883 case AMDGPU::V_ASHR_I64_e64:
3884 return 1;
3885 default:
3886 return 2;
3887 }
3888}
3889
3890constexpr unsigned MAX_SRC_OPERANDS_NUM = 6;
3892
3893// Get regular operand indices in the same order as specified
3894// in the instruction (but append mandatory literals to the end).
3896 bool AddMandatoryLiterals = false) {
3897
3898 int16_t ImmIdx =
3899 AddMandatoryLiterals ? getNamedOperandIdx(Opcode, OpName::imm) : -1;
3900
3901 if (isVOPD(Opcode)) {
3902 int16_t ImmXIdx =
3903 AddMandatoryLiterals ? getNamedOperandIdx(Opcode, OpName::immX) : -1;
3904
3905 return {getNamedOperandIdx(Opcode, OpName::src0X),
3906 getNamedOperandIdx(Opcode, OpName::vsrc1X),
3907 getNamedOperandIdx(Opcode, OpName::vsrc2X),
3908 getNamedOperandIdx(Opcode, OpName::src0Y),
3909 getNamedOperandIdx(Opcode, OpName::vsrc1Y),
3910 getNamedOperandIdx(Opcode, OpName::vsrc2Y),
3911 ImmXIdx,
3912 ImmIdx};
3913 }
3914
3915 return {getNamedOperandIdx(Opcode, OpName::src0),
3916 getNamedOperandIdx(Opcode, OpName::src1),
3917 getNamedOperandIdx(Opcode, OpName::src2), ImmIdx};
3918}
3919
3920bool AMDGPUAsmParser::usesConstantBus(const MCInst &Inst, unsigned OpIdx) {
3921 const MCOperand &MO = Inst.getOperand(OpIdx);
3922 if (MO.isImm())
3923 return !isInlineConstant(Inst, OpIdx);
3924 if (MO.isReg()) {
3925 auto Reg = MO.getReg();
3926 if (!Reg)
3927 return false;
3928 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
3929 auto PReg = mc2PseudoReg(Reg);
3930 return isSGPR(PReg, TRI) && PReg != SGPR_NULL;
3931 }
3932 return true;
3933}
3934
3935// Based on the comment for `AMDGPUInstructionSelector::selectWritelane`:
3936// Writelane is special in that it can use SGPR and M0 (which would normally
3937// count as using the constant bus twice - but in this case it is allowed since
3938// the lane selector doesn't count as a use of the constant bus). However, it is
3939// still required to abide by the 1 SGPR rule.
3940static bool checkWriteLane(const MCInst &Inst) {
3941 const unsigned Opcode = Inst.getOpcode();
3942 if (Opcode != V_WRITELANE_B32_gfx6_gfx7 && Opcode != V_WRITELANE_B32_vi)
3943 return false;
3944 const MCOperand &LaneSelOp = Inst.getOperand(2);
3945 if (!LaneSelOp.isReg())
3946 return false;
3947 auto LaneSelReg = mc2PseudoReg(LaneSelOp.getReg());
3948 return LaneSelReg == M0 || LaneSelReg == M0_gfxpre11;
3949}
3950
3951bool AMDGPUAsmParser::validateConstantBusLimitations(
3952 const MCInst &Inst, const OperandVector &Operands) {
3953 const unsigned Opcode = Inst.getOpcode();
3954 const MCInstrDesc &Desc = MII.get(Opcode);
3955 MCRegister LastSGPR;
3956 unsigned ConstantBusUseCount = 0;
3957 unsigned NumLiterals = 0;
3958 unsigned LiteralSize;
3959
3962 !SIInstrFlags::isSDWA(Desc) && !isVOPD(Opcode))
3963 return true;
3964
3965 if (checkWriteLane(Inst))
3966 return true;
3967
3968 // Check special imm operands (used by madmk, etc)
3969 if (AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::imm)) {
3970 ++NumLiterals;
3971 LiteralSize = 4;
3972 }
3973
3974 SmallDenseSet<MCRegister> SGPRsUsed;
3975 MCRegister SGPRUsed = findImplicitSGPRReadInVOP(Inst);
3976 if (SGPRUsed) {
3977 SGPRsUsed.insert(SGPRUsed);
3978 ++ConstantBusUseCount;
3979 }
3980
3981 OperandIndices OpIndices = getSrcOperandIndices(Opcode);
3982
3983 unsigned ConstantBusLimit = getConstantBusLimit(Opcode);
3984
3985 for (int OpIdx : OpIndices) {
3986 if (OpIdx == -1)
3987 continue;
3988
3989 const MCOperand &MO = Inst.getOperand(OpIdx);
3990 if (usesConstantBus(Inst, OpIdx)) {
3991 if (MO.isReg()) {
3992 LastSGPR = mc2PseudoReg(MO.getReg());
3993 // Pairs of registers with a partial intersections like these
3994 // s0, s[0:1]
3995 // flat_scratch_lo, flat_scratch
3996 // flat_scratch_lo, flat_scratch_hi
3997 // are theoretically valid but they are disabled anyway.
3998 // Note that this code mimics SIInstrInfo::verifyInstruction
3999 if (SGPRsUsed.insert(LastSGPR).second) {
4000 ++ConstantBusUseCount;
4001 }
4002 } else { // Expression or a literal
4003
4004 if (Desc.operands()[OpIdx].OperandType == MCOI::OPERAND_IMMEDIATE)
4005 continue; // special operand like VINTERP attr_chan
4006
4007 // An instruction may use only one literal.
4008 // This has been validated on the previous step.
4009 // See validateVOPLiteral.
4010 // This literal may be used as more than one operand.
4011 // If all these operands are of the same size,
4012 // this literal counts as one scalar value.
4013 // Otherwise it counts as 2 scalar values.
4014 // See "GFX10 Shader Programming", section 3.6.2.3.
4015
4016 unsigned Size = AMDGPU::getOperandSize(Desc, OpIdx);
4017 if (Size < 4)
4018 Size = 4;
4019
4020 if (NumLiterals == 0) {
4021 NumLiterals = 1;
4022 LiteralSize = Size;
4023 } else if (LiteralSize != Size) {
4024 NumLiterals = 2;
4025 }
4026 }
4027 }
4028
4029 if (ConstantBusUseCount + NumLiterals > ConstantBusLimit) {
4030 Error(getOperandLoc(Operands, OpIdx),
4031 "invalid operand (violates constant bus restrictions)");
4032 return false;
4033 }
4034 }
4035 return true;
4036}
4037
4038std::optional<unsigned>
4039AMDGPUAsmParser::checkVOPDRegBankConstraints(const MCInst &Inst, bool AsVOPD3) {
4040
4041 const unsigned Opcode = Inst.getOpcode();
4042 if (!isVOPD(Opcode))
4043 return {};
4044
4045 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
4046
4047 auto getVRegIdx = [&](unsigned, unsigned OperandIdx) {
4048 const MCOperand &Opr = Inst.getOperand(OperandIdx);
4049 return (Opr.isReg() && !isSGPR(mc2PseudoReg(Opr.getReg()), TRI))
4050 ? Opr.getReg()
4051 : MCRegister();
4052 };
4053
4054 // On GFX1170+ if both OpX and OpY are V_MOV_B32 then OPY uses SRC2
4055 // source-cache.
4056 bool SkipSrc =
4057 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx1170 ||
4058 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx12 ||
4059 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx1250 ||
4060 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx13 ||
4061 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_e96_gfx1250 ||
4062 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_e96_gfx13;
4063 bool AllowSameVGPR = isGFX12Plus();
4064
4065 if (AsVOPD3) { // Literal constants are not allowed with VOPD3.
4066 for (auto OpName : {OpName::src0X, OpName::src0Y}) {
4067 int I = getNamedOperandIdx(Opcode, OpName);
4068 const MCOperand &Op = Inst.getOperand(I);
4069 if (!Op.isImm())
4070 continue;
4071 int64_t Imm = Op.getImm();
4072 if (!AMDGPU::isInlinableLiteral32(Imm, hasInv2PiInlineImm()) &&
4073 !AMDGPU::isInlinableLiteral64(Imm, hasInv2PiInlineImm()))
4074 return (unsigned)I;
4075 }
4076
4077 for (auto OpName : {OpName::vsrc1X, OpName::vsrc1Y, OpName::vsrc2X,
4078 OpName::vsrc2Y, OpName::imm}) {
4079 int I = getNamedOperandIdx(Opcode, OpName);
4080 if (I == -1)
4081 continue;
4082 const MCOperand &Op = Inst.getOperand(I);
4083 if (Op.isImm())
4084 return (unsigned)I;
4085 }
4086 }
4087
4088 const auto &InstInfo = getVOPDInstInfo(Opcode, &MII);
4089 auto InvalidCompOprIdx = InstInfo.getInvalidCompOperandIndex(
4090 getVRegIdx, *TRI, SkipSrc, AllowSameVGPR, AsVOPD3);
4091
4092 return InvalidCompOprIdx;
4093}
4094
4095bool AMDGPUAsmParser::validateVOPD(const MCInst &Inst,
4096 const OperandVector &Operands) {
4097
4098 unsigned Opcode = Inst.getOpcode();
4099 bool AsVOPD3 = SIInstrFlags::isVOPD3(MII, Inst);
4100
4101 if (AsVOPD3) {
4102 for (const std::unique_ptr<MCParsedAsmOperand> &Operand : Operands) {
4103 AMDGPUOperand &Op = (AMDGPUOperand &)*Operand;
4104 if ((Op.isRegKind() || Op.isImmTy(AMDGPUOperand::ImmTyNone)) &&
4105 (Op.getModifiers().getFPModifiersOperand() & SISrcMods::ABS))
4106 Error(Op.getStartLoc(), "ABS not allowed in VOPD3 instructions");
4107 }
4108 }
4109
4110 auto InvalidCompOprIdx = checkVOPDRegBankConstraints(Inst, AsVOPD3);
4111 if (!InvalidCompOprIdx.has_value())
4112 return true;
4113
4114 auto CompOprIdx = *InvalidCompOprIdx;
4115 const auto &InstInfo = getVOPDInstInfo(Opcode, &MII);
4116 auto ParsedIdx =
4117 std::max(InstInfo[VOPD::X].getIndexInParsedOperands(CompOprIdx),
4118 InstInfo[VOPD::Y].getIndexInParsedOperands(CompOprIdx));
4119 assert(ParsedIdx > 0 && ParsedIdx < Operands.size());
4120
4121 auto Loc = ((AMDGPUOperand &)*Operands[ParsedIdx]).getStartLoc();
4122 if (CompOprIdx == VOPD::Component::DST) {
4123 if (AsVOPD3)
4124 Error(Loc, "dst registers must be distinct");
4125 else
4126 Error(Loc, "one dst register must be even and the other odd");
4127 } else {
4128 auto CompSrcIdx = CompOprIdx - VOPD::Component::DST_NUM;
4129 Error(Loc, Twine("src") + Twine(CompSrcIdx) +
4130 " operands must use different VGPR banks");
4131 }
4132
4133 return false;
4134}
4135
4136// \returns true if \p Inst does not satisfy VOPD constraints, but can be
4137// potentially used as VOPD3 with the same operands.
4138bool AMDGPUAsmParser::tryVOPD3(const MCInst &Inst) {
4139 // First check if it fits VOPD
4140 auto InvalidCompOprIdx = checkVOPDRegBankConstraints(Inst, false);
4141 if (!InvalidCompOprIdx.has_value())
4142 return false;
4143
4144 // Then if it fits VOPD3
4145 InvalidCompOprIdx = checkVOPDRegBankConstraints(Inst, true);
4146 if (InvalidCompOprIdx.has_value()) {
4147 // If failed operand is dst it is better to show error about VOPD3
4148 // instruction as it has more capabilities and error message will be
4149 // more informative. If the dst is not legal for VOPD3, then it is not
4150 // legal for VOPD either.
4151 if (*InvalidCompOprIdx == VOPD::Component::DST)
4152 return true;
4153
4154 // Otherwise prefer VOPD as we may find ourselves in an awkward situation
4155 // with a conflict in tied implicit src2 of fmac and no asm operand to
4156 // to point to.
4157 return false;
4158 }
4159 return true;
4160}
4161
4162// \returns true is a VOPD3 instruction can be also represented as a shorter
4163// VOPD encoding.
4164bool AMDGPUAsmParser::tryVOPD(const MCInst &Inst) {
4165 const unsigned Opcode = Inst.getOpcode();
4166 const auto &II = getVOPDInstInfo(Opcode, &MII);
4167 unsigned EncodingFamily = AMDGPU::getVOPDEncodingFamily(getSTI());
4168 if (!getCanBeVOPD(II[VOPD::X].getOpcode(), EncodingFamily, false).X ||
4169 !getCanBeVOPD(II[VOPD::Y].getOpcode(), EncodingFamily, false).Y)
4170 return false;
4171
4172 // This is an awkward exception, VOPD3 variant of V_DUAL_CNDMASK_B32 has
4173 // explicit src2 even if it is vcc_lo. If it was parsed as VOPD3 it cannot
4174 // be parsed as VOPD which does not accept src2.
4175 if (II[VOPD::X].getOpcode() == AMDGPU::V_CNDMASK_B32_e32 ||
4176 II[VOPD::Y].getOpcode() == AMDGPU::V_CNDMASK_B32_e32)
4177 return false;
4178
4179 // If any modifiers are set this cannot be VOPD.
4180 for (auto OpName : {OpName::src0X_modifiers, OpName::src0Y_modifiers,
4181 OpName::vsrc1X_modifiers, OpName::vsrc1Y_modifiers,
4182 OpName::vsrc2X_modifiers, OpName::vsrc2Y_modifiers}) {
4183 int I = getNamedOperandIdx(Opcode, OpName);
4184 if (I == -1)
4185 continue;
4186 if (Inst.getOperand(I).getImm())
4187 return false;
4188 }
4189
4190 return !tryVOPD3(Inst);
4191}
4192
4193// VOPD3 has more relaxed register constraints than VOPD. We prefer shorter VOPD
4194// form but switch to VOPD3 otherwise.
4195bool AMDGPUAsmParser::tryAnotherVOPDEncoding(const MCInst &Inst) {
4196 if (!isGFX1250Plus() || !isVOPD(Inst.getOpcode()))
4197 return false;
4198
4199 if (SIInstrFlags::isVOPD3(MII, Inst))
4200 return tryVOPD(Inst);
4201 return tryVOPD3(Inst);
4202}
4203
4204bool AMDGPUAsmParser::validateIntClampSupported(const MCInst &Inst) {
4205
4206 const unsigned Opc = Inst.getOpcode();
4207
4208 if (SIInstrFlags::hasIntClamp(MII, Inst) && !hasIntClamp()) {
4209 int ClampIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::clamp);
4210 assert(ClampIdx != -1);
4211 return Inst.getOperand(ClampIdx).getImm() == 0;
4212 }
4213
4214 return true;
4215}
4216
4217bool AMDGPUAsmParser::validateMIMGDataSize(const MCInst &Inst, SMLoc IDLoc) {
4218
4219 const unsigned Opc = Inst.getOpcode();
4220 const MCInstrDesc &Desc = MII.get(Opc);
4221
4222 if ((SIInstrFlags::isImage(Desc)) == 0)
4223 return true;
4224
4225 int VDataIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdata);
4226 int DMaskIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::dmask);
4227 int TFEIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::tfe);
4228
4229 if (VDataIdx == -1 && isGFX10Plus()) // no return image_sample
4230 return true;
4231
4232 if ((DMaskIdx == -1 || TFEIdx == -1) &&
4233 hasBVHRayTracingInsts()) // intersect_ray
4234 return true;
4235
4236 unsigned VDataSize = getRegOperandSize(Desc, VDataIdx);
4237 unsigned TFESize = (TFEIdx != -1 && Inst.getOperand(TFEIdx).getImm()) ? 1 : 0;
4238 unsigned DMask = Inst.getOperand(DMaskIdx).getImm() & 0xf;
4239 if (DMask == 0)
4240 DMask = 1;
4241
4242 bool IsPackedD16 = false;
4243 unsigned DataSize = SIInstrFlags::isGather4(Desc) ? 4 : llvm::popcount(DMask);
4244 if (hasPackedD16()) {
4245 int D16Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::d16);
4246 IsPackedD16 = D16Idx >= 0;
4247 if (IsPackedD16 && Inst.getOperand(D16Idx).getImm())
4248 DataSize = (DataSize + 1) / 2;
4249 }
4250
4251 if ((VDataSize / 4) == DataSize + TFESize)
4252 return true;
4253
4254 StringRef Modifiers;
4255 if (isGFX90A())
4256 Modifiers = IsPackedD16 ? "dmask and d16" : "dmask";
4257 else
4258 Modifiers = IsPackedD16 ? "dmask, d16 and tfe" : "dmask and tfe";
4259
4260 Error(IDLoc, Twine("image data size does not match ") + Modifiers);
4261 return false;
4262}
4263
4264bool AMDGPUAsmParser::validateMIMGAddrSize(const MCInst &Inst, SMLoc IDLoc) {
4265 const unsigned Opc = Inst.getOpcode();
4266 const MCInstrDesc &Desc = MII.get(Opc);
4267
4269 return true;
4270
4271 const AMDGPU::MIMGInfo *Info = AMDGPU::getMIMGInfo(Opc);
4272
4273 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode =
4275 int VAddr0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vaddr0);
4276 AMDGPU::OpName RSrcOpName =
4277 SIInstrFlags::isMIMG(Desc) ? AMDGPU::OpName::srsrc : AMDGPU::OpName::rsrc;
4278 int SrsrcIdx = AMDGPU::getNamedOperandIdx(Opc, RSrcOpName);
4279 int DimIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::dim);
4280 int A16Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::a16);
4281
4282 assert(VAddr0Idx != -1);
4283 assert(SrsrcIdx != -1);
4284 assert(SrsrcIdx > VAddr0Idx);
4285
4286 bool IsA16 = (A16Idx != -1 && Inst.getOperand(A16Idx).getImm());
4287 if (BaseOpcode->BVH) {
4288 if (IsA16 == BaseOpcode->A16)
4289 return true;
4290 Error(IDLoc, "image address size does not match a16");
4291 return false;
4292 }
4293
4294 unsigned Dim = Inst.getOperand(DimIdx).getImm();
4295 const AMDGPU::MIMGDimInfo *DimInfo = AMDGPU::getMIMGDimInfoByEncoding(Dim);
4296 bool IsNSA = SrsrcIdx - VAddr0Idx > 1;
4297 unsigned ActualAddrSize =
4298 IsNSA ? SrsrcIdx - VAddr0Idx : getRegOperandSize(Desc, VAddr0Idx) / 4;
4299
4300 unsigned ExpectedAddrSize =
4301 AMDGPU::getAddrSizeMIMGOp(BaseOpcode, DimInfo, IsA16, hasG16());
4302
4303 if (IsNSA) {
4304 if (hasPartialNSAEncoding() &&
4305 ExpectedAddrSize > getNSAMaxSize(SIInstrFlags::isVSAMPLE(Desc))) {
4306 int VAddrLastIdx = SrsrcIdx - 1;
4307 unsigned VAddrLastSize = getRegOperandSize(Desc, VAddrLastIdx) / 4;
4308
4309 ActualAddrSize = VAddrLastIdx - VAddr0Idx + VAddrLastSize;
4310 }
4311 } else {
4312 if (ExpectedAddrSize > 12)
4313 ExpectedAddrSize = 16;
4314
4315 // Allow oversized 8 VGPR vaddr when only 5/6/7 VGPRs are required.
4316 // This provides backward compatibility for assembly created
4317 // before 160b/192b/224b types were directly supported.
4318 if (ActualAddrSize == 8 && (ExpectedAddrSize >= 5 && ExpectedAddrSize <= 7))
4319 return true;
4320 }
4321
4322 if (ActualAddrSize == ExpectedAddrSize)
4323 return true;
4324
4325 Error(IDLoc, "image address size does not match dim and a16");
4326 return false;
4327}
4328
4329bool AMDGPUAsmParser::validateMIMGAtomicDMask(const MCInst &Inst) {
4330
4331 const unsigned Opc = Inst.getOpcode();
4332 const MCInstrDesc &Desc = MII.get(Opc);
4333
4334 if ((SIInstrFlags::isImage(Desc)) == 0)
4335 return true;
4336 if (!Desc.mayLoad() || !Desc.mayStore())
4337 return true; // Not atomic
4338
4339 int DMaskIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::dmask);
4340 unsigned DMask = Inst.getOperand(DMaskIdx).getImm() & 0xf;
4341
4342 // This is an incomplete check because image_atomic_cmpswap
4343 // may only use 0x3 and 0xf while other atomic operations
4344 // may use 0x1 and 0x3. However these limitations are
4345 // verified when we check that dmask matches dst size.
4346 return DMask == 0x1 || DMask == 0x3 || DMask == 0xf;
4347}
4348
4349bool AMDGPUAsmParser::validateMIMGGatherDMask(const MCInst &Inst) {
4350
4351 const unsigned Opc = Inst.getOpcode();
4352
4353 if (!SIInstrFlags::isGather4(MII, Inst))
4354 return true;
4355
4356 int DMaskIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::dmask);
4357 unsigned DMask = Inst.getOperand(DMaskIdx).getImm() & 0xf;
4358
4359 // GATHER4 instructions use dmask in a different fashion compared to
4360 // other MIMG instructions. The only useful DMASK values are
4361 // 1=red, 2=green, 4=blue, 8=alpha. (e.g. 1 returns
4362 // (red,red,red,red) etc.) The ISA document doesn't mention
4363 // this.
4364 return DMask == 0x1 || DMask == 0x2 || DMask == 0x4 || DMask == 0x8;
4365}
4366
4367bool AMDGPUAsmParser::validateMIMGDim(const MCInst &Inst,
4368 const OperandVector &Operands) {
4369 if (!isGFX10Plus())
4370 return true;
4371
4372 const unsigned Opc = Inst.getOpcode();
4373
4374 if ((SIInstrFlags::isImage(MII, Inst)) == 0)
4375 return true;
4376
4377 // image_bvh_intersect_ray instructions do not have dim
4379 return true;
4380
4381 for (unsigned i = 1, e = Operands.size(); i != e; ++i) {
4382 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
4383 if (Op.isDim())
4384 return true;
4385 }
4386 return false;
4387}
4388
4389bool AMDGPUAsmParser::validateMIMGMSAA(const MCInst &Inst) {
4390 const unsigned Opc = Inst.getOpcode();
4391
4392 if ((SIInstrFlags::isImage(MII, Inst)) == 0)
4393 return true;
4394
4395 const AMDGPU::MIMGInfo *Info = AMDGPU::getMIMGInfo(Opc);
4396 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode =
4398
4399 if (!BaseOpcode->MSAA)
4400 return true;
4401
4402 int DimIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::dim);
4403 assert(DimIdx != -1);
4404
4405 unsigned Dim = Inst.getOperand(DimIdx).getImm();
4406 const AMDGPU::MIMGDimInfo *DimInfo = AMDGPU::getMIMGDimInfoByEncoding(Dim);
4407
4408 return DimInfo->MSAA;
4409}
4410
4411static bool IsMovrelsSDWAOpcode(const unsigned Opcode) {
4412 switch (Opcode) {
4413 case AMDGPU::V_MOVRELS_B32_sdwa_gfx10:
4414 case AMDGPU::V_MOVRELSD_B32_sdwa_gfx10:
4415 case AMDGPU::V_MOVRELSD_2_B32_sdwa_gfx10:
4416 return true;
4417 default:
4418 return false;
4419 }
4420}
4421
4422// movrels* opcodes should only allow VGPRS as src0.
4423// This is specified in .td description for vop1/vop3,
4424// but sdwa is handled differently. See isSDWAOperand.
4425bool AMDGPUAsmParser::validateMovrels(const MCInst &Inst,
4426 const OperandVector &Operands) {
4427
4428 const unsigned Opc = Inst.getOpcode();
4429
4430 if (!SIInstrFlags::isSDWA(MII, Inst) || !IsMovrelsSDWAOpcode(Opc))
4431 return true;
4432
4433 const int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
4434 assert(Src0Idx != -1);
4435
4436 const MCOperand &Src0 = Inst.getOperand(Src0Idx);
4437 if (Src0.isReg()) {
4438 auto Reg = mc2PseudoReg(Src0.getReg());
4439 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
4440 if (!isSGPR(Reg, TRI))
4441 return true;
4442 }
4443
4444 Error(getOperandLoc(Operands, Src0Idx), "source operand must be a VGPR");
4445 return false;
4446}
4447
4448bool AMDGPUAsmParser::validateMAIAccWrite(const MCInst &Inst,
4449 const OperandVector &Operands) {
4450
4451 const unsigned Opc = Inst.getOpcode();
4452
4453 if (Opc != AMDGPU::V_ACCVGPR_WRITE_B32_vi)
4454 return true;
4455
4456 const int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
4457 assert(Src0Idx != -1);
4458
4459 const MCOperand &Src0 = Inst.getOperand(Src0Idx);
4460 if (!Src0.isReg())
4461 return true;
4462
4463 auto Reg = mc2PseudoReg(Src0.getReg());
4464 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
4465 if (!isGFX90A() && isSGPR(Reg, TRI)) {
4466 Error(getOperandLoc(Operands, Src0Idx),
4467 "source operand must be either a VGPR or an inline constant");
4468 return false;
4469 }
4470
4471 return true;
4472}
4473
4474bool AMDGPUAsmParser::validateMAISrc2(const MCInst &Inst,
4475 const OperandVector &Operands) {
4476 unsigned Opcode = Inst.getOpcode();
4477
4478 if (!SIInstrFlags::isMAI(MII, Inst) ||
4479 !getFeatureBits()[FeatureMFMAInlineLiteralBug])
4480 return true;
4481
4482 const int Src2Idx = getNamedOperandIdx(Opcode, OpName::src2);
4483 if (Src2Idx == -1)
4484 return true;
4485
4486 if (Inst.getOperand(Src2Idx).isImm() && isInlineConstant(Inst, Src2Idx)) {
4487 Error(getOperandLoc(Operands, Src2Idx),
4488 "inline constants are not allowed for this operand");
4489 return false;
4490 }
4491
4492 return true;
4493}
4494
4495bool AMDGPUAsmParser::validateMFMA(const MCInst &Inst,
4496 const OperandVector &Operands) {
4497 const unsigned Opc = Inst.getOpcode();
4498 const MCInstrDesc &Desc = MII.get(Opc);
4499
4501 return true;
4502
4503 int BlgpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::blgp);
4504 if (BlgpIdx != -1) {
4505 if (const MFMA_F8F6F4_Info *Info = AMDGPU::isMFMA_F8F6F4(Opc)) {
4506 int CbszIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::cbsz);
4507
4508 unsigned CBSZ = Inst.getOperand(CbszIdx).getImm();
4509 unsigned BLGP = Inst.getOperand(BlgpIdx).getImm();
4510
4511 // Validate the correct register size was used for the floating point
4512 // format operands
4513
4514 bool Success = true;
4515 if (Info->NumRegsSrcA != mfmaScaleF8F6F4FormatToNumRegs(CBSZ)) {
4516 int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
4517 Error(getOperandLoc(Operands, Src0Idx),
4518 "wrong register tuple size for cbsz value " + Twine(CBSZ));
4519 Success = false;
4520 }
4521
4522 if (Info->NumRegsSrcB != mfmaScaleF8F6F4FormatToNumRegs(BLGP)) {
4523 int Src1Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1);
4524 Error(getOperandLoc(Operands, Src1Idx),
4525 "wrong register tuple size for blgp value " + Twine(BLGP));
4526 Success = false;
4527 }
4528
4529 return Success;
4530 }
4531 }
4532
4533 const int Src2Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2);
4534 if (Src2Idx == -1)
4535 return true;
4536
4537 const MCOperand &Src2 = Inst.getOperand(Src2Idx);
4538 if (!Src2.isReg())
4539 return true;
4540
4541 MCRegister Src2Reg = Src2.getReg();
4542 MCRegister DstReg = Inst.getOperand(0).getReg();
4543 if (Src2Reg == DstReg)
4544 return true;
4545
4546 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
4547 if (TRI->getRegClass(MII.getOpRegClassID(Desc.operands()[0], HwMode))
4548 .getSizeInBits() <= 128)
4549 return true;
4550
4551 if (TRI->regsOverlap(Src2Reg, DstReg)) {
4552 Error(getOperandLoc(Operands, Src2Idx),
4553 "source 2 operand must not partially overlap with dst");
4554 return false;
4555 }
4556
4557 return true;
4558}
4559
4560bool AMDGPUAsmParser::validateDivScale(const MCInst &Inst) {
4561 switch (Inst.getOpcode()) {
4562 default:
4563 return true;
4564 case V_DIV_SCALE_F32_gfx6_gfx7:
4565 case V_DIV_SCALE_F32_vi:
4566 case V_DIV_SCALE_F32_gfx10:
4567 case V_DIV_SCALE_F64_gfx6_gfx7:
4568 case V_DIV_SCALE_F64_vi:
4569 case V_DIV_SCALE_F64_gfx10:
4570 break;
4571 }
4572
4573 // TODO: Check that src0 = src1 or src2.
4574
4575 for (auto Name :
4576 {AMDGPU::OpName::src0_modifiers, AMDGPU::OpName::src2_modifiers,
4577 AMDGPU::OpName::src2_modifiers}) {
4578 if (Inst.getOperand(AMDGPU::getNamedOperandIdx(Inst.getOpcode(), Name))
4579 .getImm() &
4581 return false;
4582 }
4583 }
4584
4585 return true;
4586}
4587
4588bool AMDGPUAsmParser::validateMIMGD16(const MCInst &Inst) {
4589
4590 const unsigned Opc = Inst.getOpcode();
4591
4592 if ((SIInstrFlags::isImage(MII, Inst)) == 0)
4593 return true;
4594
4595 int D16Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::d16);
4596 if (D16Idx >= 0 && Inst.getOperand(D16Idx).getImm()) {
4597 if (isCI() || isSI())
4598 return false;
4599 }
4600
4601 return true;
4602}
4603
4604bool AMDGPUAsmParser::validateTensorR128(const MCInst &Inst) {
4605 const unsigned Opc = Inst.getOpcode();
4606
4607 if (!SIInstrFlags::usesTENSOR_CNT(MII, Inst))
4608 return true;
4609
4610 int R128Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::r128);
4611
4612 return R128Idx < 0 || !Inst.getOperand(R128Idx).getImm();
4613}
4614
4615static bool IsRevOpcode(const unsigned Opcode) {
4616 switch (Opcode) {
4617 case AMDGPU::V_SUBREV_F32_e32:
4618 case AMDGPU::V_SUBREV_F32_e64:
4619 case AMDGPU::V_SUBREV_F32_e32_gfx10:
4620 case AMDGPU::V_SUBREV_F32_e32_gfx6_gfx7:
4621 case AMDGPU::V_SUBREV_F32_e32_vi:
4622 case AMDGPU::V_SUBREV_F32_e64_gfx10:
4623 case AMDGPU::V_SUBREV_F32_e64_gfx6_gfx7:
4624 case AMDGPU::V_SUBREV_F32_e64_vi:
4625
4626 case AMDGPU::V_SUBREV_CO_U32_e32:
4627 case AMDGPU::V_SUBREV_CO_U32_e64:
4628 case AMDGPU::V_SUBREV_I32_e32_gfx6_gfx7:
4629 case AMDGPU::V_SUBREV_I32_e64_gfx6_gfx7:
4630
4631 case AMDGPU::V_SUBBREV_U32_e32:
4632 case AMDGPU::V_SUBBREV_U32_e64:
4633 case AMDGPU::V_SUBBREV_U32_e32_gfx6_gfx7:
4634 case AMDGPU::V_SUBBREV_U32_e32_vi:
4635 case AMDGPU::V_SUBBREV_U32_e64_gfx6_gfx7:
4636 case AMDGPU::V_SUBBREV_U32_e64_vi:
4637
4638 case AMDGPU::V_SUBREV_U32_e32:
4639 case AMDGPU::V_SUBREV_U32_e64:
4640 case AMDGPU::V_SUBREV_U32_e32_gfx9:
4641 case AMDGPU::V_SUBREV_U32_e32_vi:
4642 case AMDGPU::V_SUBREV_U32_e64_gfx9:
4643 case AMDGPU::V_SUBREV_U32_e64_vi:
4644
4645 case AMDGPU::V_SUBREV_F16_e32:
4646 case AMDGPU::V_SUBREV_F16_e64:
4647 case AMDGPU::V_SUBREV_F16_e32_gfx10:
4648 case AMDGPU::V_SUBREV_F16_e32_vi:
4649 case AMDGPU::V_SUBREV_F16_e64_gfx10:
4650 case AMDGPU::V_SUBREV_F16_e64_vi:
4651
4652 case AMDGPU::V_SUBREV_U16_e32:
4653 case AMDGPU::V_SUBREV_U16_e64:
4654 case AMDGPU::V_SUBREV_U16_e32_vi:
4655 case AMDGPU::V_SUBREV_U16_e64_vi:
4656
4657 case AMDGPU::V_SUBREV_CO_U32_e32_gfx9:
4658 case AMDGPU::V_SUBREV_CO_U32_e64_gfx10:
4659 case AMDGPU::V_SUBREV_CO_U32_e64_gfx9:
4660
4661 case AMDGPU::V_SUBBREV_CO_U32_e32_gfx9:
4662 case AMDGPU::V_SUBBREV_CO_U32_e64_gfx9:
4663
4664 case AMDGPU::V_SUBREV_NC_U32_e32_gfx10:
4665 case AMDGPU::V_SUBREV_NC_U32_e64_gfx10:
4666
4667 case AMDGPU::V_SUBREV_CO_CI_U32_e32_gfx10:
4668 case AMDGPU::V_SUBREV_CO_CI_U32_e64_gfx10:
4669
4670 case AMDGPU::V_LSHRREV_B32_e32:
4671 case AMDGPU::V_LSHRREV_B32_e64:
4672 case AMDGPU::V_LSHRREV_B32_e32_gfx6_gfx7:
4673 case AMDGPU::V_LSHRREV_B32_e64_gfx6_gfx7:
4674 case AMDGPU::V_LSHRREV_B32_e32_vi:
4675 case AMDGPU::V_LSHRREV_B32_e64_vi:
4676 case AMDGPU::V_LSHRREV_B32_e32_gfx10:
4677 case AMDGPU::V_LSHRREV_B32_e64_gfx10:
4678
4679 case AMDGPU::V_ASHRREV_I32_e32:
4680 case AMDGPU::V_ASHRREV_I32_e64:
4681 case AMDGPU::V_ASHRREV_I32_e32_gfx10:
4682 case AMDGPU::V_ASHRREV_I32_e32_gfx6_gfx7:
4683 case AMDGPU::V_ASHRREV_I32_e32_vi:
4684 case AMDGPU::V_ASHRREV_I32_e64_gfx10:
4685 case AMDGPU::V_ASHRREV_I32_e64_gfx6_gfx7:
4686 case AMDGPU::V_ASHRREV_I32_e64_vi:
4687
4688 case AMDGPU::V_LSHLREV_B32_e32:
4689 case AMDGPU::V_LSHLREV_B32_e64:
4690 case AMDGPU::V_LSHLREV_B32_e32_gfx10:
4691 case AMDGPU::V_LSHLREV_B32_e32_gfx6_gfx7:
4692 case AMDGPU::V_LSHLREV_B32_e32_vi:
4693 case AMDGPU::V_LSHLREV_B32_e64_gfx10:
4694 case AMDGPU::V_LSHLREV_B32_e64_gfx6_gfx7:
4695 case AMDGPU::V_LSHLREV_B32_e64_vi:
4696
4697 case AMDGPU::V_LSHLREV_B16_e32:
4698 case AMDGPU::V_LSHLREV_B16_e64:
4699 case AMDGPU::V_LSHLREV_B16_e32_vi:
4700 case AMDGPU::V_LSHLREV_B16_e64_vi:
4701 case AMDGPU::V_LSHLREV_B16_gfx10:
4702
4703 case AMDGPU::V_LSHRREV_B16_e32:
4704 case AMDGPU::V_LSHRREV_B16_e64:
4705 case AMDGPU::V_LSHRREV_B16_e32_vi:
4706 case AMDGPU::V_LSHRREV_B16_e64_vi:
4707 case AMDGPU::V_LSHRREV_B16_gfx10:
4708
4709 case AMDGPU::V_ASHRREV_I16_e32:
4710 case AMDGPU::V_ASHRREV_I16_e64:
4711 case AMDGPU::V_ASHRREV_I16_e32_vi:
4712 case AMDGPU::V_ASHRREV_I16_e64_vi:
4713 case AMDGPU::V_ASHRREV_I16_gfx10:
4714
4715 case AMDGPU::V_LSHLREV_B64_e64:
4716 case AMDGPU::V_LSHLREV_B64_gfx10:
4717 case AMDGPU::V_LSHLREV_B64_vi:
4718
4719 case AMDGPU::V_LSHRREV_B64_e64:
4720 case AMDGPU::V_LSHRREV_B64_gfx10:
4721 case AMDGPU::V_LSHRREV_B64_vi:
4722
4723 case AMDGPU::V_ASHRREV_I64_e64:
4724 case AMDGPU::V_ASHRREV_I64_gfx10:
4725 case AMDGPU::V_ASHRREV_I64_vi:
4726
4727 case AMDGPU::V_PK_LSHLREV_B16:
4728 case AMDGPU::V_PK_LSHLREV_B16_gfx10:
4729 case AMDGPU::V_PK_LSHLREV_B16_vi:
4730
4731 case AMDGPU::V_PK_LSHRREV_B16:
4732 case AMDGPU::V_PK_LSHRREV_B16_gfx10:
4733 case AMDGPU::V_PK_LSHRREV_B16_vi:
4734 case AMDGPU::V_PK_ASHRREV_I16:
4735 case AMDGPU::V_PK_ASHRREV_I16_gfx10:
4736 case AMDGPU::V_PK_ASHRREV_I16_vi:
4737 return true;
4738 default:
4739 return false;
4740 }
4741}
4742
4743bool AMDGPUAsmParser::validateLdsDirect(const MCInst &Inst,
4744 const OperandVector &Operands) {
4745 const unsigned Opcode = Inst.getOpcode();
4746
4747 // lds_direct register is defined so that it can be used
4748 // with 9-bit operands only. Ignore encodings which do not accept these.
4749 if (!SIInstrFlags::isVOP1(MII, Inst) && !SIInstrFlags::isVOP2(MII, Inst) &&
4750 !SIInstrFlags::isVOP3Like(MII, Inst) &&
4751 !SIInstrFlags::isVOPC(MII, Inst) && !SIInstrFlags::isSDWA(MII, Inst))
4752 return true;
4753
4754 for (auto SrcName : {OpName::src0, OpName::src1, OpName::src2}) {
4755 auto SrcIdx = getNamedOperandIdx(Opcode, SrcName);
4756 if (SrcIdx == -1)
4757 break;
4758 const auto &Src = Inst.getOperand(SrcIdx);
4759 if (Src.isReg() && Src.getReg() == LDS_DIRECT) {
4760
4761 if (isGFX90A() || isGFX11Plus()) {
4762 Error(getOperandLoc(Operands, SrcIdx),
4763 "lds_direct is not supported on this GPU");
4764 return false;
4765 }
4766
4767 if (IsRevOpcode(Opcode) || SIInstrFlags::isSDWA(MII, Inst)) {
4768 Error(getOperandLoc(Operands, SrcIdx),
4769 "lds_direct cannot be used with this instruction");
4770 return false;
4771 }
4772
4773 if (SrcName != OpName::src0) {
4774 Error(getOperandLoc(Operands, SrcIdx),
4775 "lds_direct may be used as src0 only");
4776 return false;
4777 }
4778 }
4779 }
4780
4781 return true;
4782}
4783
4784SMLoc AMDGPUAsmParser::getFlatOffsetLoc(const OperandVector &Operands) const {
4785 for (unsigned i = 1, e = Operands.size(); i != e; ++i) {
4786 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
4787 if (Op.isFlatOffset())
4788 return Op.getStartLoc();
4789 }
4790 return getLoc();
4791}
4792
4793bool AMDGPUAsmParser::validateOffset(const MCInst &Inst,
4794 const OperandVector &Operands) {
4795 auto Opcode = Inst.getOpcode();
4796 auto OpNum = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::offset);
4797 if (OpNum == -1)
4798 return true;
4799
4800 if (SIInstrFlags::isFLAT(MII, Inst))
4801 return validateFlatOffset(Inst, Operands);
4802
4803 if (SIInstrFlags::isSMRD(MII, Inst))
4804 return validateSMEMOffset(Inst, Operands);
4805
4806 const auto &Op = Inst.getOperand(OpNum);
4807 // GFX12+ buffer ops: InstOffset is signed 24, but must not be a negative.
4808 if (isGFX12Plus() && SIInstrFlags::isBuffer(MII, Inst)) {
4809 const unsigned OffsetSize = 24;
4810 if (!isUIntN(OffsetSize - 1, Op.getImm())) {
4811 Error(getFlatOffsetLoc(Operands),
4812 Twine("expected a ") + Twine(OffsetSize - 1) +
4813 "-bit unsigned offset for buffer ops");
4814 return false;
4815 }
4816 } else {
4817 const unsigned OffsetSize = 16;
4818 if (!isUIntN(OffsetSize, Op.getImm())) {
4819 Error(getFlatOffsetLoc(Operands),
4820 Twine("expected a ") + Twine(OffsetSize) + "-bit unsigned offset");
4821 return false;
4822 }
4823 }
4824 return true;
4825}
4826
4827bool AMDGPUAsmParser::validateFlatOffset(const MCInst &Inst,
4828 const OperandVector &Operands) {
4829 if (!SIInstrFlags::isFLAT(MII, Inst))
4830 return true;
4831
4832 auto Opcode = Inst.getOpcode();
4833 auto OpNum = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::offset);
4834 assert(OpNum != -1);
4835
4836 const auto &Op = Inst.getOperand(OpNum);
4837 if (!hasFlatOffsets() && Op.getImm() != 0) {
4838 Error(getFlatOffsetLoc(Operands),
4839 "flat offset modifier is not supported on this GPU");
4840 return false;
4841 }
4842
4843 // For pre-GFX12 FLAT instructions the offset must be positive;
4844 // MSB is ignored and forced to zero.
4845 unsigned OffsetSize = AMDGPU::getNumFlatOffsetBits(getSTI());
4846 bool AllowNegative =
4848 if (!isIntN(OffsetSize, Op.getImm()) || (!AllowNegative && Op.getImm() < 0)) {
4849 Error(getFlatOffsetLoc(Operands),
4850 Twine("expected a ") +
4851 (AllowNegative ? Twine(OffsetSize) + "-bit signed offset"
4852 : Twine(OffsetSize - 1) + "-bit unsigned offset"));
4853 return false;
4854 }
4855
4856 return true;
4857}
4858
4859SMLoc AMDGPUAsmParser::getSMEMOffsetLoc(const OperandVector &Operands) const {
4860 // Start with second operand because SMEM Offset cannot be dst or src0.
4861 for (unsigned i = 2, e = Operands.size(); i != e; ++i) {
4862 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
4863 if (Op.isSMEMOffset() || Op.isSMEMOffsetMod())
4864 return Op.getStartLoc();
4865 }
4866 return getLoc();
4867}
4868
4869bool AMDGPUAsmParser::validateSMEMOffset(const MCInst &Inst,
4870 const OperandVector &Operands) {
4871 if (isCI() || isSI())
4872 return true;
4873
4874 if (!SIInstrFlags::isSMRD(MII, Inst))
4875 return true;
4876
4877 auto Opcode = Inst.getOpcode();
4878 auto OpNum = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::offset);
4879 if (OpNum == -1)
4880 return true;
4881
4882 const auto &Op = Inst.getOperand(OpNum);
4883 if (!Op.isImm())
4884 return true;
4885
4886 uint64_t Offset = Op.getImm();
4887 bool IsBuffer = AMDGPU::getSMEMIsBuffer(Opcode);
4890 return true;
4891
4892 Error(getSMEMOffsetLoc(Operands),
4893 isGFX12Plus() && IsBuffer
4894 ? "expected a 23-bit unsigned offset for buffer ops"
4895 : isGFX12Plus() ? "expected a 24-bit signed offset"
4896 : (isVI() || IsBuffer) ? "expected a 20-bit unsigned offset"
4897 : "expected a 21-bit signed offset");
4898
4899 return false;
4900}
4901
4902// On subtargets with FeatureBF16InlineConstFromUpperFP32 the hardware generates
4903// a bf16 inline constant in the high half of the corresponding fp32 inline
4904// constant. A VOP1 bf16 opcode (v_cvt_f32_bf16 and the bf16 transcendentals)
4905// reads the low half of its source, so it must use the VOP3 encoding with
4906// op_sel[0] set in order to see the constant at all.
4907bool AMDGPUAsmParser::validateBF16InlineConst(const MCInst &Inst,
4908 const OperandVector &Operands) {
4909 if (!getFeatureBits()[AMDGPU::FeatureBF16InlineConstFromUpperFP32])
4910 return true;
4911
4912 const unsigned Opc = Inst.getOpcode();
4913 const MCInstrDesc &Desc = MII.get(Opc);
4914 const bool IsVOP3 =
4916 if (!SIInstrFlags::isVOP1(Desc) && !IsVOP3)
4917 return true;
4918
4919 // Only the single-source VOP1 bf16 opcodes are affected. Multi-source and
4920 // packed bf16 instructions such as v_fma_mix*_bf16 are not.
4921 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::src1))
4922 return true;
4923
4924 const int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
4925 if (Src0Idx == -1)
4926 return true;
4927
4928 const MCOperandInfo &Src0Info = Desc.operands()[Src0Idx];
4929 if (!AMDGPU::isBF16SrcOperand(Src0Info))
4930 return true;
4931
4932 const MCOperand &Src0 = Inst.getOperand(Src0Idx);
4933 if (!Src0.isImm() ||
4934 !AMDGPU::isInlinableLiteralBF16(static_cast<int16_t>(Src0.getImm()),
4935 hasInv2PiInlineImm()))
4936 return true;
4937
4938 if (IsVOP3) {
4939 const int ModsIdx =
4940 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0_modifiers);
4941 if (ModsIdx != -1 &&
4942 (Inst.getOperand(ModsIdx).getImm() & SISrcMods::OP_SEL_0))
4943 return true;
4944 }
4945
4946 Error(getOperandLoc(Operands, Src0Idx),
4947 "bf16 inline constant is read from the high half of the fp32 inline "
4948 "constant on this GPU; use the e64 encoding with op_sel:[1,0]");
4949 return false;
4950}
4951
4952bool AMDGPUAsmParser::validateSOPLiteral(const MCInst &Inst,
4953 const OperandVector &Operands) {
4954 unsigned Opcode = Inst.getOpcode();
4955 const MCInstrDesc &Desc = MII.get(Opcode);
4957 return true;
4958
4959 const int Src0Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0);
4960 const int Src1Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src1);
4961
4962 const int OpIndices[] = {Src0Idx, Src1Idx};
4963
4964 unsigned NumExprs = 0;
4965 unsigned NumLiterals = 0;
4966 int64_t LiteralValue;
4967
4968 for (int OpIdx : OpIndices) {
4969 if (OpIdx == -1)
4970 break;
4971
4972 const MCOperand &MO = Inst.getOperand(OpIdx);
4973 // Exclude special imm operands (like that used by s_set_gpr_idx_on)
4974 if (AMDGPU::isSISrcOperand(Desc, OpIdx)) {
4975 bool IsLit = false;
4976 std::optional<int64_t> Imm;
4977 if (MO.isImm()) {
4978 Imm = MO.getImm();
4979 } else if (MO.isExpr()) {
4980 if (isLitExpr(MO.getExpr())) {
4981 IsLit = true;
4982 Imm = getLitValue(MO.getExpr());
4983 }
4984 } else {
4985 continue;
4986 }
4987
4988 if (!Imm.has_value()) {
4989 ++NumExprs;
4990 } else if (!isInlineConstant(Inst, OpIdx)) {
4991 auto OpType = static_cast<AMDGPU::OperandType>(
4992 Desc.operands()[OpIdx].OperandType);
4993 int64_t Value = encode32BitLiteral(*Imm, OpType, IsLit);
4994 if (NumLiterals == 0 || LiteralValue != Value) {
4996 ++NumLiterals;
4997 }
4998 }
4999 }
5000 }
5001
5002 if (NumLiterals + NumExprs <= 1)
5003 return true;
5004
5005 Error(getOperandLoc(Operands, Src1Idx),
5006 "only one unique literal operand is allowed");
5007 return false;
5008}
5009
5010bool AMDGPUAsmParser::validateOpSel(const MCInst &Inst) {
5011 const unsigned Opc = Inst.getOpcode();
5012 if (isPermlane16(Opc)) {
5013 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
5014 unsigned OpSel = Inst.getOperand(OpSelIdx).getImm();
5015
5016 if (OpSel & ~3)
5017 return false;
5018 }
5019
5020 if (isGFX940() && SIInstrFlags::isDOT(MII, Inst)) {
5021 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
5022 if (OpSelIdx != -1) {
5023 if (Inst.getOperand(OpSelIdx).getImm() != 0)
5024 return false;
5025 }
5026 int OpSelHiIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel_hi);
5027 if (OpSelHiIdx != -1) {
5028 if (Inst.getOperand(OpSelHiIdx).getImm() != -1)
5029 return false;
5030 }
5031 }
5032
5033 // op_sel[0:1] must be 0 for v_dot2_bf16_bf16 and v_dot2_f16_f16 (VOP3 Dot).
5034 if (isGFX11Plus() && SIInstrFlags::isDOT(MII, Inst) &&
5035 SIInstrFlags::isVOP3(MII, Inst) && !SIInstrFlags::isVOP3P(MII, Inst)) {
5036 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
5037 unsigned OpSel = Inst.getOperand(OpSelIdx).getImm();
5038 if (OpSel & 3)
5039 return false;
5040 }
5041
5042 // Packed math FP32 instructions typically accept SGPRs or VGPRs as source
5043 // operands. On gfx12+, if a source operand uses SGPRs, the HW can only read
5044 // the first SGPR and use it for both the low and high operations.
5046 int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
5047 int Src1Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1);
5048 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
5049 int OpSelHiIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel_hi);
5050
5051 const MCOperand &Src0 = Inst.getOperand(Src0Idx);
5052 const MCOperand &Src1 = Inst.getOperand(Src1Idx);
5053 unsigned OpSel = Inst.getOperand(OpSelIdx).getImm();
5054 unsigned OpSelHi = Inst.getOperand(OpSelHiIdx).getImm();
5055
5056 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
5057
5058 auto VerifyOneSGPR = [OpSel, OpSelHi](unsigned Index) -> bool {
5059 unsigned Mask = 1U << Index;
5060 return ((OpSel & Mask) == 0) && ((OpSelHi & Mask) == 0);
5061 };
5062
5063 if (Src0.isReg() && isSGPR(Src0.getReg(), TRI) &&
5064 !VerifyOneSGPR(/*Index=*/0))
5065 return false;
5066 if (Src1.isReg() && isSGPR(Src1.getReg(), TRI) &&
5067 !VerifyOneSGPR(/*Index=*/1))
5068 return false;
5069
5070 int Src2Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2);
5071 if (Src2Idx != -1) {
5072 const MCOperand &Src2 = Inst.getOperand(Src2Idx);
5073 if (Src2.isReg() && isSGPR(Src2.getReg(), TRI) &&
5074 !VerifyOneSGPR(/*Index=*/2))
5075 return false;
5076 }
5077 }
5078
5079 return true;
5080}
5081
5082bool AMDGPUAsmParser::validateTrue16OpSel(const MCInst &Inst) {
5083 if (!hasTrue16Insts())
5084 return true;
5085 const MCRegisterInfo *MRI = getMRI();
5086 const unsigned Opc = Inst.getOpcode();
5087 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
5088 if (OpSelIdx == -1)
5089 return true;
5090 unsigned OpSelOpValue = Inst.getOperand(OpSelIdx).getImm();
5091 // If the value is 0 we could have a default OpSel Operand, so conservatively
5092 // allow it.
5093 if (OpSelOpValue == 0)
5094 return true;
5095 unsigned OpCount = 0;
5096 for (AMDGPU::OpName OpName : {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
5097 AMDGPU::OpName::src2, AMDGPU::OpName::vdst}) {
5098 int OpIdx = AMDGPU::getNamedOperandIdx(Inst.getOpcode(), OpName);
5099 if (OpIdx == -1)
5100 continue;
5101 const MCOperand &Op = Inst.getOperand(OpIdx);
5102 if (Op.isReg() &&
5103 MRI->getRegClass(AMDGPU::VGPR_16RegClassID).contains(Op.getReg())) {
5104 bool VGPRSuffixIsHi = AMDGPU::isHi16Reg(Op.getReg(), *MRI);
5105 bool OpSelOpIsHi = ((OpSelOpValue & (1 << OpCount)) != 0);
5106 if (OpSelOpIsHi != VGPRSuffixIsHi)
5107 return false;
5108 }
5109 ++OpCount;
5110 }
5111
5112 return true;
5113}
5114
5115bool AMDGPUAsmParser::validateNeg(const MCInst &Inst, AMDGPU::OpName OpName) {
5116 assert(OpName == AMDGPU::OpName::neg_lo || OpName == AMDGPU::OpName::neg_hi);
5117
5118 const unsigned Opc = Inst.getOpcode();
5119
5120 // v_dot4 fp8/bf8 neg_lo/neg_hi not allowed on src0 and src1 (allowed on src2)
5121 // v_wmma iu4/iu8 neg_lo not allowed on src2 (allowed on src0, src1)
5122 // v_swmmac f16/bf16 neg_lo/neg_hi not allowed on src2 (allowed on src0, src1)
5123 // other wmma/swmmac instructions don't have neg_lo/neg_hi operand.
5124 if (!SIInstrFlags::isDOT(MII, Inst) && !SIInstrFlags::isWMMA(MII, Inst) &&
5125 !SIInstrFlags::isSWMMAC(MII, Inst))
5126 return true;
5127
5128 int NegIdx = AMDGPU::getNamedOperandIdx(Opc, OpName);
5129 if (NegIdx == -1)
5130 return true;
5131
5132 unsigned Neg = Inst.getOperand(NegIdx).getImm();
5133
5134 // Instructions that have neg_lo or neg_hi operand but neg modifier is allowed
5135 // on some src operands but not allowed on other.
5136 // It is convenient that such instructions don't have src_modifiers operand
5137 // for src operands that don't allow neg because they also don't allow opsel.
5138
5139 const AMDGPU::OpName SrcMods[3] = {AMDGPU::OpName::src0_modifiers,
5140 AMDGPU::OpName::src1_modifiers,
5141 AMDGPU::OpName::src2_modifiers};
5142
5143 for (unsigned i = 0; i < 3; ++i) {
5144 if (!AMDGPU::hasNamedOperand(Opc, SrcMods[i])) {
5145 if (Neg & (1 << i))
5146 return false;
5147 }
5148 }
5149
5150 return true;
5151}
5152
5153bool AMDGPUAsmParser::validateDPP(const MCInst &Inst,
5154 const OperandVector &Operands) {
5155 const unsigned Opc = Inst.getOpcode();
5156 int DppCtrlIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::dpp_ctrl);
5157 if (DppCtrlIdx >= 0) {
5158 unsigned DppCtrl = Inst.getOperand(DppCtrlIdx).getImm();
5159
5160 if (!AMDGPU::isLegalDPALU_DPPControl(getSTI(), DppCtrl) &&
5161 getSTI().hasFeature(AMDGPU::FeatureDPALU_DPP) &&
5162 AMDGPU::isDPALU_DPP(MII.get(Opc), MII, getSTI())) {
5163 // DP ALU DPP is supported for row_newbcast only on GFX9* and row_share
5164 // only on GFX12.
5165 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyDppCtrl, Operands);
5166 Error(S, isGFX12() ? "DP ALU dpp only supports row_share"
5167 : "DP ALU dpp only supports row_newbcast");
5168 return false;
5169 }
5170 }
5171
5172 int Dpp8Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::dpp8);
5173 bool IsDPP = DppCtrlIdx >= 0 || Dpp8Idx >= 0;
5174
5175 if (IsDPP && !hasDPPSrc1SGPR(getSTI())) {
5176 int Src1Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1);
5177 if (Src1Idx >= 0) {
5178 const MCOperand &Src1 = Inst.getOperand(Src1Idx);
5179 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
5180 if (Src1.isReg() && isSGPR(mc2PseudoReg(Src1.getReg()), TRI)) {
5181 Error(getOperandLoc(Operands, Src1Idx),
5182 "invalid operand for instruction");
5183 return false;
5184 }
5185 if (Src1.isImm()) {
5186 Error(getInstLoc(Operands),
5187 "src1 immediate operand invalid for instruction");
5188 return false;
5189 }
5190 }
5191 }
5192
5193 return true;
5194}
5195
5196// Check if VCC register matches wavefront size
5197bool AMDGPUAsmParser::validateVccOperand(MCRegister Reg) const {
5198 return (Reg == AMDGPU::VCC && isWave64()) ||
5199 (Reg == AMDGPU::VCC_LO && isWave32());
5200}
5201
5202// One unique literal can be used. VOP3 literal is only allowed in GFX10+
5203bool AMDGPUAsmParser::validateVOPLiteral(const MCInst &Inst,
5204 const OperandVector &Operands) {
5205 unsigned Opcode = Inst.getOpcode();
5206 const MCInstrDesc &Desc = MII.get(Opcode);
5207 bool HasMandatoryLiteral = getNamedOperandIdx(Opcode, OpName::imm) != -1;
5208 if (!SIInstrFlags::isVOP3Like(Desc) && !HasMandatoryLiteral &&
5209 !isVOPD(Opcode))
5210 return true;
5211
5212 OperandIndices OpIndices = getSrcOperandIndices(Opcode, HasMandatoryLiteral);
5213
5214 std::optional<unsigned> LiteralOpIdx;
5215 std::optional<uint64_t> LiteralValue;
5216
5217 for (int OpIdx : OpIndices) {
5218 if (OpIdx == -1)
5219 continue;
5220
5221 const MCOperand &MO = Inst.getOperand(OpIdx);
5222 if (!MO.isImm() && !MO.isExpr())
5223 continue;
5224 if (!isSISrcOperand(Desc, OpIdx))
5225 continue;
5226
5227 std::optional<int64_t> Imm;
5228 if (MO.isImm())
5229 Imm = MO.getImm();
5230 else if (MO.isExpr() && isLitExpr(MO.getExpr()))
5231 Imm = getLitValue(MO.getExpr());
5232
5233 bool IsAnotherLiteral = false;
5234 bool IsForcedLit = findMCOperand(Operands, OpIdx).isForcedLit();
5235 bool IsForcedLit64 = findMCOperand(Operands, OpIdx).isForcedLit64();
5236 if (!Imm.has_value()) {
5237 // Literal value not known, so we conservately assume it's different.
5238 IsAnotherLiteral = true;
5239 } else if (IsForcedLit || IsForcedLit64 || !isInlineConstant(Inst, OpIdx)) {
5240 uint64_t Value = *Imm;
5241 bool IsForcedFP64 =
5242 Desc.operands()[OpIdx].OperandType == AMDGPU::OPERAND_KIMM64 ||
5243 (Desc.operands()[OpIdx].OperandType == AMDGPU::OPERAND_REG_IMM_FP64 &&
5244 HasMandatoryLiteral);
5245 AMDGPU::OperandType OpTy =
5246 static_cast<AMDGPU::OperandType>(Desc.operands()[OpIdx].OperandType);
5247 bool IsFP64 =
5248 (IsForcedFP64 || (AMDGPU::isSISrcFPOperand(Desc, OpIdx) &&
5250 AMDGPU::getOperandSize(Desc.operands()[OpIdx]) == 8;
5251 bool IsValid32Op =
5252 IsForcedLit || AMDGPU::isValid32BitLiteral(Value, IsFP64);
5253
5254 if (((!IsValid32Op && !isInt<32>(Value) && !isUInt<32>(Value) &&
5255 !IsForcedFP64) ||
5256 (IsForcedLit64 && !HasMandatoryLiteral)) &&
5257 (!has64BitLiterals() || Desc.getSize() != 4)) {
5258 Error(getOperandLoc(Operands, OpIdx),
5259 "invalid operand for instruction");
5260 return false;
5261 }
5262
5263 // Only src0 can use lit64 in VOP* encoding.
5264 if (!IsForcedFP64 && (IsForcedLit64 || !IsValid32Op) &&
5265 OpIdx != getNamedOperandIdx(Opcode, OpName::src0)) {
5266 Error(getOperandLoc(Operands, OpIdx),
5267 "invalid operand for instruction");
5268 return false;
5269 }
5270
5271 // Compare values using the word encoded by a 32-bit literal.
5272 if (IsValid32Op && !IsForcedFP64 && !IsForcedLit64) {
5273 Value = static_cast<uint32_t>(
5274 AMDGPU::encode32BitLiteral(Value, OpTy, IsForcedLit));
5275 }
5276
5277 IsAnotherLiteral = !LiteralValue || *LiteralValue != Value;
5279 }
5280
5281 if (IsAnotherLiteral && !HasMandatoryLiteral &&
5282 !getFeatureBits()[FeatureVOP3Literal]) {
5283 Error(getOperandLoc(Operands, OpIdx),
5284 "literal operands are not supported");
5285 return false;
5286 }
5287
5288 if (LiteralOpIdx && IsAnotherLiteral) {
5289 Error(getLaterLoc(getOperandLoc(Operands, OpIdx),
5290 getOperandLoc(Operands, *LiteralOpIdx)),
5291 "only one unique literal operand is allowed");
5292 return false;
5293 }
5294
5295 if (IsAnotherLiteral)
5296 LiteralOpIdx = OpIdx;
5297 }
5298
5299 return true;
5300}
5301
5302// Returns -1 if not a register, 0 if VGPR and 1 if AGPR.
5303static int IsAGPROperand(const MCInst &Inst, AMDGPU::OpName Name,
5304 const MCRegisterInfo *MRI) {
5305 int OpIdx = AMDGPU::getNamedOperandIdx(Inst.getOpcode(), Name);
5306 if (OpIdx < 0)
5307 return -1;
5308
5309 const MCOperand &Op = Inst.getOperand(OpIdx);
5310 if (!Op.isReg())
5311 return -1;
5312
5313 MCRegister Sub = MRI->getSubReg(Op.getReg(), AMDGPU::sub0);
5314 auto Reg = Sub ? Sub : Op.getReg();
5315 const MCRegisterClass &AGPR32 = MRI->getRegClass(AMDGPU::AGPR_32RegClassID);
5316 return AGPR32.contains(Reg) ? 1 : 0;
5317}
5318
5319bool AMDGPUAsmParser::validateAGPRLdSt(const MCInst &Inst) const {
5320 if (!SIInstrFlags::isFLAT(MII, Inst) && !SIInstrFlags::isBuffer(MII, Inst) &&
5321 !SIInstrFlags::isMIMG(MII, Inst) && !SIInstrFlags::isDS(MII, Inst))
5322 return true;
5323
5324 AMDGPU::OpName DataName = SIInstrFlags::isDS(MII, Inst)
5325 ? AMDGPU::OpName::data0
5326 : AMDGPU::OpName::vdata;
5327
5328 const MCRegisterInfo *MRI = getMRI();
5329 int DstAreg = IsAGPROperand(Inst, AMDGPU::OpName::vdst, MRI);
5330 int DataAreg = IsAGPROperand(Inst, DataName, MRI);
5331
5332 if (SIInstrFlags::isDS(MII, Inst) && DataAreg >= 0) {
5333 int Data2Areg = IsAGPROperand(Inst, AMDGPU::OpName::data1, MRI);
5334 if (Data2Areg >= 0 && Data2Areg != DataAreg)
5335 return false;
5336 }
5337
5338 auto FB = getFeatureBits();
5339 if (FB[AMDGPU::FeatureGFX90AInsts]) {
5340 if (DataAreg < 0 || DstAreg < 0)
5341 return true;
5342 return DstAreg == DataAreg;
5343 }
5344
5345 return DstAreg < 1 && DataAreg < 1;
5346}
5347
5348bool AMDGPUAsmParser::validateVGPRAlign(const MCInst &Inst) const {
5349 auto FB = getFeatureBits();
5350 if (!FB[AMDGPU::FeatureRequiresAlignedVGPRs])
5351 return true;
5352
5353 unsigned Opc = Inst.getOpcode();
5354 const MCRegisterInfo *MRI = getMRI();
5355 // DS_READ_B96_TR_B6 is the only DS instruction in GFX950, that allows
5356 // unaligned VGPR. All others only allow even aligned VGPRs.
5357 if (FB[AMDGPU::FeatureGFX90AInsts] && Opc == AMDGPU::DS_READ_B96_TR_B6_vi)
5358 return true;
5359
5360 if (FB[AMDGPU::FeatureGFX1250Insts]) {
5361 switch (Opc) {
5362 default:
5363 break;
5364 case AMDGPU::DS_LOAD_TR6_B96:
5365 case AMDGPU::DS_LOAD_TR6_B96_gfx12:
5366 // DS_LOAD_TR6_B96 is the only DS instruction in GFX1250, that
5367 // allows unaligned VGPR. All others only allow even aligned VGPRs.
5368 return true;
5369 case AMDGPU::GLOBAL_LOAD_TR6_B96:
5370 case AMDGPU::GLOBAL_LOAD_TR6_B96_gfx1250: {
5371 // GLOBAL_LOAD_TR6_B96 is the only GLOBAL instruction in GFX1250, that
5372 // allows unaligned VGPR for vdst, but other operands still only allow
5373 // even aligned VGPRs.
5374 int VAddrIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vaddr);
5375 if (VAddrIdx != -1) {
5376 const MCOperand &Op = Inst.getOperand(VAddrIdx);
5377 MCRegister Sub = MRI->getSubReg(Op.getReg(), AMDGPU::sub0);
5378 if ((Sub - AMDGPU::VGPR0) & 1)
5379 return false;
5380 }
5381 return true;
5382 }
5383 case AMDGPU::GLOBAL_LOAD_TR6_B96_SADDR:
5384 case AMDGPU::GLOBAL_LOAD_TR6_B96_SADDR_gfx1250:
5385 return true;
5386 }
5387 }
5388
5389 const MCRegisterClass &VGPR32 = MRI->getRegClass(AMDGPU::VGPR_32RegClassID);
5390 const MCRegisterClass &AGPR32 = MRI->getRegClass(AMDGPU::AGPR_32RegClassID);
5391 for (unsigned I = 0, E = Inst.getNumOperands(); I != E; ++I) {
5392 const MCOperand &Op = Inst.getOperand(I);
5393 if (!Op.isReg())
5394 continue;
5395
5396 MCRegister Sub = MRI->getSubReg(Op.getReg(), AMDGPU::sub0);
5397 if (!Sub)
5398 continue;
5399
5400 if (VGPR32.contains(Sub) && ((Sub - AMDGPU::VGPR0) & 1))
5401 return false;
5402 if (AGPR32.contains(Sub) && ((Sub - AMDGPU::AGPR0) & 1))
5403 return false;
5404 }
5405
5406 return true;
5407}
5408
5409SMLoc AMDGPUAsmParser::getBLGPLoc(const OperandVector &Operands) const {
5410 for (unsigned i = 1, e = Operands.size(); i != e; ++i) {
5411 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
5412 if (Op.isBLGP())
5413 return Op.getStartLoc();
5414 }
5415 return SMLoc();
5416}
5417
5418bool AMDGPUAsmParser::validateBLGP(const MCInst &Inst,
5419 const OperandVector &Operands) {
5420 unsigned Opc = Inst.getOpcode();
5421 int BlgpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::blgp);
5422 if (BlgpIdx == -1)
5423 return true;
5424 SMLoc BLGPLoc = getBLGPLoc(Operands);
5425 if (!BLGPLoc.isValid())
5426 return true;
5427 bool IsNeg = StringRef(BLGPLoc.getPointer()).starts_with("neg:");
5428 auto FB = getFeatureBits();
5429 bool UsesNeg = false;
5430 if (FB[AMDGPU::FeatureGFX940Insts]) {
5431 switch (Opc) {
5432 case AMDGPU::V_MFMA_F64_16X16X4F64_gfx940_acd:
5433 case AMDGPU::V_MFMA_F64_16X16X4F64_gfx940_vcd:
5434 case AMDGPU::V_MFMA_F64_4X4X4F64_gfx940_acd:
5435 case AMDGPU::V_MFMA_F64_4X4X4F64_gfx940_vcd:
5436 UsesNeg = true;
5437 }
5438 }
5439
5440 if (IsNeg == UsesNeg)
5441 return true;
5442
5443 Error(BLGPLoc, UsesNeg ? "invalid modifier: blgp is not supported"
5444 : "invalid modifier: neg is not supported");
5445
5446 return false;
5447}
5448
5449bool AMDGPUAsmParser::validateWaitCnt(const MCInst &Inst,
5450 const OperandVector &Operands) {
5451 if (!isGFX11Plus())
5452 return true;
5453
5454 unsigned Opc = Inst.getOpcode();
5455 if (Opc != AMDGPU::S_WAITCNT_EXPCNT_gfx11 &&
5456 Opc != AMDGPU::S_WAITCNT_LGKMCNT_gfx11 &&
5457 Opc != AMDGPU::S_WAITCNT_VMCNT_gfx11 &&
5458 Opc != AMDGPU::S_WAITCNT_VSCNT_gfx11)
5459 return true;
5460
5461 int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::sdst);
5462 assert(Src0Idx >= 0 && Inst.getOperand(Src0Idx).isReg());
5463 auto Reg = mc2PseudoReg(Inst.getOperand(Src0Idx).getReg());
5464 if (Reg == AMDGPU::SGPR_NULL)
5465 return true;
5466
5467 Error(getOperandLoc(Operands, Src0Idx), "src0 must be null");
5468 return false;
5469}
5470
5471bool AMDGPUAsmParser::validateDS(const MCInst &Inst,
5472 const OperandVector &Operands) {
5473 if (!SIInstrFlags::isDS(MII, Inst))
5474 return true;
5475 if (SIInstrFlags::isGWS(MII, Inst))
5476 return validateGWS(Inst, Operands);
5477 // Only validate GDS for non-GWS instructions.
5478 if (hasGDS())
5479 return true;
5480 int GDSIdx =
5481 AMDGPU::getNamedOperandIdx(Inst.getOpcode(), AMDGPU::OpName::gds);
5482 if (GDSIdx < 0)
5483 return true;
5484 unsigned GDS = Inst.getOperand(GDSIdx).getImm();
5485 if (GDS) {
5486 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyGDS, Operands);
5487 Error(S, "gds modifier is not supported on this GPU");
5488 return false;
5489 }
5490 return true;
5491}
5492
5493// gfx90a has an undocumented limitation:
5494// DS_GWS opcodes must use even aligned registers.
5495bool AMDGPUAsmParser::validateGWS(const MCInst &Inst,
5496 const OperandVector &Operands) {
5497 if (!getFeatureBits()[AMDGPU::FeatureGFX90AInsts])
5498 return true;
5499
5500 int Opc = Inst.getOpcode();
5501 if (Opc != AMDGPU::DS_GWS_INIT_vi && Opc != AMDGPU::DS_GWS_BARRIER_vi &&
5502 Opc != AMDGPU::DS_GWS_SEMA_BR_vi)
5503 return true;
5504
5505 const MCRegisterInfo *MRI = getMRI();
5506 const MCRegisterClass &VGPR32 = MRI->getRegClass(AMDGPU::VGPR_32RegClassID);
5507 int Data0Pos =
5508 AMDGPU::getNamedOperandIdx(Inst.getOpcode(), AMDGPU::OpName::data0);
5509 assert(Data0Pos != -1);
5510 auto Reg = Inst.getOperand(Data0Pos).getReg();
5511 auto RegIdx = Reg - (VGPR32.contains(Reg) ? AMDGPU::VGPR0 : AMDGPU::AGPR0);
5512 if (RegIdx & 1) {
5513 Error(getOperandLoc(Operands, Data0Pos), "vgpr must be even aligned");
5514 return false;
5515 }
5516
5517 return true;
5518}
5519
5520bool AMDGPUAsmParser::validateCoherencyBits(const MCInst &Inst,
5521 const OperandVector &Operands,
5522 SMLoc IDLoc) {
5523 int CPolPos =
5524 AMDGPU::getNamedOperandIdx(Inst.getOpcode(), AMDGPU::OpName::cpol);
5525 if (CPolPos == -1)
5526 return true;
5527
5528 unsigned CPol = Inst.getOperand(CPolPos).getImm();
5529
5530 if (!isGFX1250Plus()) {
5531 if (CPol & CPol::SCAL) {
5532 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol, Operands);
5533 StringRef CStr(S.getPointer());
5534 S = SMLoc::getFromPointer(&CStr.data()[CStr.find("scale_offset")]);
5535 Error(S, "scale_offset is not supported on this GPU");
5536 }
5537 if (CPol & CPol::NV) {
5538 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol, Operands);
5539 StringRef CStr(S.getPointer());
5540 S = SMLoc::getFromPointer(&CStr.data()[CStr.find("nv")]);
5541 Error(S, "nv is not supported on this GPU");
5542 }
5543 }
5544
5545 if ((CPol & CPol::SCAL) && !supportsScaleOffset(MII, Inst.getOpcode())) {
5546 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol, Operands);
5547 StringRef CStr(S.getPointer());
5548 S = SMLoc::getFromPointer(&CStr.data()[CStr.find("scale_offset")]);
5549 Error(S, "scale_offset is not supported for this instruction");
5550 }
5551
5552 if (isGFX12Plus())
5553 return validateTHAndScopeBits(Inst, Operands, CPol);
5554
5555 if (SIInstrFlags::isSMRD(MII, Inst)) {
5556 if (CPol && (isSI() || isCI())) {
5557 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol, Operands);
5558 Error(S, "cache policy is not supported for SMRD instructions");
5559 return false;
5560 }
5561 if (CPol & ~(AMDGPU::CPol::GLC | AMDGPU::CPol::DLC)) {
5562 Error(IDLoc, "invalid cache policy for SMEM instruction");
5563 return false;
5564 }
5565 }
5566
5567 if (isGFX90A() && !isGFX940() && (CPol & CPol::SCC)) {
5568 if (!SIInstrFlags::isVMEM(MII, Inst)) {
5569 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol, Operands);
5570 StringRef CStr(S.getPointer());
5571 S = SMLoc::getFromPointer(&CStr.data()[CStr.find("scc")]);
5572 Error(S,
5573 "scc modifier is not supported for this instruction on this GPU");
5574 return false;
5575 }
5576 }
5577
5578 if (!SIInstrFlags::isAtomic(MII, Inst))
5579 return true;
5580
5581 if (SIInstrFlags::isAtomicRet(MII, Inst)) {
5582 if (!SIInstrFlags::isMIMG(MII, Inst) && !(CPol & CPol::GLC)) {
5583 Error(IDLoc, isGFX940() ? "instruction must use sc0"
5584 : "instruction must use glc");
5585 return false;
5586 }
5587 } else {
5588 if (CPol & CPol::GLC) {
5589 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol, Operands);
5590 StringRef CStr(S.getPointer());
5592 &CStr.data()[CStr.find(isGFX940() ? "sc0" : "glc")]);
5593 Error(S, isGFX940() ? "instruction must not use sc0"
5594 : "instruction must not use glc");
5595 return false;
5596 }
5597 }
5598
5599 return true;
5600}
5601
5602bool AMDGPUAsmParser::validateTHAndScopeBits(const MCInst &Inst,
5603 const OperandVector &Operands,
5604 const unsigned CPol) {
5605 const unsigned TH = CPol & AMDGPU::CPol::TH;
5606 const unsigned Scope = CPol & AMDGPU::CPol::SCOPE;
5607
5608 auto PrintError = [&](StringRef Msg) {
5609 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol, Operands);
5610 Error(S, Msg);
5611 return false;
5612 };
5613
5614 if ((TH & AMDGPU::CPol::TH_ATOMIC_RETURN) &&
5615 SIInstrFlags::isAtomicNoRet(MII, Inst))
5616 return PrintError("th:TH_ATOMIC_RETURN requires a destination operand");
5617
5618 if (SIInstrFlags::isAtomicRet(MII, Inst) &&
5619 (SIInstrFlags::isFLAT(MII, Inst) || SIInstrFlags::isMUBUF(MII, Inst)) &&
5621 return PrintError("instruction must use th:TH_ATOMIC_RETURN");
5622
5623 if (TH == 0)
5624 return true;
5625
5626 if (SIInstrFlags::isSMRD(MII, Inst) &&
5627 ((TH == AMDGPU::CPol::TH_NT_RT) || (TH == AMDGPU::CPol::TH_RT_NT) ||
5628 (TH == AMDGPU::CPol::TH_NT_HT)))
5629 return PrintError("invalid th value for SMEM instruction");
5630
5631 if (TH == AMDGPU::CPol::TH_BYPASS) {
5632 if ((Scope != AMDGPU::CPol::SCOPE_SYS &&
5634 (Scope == AMDGPU::CPol::SCOPE_SYS &&
5636 return PrintError("scope and th combination is not valid");
5637 }
5638
5639 unsigned THType = AMDGPU::getTemporalHintType(MII.get(Inst.getOpcode()));
5640 if (THType == AMDGPU::CPol::TH_TYPE_ATOMIC) {
5641 if (!(CPol & AMDGPU::CPol::TH_TYPE_ATOMIC))
5642 return PrintError("invalid th value for atomic instructions");
5643 } else if (THType == AMDGPU::CPol::TH_TYPE_STORE) {
5644 if (!(CPol & AMDGPU::CPol::TH_TYPE_STORE))
5645 return PrintError("invalid th value for store instructions");
5646 } else {
5647 if (!(CPol & AMDGPU::CPol::TH_TYPE_LOAD))
5648 return PrintError("invalid th value for load instructions");
5649 }
5650
5651 return true;
5652}
5653
5654bool AMDGPUAsmParser::validateTFE(const MCInst &Inst,
5655 const OperandVector &Operands) {
5656 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
5657 if (Desc.mayStore() && SIInstrFlags::isBuffer(Desc)) {
5658 SMLoc Loc = getImmLoc(AMDGPUOperand::ImmTyTFE, Operands);
5659 if (Loc != getInstLoc(Operands)) {
5660 Error(Loc, "TFE modifier has no meaning for store instructions");
5661 return false;
5662 }
5663 }
5664
5665 return true;
5666}
5667
5668bool AMDGPUAsmParser::validateWMMA(const MCInst &Inst,
5669 const OperandVector &Operands) {
5670 unsigned Opc = Inst.getOpcode();
5671 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
5672 const MCInstrDesc &Desc = MII.get(Opc);
5673
5674 int AFmtIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_a_fmt);
5675 if (AFmtIdx == -1)
5676 return true;
5677 unsigned AFmt = Inst.getOperand(AFmtIdx).getImm();
5678 int BFmtIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_b_fmt);
5679 unsigned BFmt = Inst.getOperand(BFmtIdx).getImm();
5680
5681 auto validateFmt = [&](unsigned Fmt, AMDGPU::OpName SrcOp) -> bool {
5682 int SrcIdx = AMDGPU::getNamedOperandIdx(Opc, SrcOp);
5683 unsigned RegSize =
5684 TRI->getRegClass(MII.getOpRegClassID(Desc.operands()[SrcIdx], HwMode))
5685 .getSizeInBits();
5686
5688 return true;
5689
5690 Error(getOperandLoc(Operands, SrcIdx),
5691 "wrong register tuple size for " +
5692 Twine(WMMAMods::ModMatrixFmt[Fmt]));
5693 return false;
5694 };
5695
5696 if (!validateFmt(AFmt, AMDGPU::OpName::src0) ||
5697 !validateFmt(BFmt, AMDGPU::OpName::src1))
5698 return false;
5699
5700 int AScaleIdx =
5701 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_a_scale_fmt);
5702 if (AScaleIdx == -1)
5703 return true;
5704 unsigned AScale = Inst.getOperand(AScaleIdx).getImm();
5705 int BScaleIdx =
5706 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_b_scale_fmt);
5707 unsigned BScale = Inst.getOperand(BScaleIdx).getImm();
5708 if (!isValidWMMAScaleFmtCombination(AFmt, AScale, BFmt, BScale)) {
5709 Error(getImmLoc(AMDGPUOperand::ImmTyMatrixAFMT, Operands),
5710 "invalid matrix and scale format combination");
5711 return false;
5712 }
5713
5714 return true;
5715}
5716
5717bool AMDGPUAsmParser::validateMonitorSleep(const MCInst &Inst,
5718 const OperandVector &Operands) {
5719 unsigned Opc = Inst.getOpcode();
5720 if (Opc != AMDGPU::S_MONITOR_SLEEP_gfx12 ||
5721 !getSTI().hasFeature(AMDGPU::FeatureNoSleepForever))
5722 return true;
5723
5724 int ImmIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::simm16);
5725 if (Inst.getOperand(ImmIdx).getImm() & 0x8000) {
5726 Error(getOperandLoc(Operands, ImmIdx),
5727 "sleep forever is unsuported on the target");
5728 return false;
5729 }
5730
5731 return true;
5732}
5733
5734bool AMDGPUAsmParser::validateClusterBarrierIsFirst(
5735 const MCInst &Inst, const OperandVector &Operands) {
5736 unsigned Opc = Inst.getOpcode();
5737 if (Opc != AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM_gfx12 &&
5738 Opc != AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM_gfx13)
5739 return true;
5740
5741 int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
5742 int BarrierID = Inst.getOperand(Src0Idx).getImm();
5743 if (BarrierID != AMDGPU::Barrier::CLUSTER)
5744 return true;
5745
5746 Error(
5747 getOperandLoc(Operands, Src0Idx),
5748 "s_barrier_signal_isfirst does not support user_cluster_barrier_id (-3)");
5749 return false;
5750}
5751
5752bool AMDGPUAsmParser::validateScaleSel(const MCInst &Inst,
5753 const OperandVector &Operands) {
5754 unsigned Opc = Inst.getOpcode();
5755 int ScaleSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::scale_sel);
5756 if (ScaleSelIdx == -1)
5757 return true;
5758 int MaxSel = 0;
5759 switch (Opc) {
5760 case AMDGPU::V_CVT_SCALE_PK16_F16_FP6_e64_gfx1250:
5761 case AMDGPU::V_CVT_SCALE_PK16_BF16_FP6_e64_gfx1250:
5762 case AMDGPU::V_CVT_SCALE_PK16_F16_BF6_e64_gfx1250:
5763 case AMDGPU::V_CVT_SCALE_PK16_BF16_BF6_e64_gfx1250:
5764 case AMDGPU::V_CVT_SCALE_PK16_F32_FP6_e64_gfx1250:
5765 case AMDGPU::V_CVT_SCALE_PK16_F32_BF6_e64_gfx1250:
5766 case AMDGPU::V_CVT_SCALE_PK8_F16_FP4_e64_gfx1250:
5767 case AMDGPU::V_CVT_SCALE_PK8_BF16_FP4_e64_gfx1250:
5768 case AMDGPU::V_CVT_SCALE_PK8_F32_FP4_e64_gfx1250:
5769 MaxSel = 4;
5770 break;
5771 case AMDGPU::V_CVT_SCALE_PK8_F16_FP8_e64_gfx1250:
5772 case AMDGPU::V_CVT_SCALE_PK8_BF16_FP8_e64_gfx1250:
5773 case AMDGPU::V_CVT_SCALE_PK8_F16_BF8_e64_gfx1250:
5774 case AMDGPU::V_CVT_SCALE_PK8_BF16_BF8_e64_gfx1250:
5775 case AMDGPU::V_CVT_SCALE_PK8_F32_FP8_e64_gfx1250:
5776 case AMDGPU::V_CVT_SCALE_PK8_F32_BF8_e64_gfx1250:
5777 MaxSel = 8;
5778 break;
5779 default:
5780 return true;
5781 }
5782
5783 if (getSTI().hasFeature(AMDGPU::FeatureBlock16ConversionScaleInsts))
5784 MaxSel *= 2;
5785
5786 int ScaleSel = Inst.getOperand(ScaleSelIdx).getImm();
5787 if (ScaleSel < MaxSel)
5788 return true;
5789
5790 Error(getOperandLoc(Operands, ScaleSelIdx),
5791 "scale_sel maximum supported value is " + Twine(MaxSel - 1));
5792 return false;
5793}
5794
5795bool AMDGPUAsmParser::validateInstruction(const MCInst &Inst, SMLoc IDLoc,
5796 const OperandVector &Operands) {
5797 if (!validateLdsDirect(Inst, Operands))
5798 return false;
5799 if (!validateTrue16OpSel(Inst)) {
5800 Error(getImmLoc(AMDGPUOperand::ImmTyOpSel, Operands),
5801 "op_sel operand conflicts with 16-bit operand suffix");
5802 return false;
5803 }
5804 if (!validateSOPLiteral(Inst, Operands))
5805 return false;
5806 if (!validateVOPLiteral(Inst, Operands)) {
5807 return false;
5808 }
5809 if (!validateConstantBusLimitations(Inst, Operands)) {
5810 return false;
5811 }
5812 if (!validateVOPD(Inst, Operands)) {
5813 return false;
5814 }
5815 if (!validateIntClampSupported(Inst)) {
5816 Error(getImmLoc(AMDGPUOperand::ImmTyClamp, Operands),
5817 "integer clamping is not supported on this GPU");
5818 return false;
5819 }
5820 if (!validateOpSel(Inst)) {
5821 Error(getImmLoc(AMDGPUOperand::ImmTyOpSel, Operands),
5822 "invalid op_sel operand");
5823 return false;
5824 }
5825 if (!validateNeg(Inst, AMDGPU::OpName::neg_lo)) {
5826 Error(getImmLoc(AMDGPUOperand::ImmTyNegLo, Operands),
5827 "invalid neg_lo operand");
5828 return false;
5829 }
5830 if (!validateNeg(Inst, AMDGPU::OpName::neg_hi)) {
5831 Error(getImmLoc(AMDGPUOperand::ImmTyNegHi, Operands),
5832 "invalid neg_hi operand");
5833 return false;
5834 }
5835 if (!validateDPP(Inst, Operands)) {
5836 return false;
5837 }
5838 // For MUBUF/MTBUF d16 is a part of opcode, so there is nothing to validate.
5839 if (!validateMIMGD16(Inst)) {
5840 Error(getImmLoc(AMDGPUOperand::ImmTyD16, Operands),
5841 "d16 modifier is not supported on this GPU");
5842 return false;
5843 }
5844 if (!validateMIMGDim(Inst, Operands)) {
5845 Error(IDLoc, "missing dim operand");
5846 return false;
5847 }
5848 if (!validateTensorR128(Inst)) {
5849 Error(getImmLoc(AMDGPUOperand::ImmTyD16, Operands),
5850 "instruction must set modifier r128=0");
5851 return false;
5852 }
5853 if (!validateMIMGMSAA(Inst)) {
5854 Error(getImmLoc(AMDGPUOperand::ImmTyDim, Operands),
5855 "invalid dim; must be MSAA type");
5856 return false;
5857 }
5858 if (!validateMIMGDataSize(Inst, IDLoc)) {
5859 return false;
5860 }
5861 if (!validateMIMGAddrSize(Inst, IDLoc))
5862 return false;
5863 if (!validateMIMGAtomicDMask(Inst)) {
5864 Error(getImmLoc(AMDGPUOperand::ImmTyDMask, Operands),
5865 "invalid atomic image dmask");
5866 return false;
5867 }
5868 if (!validateMIMGGatherDMask(Inst)) {
5869 Error(getImmLoc(AMDGPUOperand::ImmTyDMask, Operands),
5870 "invalid image_gather dmask: only one bit must be set");
5871 return false;
5872 }
5873 if (!validateMovrels(Inst, Operands)) {
5874 return false;
5875 }
5876 if (!validateOffset(Inst, Operands)) {
5877 return false;
5878 }
5879 if (!validateBF16InlineConst(Inst, Operands)) {
5880 return false;
5881 }
5882 if (!validateMAIAccWrite(Inst, Operands)) {
5883 return false;
5884 }
5885 if (!validateMAISrc2(Inst, Operands)) {
5886 return false;
5887 }
5888 if (!validateMFMA(Inst, Operands)) {
5889 return false;
5890 }
5891 if (!validateCoherencyBits(Inst, Operands, IDLoc)) {
5892 return false;
5893 }
5894
5895 if (!validateAGPRLdSt(Inst)) {
5896 Error(
5897 IDLoc,
5898 getFeatureBits()[AMDGPU::FeatureGFX90AInsts]
5899 ? "invalid register class: data and dst should be all VGPR or AGPR"
5900 : "invalid register class: agpr loads and stores not supported on "
5901 "this GPU");
5902 return false;
5903 }
5904 if (!validateVGPRAlign(Inst)) {
5905 Error(IDLoc, "invalid register class: vgpr tuples must be 64 bit aligned");
5906 return false;
5907 }
5908 if (!validateDS(Inst, Operands)) {
5909 return false;
5910 }
5911
5912 if (!validateBLGP(Inst, Operands)) {
5913 return false;
5914 }
5915
5916 if (!validateDivScale(Inst)) {
5917 Error(IDLoc, "ABS not allowed in VOP3B instructions");
5918 return false;
5919 }
5920 if (!validateWaitCnt(Inst, Operands)) {
5921 return false;
5922 }
5923 if (!validateTFE(Inst, Operands)) {
5924 return false;
5925 }
5926 if (!validateWMMA(Inst, Operands)) {
5927 return false;
5928 }
5929 if (!validateMonitorSleep(Inst, Operands)) {
5930 return false;
5931 }
5932 if (!validateClusterBarrierIsFirst(Inst, Operands)) {
5933 return false;
5934 }
5935 if (!validateScaleSel(Inst, Operands)) {
5936 return false;
5937 }
5938
5939 return true;
5940}
5941
5943 const FeatureBitset &FBS,
5944 unsigned VariantID = 0);
5945
5946static bool AMDGPUCheckMnemonic(StringRef Mnemonic,
5947 const FeatureBitset &AvailableFeatures,
5948 unsigned VariantID);
5949
5950bool AMDGPUAsmParser::isSupportedMnemo(StringRef Mnemo,
5951 const FeatureBitset &FBS) {
5952 return isSupportedMnemo(Mnemo, FBS, getAllVariants());
5953}
5954
5955bool AMDGPUAsmParser::isSupportedMnemo(StringRef Mnemo,
5956 const FeatureBitset &FBS,
5957 ArrayRef<unsigned> Variants) {
5958 for (auto Variant : Variants) {
5959 if (AMDGPUCheckMnemonic(Mnemo, FBS, Variant))
5960 return true;
5961 }
5962
5963 return false;
5964}
5965
5966bool AMDGPUAsmParser::checkUnsupportedInstruction(StringRef Mnemo,
5967 SMLoc IDLoc) {
5968 FeatureBitset FBS = ComputeAvailableFeatures(getFeatureBits());
5969
5970 // Check if requested instruction variant is supported.
5971 if (isSupportedMnemo(Mnemo, FBS, getMatchedVariants()))
5972 return false;
5973
5974 // This instruction is not supported.
5975 // Clear any other pending errors because they are no longer relevant.
5976 getParser().clearPendingErrors();
5977
5978 // Requested instruction variant is not supported.
5979 // Check if any other variants are supported.
5980 StringRef VariantName = getMatchedVariantName();
5981 if (!VariantName.empty() && isSupportedMnemo(Mnemo, FBS)) {
5982 return Error(IDLoc, Twine(VariantName,
5983 " variant of this instruction is not supported"));
5984 }
5985
5986 // Check if this instruction may be used with a different wavesize.
5987 if (isGFX10Plus() && getFeatureBits()[AMDGPU::FeatureWavefrontSize64] &&
5988 !getFeatureBits()[AMDGPU::FeatureWavefrontSize32]) {
5989 // FIXME: Use getAvailableFeatures, and do not manually recompute
5990 FeatureBitset FeaturesWS32 = getFeatureBits();
5991 FeaturesWS32.flip(AMDGPU::FeatureWavefrontSize64)
5992 .flip(AMDGPU::FeatureWavefrontSize32);
5993 FeatureBitset AvailableFeaturesWS32 =
5994 ComputeAvailableFeatures(FeaturesWS32);
5995
5996 if (isSupportedMnemo(Mnemo, AvailableFeaturesWS32, getMatchedVariants()))
5997 return Error(IDLoc, "instruction requires wavesize=32");
5998 }
5999
6000 // Finally check if this instruction is supported on any other GPU.
6001 if (isSupportedMnemo(Mnemo, FeatureBitset().set())) {
6002 return Error(IDLoc, "instruction not supported on this GPU (" +
6003 getSTI().getCPU() + ")" + ": " + Mnemo);
6004 }
6005
6006 // Instruction not supported on any GPU. Probably a typo.
6007 std::string Suggestion = AMDGPUMnemonicSpellCheck(Mnemo, FBS);
6008 return Error(IDLoc, "invalid instruction" + Suggestion);
6009}
6010
6012 uint64_t InvalidOprIdx) {
6013 assert(InvalidOprIdx < Operands.size());
6014 const auto &Op = ((AMDGPUOperand &)*Operands[InvalidOprIdx]);
6015 if (Op.isToken() && InvalidOprIdx > 1) {
6016 const auto &PrevOp = ((AMDGPUOperand &)*Operands[InvalidOprIdx - 1]);
6017 return PrevOp.isToken() && PrevOp.getToken() == "::";
6018 }
6019 return false;
6020}
6021
6022bool AMDGPUAsmParser::matchAndEmitInstruction(SMLoc IDLoc, unsigned &Opcode,
6024 MCStreamer &Out,
6025 uint64_t &ErrorInfo,
6026 bool MatchingInlineAsm) {
6027 MCInst Inst;
6028 Inst.setLoc(IDLoc);
6029 unsigned Result = Match_Success;
6030
6031 // Order match statuses from least to most specific and keep the most
6032 // specific one:
6033 // Match_MnemonicFail < Match_InvalidOperand < Match_MissingFeature
6034 auto atLeastAsSpecific = [](unsigned New, unsigned Cur) {
6035 auto rank = [](unsigned M) {
6036 return M == Match_MnemonicFail ? 1
6037 : M == Match_InvalidOperand ? 2
6038 : M == Match_MissingFeature ? 3
6039 : 0; // Match_Success sentinel
6040 };
6041 return rank(New) >= rank(Cur);
6042 };
6043
6044 for (auto Variant : getMatchedVariants()) {
6045 uint64_t EI;
6046 auto R =
6047 MatchInstructionImpl(Operands, Inst, EI, MatchingInlineAsm, Variant);
6048 if (R == Match_Success || atLeastAsSpecific(R, Result)) {
6049 Result = R;
6050 ErrorInfo = EI;
6051 }
6052 if (R == Match_Success)
6053 break;
6054 }
6055
6056 if (Result == Match_Success) {
6057 if (!validateInstruction(Inst, IDLoc, Operands)) {
6058 return true;
6059 }
6060 emitTargetDirective();
6061 Out.emitInstruction(Inst, getSTI());
6062 // Record for kernel prologue checking.
6063 OpcodeStream.push_back(Inst.getOpcode());
6064 return false;
6065 }
6066
6067 StringRef Mnemo = ((AMDGPUOperand &)*Operands[0]).getToken();
6068 if (checkUnsupportedInstruction(Mnemo, IDLoc)) {
6069 return true;
6070 }
6071
6072 switch (Result) {
6073 default:
6074 break;
6075 case Match_MissingFeature:
6076 // It has been verified that the specified instruction
6077 // mnemonic is valid. A match was found but it requires
6078 // features which are not supported on this GPU.
6079 return Error(IDLoc, "operands are not valid for this GPU or mode");
6080
6081 case Match_InvalidOperand: {
6082 SMLoc ErrorLoc = IDLoc;
6083 if (ErrorInfo != ~0ULL) {
6084 if (ErrorInfo >= Operands.size()) {
6085 return Error(IDLoc, "too few operands for instruction");
6086 }
6087 AMDGPUOperand &ErrorOp = (AMDGPUOperand &)*Operands[ErrorInfo];
6088 ErrorLoc = ErrorOp.getStartLoc();
6089 if (ErrorLoc == SMLoc())
6090 ErrorLoc = IDLoc;
6091
6092 if (isInvalidVOPDY(Operands, ErrorInfo))
6093 return Error(ErrorLoc, "invalid VOPDY instruction");
6094 }
6095 return Error(ErrorLoc, "invalid operand for instruction");
6096 }
6097
6098 case Match_MnemonicFail:
6099 llvm_unreachable("Invalid instructions should have been handled already");
6100 }
6101 llvm_unreachable("Implement any new match types added!");
6102}
6103
6104bool AMDGPUAsmParser::ParseAsAbsoluteExpression(uint32_t &Ret) {
6105 int64_t Tmp = -1;
6106 if (!isToken(AsmToken::Integer) && !isToken(AsmToken::Identifier)) {
6107 return true;
6108 }
6109 if (getParser().parseAbsoluteExpression(Tmp)) {
6110 return true;
6111 }
6112 Ret = static_cast<uint32_t>(Tmp);
6113 return false;
6114}
6115
6116bool AMDGPUAsmParser::ParseDirectiveAMDGCNTarget() {
6117 if (!getSTI().getTargetTriple().isAMDGCN())
6118 return TokError("directive only supported for amdgcn architecture");
6119
6120 std::string TargetIDDirective;
6121 SMLoc TargetStart = getTok().getLoc();
6122 if (getParser().parseEscapedString(TargetIDDirective))
6123 return true;
6124
6125 std::optional<AMDGPU::TargetID> MaybeParsed =
6126 AMDGPU::TargetID::parseTargetIDString(TargetIDDirective);
6127 if (!MaybeParsed)
6128 return getParser().Error(TargetStart,
6129 "malformed target id '" + TargetIDDirective + "'");
6130
6131 const AMDGPU::TargetID &ParsedTargetID = *MaybeParsed;
6132 const Triple &TT = getSTI().getTargetTriple();
6133
6134 // The processor named in the target id must be covered by the triple's
6135 // subarch.
6136 if (!AMDGPU::isCPUValidForSubArch(TT.getSubArch(),
6137 ParsedTargetID.getGPUKind())) {
6138 return getParser().Error(
6139 TargetStart, "target id '" + TargetIDDirective +
6140 "' specifies a processor that is not valid for "
6141 "subarch '" +
6142 TT.getArchName() + "'");
6143 }
6144
6145 const std::optional<AMDGPU::TargetID> &CurrentTargetID =
6146 getTargetStreamer().getTargetID();
6147
6148 Triple DirectiveTriple(ParsedTargetID.getTargetTripleString());
6149 const Triple &STITriple = getSTI().getTargetTriple();
6150 if (!DirectiveTriple.isCompatibleWith(STITriple)) {
6151 return getParser().Error(
6152 TargetStart, ".amdgcn_target " + Twine(ParsedTargetID.toString()) +
6153 " is incompatible with " +
6154 Twine(CurrentTargetID->toString()));
6155 }
6156
6157 // Error if the ISA version doesn't match
6158 StringRef DirectiveProcessor =
6159 AMDGPU::getArchNameAMDGCN(ParsedTargetID.getGPUKind());
6160 AMDGPU::IsaVersion DirectiveISA = AMDGPU::getIsaVersion(DirectiveProcessor);
6161 if (DirectiveISA != ISA) {
6162 return getParser().Error(TargetStart,
6163 ".amdgcn_target directive processor " +
6164 Twine(DirectiveProcessor) +
6165 " does not match the specified processor " +
6166 Twine(getSTI().getCPU()));
6167 }
6168
6169 // Warn if sramecc or xnack mismatch. These do not change the encoding.
6171 ParsedTargetID.getXnackSetting(),
6172 CurrentTargetID->getXnackSetting())) {
6173 Warning(TargetStart,
6174 ".amdgcn_target directive has conflicting xnack settings");
6175 }
6177 ParsedTargetID.getSramEccSetting(),
6178 CurrentTargetID->getSramEccSetting())) {
6179 Warning(TargetStart,
6180 ".amdgcn_target directive has conflicting sramecc settings");
6181 }
6182
6183 // Update the target streamer's TargetID with settings from the directive.
6184 // We don't update the MCSubtargetInfo because we've already validated
6185 // that the directive matches the command-line CPU.
6186 getTargetStreamer().getTargetID()->setXnackSetting(
6187 ParsedTargetID.getXnackSetting());
6188 getTargetStreamer().getTargetID()->setSramEccSetting(
6189 ParsedTargetID.getSramEccSetting());
6190
6191 return false;
6192}
6193
6194bool AMDGPUAsmParser::OutOfRangeError(SMRange Range) {
6195 return Error(Range.Start, "value out of range", Range);
6196}
6197
6198bool AMDGPUAsmParser::calculateGPRBlocks(
6199 const FeatureBitset &Features, const MCExpr *VCCUsed,
6200 const MCExpr *FlatScrUsed, bool XNACKUsed,
6201 std::optional<bool> EnableWavefrontSize32, const MCExpr *NextFreeVGPR,
6202 SMRange VGPRRange, const MCExpr *NextFreeSGPR, SMRange SGPRRange,
6203 const MCExpr *&VGPRBlocks, const MCExpr *&SGPRBlocks) {
6204 // TODO(scott.linder): These calculations are duplicated from
6205 // AMDGPUAsmPrinter::getSIProgramInfo and could be unified.
6206 MCContext &Ctx = getContext();
6207
6208 const MCExpr *NumSGPRs = NextFreeSGPR;
6209 int64_t EvaluatedSGPRs;
6210
6211 if (ISA.Major >= 10)
6213 else {
6214 unsigned MaxAddressableNumSGPRs = AMDGPU::getAddressableNumSGPRs(Gfx);
6215
6216 if (NumSGPRs->evaluateAsAbsolute(EvaluatedSGPRs) && ISA.Major >= 8 &&
6217 !Features.test(FeatureSGPRInitBug) &&
6218 static_cast<uint64_t>(EvaluatedSGPRs) > MaxAddressableNumSGPRs)
6219 return OutOfRangeError(SGPRRange);
6220
6221 const MCExpr *ExtraSGPRs =
6222 AMDGPUMCExpr::createExtraSGPRs(VCCUsed, FlatScrUsed, XNACKUsed, Ctx);
6223 NumSGPRs = MCBinaryExpr::createAdd(NumSGPRs, ExtraSGPRs, Ctx);
6224
6225 if (NumSGPRs->evaluateAsAbsolute(EvaluatedSGPRs) &&
6226 (ISA.Major <= 7 || Features.test(FeatureSGPRInitBug)) &&
6227 static_cast<uint64_t>(EvaluatedSGPRs) > MaxAddressableNumSGPRs)
6228 return OutOfRangeError(SGPRRange);
6229
6230 if (Features.test(FeatureSGPRInitBug))
6231 NumSGPRs =
6233 }
6234
6235 // The MCExpr equivalent of getNumSGPRBlocks/getNumVGPRBlocks:
6236 // (alignTo(max(1u, NumGPR), GPREncodingGranule) / GPREncodingGranule) - 1
6237 auto GetNumGPRBlocks = [&Ctx](const MCExpr *NumGPR,
6238 unsigned Granule) -> const MCExpr * {
6239 const MCExpr *OneConst = MCConstantExpr::create(1ul, Ctx);
6240 const MCExpr *GranuleConst = MCConstantExpr::create(Granule, Ctx);
6241 const MCExpr *MaxNumGPR = AMDGPUMCExpr::createMax({NumGPR, OneConst}, Ctx);
6242 const MCExpr *AlignToGPR =
6243 AMDGPUMCExpr::createAlignTo(MaxNumGPR, GranuleConst, Ctx);
6244 const MCExpr *DivGPR =
6245 MCBinaryExpr::createDiv(AlignToGPR, GranuleConst, Ctx);
6246 const MCExpr *SubGPR = MCBinaryExpr::createSub(DivGPR, OneConst, Ctx);
6247 return SubGPR;
6248 };
6249
6250 VGPRBlocks = GetNumGPRBlocks(
6251 NextFreeVGPR,
6252 IsaInfo::getVGPREncodingGranule(getSTI(), EnableWavefrontSize32));
6253 SGPRBlocks =
6254 GetNumGPRBlocks(NumSGPRs, IsaInfo::getSGPREncodingGranule(getSTI()));
6255
6256 return false;
6257}
6258
6259bool AMDGPUAsmParser::ParseDirectiveAMDHSAKernel() {
6260 if (!getSTI().getTargetTriple().isAMDGCN())
6261 return TokError("directive only supported for amdgcn architecture");
6262
6263 if (!isHsaAbi(getSTI()))
6264 return TokError("directive only supported for amdhsa OS");
6265
6266 StringRef KernelName;
6267 if (getParser().parseIdentifier(KernelName))
6268 return true;
6269
6270 // Remember the kernel name so its prologue can be checked at end of file.
6271 // The matching label may have been parsed already or may follow later.
6272 AMDHSAKernelSymbols.insert(getContext().getOrCreateSymbol(KernelName));
6273
6274 AMDGPU::MCKernelDescriptor KD =
6276 &getSTI(), getContext());
6277
6278 StringSet<> Seen;
6279
6280 const MCExpr *ZeroExpr = MCConstantExpr::create(0, getContext());
6281 const MCExpr *OneExpr = MCConstantExpr::create(1, getContext());
6282
6283 SMRange VGPRRange;
6284 const MCExpr *NextFreeVGPR = ZeroExpr;
6285 const MCExpr *AccumOffset = MCConstantExpr::create(0, getContext());
6286 const MCExpr *NamedBarCnt = ZeroExpr;
6287 uint64_t SharedVGPRCount = 0;
6288 uint64_t PreloadLength = 0;
6289 uint64_t PreloadOffset = 0;
6290 SMRange SGPRRange;
6291 const MCExpr *NextFreeSGPR = ZeroExpr;
6292
6293 // Count the number of user SGPRs implied from the enabled feature bits.
6294 unsigned ImpliedUserSGPRCount = 0;
6295
6296 // Track if the asm explicitly contains the directive for the user SGPR
6297 // count.
6298 std::optional<unsigned> ExplicitUserSGPRCount;
6299 const MCExpr *ReserveVCC = OneExpr;
6300 const MCExpr *ReserveFlatScr = OneExpr;
6301 std::optional<bool> EnableWavefrontSize32;
6302
6303 while (true) {
6304 while (trySkipToken(AsmToken::EndOfStatement))
6305 ;
6306
6307 StringRef ID;
6308 SMRange IDRange = getTok().getLocRange();
6309 if (!parseId(ID, "expected .amdhsa_ directive or .end_amdhsa_kernel"))
6310 return true;
6311
6312 if (ID == ".end_amdhsa_kernel")
6313 break;
6314
6315 if (!Seen.insert(ID).second)
6316 return TokError(".amdhsa_ directives cannot be repeated");
6317
6318 SMLoc ValStart = getLoc();
6319 const MCExpr *ExprVal;
6320 if (getParser().parseExpression(ExprVal))
6321 return true;
6322 SMLoc ValEnd = getLoc();
6323 SMRange ValRange = SMRange(ValStart, ValEnd);
6324
6325 int64_t IVal = 0;
6326 uint64_t Val = IVal;
6327 bool EvaluatableExpr;
6328 if ((EvaluatableExpr = ExprVal->evaluateAsAbsolute(IVal))) {
6329 if (IVal < 0)
6330 return OutOfRangeError(ValRange);
6331 Val = IVal;
6332 }
6333
6334#define PARSE_BITS_ENTRY(FIELD, ENTRY, VALUE, RANGE) \
6335 if (!isUInt<ENTRY##_WIDTH>(Val)) \
6336 return OutOfRangeError(RANGE); \
6337 AMDGPU::MCKernelDescriptor::bits_set(FIELD, VALUE, ENTRY##_SHIFT, ENTRY, \
6338 getContext());
6339
6340// Some fields use the parsed value immediately which requires the expression to
6341// be solvable.
6342#define EXPR_RESOLVE_OR_ERROR(RESOLVED) \
6343 if (!(RESOLVED)) \
6344 return Error(IDRange.Start, "directive should have resolvable expression", \
6345 IDRange);
6346
6347 if (ID == ".amdhsa_group_segment_fixed_size") {
6349 CHAR_BIT>(Val))
6350 return OutOfRangeError(ValRange);
6351 KD.group_segment_fixed_size = ExprVal;
6352 } else if (ID == ".amdhsa_private_segment_fixed_size") {
6354 CHAR_BIT>(Val))
6355 return OutOfRangeError(ValRange);
6356 KD.private_segment_fixed_size = ExprVal;
6357 } else if (ID == ".amdhsa_kernarg_size") {
6358 if (!isUInt<sizeof(kernel_descriptor_t::kernarg_size) * CHAR_BIT>(Val))
6359 return OutOfRangeError(ValRange);
6360 KD.kernarg_size = ExprVal;
6361 } else if (ID == ".amdhsa_user_sgpr_count") {
6362 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6363 ExplicitUserSGPRCount = Val;
6364 } else if (ID == ".amdhsa_user_sgpr_private_segment_buffer") {
6365 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6367 return Error(IDRange.Start,
6368 "directive is not supported with architected flat scratch",
6369 IDRange);
6371 KERNEL_CODE_PROPERTY_ENABLE_SGPR_PRIVATE_SEGMENT_BUFFER,
6372 ExprVal, ValRange);
6373 if (Val)
6374 ImpliedUserSGPRCount += 4;
6375 } else if (ID == ".amdhsa_user_sgpr_kernarg_preload_length") {
6376 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6377 if (!hasKernargPreload())
6378 return Error(IDRange.Start, "directive requires gfx90a+", IDRange);
6379
6380 if (Val > getMaxNumUserSGPRs())
6381 return OutOfRangeError(ValRange);
6382 PARSE_BITS_ENTRY(KD.kernarg_preload, KERNARG_PRELOAD_SPEC_LENGTH, ExprVal,
6383 ValRange);
6384 if (Val) {
6385 ImpliedUserSGPRCount += Val;
6386 PreloadLength = Val;
6387 }
6388 } else if (ID == ".amdhsa_user_sgpr_kernarg_preload_offset") {
6389 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6390 if (!hasKernargPreload())
6391 return Error(IDRange.Start, "directive requires gfx90a+", IDRange);
6392
6393 if (Val >= 1024)
6394 return OutOfRangeError(ValRange);
6395 PARSE_BITS_ENTRY(KD.kernarg_preload, KERNARG_PRELOAD_SPEC_OFFSET, ExprVal,
6396 ValRange);
6397 if (Val)
6398 PreloadOffset = Val;
6399 } else if (ID == ".amdhsa_user_sgpr_dispatch_ptr") {
6400 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6402 KERNEL_CODE_PROPERTY_ENABLE_SGPR_DISPATCH_PTR, ExprVal,
6403 ValRange);
6404 if (Val)
6405 ImpliedUserSGPRCount += 2;
6406 } else if (ID == ".amdhsa_user_sgpr_queue_ptr") {
6407 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6409 KERNEL_CODE_PROPERTY_ENABLE_SGPR_QUEUE_PTR, ExprVal,
6410 ValRange);
6411 if (Val)
6412 ImpliedUserSGPRCount += 2;
6413 } else if (ID == ".amdhsa_user_sgpr_kernarg_segment_ptr") {
6414 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6416 KERNEL_CODE_PROPERTY_ENABLE_SGPR_KERNARG_SEGMENT_PTR,
6417 ExprVal, ValRange);
6418 if (Val)
6419 ImpliedUserSGPRCount += 2;
6420 } else if (ID == ".amdhsa_user_sgpr_dispatch_id") {
6421 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6423 KERNEL_CODE_PROPERTY_ENABLE_SGPR_DISPATCH_ID, ExprVal,
6424 ValRange);
6425 if (Val)
6426 ImpliedUserSGPRCount += 2;
6427 } else if (ID == ".amdhsa_user_sgpr_flat_scratch_init") {
6429 return Error(IDRange.Start,
6430 "directive is not supported with architected flat scratch",
6431 IDRange);
6432 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6434 KERNEL_CODE_PROPERTY_ENABLE_SGPR_FLAT_SCRATCH_INIT,
6435 ExprVal, ValRange);
6436 if (Val)
6437 ImpliedUserSGPRCount += 2;
6438 } else if (ID == ".amdhsa_user_sgpr_private_segment_size") {
6439 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6441 KERNEL_CODE_PROPERTY_ENABLE_SGPR_PRIVATE_SEGMENT_SIZE,
6442 ExprVal, ValRange);
6443 if (Val)
6444 ImpliedUserSGPRCount += 1;
6445 } else if (ID == ".amdhsa_wavefront_size32") {
6446 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6447 if (ISA.Major < 10)
6448 return Error(IDRange.Start, "directive requires gfx10+", IDRange);
6449 EnableWavefrontSize32 = Val;
6451 KERNEL_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32, ExprVal,
6452 ValRange);
6453 } else if (ID == ".amdhsa_uses_dynamic_stack") {
6455 KERNEL_CODE_PROPERTY_USES_DYNAMIC_STACK, ExprVal,
6456 ValRange);
6457 } else if (ID == ".amdhsa_system_sgpr_private_segment_wavefront_offset") {
6459 return Error(IDRange.Start,
6460 "directive is not supported with architected flat scratch",
6461 IDRange);
6463 COMPUTE_PGM_RSRC2_ENABLE_PRIVATE_SEGMENT, ExprVal,
6464 ValRange);
6465 } else if (ID == ".amdhsa_enable_private_segment") {
6467 return Error(
6468 IDRange.Start,
6469 "directive is not supported without architected flat scratch",
6470 IDRange);
6472 COMPUTE_PGM_RSRC2_ENABLE_PRIVATE_SEGMENT, ExprVal,
6473 ValRange);
6474 } else if (ID == ".amdhsa_system_sgpr_workgroup_id_x") {
6476 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_ID_X, ExprVal,
6477 ValRange);
6478 } else if (ID == ".amdhsa_system_sgpr_workgroup_id_y") {
6480 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_ID_Y, ExprVal,
6481 ValRange);
6482 } else if (ID == ".amdhsa_system_sgpr_workgroup_id_z") {
6484 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_ID_Z, ExprVal,
6485 ValRange);
6486 } else if (ID == ".amdhsa_system_sgpr_workgroup_info") {
6488 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_INFO, ExprVal,
6489 ValRange);
6490 } else if (ID == ".amdhsa_system_vgpr_workitem_id") {
6492 COMPUTE_PGM_RSRC2_ENABLE_VGPR_WORKITEM_ID, ExprVal,
6493 ValRange);
6494 } else if (ID == ".amdhsa_next_free_vgpr") {
6495 VGPRRange = ValRange;
6496 NextFreeVGPR = ExprVal;
6497 } else if (ID == ".amdhsa_next_free_sgpr") {
6498 SGPRRange = ValRange;
6499 NextFreeSGPR = ExprVal;
6500 } else if (ID == ".amdhsa_accum_offset") {
6501 if (!isGFX90A())
6502 return Error(IDRange.Start, "directive requires gfx90a+", IDRange);
6503 AccumOffset = ExprVal;
6504 } else if (ID == ".amdhsa_named_barrier_count") {
6505 if (!isGFX1250Plus())
6506 return Error(IDRange.Start, "directive requires gfx1250+", IDRange);
6507 NamedBarCnt = ExprVal;
6508 } else if (ID == ".amdhsa_reserve_vcc") {
6509 if (EvaluatableExpr && !isUInt<1>(Val))
6510 return OutOfRangeError(ValRange);
6511 ReserveVCC = ExprVal;
6512 } else if (ID == ".amdhsa_reserve_flat_scratch") {
6513 if (ISA.Major < 7)
6514 return Error(IDRange.Start, "directive requires gfx7+", IDRange);
6516 return Error(IDRange.Start,
6517 "directive is not supported with architected flat scratch",
6518 IDRange);
6519 if (EvaluatableExpr && !isUInt<1>(Val))
6520 return OutOfRangeError(ValRange);
6521 ReserveFlatScr = ExprVal;
6522 } else if (ID == ".amdhsa_reserve_xnack_mask") {
6523 if (ISA.Major < 8)
6524 return Error(IDRange.Start, "directive requires gfx8+", IDRange);
6525 if (!isUInt<1>(Val))
6526 return OutOfRangeError(ValRange);
6527 bool XnackOn = getTargetStreamer().getTargetID()->isXnackOnOrAny();
6528 if (Val != XnackOn) {
6529 return getParser().Error(
6530 IDRange.Start,
6531 ".amdhsa_reserve_xnack_mask does not match target id", IDRange);
6532 }
6533 } else if (ID == ".amdhsa_float_round_mode_32") {
6535 COMPUTE_PGM_RSRC1_FLOAT_ROUND_MODE_32, ExprVal,
6536 ValRange);
6537 } else if (ID == ".amdhsa_float_round_mode_16_64") {
6539 COMPUTE_PGM_RSRC1_FLOAT_ROUND_MODE_16_64, ExprVal,
6540 ValRange);
6541 } else if (ID == ".amdhsa_float_denorm_mode_32") {
6543 COMPUTE_PGM_RSRC1_FLOAT_DENORM_MODE_32, ExprVal,
6544 ValRange);
6545 } else if (ID == ".amdhsa_float_denorm_mode_16_64") {
6547 COMPUTE_PGM_RSRC1_FLOAT_DENORM_MODE_16_64, ExprVal,
6548 ValRange);
6549 } else if (ID == ".amdhsa_dx10_clamp") {
6550 if (!getSTI().hasFeature(AMDGPU::FeatureDX10ClampAndIEEEMode))
6551 return Error(IDRange.Start, "directive unsupported on gfx1170+",
6552 IDRange);
6554 COMPUTE_PGM_RSRC1_GFX6_GFX11_ENABLE_DX10_CLAMP, ExprVal,
6555 ValRange);
6556 } else if (ID == ".amdhsa_ieee_mode") {
6557 if (!getSTI().hasFeature(AMDGPU::FeatureDX10ClampAndIEEEMode))
6558 return Error(IDRange.Start, "directive unsupported on gfx1170+",
6559 IDRange);
6561 COMPUTE_PGM_RSRC1_GFX6_GFX11_ENABLE_IEEE_MODE, ExprVal,
6562 ValRange);
6563 } else if (ID == ".amdhsa_fp16_overflow") {
6564 if (ISA.Major < 9)
6565 return Error(IDRange.Start, "directive requires gfx9+", IDRange);
6567 COMPUTE_PGM_RSRC1_GFX9_PLUS_FP16_OVFL, ExprVal,
6568 ValRange);
6569 } else if (ID == ".amdhsa_tg_split") {
6570 if (!isGFX90A())
6571 return Error(IDRange.Start, "directive requires gfx90a+", IDRange);
6572 PARSE_BITS_ENTRY(KD.compute_pgm_rsrc3, COMPUTE_PGM_RSRC3_GFX90A_TG_SPLIT,
6573 ExprVal, ValRange);
6574 } else if (ID == ".amdhsa_workgroup_processor_mode") {
6575 if (!supportsWGP(getSTI()))
6576 return Error(IDRange.Start,
6577 "directive unsupported on " + getSTI().getCPU(), IDRange);
6579 COMPUTE_PGM_RSRC1_GFX10_PLUS_WGP_MODE, ExprVal,
6580 ValRange);
6581 } else if (ID == ".amdhsa_memory_ordered") {
6582 if (ISA.Major < 10)
6583 return Error(IDRange.Start, "directive requires gfx10+", IDRange);
6585 COMPUTE_PGM_RSRC1_GFX10_PLUS_MEM_ORDERED, ExprVal,
6586 ValRange);
6587 } else if (ID == ".amdhsa_forward_progress") {
6588 if (ISA.Major < 10)
6589 return Error(IDRange.Start, "directive requires gfx10+", IDRange);
6591 COMPUTE_PGM_RSRC1_GFX10_PLUS_FWD_PROGRESS, ExprVal,
6592 ValRange);
6593 } else if (ID == ".amdhsa_shared_vgpr_count") {
6594 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6595 if (ISA.Major < 10 || ISA.Major >= 12)
6596 return Error(IDRange.Start, "directive requires gfx10 or gfx11",
6597 IDRange);
6598 SharedVGPRCount = Val;
6600 COMPUTE_PGM_RSRC3_GFX10_GFX11_SHARED_VGPR_COUNT, ExprVal,
6601 ValRange);
6602 } else if (ID == ".amdhsa_inst_pref_size") {
6603 if (ISA.Major < 11)
6604 return Error(IDRange.Start, "directive requires gfx11+", IDRange);
6605 if (ISA.Major == 11) {
6607 COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE, ExprVal,
6608 ValRange);
6609 } else {
6611 COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE, ExprVal,
6612 ValRange);
6613 }
6614 } else if (ID == ".amdhsa_exception_fp_ieee_invalid_op") {
6617 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_INVALID_OPERATION,
6618 ExprVal, ValRange);
6619 } else if (ID == ".amdhsa_exception_fp_denorm_src") {
6621 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_FP_DENORMAL_SOURCE,
6622 ExprVal, ValRange);
6623 } else if (ID == ".amdhsa_exception_fp_ieee_div_zero") {
6626 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_DIVISION_BY_ZERO,
6627 ExprVal, ValRange);
6628 } else if (ID == ".amdhsa_exception_fp_ieee_overflow") {
6630 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_OVERFLOW,
6631 ExprVal, ValRange);
6632 } else if (ID == ".amdhsa_exception_fp_ieee_underflow") {
6634 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_UNDERFLOW,
6635 ExprVal, ValRange);
6636 } else if (ID == ".amdhsa_exception_fp_ieee_inexact") {
6638 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_INEXACT,
6639 ExprVal, ValRange);
6640 } else if (ID == ".amdhsa_exception_int_div_zero") {
6642 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_INT_DIVIDE_BY_ZERO,
6643 ExprVal, ValRange);
6644 } else if (ID == ".amdhsa_round_robin_scheduling") {
6645 if (ISA.Major < 12)
6646 return Error(IDRange.Start, "directive requires gfx12+", IDRange);
6648 COMPUTE_PGM_RSRC1_GFX12_PLUS_ENABLE_WG_RR_EN, ExprVal,
6649 ValRange);
6650 } else {
6651 return Error(IDRange.Start, "unknown .amdhsa_kernel directive", IDRange);
6652 }
6653
6654#undef PARSE_BITS_ENTRY
6655 }
6656
6657 if (!Seen.contains(".amdhsa_next_free_vgpr"))
6658 return TokError(".amdhsa_next_free_vgpr directive is required");
6659
6660 if (!Seen.contains(".amdhsa_next_free_sgpr"))
6661 return TokError(".amdhsa_next_free_sgpr directive is required");
6662
6663 unsigned UserSGPRCount = ExplicitUserSGPRCount.value_or(ImpliedUserSGPRCount);
6664 if (UserSGPRCount > getMaxNumUserSGPRs())
6665 return TokError("too many user SGPRs enabled, found " +
6666 Twine(UserSGPRCount) + ", but only " +
6667 Twine(getMaxNumUserSGPRs()) + " are supported.");
6668
6669 // Consider the case where the total number of UserSGPRs with trailing
6670 // allocated preload SGPRs, is greater than the number of explicitly
6671 // referenced SGPRs.
6672 if (PreloadLength) {
6673 MCContext &Ctx = getContext();
6674 NextFreeSGPR = AMDGPUMCExpr::createMax(
6675 {NextFreeSGPR, MCConstantExpr::create(UserSGPRCount, Ctx)}, Ctx);
6676 }
6677
6678 const MCExpr *VGPRBlocks;
6679 const MCExpr *SGPRBlocks;
6680 if (calculateGPRBlocks(getFeatureBits(), ReserveVCC, ReserveFlatScr,
6681 getTargetStreamer().getTargetID()->isXnackOnOrAny(),
6682 EnableWavefrontSize32, NextFreeVGPR, VGPRRange,
6683 NextFreeSGPR, SGPRRange, VGPRBlocks, SGPRBlocks))
6684 return true;
6685
6686 int64_t EvaluatedVGPRBlocks;
6687 bool VGPRBlocksEvaluatable =
6688 VGPRBlocks->evaluateAsAbsolute(EvaluatedVGPRBlocks);
6689 if (VGPRBlocksEvaluatable &&
6691 static_cast<uint64_t>(EvaluatedVGPRBlocks))) {
6692 return OutOfRangeError(VGPRRange);
6693 }
6695 KD.compute_pgm_rsrc1, VGPRBlocks,
6696 COMPUTE_PGM_RSRC1_GRANULATED_WORKITEM_VGPR_COUNT_SHIFT,
6697 COMPUTE_PGM_RSRC1_GRANULATED_WORKITEM_VGPR_COUNT, getContext());
6698
6699 int64_t EvaluatedSGPRBlocks;
6700 if (SGPRBlocks->evaluateAsAbsolute(EvaluatedSGPRBlocks) &&
6702 static_cast<uint64_t>(EvaluatedSGPRBlocks)))
6703 return OutOfRangeError(SGPRRange);
6705 KD.compute_pgm_rsrc1, SGPRBlocks,
6706 COMPUTE_PGM_RSRC1_GRANULATED_WAVEFRONT_SGPR_COUNT_SHIFT,
6707 COMPUTE_PGM_RSRC1_GRANULATED_WAVEFRONT_SGPR_COUNT, getContext());
6708
6709 if (ExplicitUserSGPRCount && ImpliedUserSGPRCount > *ExplicitUserSGPRCount)
6710 return TokError("amdgpu_user_sgpr_count smaller than implied by "
6711 "enabled user SGPRs");
6712
6713 if (isGFX1250Plus()) {
6716 MCConstantExpr::create(UserSGPRCount, getContext()),
6717 COMPUTE_PGM_RSRC2_GFX125_USER_SGPR_COUNT_SHIFT,
6718 COMPUTE_PGM_RSRC2_GFX125_USER_SGPR_COUNT, getContext());
6719 } else {
6722 MCConstantExpr::create(UserSGPRCount, getContext()),
6723 COMPUTE_PGM_RSRC2_GFX6_GFX120_USER_SGPR_COUNT_SHIFT,
6724 COMPUTE_PGM_RSRC2_GFX6_GFX120_USER_SGPR_COUNT, getContext());
6725 }
6726
6727 int64_t IVal = 0;
6728 if (!KD.kernarg_size->evaluateAsAbsolute(IVal))
6729 return TokError("Kernarg size should be resolvable");
6730 uint64_t kernarg_size = IVal;
6731 if (PreloadLength && kernarg_size &&
6732 (PreloadLength * 4 + PreloadOffset * 4 > kernarg_size))
6733 return TokError("Kernarg preload length + offset is larger than the "
6734 "kernarg segment size");
6735
6736 if (isGFX90A()) {
6737 if (!Seen.contains(".amdhsa_accum_offset"))
6738 return TokError(".amdhsa_accum_offset directive is required");
6739 int64_t EvaluatedAccum;
6740 bool AccumEvaluatable = AccumOffset->evaluateAsAbsolute(EvaluatedAccum);
6741 uint64_t UEvaluatedAccum = EvaluatedAccum;
6742 if (AccumEvaluatable &&
6743 (UEvaluatedAccum < 4 || UEvaluatedAccum > 256 || (UEvaluatedAccum & 3)))
6744 return TokError("accum_offset should be in range [4..256] in "
6745 "increments of 4");
6746
6747 int64_t EvaluatedNumVGPR;
6748 if (NextFreeVGPR->evaluateAsAbsolute(EvaluatedNumVGPR) &&
6749 AccumEvaluatable &&
6750 UEvaluatedAccum >
6751 alignTo(std::max((uint64_t)1, (uint64_t)EvaluatedNumVGPR), 4))
6752 return TokError("accum_offset exceeds total VGPR allocation");
6753 const MCExpr *AdjustedAccum = MCBinaryExpr::createSub(
6755 AccumOffset, MCConstantExpr::create(4, getContext()), getContext()),
6758 COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET_SHIFT,
6759 COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET,
6760 getContext());
6761 }
6762
6763 if (isGFX1250Plus())
6765 COMPUTE_PGM_RSRC3_GFX125_NAMED_BAR_CNT_SHIFT,
6766 COMPUTE_PGM_RSRC3_GFX125_NAMED_BAR_CNT,
6767 getContext());
6768
6769 if (ISA.Major >= 10 && ISA.Major < 12) {
6770 // SharedVGPRCount < 16 checked by PARSE_ENTRY_BITS
6771 if (SharedVGPRCount && EnableWavefrontSize32 && *EnableWavefrontSize32) {
6772 return TokError("shared_vgpr_count directive not valid on "
6773 "wavefront size 32");
6774 }
6775
6776 if (VGPRBlocksEvaluatable &&
6777 (SharedVGPRCount * 2 + static_cast<uint64_t>(EvaluatedVGPRBlocks) >
6778 63)) {
6779 return TokError("shared_vgpr_count*2 + "
6780 "compute_pgm_rsrc1.GRANULATED_WORKITEM_VGPR_COUNT cannot "
6781 "exceed 63\n");
6782 }
6783 }
6784
6785 emitTargetDirective();
6786 getTargetStreamer().EmitAmdhsaKernelDescriptor(getSTI(), KernelName, KD,
6787 NextFreeVGPR, NextFreeSGPR,
6788 ReserveVCC, ReserveFlatScr);
6789 return false;
6790}
6791
6792bool AMDGPUAsmParser::ParseDirectiveAMDHSACodeObjectVersion() {
6793 uint32_t Version;
6794 if (ParseAsAbsoluteExpression(Version))
6795 return true;
6796
6797 getTargetStreamer().EmitDirectiveAMDHSACodeObjectVersion(Version);
6798 emitTargetDirective();
6799 return false;
6800}
6801
6802bool AMDGPUAsmParser::ParseAMDKernelCodeTValue(StringRef ID,
6803 AMDGPUMCKernelCodeT &C) {
6804 // max_scratch_backing_memory_byte_size is deprecated. Ignore it while parsing
6805 // assembly for backwards compatibility.
6806 if (ID == "max_scratch_backing_memory_byte_size") {
6807 Parser.eatToEndOfStatement();
6808 return false;
6809 }
6810
6811 SmallString<40> ErrStr;
6812 raw_svector_ostream Err(ErrStr);
6813 if (!C.ParseKernelCodeT(ID, getParser(), Err)) {
6814 return TokError(Err.str());
6815 }
6816 Lex();
6817
6818 if (ID == "enable_wavefront_size32") {
6819 if (C.code_properties & AMD_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32) {
6820 if (!isGFX10Plus())
6821 return TokError("enable_wavefront_size32=1 is only allowed on GFX10+");
6822 if (!isWave32())
6823 return TokError("enable_wavefront_size32=1 requires +WavefrontSize32");
6824 } else {
6825 if (!isWave64())
6826 return TokError("enable_wavefront_size32=0 requires +WavefrontSize64");
6827 }
6828 }
6829
6830 if (ID == "wavefront_size") {
6831 if (C.wavefront_size == 5) {
6832 if (!isGFX10Plus())
6833 return TokError("wavefront_size=5 is only allowed on GFX10+");
6834 if (!isWave32())
6835 return TokError("wavefront_size=5 requires +WavefrontSize32");
6836 } else if (C.wavefront_size == 6) {
6837 if (!isWave64())
6838 return TokError("wavefront_size=6 requires +WavefrontSize64");
6839 }
6840 }
6841
6842 return false;
6843}
6844
6845bool AMDGPUAsmParser::ParseDirectiveAMDKernelCodeT() {
6846 AMDGPUMCKernelCodeT KernelCode;
6847 KernelCode.initDefault(getSTI(), getContext());
6848
6849 while (true) {
6850 // Lex EndOfStatement. This is in a while loop, because lexing a comment
6851 // will set the current token to EndOfStatement.
6852 while (trySkipToken(AsmToken::EndOfStatement))
6853 ;
6854
6855 StringRef ID;
6856 if (!parseId(ID, "expected value identifier or .end_amd_kernel_code_t"))
6857 return true;
6858
6859 if (ID == ".end_amd_kernel_code_t")
6860 break;
6861
6862 if (ParseAMDKernelCodeTValue(ID, KernelCode))
6863 return true;
6864 }
6865
6866 KernelCode.validate(&getSTI(), getContext());
6867 getTargetStreamer().EmitAMDKernelCodeT(KernelCode);
6868
6869 return false;
6870}
6871
6872bool AMDGPUAsmParser::ParseDirectiveAMDGPUHsaKernel() {
6873 StringRef KernelName;
6874 if (!parseId(KernelName, "expected symbol name"))
6875 return true;
6876
6877 getTargetStreamer().EmitAMDGPUSymbolType(KernelName,
6879
6880 KernelScope.initialize(getContext());
6881 return false;
6882}
6883
6884bool AMDGPUAsmParser::ParseDirectiveISAVersion() {
6885 if (!getSTI().getTargetTriple().isAMDGCN()) {
6886 return Error(getLoc(),
6887 ".amd_amdgpu_isa directive is not available on non-amdgcn "
6888 "architectures");
6889 }
6890
6891 StringRef TargetIDDirective = getLexer().getTok().getStringContents();
6892
6893 std::optional<AMDGPU::TargetID> MaybeParsed =
6894 AMDGPU::TargetID::parseTargetIDString(TargetIDDirective);
6895 if (!MaybeParsed)
6896 return Error(getParser().getTok().getLoc(),
6897 "malformed target id '" + TargetIDDirective + "'");
6898
6899 const AMDGPU::TargetID &ParsedTargetID = *MaybeParsed;
6900 const Triple &TT = getSTI().getTargetTriple();
6901
6902 // The processor named in the target id must be covered by the triple's
6903 // subarch.
6904 if (!AMDGPU::isCPUValidForSubArch(TT.getSubArch(),
6905 ParsedTargetID.getGPUKind())) {
6906 return Error(getParser().getTok().getLoc(),
6907 "target id '" + TargetIDDirective +
6908 "' specifies a processor that is not valid for subarch '" +
6909 TT.getArchName() + "'");
6910 }
6911
6912 const std::optional<AMDGPU::TargetID> &CurrentTargetID =
6913 getTargetStreamer().getTargetID();
6914
6915 Triple DirectiveTriple(ParsedTargetID.getTargetTripleString());
6916 const Triple &STITriple = getSTI().getTargetTriple();
6917 if (!DirectiveTriple.isCompatibleWith(STITriple)) {
6918 return Error(getParser().getTok().getLoc(),
6919 ".amd_amdgpu_isa " + Twine(ParsedTargetID.toString()) +
6920 " is incompatible with " +
6921 Twine(CurrentTargetID->toString()));
6922 }
6923
6924 // Error if the ISA version doesn't match
6925 StringRef DirectiveProcessor =
6926 AMDGPU::getArchNameAMDGCN(ParsedTargetID.getGPUKind());
6927 AMDGPU::IsaVersion DirectiveISA = AMDGPU::getIsaVersion(DirectiveProcessor);
6928 if (DirectiveISA != ISA) {
6929 return Error(getParser().getTok().getLoc(),
6930 ".amd_amdgpu_isa directive processor " +
6931 Twine(DirectiveProcessor) +
6932 " does not match the specified processor " +
6933 Twine(getSTI().getCPU()));
6934 }
6935
6936 getTargetStreamer().EmitISAVersion();
6937 Lex();
6938
6939 return false;
6940}
6941
6942bool AMDGPUAsmParser::ParseDirectiveHSAMetadata() {
6943 assert(isHsaAbi(getSTI()));
6944
6945 std::string HSAMetadataString;
6946 if (ParseToEndDirective(HSAMD::V3::AssemblerDirectiveBegin,
6947 HSAMD::V3::AssemblerDirectiveEnd, HSAMetadataString))
6948 return true;
6949
6950 if (!getTargetStreamer().EmitHSAMetadataV3(HSAMetadataString))
6951 return Error(getLoc(), "invalid HSA metadata");
6952
6953 return false;
6954}
6955
6956/// Common code to parse out a block of text (typically YAML) between start and
6957/// end directives.
6958bool AMDGPUAsmParser::ParseToEndDirective(const char *AssemblerDirectiveBegin,
6959 const char *AssemblerDirectiveEnd,
6960 std::string &CollectString) {
6961
6962 raw_string_ostream CollectStream(CollectString);
6963
6964 getLexer().setSkipSpace(false);
6965
6966 bool FoundEnd = false;
6967 while (!isToken(AsmToken::Eof)) {
6968 while (isToken(AsmToken::Space)) {
6969 CollectStream << getTokenStr();
6970 Lex();
6971 }
6972
6973 if (trySkipId(AssemblerDirectiveEnd)) {
6974 FoundEnd = true;
6975 break;
6976 }
6977
6978 CollectStream << Parser.parseStringToEndOfStatement()
6979 << getContext().getAsmInfo().getSeparatorString();
6980
6981 Parser.eatToEndOfStatement();
6982 }
6983
6984 getLexer().setSkipSpace(true);
6985
6986 if (isToken(AsmToken::Eof) && !FoundEnd) {
6987 return TokError(Twine("expected directive ") +
6988 Twine(AssemblerDirectiveEnd) + Twine(" not found"));
6989 }
6990
6991 return false;
6992}
6993
6994/// Parse the assembler directive for new MsgPack-format PAL metadata.
6995bool AMDGPUAsmParser::ParseDirectivePALMetadataBegin() {
6996 std::string String;
6997 if (ParseToEndDirective(AMDGPU::PALMD::AssemblerDirectiveBegin,
6999 return true;
7000
7001 auto *PALMetadata = getTargetStreamer().getPALMetadata();
7002 if (!PALMetadata->setFromString(String))
7003 return Error(getLoc(), "invalid PAL metadata");
7004 return false;
7005}
7006
7007/// Parse the assembler directive for old linear-format PAL metadata.
7008bool AMDGPUAsmParser::ParseDirectivePALMetadata() {
7009 if (getSTI().getTargetTriple().getOS() != Triple::AMDPAL) {
7010 return Error(getLoc(), (Twine(PALMD::AssemblerDirective) +
7011 Twine(" directive is "
7012 "not available on non-amdpal OSes"))
7013 .str());
7014 }
7015
7016 auto *PALMetadata = getTargetStreamer().getPALMetadata();
7017 PALMetadata->setLegacy();
7018 for (;;) {
7019 uint32_t Key, Value;
7020 if (ParseAsAbsoluteExpression(Key)) {
7021 return TokError(Twine("invalid value in ") +
7023 }
7024 if (!trySkipToken(AsmToken::Comma)) {
7025 return TokError(Twine("expected an even number of values in ") +
7027 }
7028 if (ParseAsAbsoluteExpression(Value)) {
7029 return TokError(Twine("invalid value in ") +
7031 }
7032 PALMetadata->setRegister(Key, Value);
7033 if (!trySkipToken(AsmToken::Comma))
7034 break;
7035 }
7036 return false;
7037}
7038
7039/// ParseDirectiveAMDGPULDS
7040/// ::= .amdgpu_lds identifier ',' size_expression [',' align_expression]
7041bool AMDGPUAsmParser::ParseDirectiveAMDGPULDS() {
7042 if (getParser().checkForValidSection())
7043 return true;
7044
7045 StringRef Name;
7046 SMLoc NameLoc = getLoc();
7047 if (getParser().parseIdentifier(Name))
7048 return TokError("expected identifier in directive");
7049
7050 MCSymbol *Symbol = getContext().getOrCreateSymbol(Name);
7051 if (getParser().parseComma())
7052 return true;
7053
7054 unsigned LocalMemorySize = AMDGPU::IsaInfo::getLocalMemorySize(getSTI());
7055
7056 int64_t Size;
7057 SMLoc SizeLoc = getLoc();
7058 if (getParser().parseAbsoluteExpression(Size))
7059 return true;
7060 if (Size < 0)
7061 return Error(SizeLoc, "size must be non-negative");
7062 if (Size > LocalMemorySize)
7063 return Error(SizeLoc, "size is too large");
7064
7065 int64_t Alignment = 4;
7066 if (trySkipToken(AsmToken::Comma)) {
7067 SMLoc AlignLoc = getLoc();
7068 if (getParser().parseAbsoluteExpression(Alignment))
7069 return true;
7070 if (Alignment < 0 || !isPowerOf2_64(Alignment))
7071 return Error(AlignLoc, "alignment must be a power of two");
7072
7073 // Alignment larger than the size of LDS is possible in theory, as long
7074 // as the linker manages to place to symbol at address 0, but we do want
7075 // to make sure the alignment fits nicely into a 32-bit integer.
7076 if (Alignment >= 1u << 31)
7077 return Error(AlignLoc, "alignment is too large");
7078 }
7079
7080 if (parseEOL())
7081 return true;
7082
7083 Symbol->redefineIfPossible();
7084 if (!Symbol->isUndefined())
7085 return Error(NameLoc, "invalid symbol redefinition");
7086
7087 getTargetStreamer().emitAMDGPULDS(Symbol, Size, Align(Alignment));
7088 return false;
7089}
7090
7091bool AMDGPUAsmParser::ParseDirectiveAMDGPUInfo() {
7092 if (getParser().checkForValidSection())
7093 return true;
7094
7095 StringRef FuncName;
7096 if (getParser().parseIdentifier(FuncName))
7097 return TokError("expected symbol name after .amdgpu_info");
7098
7099 MCSymbol *FuncSym = getContext().getOrCreateSymbol(FuncName);
7100 AMDGPU::InfoSectionData ParsedInfoData;
7101 AMDGPU::FuncInfo FI;
7102 FI.Sym = FuncSym;
7103 bool HasScalarAttrs = false;
7104
7105 while (true) {
7106 while (trySkipToken(AsmToken::EndOfStatement))
7107 ;
7108
7109 StringRef ID;
7110 SMLoc IDLoc = getLoc();
7111 if (!parseId(ID, "expected directive or .end_amdgpu_info"))
7112 return true;
7113
7114 if (ID == ".end_amdgpu_info")
7115 break;
7116
7117 // Every per-entry directive shares the `.amdgpu_` namespace prefix; strip
7118 // it once and dispatch on the distinguishing suffix below. The unstripped
7119 // ID is preserved for diagnostics.
7120 StringRef Dir = ID;
7121 if (!Dir.consume_front(".amdgpu_"))
7122 return Error(IDLoc, "unknown .amdgpu_info directive '" + ID + "'");
7123
7124 if (Dir == "flags") {
7125 int64_t Val;
7126 if (getParser().parseAbsoluteExpression(Val))
7127 return true;
7128 auto Flags = static_cast<AMDGPU::FuncInfoFlags>(Val);
7129 FI.UsesVCC = !!(Flags & AMDGPU::FuncInfoFlags::FUNC_USES_VCC);
7130 FI.UsesFlatScratch =
7131 !!(Flags & AMDGPU::FuncInfoFlags::FUNC_USES_FLAT_SCRATCH);
7132 FI.HasDynStack = !!(Flags & AMDGPU::FuncInfoFlags::FUNC_HAS_DYN_STACK);
7133 HasScalarAttrs = true;
7134 } else if (Dir == "num_sgpr") {
7135 int64_t Val;
7136 if (getParser().parseAbsoluteExpression(Val))
7137 return true;
7138 FI.NumSGPR = static_cast<uint32_t>(Val);
7139 HasScalarAttrs = true;
7140 } else if (Dir == "num_vgpr") {
7141 int64_t Val;
7142 if (getParser().parseAbsoluteExpression(Val))
7143 return true;
7144 FI.NumArchVGPR = static_cast<uint32_t>(Val);
7145 HasScalarAttrs = true;
7146 } else if (Dir == "num_agpr") {
7147 int64_t Val;
7148 if (getParser().parseAbsoluteExpression(Val))
7149 return true;
7150 FI.NumAccVGPR = static_cast<uint32_t>(Val);
7151 HasScalarAttrs = true;
7152 } else if (Dir == "private_segment_size") {
7153 int64_t Val;
7154 if (getParser().parseAbsoluteExpression(Val))
7155 return true;
7156 FI.PrivateSegmentSize = static_cast<uint32_t>(Val);
7157 HasScalarAttrs = true;
7158 } else if (Dir == "use") {
7159 StringRef ResName;
7160 if (getParser().parseIdentifier(ResName))
7161 return TokError("expected resource symbol for .amdgpu_use");
7162 ParsedInfoData.Uses.push_back(
7163 {FuncSym, getContext().getOrCreateSymbol(ResName)});
7164 } else if (Dir == "call") {
7165 StringRef DstName;
7166 if (getParser().parseIdentifier(DstName))
7167 return TokError("expected callee symbol for .amdgpu_call");
7168 ParsedInfoData.Calls.push_back(
7169 {FuncSym, getContext().getOrCreateSymbol(DstName)});
7170 } else if (Dir == "indirect_call") {
7171 std::string TypeId;
7172 if (getParser().parseEscapedString(TypeId))
7173 return TokError("expected type ID string for .amdgpu_indirect_call");
7174 ParsedInfoData.IndirectCalls.push_back({FuncSym, std::move(TypeId)});
7175 } else if (Dir == "typeid") {
7176 std::string TypeId;
7177 if (getParser().parseEscapedString(TypeId))
7178 return TokError("expected type ID string for .amdgpu_typeid");
7179 ParsedInfoData.TypeIds.push_back({FuncSym, std::move(TypeId)});
7180 } else {
7181 return Error(IDLoc, "unknown .amdgpu_info directive '" + ID + "'");
7182 }
7183 }
7184
7185 if (HasScalarAttrs)
7186 ParsedInfoData.Funcs.push_back(std::move(FI));
7187
7188 AMDGPU::InfoSectionData &Data = InfoData ? *InfoData : InfoData.emplace();
7189 for (AMDGPU::FuncInfo &Func : ParsedInfoData.Funcs)
7190 Data.Funcs.push_back(std::move(Func));
7191 for (std::pair<MCSymbol *, MCSymbol *> &Use : ParsedInfoData.Uses)
7192 Data.Uses.push_back(Use);
7193 for (std::pair<MCSymbol *, MCSymbol *> &Call : ParsedInfoData.Calls)
7194 Data.Calls.push_back(Call);
7195 for (std::pair<MCSymbol *, std::string> &IndirectCall :
7196 ParsedInfoData.IndirectCalls)
7197 Data.IndirectCalls.push_back(std::move(IndirectCall));
7198 for (std::pair<MCSymbol *, std::string> &TypeId : ParsedInfoData.TypeIds)
7199 Data.TypeIds.push_back(std::move(TypeId));
7200
7201 return false;
7202}
7203
7204void AMDGPUAsmParser::doBeforeLabelEmit(MCSymbol *Symbol, SMLoc IDLoc) {
7205 // Record every parsed label in the timeline so that, at end of file, the
7206 // instructions following a kernel's label can be located regardless of
7207 // whether the .amdhsa_kernel directive came before or after the label.
7208 OpcodeStreamSymbols.emplace_back(Symbol, IDLoc, OpcodeStream.size());
7209}
7210
7211void AMDGPUAsmParser::checkKernelPrologues() {
7212 if (getFeatureBits()[AMDGPU::FeatureRequiresInitialUnclausedVmem]) {
7213 static const unsigned Required[] = {S_MOV_B64_gfx12, V_NOP_e32_gfx12,
7214 GLOBAL_PREFETCH_B8_SADDR_gfx1250};
7215 for (auto [Sym, Loc, Offset] : OpcodeStreamSymbols) {
7216 if (!AMDHSAKernelSymbols.contains(Sym))
7217 continue;
7218 ArrayRef<unsigned> Prologue = ArrayRef(OpcodeStream).drop_front(Offset);
7219 if (!Prologue.empty() && Prologue.front() == S_SETREG_IMM32_B32_gfx12)
7220 Prologue = Prologue.drop_front();
7221 if (Prologue.take_front(std::size(Required)) != ArrayRef(Required)) {
7222 Warning(Loc, "kernel '" + Sym->getName() +
7223 "' does not begin with the required prologue "
7224 "sequence: s_mov_b64 followed by v_nop and "
7225 "global_prefetch_b8");
7226 }
7227 }
7228 }
7229 OpcodeStream.clear();
7230 OpcodeStreamSymbols.clear();
7231 AMDHSAKernelSymbols.clear();
7232}
7233
7234void AMDGPUAsmParser::onEndOfFile() {
7235 emitTargetDirective();
7236 checkKernelPrologues();
7237 if (InfoData)
7238 getTargetStreamer().emitAMDGPUInfo(*InfoData);
7239}
7240
7241bool AMDGPUAsmParser::ParseDirective(AsmToken DirectiveID) {
7242 StringRef IDVal = DirectiveID.getString();
7243
7244 if (isHsaAbi(getSTI())) {
7245 if (IDVal == ".amdhsa_kernel")
7246 return ParseDirectiveAMDHSAKernel();
7247
7248 if (IDVal == ".amdhsa_code_object_version")
7249 return ParseDirectiveAMDHSACodeObjectVersion();
7250
7251 // TODO: Restructure/combine with PAL metadata directive.
7253 return ParseDirectiveHSAMetadata();
7254 } else {
7255 if (IDVal == ".amd_kernel_code_t")
7256 return ParseDirectiveAMDKernelCodeT();
7257
7258 if (IDVal == ".amdgpu_hsa_kernel")
7259 return ParseDirectiveAMDGPUHsaKernel();
7260
7261 if (IDVal == ".amd_amdgpu_isa")
7262 return ParseDirectiveISAVersion();
7263
7265 return Error(getLoc(), (Twine(HSAMD::AssemblerDirectiveBegin) +
7266 Twine(" directive is "
7267 "not available on non-amdhsa OSes"))
7268 .str());
7269 }
7270 }
7271
7272 if (IDVal == ".amdgcn_target")
7273 return ParseDirectiveAMDGCNTarget();
7274
7275 if (IDVal == ".amdgpu_lds")
7276 return ParseDirectiveAMDGPULDS();
7277
7278 if (IDVal == ".amdgpu_info")
7279 return ParseDirectiveAMDGPUInfo();
7280
7281 if (IDVal == PALMD::AssemblerDirectiveBegin)
7282 return ParseDirectivePALMetadataBegin();
7283
7284 if (IDVal == PALMD::AssemblerDirective)
7285 return ParseDirectivePALMetadata();
7286
7287 return true;
7288}
7289
7290bool AMDGPUAsmParser::subtargetHasRegister(const MCRegisterInfo &MRI,
7291 MCRegister Reg) {
7292 if (MRI.regsOverlap(TTMP12_TTMP13_TTMP14_TTMP15, Reg))
7293 return isGFX9Plus();
7294
7295 // GFX10+ has 2 more SGPRs 104 and 105.
7296 if (MRI.regsOverlap(SGPR104_SGPR105, Reg))
7297 return hasSGPR104_SGPR105();
7298
7299 switch (Reg.id()) {
7300 case SRC_SHARED_BASE_LO:
7301 case SRC_SHARED_BASE:
7302 case SRC_SHARED_LIMIT_LO:
7303 case SRC_SHARED_LIMIT:
7304 return isGFX9Plus();
7305 case SRC_PRIVATE_BASE_LO:
7306 case SRC_PRIVATE_BASE:
7307 case SRC_PRIVATE_LIMIT_LO:
7308 case SRC_PRIVATE_LIMIT:
7309 return AMDGPU::hasPrivateApertureRegs(getSTI());
7310 case SRC_FLAT_SCRATCH_BASE_LO:
7311 case SRC_FLAT_SCRATCH_BASE_HI:
7312 return hasGloballyAddressableScratch();
7313 case SRC_POPS_EXITING_WAVE_ID:
7314 return hasPopsExitingWaveID(getSTI());
7315 case TBA:
7316 case TBA_LO:
7317 case TBA_HI:
7318 case TMA:
7319 case TMA_LO:
7320 case TMA_HI:
7321 return !isGFX9Plus();
7322 case XNACK_MASK:
7323 case XNACK_MASK_LO:
7324 case XNACK_MASK_HI:
7325 return (isVI() || isGFX9()) &&
7326 getTargetStreamer().getTargetID()->isXnackSupported();
7327 case SGPR_NULL:
7328 return isGFX10Plus();
7329 case SRC_EXECZ:
7330 case SRC_VCCZ:
7331 return !isGFX11Plus();
7332 default:
7333 break;
7334 }
7335
7336 if (isCI())
7337 return true;
7338
7339 if (isSI() || isGFX10Plus()) {
7340 // No flat_scr on SI.
7341 // On GFX10Plus flat scratch is not a valid register operand and can only be
7342 // accessed with s_setreg/s_getreg.
7343 switch (Reg.id()) {
7344 case FLAT_SCR:
7345 case FLAT_SCR_LO:
7346 case FLAT_SCR_HI:
7347 return false;
7348 default:
7349 return true;
7350 }
7351 }
7352
7353 // VI only has 102 SGPRs, so make sure we aren't trying to use the 2 more that
7354 // SI/CI have.
7355 if (MRI.regsOverlap(SGPR102_SGPR103, Reg))
7356 return hasSGPR102_SGPR103();
7357
7358 return true;
7359}
7360
7361ParseStatus AMDGPUAsmParser::parseOperand(OperandVector &Operands,
7362 StringRef Mnemonic,
7363 OperandMode Mode) {
7364 ParseStatus Res = parseVOPD(Operands);
7365 if (Res.isSuccess() || Res.isFailure() || isToken(AsmToken::EndOfStatement))
7366 return Res;
7367
7368 // Try to parse with a custom parser
7369 Res = MatchOperandParserImpl(Operands, Mnemonic);
7370
7371 // If we successfully parsed the operand or if there as an error parsing,
7372 // we are done.
7373 //
7374 // If we are parsing after we reach EndOfStatement then this means we
7375 // are appending default values to the Operands list. This is only done
7376 // by custom parser, so we shouldn't continue on to the generic parsing.
7377 if (Res.isSuccess() || Res.isFailure() || isToken(AsmToken::EndOfStatement))
7378 return Res;
7379
7380 SMLoc RBraceLoc;
7381 SMLoc LBraceLoc = getLoc();
7382 if (Mode == OperandMode_NSA && trySkipToken(AsmToken::LBrac)) {
7383 unsigned Prefix = Operands.size();
7384
7385 for (;;) {
7386 auto Loc = getLoc();
7387 Res = parseReg(Operands);
7388 if (Res.isNoMatch())
7389 Error(Loc, "expected a register");
7390 if (!Res.isSuccess())
7391 return ParseStatus::Failure;
7392
7393 RBraceLoc = getLoc();
7394 if (trySkipToken(AsmToken::RBrac))
7395 break;
7396
7397 if (!skipToken(AsmToken::Comma,
7398 "expected a comma or a closing square bracket"))
7399 return ParseStatus::Failure;
7400 }
7401
7402 if (Operands.size() - Prefix > 1) {
7403 Operands.insert(Operands.begin() + Prefix,
7404 AMDGPUOperand::CreateToken(this, "[", LBraceLoc));
7405 Operands.push_back(AMDGPUOperand::CreateToken(this, "]", RBraceLoc));
7406 }
7407
7408 return ParseStatus::Success;
7409 }
7410
7411 return parseRegOrImm(Operands);
7412}
7413
7414StringRef AMDGPUAsmParser::parseMnemonicSuffix(StringRef Name) {
7415 // Clear any forced encodings from the previous instruction.
7416 setForcedEncodingSize(0);
7417 setForcedDPP(false);
7418 setForcedSDWA(false);
7419
7420 if (Name.consume_back("_e64_dpp")) {
7421 setForcedDPP(true);
7422 setForcedEncodingSize(64);
7423 return Name;
7424 }
7425 if (Name.consume_back("_e64")) {
7426 setForcedEncodingSize(64);
7427 return Name;
7428 }
7429 if (Name.consume_back("_e32")) {
7430 setForcedEncodingSize(32);
7431 return Name;
7432 }
7433 if (Name.consume_back("_dpp")) {
7434 setForcedDPP(true);
7435 return Name;
7436 }
7437 if (Name.consume_back("_sdwa")) {
7438 setForcedSDWA(true);
7439 return Name;
7440 }
7441 return Name;
7442}
7443
7444static void applyMnemonicAliases(StringRef &Mnemonic,
7445 const FeatureBitset &Features,
7446 unsigned VariantID);
7447
7448bool AMDGPUAsmParser::parseInstruction(ParseInstructionInfo &Info,
7449 StringRef Name, SMLoc NameLoc,
7451 // Add the instruction mnemonic
7452 Name = parseMnemonicSuffix(Name);
7453
7454 // If the target architecture uses MnemonicAlias, call it here to parse
7455 // operands correctly.
7456 applyMnemonicAliases(Name, getAvailableFeatures(), 0);
7457
7458 Operands.push_back(AMDGPUOperand::CreateToken(this, Name, NameLoc));
7459
7460 bool IsMIMG = Name.starts_with("image_");
7461
7462 while (!trySkipToken(AsmToken::EndOfStatement)) {
7463 OperandMode Mode = OperandMode_Default;
7464 if (IsMIMG && isGFX10Plus() && Operands.size() == 2)
7465 Mode = OperandMode_NSA;
7466 ParseStatus Res = parseOperand(Operands, Name, Mode);
7467
7468 if (!Res.isSuccess()) {
7469 checkUnsupportedInstruction(Name, NameLoc);
7470 if (!Parser.hasPendingError()) {
7471 // FIXME: use real operand location rather than the current location.
7472 StringRef Msg = Res.isFailure() ? "failed parsing operand."
7473 : "not a valid operand.";
7474 Error(getLoc(), Msg);
7475 }
7476 while (!trySkipToken(AsmToken::EndOfStatement)) {
7477 lex();
7478 }
7479 return true;
7480 }
7481
7482 // Eat the comma or space if there is one.
7483 trySkipToken(AsmToken::Comma);
7484 }
7485
7486 return false;
7487}
7488
7489//===----------------------------------------------------------------------===//
7490// Utility functions
7491//===----------------------------------------------------------------------===//
7492
7493ParseStatus AMDGPUAsmParser::parseTokenOp(StringRef Name,
7495 SMLoc S = getLoc();
7496 if (!trySkipId(Name))
7497 return ParseStatus::NoMatch;
7498
7499 Operands.push_back(AMDGPUOperand::CreateToken(this, Name, S));
7500 return ParseStatus::Success;
7501}
7502
7503ParseStatus AMDGPUAsmParser::parseIntWithPrefix(const char *Prefix,
7504 int64_t &IntVal) {
7505
7506 if (!trySkipId(Prefix, AsmToken::Colon))
7507 return ParseStatus::NoMatch;
7508
7510}
7511
7512ParseStatus AMDGPUAsmParser::parseIntWithPrefix(
7513 const char *Prefix, OperandVector &Operands, AMDGPUOperand::ImmTy ImmTy,
7514 std::function<bool(int64_t &)> ConvertResult) {
7515 SMLoc S = getLoc();
7516 int64_t Value = 0;
7517
7518 ParseStatus Res = parseIntWithPrefix(Prefix, Value);
7519 if (!Res.isSuccess())
7520 return Res;
7521
7522 if (ConvertResult && !ConvertResult(Value)) {
7523 Error(S, "invalid " + StringRef(Prefix) + " value.");
7524 }
7525
7526 Operands.push_back(AMDGPUOperand::CreateImm(this, Value, S, ImmTy));
7527 return ParseStatus::Success;
7528}
7529
7530ParseStatus AMDGPUAsmParser::parseOperandArrayWithPrefix(
7531 const char *Prefix, OperandVector &Operands, AMDGPUOperand::ImmTy ImmTy,
7532 bool (*ConvertResult)(int64_t &)) {
7533 SMLoc S = getLoc();
7534 if (!trySkipId(Prefix, AsmToken::Colon))
7535 return ParseStatus::NoMatch;
7536
7537 if (!skipToken(AsmToken::LBrac, "expected a left square bracket"))
7538 return ParseStatus::Failure;
7539
7540 unsigned Val = 0;
7541 const unsigned MaxSize = 4;
7542
7543 // FIXME: How to verify the number of elements matches the number of src
7544 // operands?
7545 for (int I = 0;; ++I) {
7546 int64_t Op;
7547 SMLoc Loc = getLoc();
7548 if (!parseExpr(Op))
7549 return ParseStatus::Failure;
7550
7551 if (Op != 0 && Op != 1)
7552 return Error(Loc, "invalid " + StringRef(Prefix) + " value.");
7553
7554 Val |= (Op << I);
7555
7556 if (trySkipToken(AsmToken::RBrac))
7557 break;
7558
7559 if (I + 1 == MaxSize)
7560 return Error(getLoc(), "expected a closing square bracket");
7561
7562 if (!skipToken(AsmToken::Comma, "expected a comma"))
7563 return ParseStatus::Failure;
7564 }
7565
7566 Operands.push_back(AMDGPUOperand::CreateImm(this, Val, S, ImmTy));
7567 return ParseStatus::Success;
7568}
7569
7570ParseStatus AMDGPUAsmParser::parseNamedBit(StringRef Name,
7572 AMDGPUOperand::ImmTy ImmTy,
7573 bool IgnoreNegative) {
7574 int64_t Bit;
7575 SMLoc S = getLoc();
7576
7577 if (trySkipId(Name)) {
7578 Bit = 1;
7579 } else if (trySkipId("no", Name)) {
7580 if (IgnoreNegative)
7581 return ParseStatus::Success;
7582 Bit = 0;
7583 } else {
7584 return ParseStatus::NoMatch;
7585 }
7586
7587 if (Name == "r128" && !hasMIMG_R128())
7588 return Error(S, "r128 modifier is not supported on this GPU");
7589 if (Name == "a16" && !hasA16())
7590 return Error(S, "a16 modifier is not supported on this GPU");
7591
7592 if (Bit == 0 && Name == "gds") {
7593 StringRef Mnemo = ((AMDGPUOperand &)*Operands[0]).getToken();
7594 if (Mnemo.starts_with("ds_gws"))
7595 return Error(S, "nogds is not allowed");
7596 }
7597
7598 if (isGFX9() && ImmTy == AMDGPUOperand::ImmTyA16)
7599 ImmTy = AMDGPUOperand::ImmTyR128A16;
7600
7601 Operands.push_back(AMDGPUOperand::CreateImm(this, Bit, S, ImmTy));
7602 return ParseStatus::Success;
7603}
7604
7605unsigned AMDGPUAsmParser::getCPolKind(StringRef Id, StringRef Mnemo,
7606 bool &Disabling) const {
7607 Disabling = Id.consume_front("no");
7608
7609 if (isGFX940() && !Mnemo.starts_with("s_")) {
7610 return StringSwitch<unsigned>(Id)
7611 .Case("nt", AMDGPU::CPol::NT)
7612 .Case("sc0", AMDGPU::CPol::SC0)
7613 .Case("sc1", AMDGPU::CPol::SC1)
7614 .Default(0);
7615 }
7616
7617 return StringSwitch<unsigned>(Id)
7618 .Case("dlc", AMDGPU::CPol::DLC)
7619 .Case("glc", AMDGPU::CPol::GLC)
7620 .Case("scc", AMDGPU::CPol::SCC)
7621 .Case("slc", AMDGPU::CPol::SLC)
7622 .Default(0);
7623}
7624
7625ParseStatus AMDGPUAsmParser::parseCPol(OperandVector &Operands) {
7626 if (isGFX12Plus()) {
7627 SMLoc StringLoc = getLoc();
7628
7629 int64_t CPolVal = 0;
7630 ParseStatus ResTH = ParseStatus::NoMatch;
7631 ParseStatus ResScope = ParseStatus::NoMatch;
7632 ParseStatus ResNV = ParseStatus::NoMatch;
7633 ParseStatus ResScal = ParseStatus::NoMatch;
7634
7635 for (;;) {
7636 if (ResTH.isNoMatch()) {
7637 int64_t TH;
7638 ResTH = parseTH(Operands, TH);
7639 if (ResTH.isFailure())
7640 return ResTH;
7641 if (ResTH.isSuccess()) {
7642 CPolVal |= TH;
7643 continue;
7644 }
7645 }
7646
7647 if (ResScope.isNoMatch()) {
7648 int64_t Scope;
7649 ResScope = parseScope(Operands, Scope);
7650 if (ResScope.isFailure())
7651 return ResScope;
7652 if (ResScope.isSuccess()) {
7653 CPolVal |= Scope;
7654 continue;
7655 }
7656 }
7657
7658 // NV bit exists on GFX12+, but does something starting from GFX1250.
7659 // Allow parsing on all GFX12 and fail on validation for better
7660 // diagnostics.
7661 if (ResNV.isNoMatch()) {
7662 if (trySkipId("nv")) {
7663 ResNV = ParseStatus::Success;
7664 CPolVal |= CPol::NV;
7665 continue;
7666 } else if (trySkipId("no", "nv")) {
7667 ResNV = ParseStatus::Success;
7668 continue;
7669 }
7670 }
7671
7672 if (ResScal.isNoMatch()) {
7673 if (trySkipId("scale_offset")) {
7674 ResScal = ParseStatus::Success;
7675 CPolVal |= CPol::SCAL;
7676 continue;
7677 } else if (trySkipId("no", "scale_offset")) {
7678 ResScal = ParseStatus::Success;
7679 continue;
7680 }
7681 }
7682
7683 break;
7684 }
7685
7686 if (ResTH.isNoMatch() && ResScope.isNoMatch() && ResNV.isNoMatch() &&
7687 ResScal.isNoMatch())
7688 return ParseStatus::NoMatch;
7689
7690 Operands.push_back(AMDGPUOperand::CreateImm(this, CPolVal, StringLoc,
7691 AMDGPUOperand::ImmTyCPol));
7692 return ParseStatus::Success;
7693 }
7694
7695 StringRef Mnemo = ((AMDGPUOperand &)*Operands[0]).getToken();
7696 SMLoc OpLoc = getLoc();
7697 unsigned Enabled = 0, Seen = 0;
7698 for (;;) {
7699 SMLoc S = getLoc();
7700 bool Disabling;
7701 unsigned CPol = getCPolKind(getId(), Mnemo, Disabling);
7702 if (!CPol)
7703 break;
7704
7705 lex();
7706
7707 if (!isGFX10Plus() && CPol == AMDGPU::CPol::DLC)
7708 return Error(S, "dlc modifier is not supported on this GPU");
7709
7710 if (!isGFX90A() && CPol == AMDGPU::CPol::SCC)
7711 return Error(S, "scc modifier is not supported on this GPU");
7712
7713 if (Seen & CPol)
7714 return Error(S, "duplicate cache policy modifier");
7715
7716 if (!Disabling)
7717 Enabled |= CPol;
7718
7719 Seen |= CPol;
7720 }
7721
7722 if (!Seen)
7723 return ParseStatus::NoMatch;
7724
7725 Operands.push_back(
7726 AMDGPUOperand::CreateImm(this, Enabled, OpLoc, AMDGPUOperand::ImmTyCPol));
7727 return ParseStatus::Success;
7728}
7729
7730ParseStatus AMDGPUAsmParser::parseScope(OperandVector &Operands,
7731 int64_t &Scope) {
7732 static const unsigned Scopes[] = {CPol::SCOPE_CU, CPol::SCOPE_SE,
7734
7735 ParseStatus Res = parseStringOrIntWithPrefix(
7736 Operands, "scope", {"SCOPE_CU", "SCOPE_SE", "SCOPE_DEV", "SCOPE_SYS"},
7737 Scope);
7738
7739 if (Res.isSuccess())
7740 Scope = Scopes[Scope];
7741
7742 return Res;
7743}
7744
7745ParseStatus AMDGPUAsmParser::parseTH(OperandVector &Operands, int64_t &TH) {
7746 TH = AMDGPU::CPol::TH_RT; // default
7747
7748 StringRef Value;
7749 SMLoc StringLoc;
7750 ParseStatus Res = parseStringWithPrefix("th", Value, StringLoc);
7751 if (!Res.isSuccess())
7752 return Res;
7753
7754 if (Value == "TH_DEFAULT")
7756 else if (Value == "TH_STORE_LU" || Value == "TH_LOAD_WB" ||
7757 Value == "TH_LOAD_NT_WB") {
7758 return Error(StringLoc, "invalid th value");
7759 } else if (Value.consume_front("TH_ATOMIC_")) {
7761 } else if (Value.consume_front("TH_LOAD_")) {
7763 } else if (Value.consume_front("TH_STORE_")) {
7765 } else {
7766 return Error(StringLoc, "invalid th value");
7767 }
7768
7769 if (Value == "BYPASS")
7771
7772 if (TH != 0) {
7774 TH |= StringSwitch<int64_t>(Value)
7775 .Case("RETURN", AMDGPU::CPol::TH_ATOMIC_RETURN)
7776 .Case("RT", AMDGPU::CPol::TH_RT)
7777 .Case("RT_RETURN", AMDGPU::CPol::TH_ATOMIC_RETURN)
7778 .Case("NT", AMDGPU::CPol::TH_ATOMIC_NT)
7779 .Case("NT_RETURN", AMDGPU::CPol::TH_ATOMIC_NT |
7781 .Case("CASCADE_RT", AMDGPU::CPol::TH_ATOMIC_CASCADE)
7782 .Case("CASCADE_NT", AMDGPU::CPol::TH_ATOMIC_CASCADE |
7784 .Default(0xffffffff);
7785 else
7786 TH |= StringSwitch<int64_t>(Value)
7787 .Case("RT", AMDGPU::CPol::TH_RT)
7788 .Case("NT", AMDGPU::CPol::TH_NT)
7789 .Case("HT", AMDGPU::CPol::TH_HT)
7790 .Case("LU", AMDGPU::CPol::TH_LU)
7791 .Case("WB", AMDGPU::CPol::TH_WB)
7792 .Case("NT_RT", AMDGPU::CPol::TH_NT_RT)
7793 .Case("RT_NT", AMDGPU::CPol::TH_RT_NT)
7794 .Case("NT_HT", AMDGPU::CPol::TH_NT_HT)
7795 .Case("NT_WB", AMDGPU::CPol::TH_NT_WB)
7796 .Case("BYPASS", AMDGPU::CPol::TH_BYPASS)
7797 .Default(0xffffffff);
7798 }
7799
7800 if (TH == 0xffffffff)
7801 return Error(StringLoc, "invalid th value");
7802
7803 return ParseStatus::Success;
7804}
7805
7806static void
7808 AMDGPUAsmParser::OptionalImmIndexMap &OptionalIdx,
7809 AMDGPUOperand::ImmTy ImmT, int64_t Default = 0,
7810 std::optional<unsigned> InsertAt = std::nullopt) {
7811 auto i = OptionalIdx.find(ImmT);
7812 if (i != OptionalIdx.end()) {
7813 unsigned Idx = i->second;
7814 const AMDGPUOperand &Op =
7815 static_cast<const AMDGPUOperand &>(*Operands[Idx]);
7816 if (InsertAt)
7817 Inst.insert(Inst.begin() + *InsertAt, MCOperand::createImm(Op.getImm()));
7818 else
7819 Op.addImmOperands(Inst, 1);
7820 } else {
7821 if (InsertAt.has_value())
7822 Inst.insert(Inst.begin() + *InsertAt, MCOperand::createImm(Default));
7823 else
7825 }
7826}
7827
7828ParseStatus AMDGPUAsmParser::parseStringWithPrefix(StringRef Prefix,
7829 StringRef &Value,
7830 SMLoc &StringLoc) {
7831 if (!trySkipId(Prefix, AsmToken::Colon))
7832 return ParseStatus::NoMatch;
7833
7834 StringLoc = getLoc();
7835 return parseId(Value, "expected an identifier") ? ParseStatus::Success
7837}
7838
7839ParseStatus AMDGPUAsmParser::parseStringOrIntWithPrefix(
7840 OperandVector &Operands, StringRef Name, ArrayRef<const char *> Ids,
7841 int64_t &IntVal) {
7842 if (!trySkipId(Name, AsmToken::Colon))
7843 return ParseStatus::NoMatch;
7844
7845 SMLoc StringLoc = getLoc();
7846
7847 StringRef Value;
7848 if (isToken(AsmToken::Identifier)) {
7849 Value = getTokenStr();
7850 lex();
7851
7852 for (IntVal = 0; IntVal < (int64_t)Ids.size(); ++IntVal)
7853 if (Value == Ids[IntVal])
7854 break;
7855 } else if (!parseExpr(IntVal))
7856 return ParseStatus::Failure;
7857
7858 if (IntVal < 0 || IntVal >= (int64_t)Ids.size())
7859 return Error(StringLoc, "invalid " + Twine(Name) + " value");
7860
7861 return ParseStatus::Success;
7862}
7863
7864ParseStatus AMDGPUAsmParser::parseStringOrIntWithPrefix(
7865 OperandVector &Operands, StringRef Name, ArrayRef<const char *> Ids,
7866 AMDGPUOperand::ImmTy Type) {
7867 SMLoc S = getLoc();
7868 int64_t IntVal;
7869
7870 ParseStatus Res = parseStringOrIntWithPrefix(Operands, Name, Ids, IntVal);
7871 if (Res.isSuccess())
7872 Operands.push_back(AMDGPUOperand::CreateImm(this, IntVal, S, Type));
7873
7874 return Res;
7875}
7876
7877//===----------------------------------------------------------------------===//
7878// MTBUF format
7879//===----------------------------------------------------------------------===//
7880
7881bool AMDGPUAsmParser::tryParseFmt(const char *Pref, int64_t MaxVal,
7882 int64_t &Fmt) {
7883 int64_t Val;
7884 SMLoc Loc = getLoc();
7885
7886 auto Res = parseIntWithPrefix(Pref, Val);
7887 if (Res.isFailure())
7888 return false;
7889 if (Res.isNoMatch())
7890 return true;
7891
7892 if (Val < 0 || Val > MaxVal) {
7893 Error(Loc, Twine("out of range ", StringRef(Pref)));
7894 return false;
7895 }
7896
7897 Fmt = Val;
7898 return true;
7899}
7900
7901ParseStatus AMDGPUAsmParser::tryParseIndexKey(OperandVector &Operands,
7902 AMDGPUOperand::ImmTy ImmTy) {
7903 const char *Pref = "index_key";
7904 int64_t ImmVal = 0;
7905 SMLoc Loc = getLoc();
7906 auto Res = parseIntWithPrefix(Pref, ImmVal);
7907 if (!Res.isSuccess())
7908 return Res;
7909
7910 if ((ImmTy == AMDGPUOperand::ImmTyIndexKey16bit ||
7911 ImmTy == AMDGPUOperand::ImmTyIndexKey32bit) &&
7912 (ImmVal < 0 || ImmVal > 1))
7913 return Error(Loc, Twine("out of range ", StringRef(Pref)));
7914
7915 if (ImmTy == AMDGPUOperand::ImmTyIndexKey8bit && (ImmVal < 0 || ImmVal > 3))
7916 return Error(Loc, Twine("out of range ", StringRef(Pref)));
7917
7918 Operands.push_back(AMDGPUOperand::CreateImm(this, ImmVal, Loc, ImmTy));
7919 return ParseStatus::Success;
7920}
7921
7922ParseStatus AMDGPUAsmParser::parseIndexKey8bit(OperandVector &Operands) {
7923 return tryParseIndexKey(Operands, AMDGPUOperand::ImmTyIndexKey8bit);
7924}
7925
7926ParseStatus AMDGPUAsmParser::parseIndexKey16bit(OperandVector &Operands) {
7927 return tryParseIndexKey(Operands, AMDGPUOperand::ImmTyIndexKey16bit);
7928}
7929
7930ParseStatus AMDGPUAsmParser::parseIndexKey32bit(OperandVector &Operands) {
7931 return tryParseIndexKey(Operands, AMDGPUOperand::ImmTyIndexKey32bit);
7932}
7933
7934ParseStatus AMDGPUAsmParser::tryParseMatrixFMT(OperandVector &Operands,
7935 StringRef Name,
7936 AMDGPUOperand::ImmTy Type) {
7937 return parseStringOrIntWithPrefix(Operands, Name, WMMAMods::ModMatrixFmt,
7938 Type);
7939}
7940
7941ParseStatus AMDGPUAsmParser::parseMatrixAFMT(OperandVector &Operands) {
7942 return tryParseMatrixFMT(Operands, "matrix_a_fmt",
7943 AMDGPUOperand::ImmTyMatrixAFMT);
7944}
7945
7946ParseStatus AMDGPUAsmParser::parseMatrixBFMT(OperandVector &Operands) {
7947 return tryParseMatrixFMT(Operands, "matrix_b_fmt",
7948 AMDGPUOperand::ImmTyMatrixBFMT);
7949}
7950
7951ParseStatus AMDGPUAsmParser::tryParseMatrixScale(OperandVector &Operands,
7952 StringRef Name,
7953 AMDGPUOperand::ImmTy Type) {
7954 return parseStringOrIntWithPrefix(Operands, Name, WMMAMods::ModMatrixScale,
7955 Type);
7956}
7957
7958ParseStatus AMDGPUAsmParser::parseMatrixAScale(OperandVector &Operands) {
7959 return tryParseMatrixScale(Operands, "matrix_a_scale",
7960 AMDGPUOperand::ImmTyMatrixAScale);
7961}
7962
7963ParseStatus AMDGPUAsmParser::parseMatrixBScale(OperandVector &Operands) {
7964 return tryParseMatrixScale(Operands, "matrix_b_scale",
7965 AMDGPUOperand::ImmTyMatrixBScale);
7966}
7967
7968ParseStatus AMDGPUAsmParser::tryParseMatrixScaleFmt(OperandVector &Operands,
7969 StringRef Name,
7970 AMDGPUOperand::ImmTy Type) {
7971 return parseStringOrIntWithPrefix(Operands, Name, WMMAMods::ModMatrixScaleFmt,
7972 Type);
7973}
7974
7975ParseStatus AMDGPUAsmParser::parseMatrixAScaleFmt(OperandVector &Operands) {
7976 return tryParseMatrixScaleFmt(Operands, "matrix_a_scale_fmt",
7977 AMDGPUOperand::ImmTyMatrixAScaleFmt);
7978}
7979
7980ParseStatus AMDGPUAsmParser::parseMatrixBScaleFmt(OperandVector &Operands) {
7981 return tryParseMatrixScaleFmt(Operands, "matrix_b_scale_fmt",
7982 AMDGPUOperand::ImmTyMatrixBScaleFmt);
7983}
7984
7985// dfmt and nfmt (in a tbuffer instruction) are parsed as one to allow their
7986// values to live in a joint format operand in the MCInst encoding.
7987ParseStatus AMDGPUAsmParser::parseDfmtNfmt(int64_t &Format) {
7988 using namespace llvm::AMDGPU::MTBUFFormat;
7989
7990 int64_t Dfmt = DFMT_UNDEF;
7991 int64_t Nfmt = NFMT_UNDEF;
7992
7993 // dfmt and nfmt can appear in either order, and each is optional.
7994 for (int I = 0; I < 2; ++I) {
7995 if (Dfmt == DFMT_UNDEF && !tryParseFmt("dfmt", DFMT_MAX, Dfmt))
7996 return ParseStatus::Failure;
7997
7998 if (Nfmt == NFMT_UNDEF && !tryParseFmt("nfmt", NFMT_MAX, Nfmt))
7999 return ParseStatus::Failure;
8000
8001 // Skip optional comma between dfmt/nfmt
8002 // but guard against 2 commas following each other.
8003 if ((Dfmt == DFMT_UNDEF) != (Nfmt == NFMT_UNDEF) &&
8004 !peekToken().is(AsmToken::Comma)) {
8005 trySkipToken(AsmToken::Comma);
8006 }
8007 }
8008
8009 if (Dfmt == DFMT_UNDEF && Nfmt == NFMT_UNDEF)
8010 return ParseStatus::NoMatch;
8011
8012 Dfmt = (Dfmt == DFMT_UNDEF) ? DFMT_DEFAULT : Dfmt;
8013 Nfmt = (Nfmt == NFMT_UNDEF) ? NFMT_DEFAULT : Nfmt;
8014
8015 Format = encodeDfmtNfmt(Dfmt, Nfmt);
8016 return ParseStatus::Success;
8017}
8018
8019ParseStatus AMDGPUAsmParser::parseUfmt(int64_t &Format) {
8020 using namespace llvm::AMDGPU::MTBUFFormat;
8021
8022 int64_t Fmt = UFMT_UNDEF;
8023
8024 if (!tryParseFmt("format", UFMT_MAX, Fmt))
8025 return ParseStatus::Failure;
8026
8027 if (Fmt == UFMT_UNDEF)
8028 return ParseStatus::NoMatch;
8029
8030 Format = Fmt;
8031 return ParseStatus::Success;
8032}
8033
8034bool AMDGPUAsmParser::matchDfmtNfmt(int64_t &Dfmt, int64_t &Nfmt,
8035 StringRef FormatStr, SMLoc Loc) {
8036 using namespace llvm::AMDGPU::MTBUFFormat;
8037 int64_t Format;
8038
8039 Format = getDfmt(FormatStr);
8040 if (Format != DFMT_UNDEF) {
8041 Dfmt = Format;
8042 return true;
8043 }
8044
8045 Format = getNfmt(FormatStr, getSTI());
8046 if (Format != NFMT_UNDEF) {
8047 Nfmt = Format;
8048 return true;
8049 }
8050
8051 Error(Loc, "unsupported format");
8052 return false;
8053}
8054
8055ParseStatus AMDGPUAsmParser::parseSymbolicSplitFormat(StringRef FormatStr,
8056 SMLoc FormatLoc,
8057 int64_t &Format) {
8058 using namespace llvm::AMDGPU::MTBUFFormat;
8059
8060 int64_t Dfmt = DFMT_UNDEF;
8061 int64_t Nfmt = NFMT_UNDEF;
8062 if (!matchDfmtNfmt(Dfmt, Nfmt, FormatStr, FormatLoc))
8063 return ParseStatus::Failure;
8064
8065 if (trySkipToken(AsmToken::Comma)) {
8066 StringRef Str;
8067 SMLoc Loc = getLoc();
8068 if (!parseId(Str, "expected a format string") ||
8069 !matchDfmtNfmt(Dfmt, Nfmt, Str, Loc))
8070 return ParseStatus::Failure;
8071 if (Dfmt == DFMT_UNDEF)
8072 return Error(Loc, "duplicate numeric format");
8073 if (Nfmt == NFMT_UNDEF)
8074 return Error(Loc, "duplicate data format");
8075 }
8076
8077 Dfmt = (Dfmt == DFMT_UNDEF) ? DFMT_DEFAULT : Dfmt;
8078 Nfmt = (Nfmt == NFMT_UNDEF) ? NFMT_DEFAULT : Nfmt;
8079
8080 if (isGFX10Plus()) {
8081 auto Ufmt = convertDfmtNfmt2Ufmt(Dfmt, Nfmt, getSTI());
8082 if (Ufmt == UFMT_UNDEF)
8083 return Error(FormatLoc, "unsupported format");
8084 Format = Ufmt;
8085 } else {
8086 Format = encodeDfmtNfmt(Dfmt, Nfmt);
8087 }
8088
8089 return ParseStatus::Success;
8090}
8091
8092ParseStatus AMDGPUAsmParser::parseSymbolicUnifiedFormat(StringRef FormatStr,
8093 SMLoc Loc,
8094 int64_t &Format) {
8095 using namespace llvm::AMDGPU::MTBUFFormat;
8096
8097 auto Id = getUnifiedFormat(FormatStr, getSTI());
8098 if (Id == UFMT_UNDEF)
8099 return ParseStatus::NoMatch;
8100
8101 if (!isGFX10Plus())
8102 return Error(Loc, "unified format is not supported on this GPU");
8103
8104 Format = Id;
8105 return ParseStatus::Success;
8106}
8107
8108ParseStatus AMDGPUAsmParser::parseNumericFormat(int64_t &Format) {
8109 using namespace llvm::AMDGPU::MTBUFFormat;
8110 SMLoc Loc = getLoc();
8111
8112 if (!parseExpr(Format))
8113 return ParseStatus::Failure;
8114 if (!isValidFormatEncoding(Format, getSTI()))
8115 return Error(Loc, "out of range format");
8116
8117 return ParseStatus::Success;
8118}
8119
8120ParseStatus AMDGPUAsmParser::parseSymbolicOrNumericFormat(int64_t &Format) {
8121 using namespace llvm::AMDGPU::MTBUFFormat;
8122
8123 if (!trySkipId("format", AsmToken::Colon))
8124 return ParseStatus::NoMatch;
8125
8126 if (trySkipToken(AsmToken::LBrac)) {
8127 StringRef FormatStr;
8128 SMLoc Loc = getLoc();
8129 if (!parseId(FormatStr, "expected a format string"))
8130 return ParseStatus::Failure;
8131
8132 auto Res = parseSymbolicUnifiedFormat(FormatStr, Loc, Format);
8133 if (Res.isNoMatch())
8134 Res = parseSymbolicSplitFormat(FormatStr, Loc, Format);
8135 if (!Res.isSuccess())
8136 return Res;
8137
8138 if (!skipToken(AsmToken::RBrac, "expected a closing square bracket"))
8139 return ParseStatus::Failure;
8140
8141 return ParseStatus::Success;
8142 }
8143
8144 return parseNumericFormat(Format);
8145}
8146
8147ParseStatus AMDGPUAsmParser::parseFORMAT(OperandVector &Operands) {
8148 using namespace llvm::AMDGPU::MTBUFFormat;
8149
8150 int64_t Format = getDefaultFormatEncoding(getSTI());
8151 ParseStatus Res;
8152 SMLoc Loc = getLoc();
8153
8154 // Parse legacy format syntax.
8155 Res = isGFX10Plus() ? parseUfmt(Format) : parseDfmtNfmt(Format);
8156 if (Res.isFailure())
8157 return Res;
8158
8159 bool FormatFound = Res.isSuccess();
8160
8161 Operands.push_back(
8162 AMDGPUOperand::CreateImm(this, Format, Loc, AMDGPUOperand::ImmTyFORMAT));
8163
8164 if (FormatFound)
8165 trySkipToken(AsmToken::Comma);
8166
8167 if (isToken(AsmToken::EndOfStatement)) {
8168 // We are expecting an soffset operand,
8169 // but let matcher handle the error.
8170 return ParseStatus::Success;
8171 }
8172
8173 // Parse soffset.
8174 Res = parseRegOrImm(Operands);
8175 if (!Res.isSuccess())
8176 return Res;
8177
8178 trySkipToken(AsmToken::Comma);
8179
8180 if (!FormatFound) {
8181 Res = parseSymbolicOrNumericFormat(Format);
8182 if (Res.isFailure())
8183 return Res;
8184 if (Res.isSuccess()) {
8185 auto Size = Operands.size();
8186 AMDGPUOperand &Op = static_cast<AMDGPUOperand &>(*Operands[Size - 2]);
8187 assert(Op.isImm() && Op.getImmTy() == AMDGPUOperand::ImmTyFORMAT);
8188 Op.setImm(Format);
8189 }
8190 return ParseStatus::Success;
8191 }
8192
8193 if (isId("format") && peekToken().is(AsmToken::Colon))
8194 return Error(getLoc(), "duplicate format");
8195 return ParseStatus::Success;
8196}
8197
8198ParseStatus AMDGPUAsmParser::parseFlatOffset(OperandVector &Operands) {
8199 ParseStatus Res =
8200 parseIntWithPrefix("offset", Operands, AMDGPUOperand::ImmTyOffset);
8201 if (Res.isNoMatch()) {
8202 Res = parseIntWithPrefix("inst_offset", Operands,
8203 AMDGPUOperand::ImmTyInstOffset);
8204 }
8205 return Res;
8206}
8207
8208ParseStatus AMDGPUAsmParser::parseR128A16(OperandVector &Operands) {
8209 ParseStatus Res =
8210 parseNamedBit("r128", Operands, AMDGPUOperand::ImmTyR128A16);
8211 if (Res.isNoMatch())
8212 Res = parseNamedBit("a16", Operands, AMDGPUOperand::ImmTyA16);
8213 return Res;
8214}
8215
8216ParseStatus AMDGPUAsmParser::parseBLGP(OperandVector &Operands) {
8217 ParseStatus Res =
8218 parseIntWithPrefix("blgp", Operands, AMDGPUOperand::ImmTyBLGP);
8219 if (Res.isNoMatch()) {
8220 Res =
8221 parseOperandArrayWithPrefix("neg", Operands, AMDGPUOperand::ImmTyBLGP);
8222 }
8223 return Res;
8224}
8225
8226//===----------------------------------------------------------------------===//
8227// Exp
8228//===----------------------------------------------------------------------===//
8229
8230void AMDGPUAsmParser::cvtExp(MCInst &Inst, const OperandVector &Operands) {
8231 OptionalImmIndexMap OptionalIdx;
8232
8233 unsigned OperandIdx[4];
8234 unsigned EnMask = 0;
8235 int SrcIdx = 0;
8236
8237 for (unsigned i = 1, e = Operands.size(); i != e; ++i) {
8238 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
8239
8240 // Add the register arguments
8241 if (Op.isReg()) {
8242 assert(SrcIdx < 4);
8243 OperandIdx[SrcIdx] = Inst.size();
8244 Op.addRegOperands(Inst, 1);
8245 ++SrcIdx;
8246 continue;
8247 }
8248
8249 if (Op.isOff()) {
8250 assert(SrcIdx < 4);
8251 OperandIdx[SrcIdx] = Inst.size();
8252 Inst.addOperand(MCOperand::createReg(MCRegister()));
8253 ++SrcIdx;
8254 continue;
8255 }
8256
8257 if (Op.isImm() && Op.getImmTy() == AMDGPUOperand::ImmTyExpTgt) {
8258 Op.addImmOperands(Inst, 1);
8259 continue;
8260 }
8261
8262 if (Op.isToken() && (Op.getToken() == "done" || Op.getToken() == "row_en"))
8263 continue;
8264
8265 // Handle optional arguments
8266 OptionalIdx[Op.getImmTy()] = i;
8267 }
8268
8269 assert(SrcIdx == 4);
8270
8271 bool Compr = false;
8272 if (OptionalIdx.find(AMDGPUOperand::ImmTyExpCompr) != OptionalIdx.end()) {
8273 Compr = true;
8274 Inst.getOperand(OperandIdx[1]) = Inst.getOperand(OperandIdx[2]);
8275 Inst.getOperand(OperandIdx[2]).setReg(MCRegister());
8276 Inst.getOperand(OperandIdx[3]).setReg(MCRegister());
8277 }
8278
8279 for (auto i = 0; i < SrcIdx; ++i) {
8280 if (Inst.getOperand(OperandIdx[i]).getReg()) {
8281 EnMask |= Compr ? (0x3 << i * 2) : (0x1 << i);
8282 }
8283 }
8284
8285 addOptionalImmOperand(Inst, Operands, OptionalIdx, AMDGPUOperand::ImmTyExpVM);
8286 addOptionalImmOperand(Inst, Operands, OptionalIdx,
8287 AMDGPUOperand::ImmTyExpCompr);
8288
8289 Inst.addOperand(MCOperand::createImm(EnMask));
8290}
8291
8292//===----------------------------------------------------------------------===//
8293// s_waitcnt
8294//===----------------------------------------------------------------------===//
8295
8296static bool encodeCnt(const AMDGPU::IsaVersion ISA, int64_t &IntVal,
8297 int64_t CntVal, bool Saturate,
8298 unsigned (*encode)(const IsaVersion &Version, unsigned,
8299 unsigned),
8300 unsigned (*decode)(const IsaVersion &Version, unsigned)) {
8301 bool Failed = false;
8302
8303 IntVal = encode(ISA, IntVal, CntVal);
8304 if (CntVal != decode(ISA, IntVal)) {
8305 if (Saturate) {
8306 IntVal = encode(ISA, IntVal, -1);
8307 } else {
8308 Failed = true;
8309 }
8310 }
8311 return Failed;
8312}
8313
8314bool AMDGPUAsmParser::parseCnt(int64_t &IntVal) {
8315
8316 SMLoc CntLoc = getLoc();
8317 StringRef CntName = getTokenStr();
8318
8319 if (!skipToken(AsmToken::Identifier, "expected a counter name") ||
8320 !skipToken(AsmToken::LParen, "expected a left parenthesis"))
8321 return false;
8322
8323 int64_t CntVal;
8324 SMLoc ValLoc = getLoc();
8325 if (!parseExpr(CntVal))
8326 return false;
8327
8328 bool Failed = true;
8329 bool Sat = CntName.ends_with("_sat");
8330
8331 if (CntName == "vmcnt" || CntName == "vmcnt_sat") {
8332 Failed = encodeCnt(ISA, IntVal, CntVal, Sat, encodeVmcnt, decodeVmcnt);
8333 } else if (CntName == "expcnt" || CntName == "expcnt_sat") {
8334 Failed = encodeCnt(ISA, IntVal, CntVal, Sat, encodeExpcnt, decodeExpcnt);
8335 } else if (CntName == "lgkmcnt" || CntName == "lgkmcnt_sat") {
8336 Failed = encodeCnt(ISA, IntVal, CntVal, Sat, encodeLgkmcnt, decodeLgkmcnt);
8337 } else {
8338 Error(CntLoc, "invalid counter name " + CntName);
8339 return false;
8340 }
8341
8342 if (Failed) {
8343 Error(ValLoc, "too large value for " + CntName);
8344 return false;
8345 }
8346
8347 if (!skipToken(AsmToken::RParen, "expected a closing parenthesis"))
8348 return false;
8349
8350 if (trySkipToken(AsmToken::Amp) || trySkipToken(AsmToken::Comma)) {
8351 if (isToken(AsmToken::EndOfStatement)) {
8352 Error(getLoc(), "expected a counter name");
8353 return false;
8354 }
8355 }
8356
8357 return true;
8358}
8359
8360ParseStatus AMDGPUAsmParser::parseSWaitCnt(OperandVector &Operands) {
8361 int64_t Waitcnt = getWaitcntBitMask(ISA);
8362 SMLoc S = getLoc();
8363
8364 if (isToken(AsmToken::Identifier) && peekToken().is(AsmToken::LParen)) {
8365 while (!isToken(AsmToken::EndOfStatement)) {
8366 if (!parseCnt(Waitcnt))
8367 return ParseStatus::Failure;
8368 }
8369 } else {
8370 if (!parseExpr(Waitcnt))
8371 return ParseStatus::Failure;
8372 }
8373
8374 Operands.push_back(AMDGPUOperand::CreateImm(this, Waitcnt, S));
8375 return ParseStatus::Success;
8376}
8377
8378bool AMDGPUAsmParser::parseDelay(int64_t &Delay) {
8379 SMLoc FieldLoc = getLoc();
8380 StringRef FieldName = getTokenStr();
8381 if (!skipToken(AsmToken::Identifier, "expected a field name") ||
8382 !skipToken(AsmToken::LParen, "expected a left parenthesis"))
8383 return false;
8384
8385 SMLoc ValueLoc = getLoc();
8386 StringRef ValueName = getTokenStr();
8387 if (!skipToken(AsmToken::Identifier, "expected a value name") ||
8388 !skipToken(AsmToken::RParen, "expected a right parenthesis"))
8389 return false;
8390
8391 unsigned Shift;
8392 if (FieldName == "instid0") {
8393 Shift = 0;
8394 } else if (FieldName == "instskip") {
8395 Shift = 4;
8396 } else if (FieldName == "instid1") {
8397 Shift = 7;
8398 } else {
8399 Error(FieldLoc, "invalid field name " + FieldName);
8400 return false;
8401 }
8402
8403 int Value;
8404 if (Shift == 4) {
8405 // Parse values for instskip.
8406 Value = StringSwitch<int>(ValueName)
8407 .Case("SAME", 0)
8408 .Case("NEXT", 1)
8409 .Case("SKIP_1", 2)
8410 .Case("SKIP_2", 3)
8411 .Case("SKIP_3", 4)
8412 .Case("SKIP_4", 5)
8413 .Default(-1);
8414 } else {
8415 // Parse values for instid0 and instid1.
8416 Value = StringSwitch<int>(ValueName)
8417 .Case("NO_DEP", 0)
8418 .Case("VALU_DEP_1", 1)
8419 .Case("VALU_DEP_2", 2)
8420 .Case("VALU_DEP_3", 3)
8421 .Case("VALU_DEP_4", 4)
8422 .Case("TRANS32_DEP_1", 5)
8423 .Case("TRANS32_DEP_2", 6)
8424 .Case("TRANS32_DEP_3", 7)
8425 .Case("FMA_ACCUM_CYCLE_1", 8)
8426 .Case("SALU_CYCLE_1", 9)
8427 .Case("SALU_CYCLE_2", 10)
8428 .Case("SALU_CYCLE_3", 11)
8429 .Default(-1);
8430 }
8431 if (Value < 0) {
8432 Error(ValueLoc, "invalid value name " + ValueName);
8433 return false;
8434 }
8435
8436 Delay |= Value << Shift;
8437 return true;
8438}
8439
8440ParseStatus AMDGPUAsmParser::parseSDelayALU(OperandVector &Operands) {
8441 int64_t Delay = 0;
8442 SMLoc S = getLoc();
8443
8444 if (isToken(AsmToken::Identifier) && peekToken().is(AsmToken::LParen)) {
8445 do {
8446 if (!parseDelay(Delay))
8447 return ParseStatus::Failure;
8448 } while (trySkipToken(AsmToken::Pipe));
8449 } else {
8450 if (!parseExpr(Delay))
8451 return ParseStatus::Failure;
8452 }
8453
8454 Operands.push_back(AMDGPUOperand::CreateImm(this, Delay, S));
8455 return ParseStatus::Success;
8456}
8457
8458bool AMDGPUOperand::isSWaitCnt() const { return isImm(); }
8459
8460bool AMDGPUOperand::isSDelayALU() const { return isImm(); }
8461
8462//===----------------------------------------------------------------------===//
8463// DepCtr
8464//===----------------------------------------------------------------------===//
8465
8466void AMDGPUAsmParser::depCtrError(SMLoc Loc, int ErrorId,
8467 StringRef DepCtrName) {
8468 switch (ErrorId) {
8469 case OPR_ID_UNKNOWN:
8470 Error(Loc, Twine("invalid counter name ", DepCtrName));
8471 return;
8472 case OPR_ID_UNSUPPORTED:
8473 Error(Loc, Twine(DepCtrName, " is not supported on this GPU"));
8474 return;
8475 case OPR_ID_DUPLICATE:
8476 Error(Loc, Twine("duplicate counter name ", DepCtrName));
8477 return;
8478 case OPR_VAL_INVALID:
8479 Error(Loc, Twine("invalid value for ", DepCtrName));
8480 return;
8481 default:
8482 assert(false);
8483 }
8484}
8485
8486bool AMDGPUAsmParser::parseDepCtr(int64_t &DepCtr, unsigned &UsedOprMask) {
8487
8488 using namespace llvm::AMDGPU::DepCtr;
8489
8490 SMLoc DepCtrLoc = getLoc();
8491 StringRef DepCtrName = getTokenStr();
8492
8493 if (!skipToken(AsmToken::Identifier, "expected a counter name") ||
8494 !skipToken(AsmToken::LParen, "expected a left parenthesis"))
8495 return false;
8496
8497 int64_t ExprVal;
8498 if (!parseExpr(ExprVal))
8499 return false;
8500
8501 unsigned PrevOprMask = UsedOprMask;
8502 int CntVal = encodeDepCtr(DepCtrName, ExprVal, UsedOprMask, getSTI());
8503
8504 if (CntVal < 0) {
8505 depCtrError(DepCtrLoc, CntVal, DepCtrName);
8506 return false;
8507 }
8508
8509 if (!skipToken(AsmToken::RParen, "expected a closing parenthesis"))
8510 return false;
8511
8512 if (trySkipToken(AsmToken::Amp) || trySkipToken(AsmToken::Comma)) {
8513 if (isToken(AsmToken::EndOfStatement)) {
8514 Error(getLoc(), "expected a counter name");
8515 return false;
8516 }
8517 }
8518
8519 int64_t CntValMask = PrevOprMask ^ UsedOprMask;
8520 DepCtr = (DepCtr & ~CntValMask) | CntVal;
8521 return true;
8522}
8523
8524ParseStatus AMDGPUAsmParser::parseDepCtr(OperandVector &Operands) {
8525 using namespace llvm::AMDGPU::DepCtr;
8526
8527 int64_t DepCtr = getDefaultDepCtrEncoding(getSTI());
8528 SMLoc Loc = getLoc();
8529
8530 if (isToken(AsmToken::Identifier) && peekToken().is(AsmToken::LParen)) {
8531 unsigned UsedOprMask = 0;
8532 while (!isToken(AsmToken::EndOfStatement)) {
8533 if (!parseDepCtr(DepCtr, UsedOprMask))
8534 return ParseStatus::Failure;
8535 }
8536 } else {
8537 if (!parseExpr(DepCtr))
8538 return ParseStatus::Failure;
8539 }
8540
8541 Operands.push_back(AMDGPUOperand::CreateImm(this, DepCtr, Loc));
8542 return ParseStatus::Success;
8543}
8544
8545bool AMDGPUOperand::isDepCtr() const { return isS16Imm(); }
8546
8547//===----------------------------------------------------------------------===//
8548// hwreg
8549//===----------------------------------------------------------------------===//
8550
8551ParseStatus AMDGPUAsmParser::parseHwregFunc(OperandInfoTy &HwReg,
8552 OperandInfoTy &Offset,
8553 OperandInfoTy &Width) {
8554 using namespace llvm::AMDGPU::Hwreg;
8555
8556 if (!trySkipId("hwreg", AsmToken::LParen))
8557 return ParseStatus::NoMatch;
8558
8559 // The register may be specified by name or using a numeric code
8560 HwReg.Loc = getLoc();
8561 if (isToken(AsmToken::Identifier) &&
8562 (HwReg.Val = getHwregId(getTokenStr(), getSTI())) != OPR_ID_UNKNOWN) {
8563 HwReg.IsSymbolic = true;
8564 lex(); // skip register name
8565 } else if (!parseExpr(HwReg.Val, "a register name")) {
8566 return ParseStatus::Failure;
8567 }
8568
8569 if (trySkipToken(AsmToken::RParen))
8570 return ParseStatus::Success;
8571
8572 // parse optional params
8573 if (!skipToken(AsmToken::Comma, "expected a comma or a closing parenthesis"))
8574 return ParseStatus::Failure;
8575
8576 Offset.Loc = getLoc();
8577 if (!parseExpr(Offset.Val))
8578 return ParseStatus::Failure;
8579
8580 if (!skipToken(AsmToken::Comma, "expected a comma"))
8581 return ParseStatus::Failure;
8582
8583 Width.Loc = getLoc();
8584 if (!parseExpr(Width.Val) ||
8585 !skipToken(AsmToken::RParen, "expected a closing parenthesis"))
8586 return ParseStatus::Failure;
8587
8588 return ParseStatus::Success;
8589}
8590
8591ParseStatus AMDGPUAsmParser::parseHwreg(OperandVector &Operands) {
8592 using namespace llvm::AMDGPU::Hwreg;
8593
8594 int64_t ImmVal = 0;
8595 SMLoc Loc = getLoc();
8596
8597 StructuredOpField HwReg("id", "hardware register", HwregId::Width,
8598 HwregId::Default);
8599 StructuredOpField Offset("offset", "bit offset", HwregOffset::Width,
8600 HwregOffset::Default);
8601 struct : StructuredOpField {
8602 using StructuredOpField::StructuredOpField;
8603 bool validate(AMDGPUAsmParser &Parser) const override {
8604 if (!isUIntN(Width, Val - 1))
8605 return Error(Parser, "only values from 1 to 32 are legal");
8606 return true;
8607 }
8608 } Width("size", "bitfield width", HwregSize::Width, HwregSize::Default);
8609 ParseStatus Res = parseStructuredOpFields({&HwReg, &Offset, &Width});
8610
8611 if (Res.isNoMatch())
8612 Res = parseHwregFunc(HwReg, Offset, Width);
8613
8614 if (Res.isSuccess()) {
8615 if (!validateStructuredOpFields({&HwReg, &Offset, &Width}))
8616 return ParseStatus::Failure;
8617 ImmVal = HwregEncoding::encode(HwReg.Val, Offset.Val, Width.Val);
8618 }
8619
8620 if (Res.isNoMatch() &&
8621 parseExpr(ImmVal, "a hwreg macro, structured immediate"))
8623
8624 if (!Res.isSuccess())
8625 return ParseStatus::Failure;
8626
8627 if (!isUInt<16>(ImmVal))
8628 return Error(Loc, "invalid immediate: only 16-bit values are legal");
8629 Operands.push_back(
8630 AMDGPUOperand::CreateImm(this, ImmVal, Loc, AMDGPUOperand::ImmTyHwreg));
8631 return ParseStatus::Success;
8632}
8633
8634bool AMDGPUOperand::isHwreg() const { return isImmTy(ImmTyHwreg); }
8635
8636//===----------------------------------------------------------------------===//
8637// sendmsg
8638//===----------------------------------------------------------------------===//
8639
8640bool AMDGPUAsmParser::parseSendMsgBody(OperandInfoTy &Msg, OperandInfoTy &Op,
8641 OperandInfoTy &Stream) {
8642 using namespace llvm::AMDGPU::SendMsg;
8643
8644 Msg.Loc = getLoc();
8645 if (isToken(AsmToken::Identifier) &&
8646 (Msg.Val = getMsgId(getTokenStr(), getSTI())) != OPR_ID_UNKNOWN) {
8647 Msg.IsSymbolic = true;
8648 lex(); // skip message name
8649 } else if (!parseExpr(Msg.Val, "a message name")) {
8650 return false;
8651 }
8652
8653 if (trySkipToken(AsmToken::Comma)) {
8654 Op.IsDefined = true;
8655 Op.Loc = getLoc();
8656 if (isToken(AsmToken::Identifier) &&
8657 (Op.Val = getMsgOpId(Msg.Val, getTokenStr(), getSTI())) !=
8659 lex(); // skip operation name
8660 } else if (!parseExpr(Op.Val, "an operation name")) {
8661 return false;
8662 }
8663
8664 if (trySkipToken(AsmToken::Comma)) {
8665 Stream.IsDefined = true;
8666 Stream.Loc = getLoc();
8667 if (!parseExpr(Stream.Val))
8668 return false;
8669 }
8670 }
8671
8672 return skipToken(AsmToken::RParen, "expected a closing parenthesis");
8673}
8674
8675bool AMDGPUAsmParser::validateSendMsg(const OperandInfoTy &Msg,
8676 const OperandInfoTy &Op,
8677 const OperandInfoTy &Stream) {
8678 using namespace llvm::AMDGPU::SendMsg;
8679
8680 // Validation strictness depends on whether message is specified
8681 // in a symbolic or in a numeric form. In the latter case
8682 // only encoding possibility is checked.
8683 bool Strict = Msg.IsSymbolic;
8684
8685 if (Strict) {
8686 if (Msg.Val == OPR_ID_UNSUPPORTED) {
8687 Error(Msg.Loc, "specified message id is not supported on this GPU");
8688 return false;
8689 }
8690 } else {
8691 if (!isValidMsgId(Msg.Val, getSTI())) {
8692 Error(Msg.Loc, "invalid message id");
8693 return false;
8694 }
8695 }
8696 if (Strict && (msgRequiresOp(Msg.Val, getSTI()) != Op.IsDefined)) {
8697 if (Op.IsDefined) {
8698 Error(Op.Loc, "message does not support operations");
8699 } else {
8700 Error(Msg.Loc, "missing message operation");
8701 }
8702 return false;
8703 }
8704 if (!isValidMsgOp(Msg.Val, Op.Val, getSTI(), Strict)) {
8705 if (Op.Val == OPR_ID_UNSUPPORTED)
8706 Error(Op.Loc, "specified operation id is not supported on this GPU");
8707 else
8708 Error(Op.Loc, "invalid operation id");
8709 return false;
8710 }
8711 if (Strict && !msgSupportsStream(Msg.Val, Op.Val, getSTI()) &&
8712 Stream.IsDefined) {
8713 Error(Stream.Loc, "message operation does not support streams");
8714 return false;
8715 }
8716 if (!isValidMsgStream(Msg.Val, Op.Val, Stream.Val, getSTI(), Strict)) {
8717 Error(Stream.Loc, "invalid message stream id");
8718 return false;
8719 }
8720 return true;
8721}
8722
8723ParseStatus AMDGPUAsmParser::parseSendMsg(OperandVector &Operands) {
8724 using namespace llvm::AMDGPU::SendMsg;
8725
8726 int64_t ImmVal = 0;
8727 SMLoc Loc = getLoc();
8728
8729 if (trySkipId("sendmsg", AsmToken::LParen)) {
8730 OperandInfoTy Msg(OPR_ID_UNKNOWN);
8731 OperandInfoTy Op(OP_NONE_);
8732 OperandInfoTy Stream(STREAM_ID_NONE_);
8733 if (parseSendMsgBody(Msg, Op, Stream) && validateSendMsg(Msg, Op, Stream)) {
8734 ImmVal = encodeMsg(Msg.Val, Op.Val, Stream.Val);
8735 } else {
8736 return ParseStatus::Failure;
8737 }
8738 } else if (parseExpr(ImmVal, "a sendmsg macro")) {
8739 if (ImmVal < 0 || !isUInt<16>(ImmVal))
8740 return Error(Loc, "invalid immediate: only 16-bit values are legal");
8741 } else {
8742 return ParseStatus::Failure;
8743 }
8744
8745 Operands.push_back(
8746 AMDGPUOperand::CreateImm(this, ImmVal, Loc, AMDGPUOperand::ImmTySendMsg));
8747 return ParseStatus::Success;
8748}
8749
8750bool AMDGPUOperand::isSendMsg() const { return isImmTy(ImmTySendMsg); }
8751
8752ParseStatus AMDGPUAsmParser::parseWaitEvent(OperandVector &Operands) {
8753 using namespace llvm::AMDGPU::WaitEvent;
8754
8755 SMLoc Loc = getLoc();
8756 int64_t ImmVal = 0;
8757
8758 StructuredOpField DontWaitExportReady("dont_wait_export_ready", "bit value",
8759 1, 0);
8760 StructuredOpField ExportReady("export_ready", "bit value", 1, 0);
8761
8762 StructuredOpField *TargetBitfield =
8763 isGFX11() ? &DontWaitExportReady : &ExportReady;
8764
8765 ParseStatus Res = parseStructuredOpFields({TargetBitfield});
8766 if (Res.isNoMatch() && parseExpr(ImmVal, "structured immediate"))
8768 else if (Res.isSuccess()) {
8769 if (!validateStructuredOpFields({TargetBitfield}))
8770 return ParseStatus::Failure;
8771 ImmVal = TargetBitfield->Val;
8772 }
8773
8774 if (!Res.isSuccess())
8775 return ParseStatus::Failure;
8776
8777 if (!isUInt<16>(ImmVal))
8778 return Error(Loc, "invalid immediate: only 16-bit values are legal");
8779
8780 Operands.push_back(AMDGPUOperand::CreateImm(this, ImmVal, Loc,
8781 AMDGPUOperand::ImmTyWaitEvent));
8782 return ParseStatus::Success;
8783}
8784
8785bool AMDGPUOperand::isWaitEvent() const { return isImmTy(ImmTyWaitEvent); }
8786
8787//===----------------------------------------------------------------------===//
8788// v_interp
8789//===----------------------------------------------------------------------===//
8790
8791ParseStatus AMDGPUAsmParser::parseInterpSlot(OperandVector &Operands) {
8792 StringRef Str;
8793 SMLoc S = getLoc();
8794
8795 if (!parseId(Str))
8796 return ParseStatus::NoMatch;
8797
8798 int Slot = StringSwitch<int>(Str)
8799 .Case("p10", 0)
8800 .Case("p20", 1)
8801 .Case("p0", 2)
8802 .Default(-1);
8803
8804 if (Slot == -1)
8805 return Error(S, "invalid interpolation slot");
8806
8807 Operands.push_back(
8808 AMDGPUOperand::CreateImm(this, Slot, S, AMDGPUOperand::ImmTyInterpSlot));
8809 return ParseStatus::Success;
8810}
8811
8812ParseStatus AMDGPUAsmParser::parseInterpAttr(OperandVector &Operands) {
8813 StringRef Str;
8814 SMLoc S = getLoc();
8815
8816 if (!parseId(Str))
8817 return ParseStatus::NoMatch;
8818
8819 if (!Str.starts_with("attr"))
8820 return Error(S, "invalid interpolation attribute");
8821
8822 StringRef Chan = Str.take_back(2);
8823 int AttrChan = StringSwitch<int>(Chan)
8824 .Case(".x", 0)
8825 .Case(".y", 1)
8826 .Case(".z", 2)
8827 .Case(".w", 3)
8828 .Default(-1);
8829 if (AttrChan == -1)
8830 return Error(S, "invalid or missing interpolation attribute channel");
8831
8832 Str = Str.drop_back(2).drop_front(4);
8833
8834 uint8_t Attr;
8835 if (Str.getAsInteger(10, Attr))
8836 return Error(S, "invalid or missing interpolation attribute number");
8837
8838 if (Attr > 32)
8839 return Error(S, "out of bounds interpolation attribute number");
8840
8841 SMLoc SChan = SMLoc::getFromPointer(Chan.data());
8842
8843 Operands.push_back(
8844 AMDGPUOperand::CreateImm(this, Attr, S, AMDGPUOperand::ImmTyInterpAttr));
8845 Operands.push_back(AMDGPUOperand::CreateImm(
8846 this, AttrChan, SChan, AMDGPUOperand::ImmTyInterpAttrChan));
8847 return ParseStatus::Success;
8848}
8849
8850//===----------------------------------------------------------------------===//
8851// exp
8852//===----------------------------------------------------------------------===//
8853
8854ParseStatus AMDGPUAsmParser::parseExpTgt(OperandVector &Operands) {
8855 using namespace llvm::AMDGPU::Exp;
8856
8857 StringRef Str;
8858 SMLoc S = getLoc();
8859
8860 if (!parseId(Str))
8861 return ParseStatus::NoMatch;
8862
8863 unsigned Id = getTgtId(Str);
8864 if (Id == ET_INVALID || !isSupportedTgtId(Id, getSTI()))
8865 return Error(S, (Id == ET_INVALID)
8866 ? "invalid exp target"
8867 : "exp target is not supported on this GPU");
8868
8869 Operands.push_back(
8870 AMDGPUOperand::CreateImm(this, Id, S, AMDGPUOperand::ImmTyExpTgt));
8871 return ParseStatus::Success;
8872}
8873
8874//===----------------------------------------------------------------------===//
8875// parser helpers
8876//===----------------------------------------------------------------------===//
8877
8878bool AMDGPUAsmParser::isId(const AsmToken &Token, const StringRef Id) const {
8879 return Token.is(AsmToken::Identifier) && Token.getString() == Id;
8880}
8881
8882bool AMDGPUAsmParser::isId(const StringRef Id) const {
8883 return isId(getToken(), Id);
8884}
8885
8886bool AMDGPUAsmParser::isToken(const AsmToken::TokenKind Kind) const {
8887 return getTokenKind() == Kind;
8888}
8889
8890StringRef AMDGPUAsmParser::getId() const {
8891 return isToken(AsmToken::Identifier) ? getTokenStr() : StringRef();
8892}
8893
8894bool AMDGPUAsmParser::trySkipId(const StringRef Id) {
8895 if (isId(Id)) {
8896 lex();
8897 return true;
8898 }
8899 return false;
8900}
8901
8902bool AMDGPUAsmParser::trySkipId(const StringRef Pref, const StringRef Id) {
8903 if (isToken(AsmToken::Identifier)) {
8904 StringRef Tok = getTokenStr();
8905 if (Tok.starts_with(Pref) && Tok.drop_front(Pref.size()) == Id) {
8906 lex();
8907 return true;
8908 }
8909 }
8910 return false;
8911}
8912
8913bool AMDGPUAsmParser::trySkipId(const StringRef Id,
8914 const AsmToken::TokenKind Kind) {
8915 if (isId(Id) && peekToken().is(Kind)) {
8916 lex();
8917 lex();
8918 return true;
8919 }
8920 return false;
8921}
8922
8923bool AMDGPUAsmParser::trySkipToken(const AsmToken::TokenKind Kind) {
8924 if (isToken(Kind)) {
8925 lex();
8926 return true;
8927 }
8928 return false;
8929}
8930
8931bool AMDGPUAsmParser::skipToken(const AsmToken::TokenKind Kind,
8932 const StringRef ErrMsg) {
8933 if (!trySkipToken(Kind)) {
8934 Error(getLoc(), ErrMsg);
8935 return false;
8936 }
8937 return true;
8938}
8939
8940bool AMDGPUAsmParser::parseExpr(int64_t &Imm, StringRef Expected) {
8941 SMLoc S = getLoc();
8942
8943 const MCExpr *Expr;
8944 if (Parser.parseExpression(Expr))
8945 return false;
8946
8947 if (Expr->evaluateAsAbsolute(Imm))
8948 return true;
8949
8950 if (Expected.empty()) {
8951 Error(S, "expected absolute expression");
8952 } else {
8953 Error(S,
8954 Twine("expected ", Expected) + Twine(" or an absolute expression"));
8955 }
8956 return false;
8957}
8958
8959bool AMDGPUAsmParser::parseExpr(OperandVector &Operands) {
8960 SMLoc S = getLoc();
8961
8962 const MCExpr *Expr;
8963 if (Parser.parseExpression(Expr))
8964 return false;
8965
8966 int64_t IntVal;
8967 if (Expr->evaluateAsAbsolute(IntVal)) {
8968 Operands.push_back(AMDGPUOperand::CreateImm(this, IntVal, S));
8969 } else {
8970 Operands.push_back(AMDGPUOperand::CreateExpr(this, Expr, S));
8971 }
8972 return true;
8973}
8974
8975bool AMDGPUAsmParser::parseString(StringRef &Val, const StringRef ErrMsg) {
8976 if (isToken(AsmToken::String)) {
8977 Val = getToken().getStringContents();
8978 lex();
8979 return true;
8980 }
8981 Error(getLoc(), ErrMsg);
8982 return false;
8983}
8984
8985bool AMDGPUAsmParser::parseId(StringRef &Val, const StringRef ErrMsg) {
8986 if (isToken(AsmToken::Identifier)) {
8987 Val = getTokenStr();
8988 lex();
8989 return true;
8990 }
8991 if (!ErrMsg.empty())
8992 Error(getLoc(), ErrMsg);
8993 return false;
8994}
8995
8996AsmToken AMDGPUAsmParser::getToken() const { return Parser.getTok(); }
8997
8998AsmToken AMDGPUAsmParser::peekToken(bool ShouldSkipSpace) {
8999 return isToken(AsmToken::EndOfStatement)
9000 ? getToken()
9001 : getLexer().peekTok(ShouldSkipSpace);
9002}
9003
9004void AMDGPUAsmParser::peekTokens(MutableArrayRef<AsmToken> Tokens) {
9005 auto TokCount = getLexer().peekTokens(Tokens);
9006
9007 for (auto Idx = TokCount; Idx < Tokens.size(); ++Idx)
9008 Tokens[Idx] = AsmToken(AsmToken::Error, "");
9009}
9010
9011AsmToken::TokenKind AMDGPUAsmParser::getTokenKind() const {
9012 return getLexer().getKind();
9013}
9014
9015SMLoc AMDGPUAsmParser::getLoc() const { return getToken().getLoc(); }
9016
9017StringRef AMDGPUAsmParser::getTokenStr() const {
9018 return getToken().getString();
9019}
9020
9021void AMDGPUAsmParser::lex() { Parser.Lex(); }
9022
9023const AMDGPUOperand &
9024AMDGPUAsmParser::findMCOperand(const OperandVector &Operands,
9025 int MCOpIdx) const {
9026 for (const auto &Op : Operands) {
9027 const AMDGPUOperand &TargetOp = static_cast<AMDGPUOperand &>(*Op);
9028 if (TargetOp.getMCOpIdx() == MCOpIdx)
9029 return TargetOp;
9030 }
9031 llvm_unreachable("no such MC operand!");
9032}
9033
9034SMLoc AMDGPUAsmParser::getInstLoc(const OperandVector &Operands) const {
9035 return ((AMDGPUOperand &)*Operands[0]).getStartLoc();
9036}
9037
9038// Returns one of the given locations that comes later in the source.
9039SMLoc AMDGPUAsmParser::getLaterLoc(SMLoc a, SMLoc b) {
9040 return a.getPointer() < b.getPointer() ? b : a;
9041}
9042
9043SMLoc AMDGPUAsmParser::getOperandLoc(const OperandVector &Operands,
9044 int MCOpIdx) const {
9045 return findMCOperand(Operands, MCOpIdx).getStartLoc();
9046}
9047
9048SMLoc AMDGPUAsmParser::getOperandLoc(
9049 std::function<bool(const AMDGPUOperand &)> Test,
9050 const OperandVector &Operands) const {
9051 for (unsigned i = Operands.size() - 1; i > 0; --i) {
9052 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
9053 if (Test(Op))
9054 return Op.getStartLoc();
9055 }
9056 return getInstLoc(Operands);
9057}
9058
9059SMLoc AMDGPUAsmParser::getImmLoc(AMDGPUOperand::ImmTy Type,
9060 const OperandVector &Operands) const {
9061 auto Test = [=](const AMDGPUOperand &Op) { return Op.isImmTy(Type); };
9062 return getOperandLoc(Test, Operands);
9063}
9064
9065ParseStatus
9066AMDGPUAsmParser::parseStructuredOpFields(ArrayRef<StructuredOpField *> Fields) {
9067 if (!trySkipToken(AsmToken::LCurly))
9068 return ParseStatus::NoMatch;
9069
9070 bool First = true;
9071 while (!trySkipToken(AsmToken::RCurly)) {
9072 if (!First &&
9073 !skipToken(AsmToken::Comma, "comma or closing brace expected"))
9074 return ParseStatus::Failure;
9075
9076 StringRef Id = getTokenStr();
9077 SMLoc IdLoc = getLoc();
9078 if (!skipToken(AsmToken::Identifier, "field name expected") ||
9079 !skipToken(AsmToken::Colon, "colon expected"))
9080 return ParseStatus::Failure;
9081
9082 const auto *I =
9083 find_if(Fields, [Id](StructuredOpField *F) { return F->Id == Id; });
9084 if (I == Fields.end())
9085 return Error(IdLoc, "unknown field");
9086 if ((*I)->IsDefined)
9087 return Error(IdLoc, "duplicate field");
9088
9089 // TODO: Support symbolic values.
9090 (*I)->Loc = getLoc();
9091 if (!parseExpr((*I)->Val))
9092 return ParseStatus::Failure;
9093 (*I)->IsDefined = true;
9094
9095 First = false;
9096 }
9097 return ParseStatus::Success;
9098}
9099
9100bool AMDGPUAsmParser::validateStructuredOpFields(
9102 return all_of(Fields, [this](const StructuredOpField *F) {
9103 return F->validate(*this);
9104 });
9105}
9106
9107//===----------------------------------------------------------------------===//
9108// swizzle
9109//===----------------------------------------------------------------------===//
9110
9112static unsigned encodeBitmaskPerm(const unsigned AndMask, const unsigned OrMask,
9113 const unsigned XorMask) {
9114 using namespace llvm::AMDGPU::Swizzle;
9115
9116 return BITMASK_PERM_ENC | (AndMask << BITMASK_AND_SHIFT) |
9117 (OrMask << BITMASK_OR_SHIFT) | (XorMask << BITMASK_XOR_SHIFT);
9118}
9119
9120bool AMDGPUAsmParser::parseSwizzleOperand(int64_t &Op, const unsigned MinVal,
9121 const unsigned MaxVal,
9122 const Twine &ErrMsg, SMLoc &Loc) {
9123 if (!skipToken(AsmToken::Comma, "expected a comma")) {
9124 return false;
9125 }
9126 Loc = getLoc();
9127 if (!parseExpr(Op)) {
9128 return false;
9129 }
9130 if (Op < MinVal || Op > MaxVal) {
9131 Error(Loc, ErrMsg);
9132 return false;
9133 }
9134
9135 return true;
9136}
9137
9138bool AMDGPUAsmParser::parseSwizzleOperands(const unsigned OpNum, int64_t *Op,
9139 const unsigned MinVal,
9140 const unsigned MaxVal,
9141 const StringRef ErrMsg) {
9142 SMLoc Loc;
9143 for (unsigned i = 0; i < OpNum; ++i) {
9144 if (!parseSwizzleOperand(Op[i], MinVal, MaxVal, ErrMsg, Loc))
9145 return false;
9146 }
9147
9148 return true;
9149}
9150
9151bool AMDGPUAsmParser::parseSwizzleQuadPerm(int64_t &Imm) {
9152 using namespace llvm::AMDGPU::Swizzle;
9153
9154 int64_t Lane[LANE_NUM];
9155 if (parseSwizzleOperands(LANE_NUM, Lane, 0, LANE_MAX,
9156 "expected a 2-bit lane id")) {
9158 for (unsigned I = 0; I < LANE_NUM; ++I) {
9159 Imm |= Lane[I] << (LANE_SHIFT * I);
9160 }
9161 return true;
9162 }
9163 return false;
9164}
9165
9166bool AMDGPUAsmParser::parseSwizzleBroadcast(int64_t &Imm) {
9167 using namespace llvm::AMDGPU::Swizzle;
9168
9169 SMLoc Loc;
9170 int64_t GroupSize;
9171 int64_t LaneIdx;
9172
9173 if (!parseSwizzleOperand(GroupSize, 2, 32,
9174 "group size must be in the interval [2,32]", Loc)) {
9175 return false;
9176 }
9177 if (!isPowerOf2_64(GroupSize)) {
9178 Error(Loc, "group size must be a power of two");
9179 return false;
9180 }
9181 if (parseSwizzleOperand(LaneIdx, 0, GroupSize - 1,
9182 "lane id must be in the interval [0,group size - 1]",
9183 Loc)) {
9184 Imm = encodeBitmaskPerm(BITMASK_MAX - GroupSize + 1, LaneIdx, 0);
9185 return true;
9186 }
9187 return false;
9188}
9189
9190bool AMDGPUAsmParser::parseSwizzleReverse(int64_t &Imm) {
9191 using namespace llvm::AMDGPU::Swizzle;
9192
9193 SMLoc Loc;
9194 int64_t GroupSize;
9195
9196 if (!parseSwizzleOperand(GroupSize, 2, 32,
9197 "group size must be in the interval [2,32]", Loc)) {
9198 return false;
9199 }
9200 if (!isPowerOf2_64(GroupSize)) {
9201 Error(Loc, "group size must be a power of two");
9202 return false;
9203 }
9204
9205 Imm = encodeBitmaskPerm(BITMASK_MAX, 0, GroupSize - 1);
9206 return true;
9207}
9208
9209bool AMDGPUAsmParser::parseSwizzleSwap(int64_t &Imm) {
9210 using namespace llvm::AMDGPU::Swizzle;
9211
9212 SMLoc Loc;
9213 int64_t GroupSize;
9214
9215 if (!parseSwizzleOperand(GroupSize, 1, 16,
9216 "group size must be in the interval [1,16]", Loc)) {
9217 return false;
9218 }
9219 if (!isPowerOf2_64(GroupSize)) {
9220 Error(Loc, "group size must be a power of two");
9221 return false;
9222 }
9223
9224 Imm = encodeBitmaskPerm(BITMASK_MAX, 0, GroupSize);
9225 return true;
9226}
9227
9228bool AMDGPUAsmParser::parseSwizzleBitmaskPerm(int64_t &Imm) {
9229 using namespace llvm::AMDGPU::Swizzle;
9230
9231 if (!skipToken(AsmToken::Comma, "expected a comma")) {
9232 return false;
9233 }
9234
9235 StringRef Ctl;
9236 SMLoc StrLoc = getLoc();
9237 if (!parseString(Ctl)) {
9238 return false;
9239 }
9240 if (Ctl.size() != BITMASK_WIDTH) {
9241 Error(StrLoc, "expected a 5-character mask");
9242 return false;
9243 }
9244
9245 unsigned AndMask = 0;
9246 unsigned OrMask = 0;
9247 unsigned XorMask = 0;
9248
9249 for (size_t i = 0; i < Ctl.size(); ++i) {
9250 unsigned Mask = 1 << (BITMASK_WIDTH - 1 - i);
9251 switch (Ctl[i]) {
9252 default:
9253 Error(StrLoc, "invalid mask");
9254 return false;
9255 case '0':
9256 break;
9257 case '1':
9258 OrMask |= Mask;
9259 break;
9260 case 'p':
9261 AndMask |= Mask;
9262 break;
9263 case 'i':
9264 AndMask |= Mask;
9265 XorMask |= Mask;
9266 break;
9267 }
9268 }
9269
9270 Imm = encodeBitmaskPerm(AndMask, OrMask, XorMask);
9271 return true;
9272}
9273
9274bool AMDGPUAsmParser::parseSwizzleFFT(int64_t &Imm) {
9275 using namespace llvm::AMDGPU::Swizzle;
9276
9277 if (!AMDGPU::isGFX9Plus(getSTI())) {
9278 Error(getLoc(), "FFT mode swizzle not supported on this GPU");
9279 return false;
9280 }
9281
9282 int64_t Swizzle;
9283 SMLoc Loc;
9284 if (!parseSwizzleOperand(Swizzle, 0, FFT_SWIZZLE_MAX,
9285 "FFT swizzle must be in the interval [0," +
9286 Twine(FFT_SWIZZLE_MAX) + Twine(']'),
9287 Loc))
9288 return false;
9289
9290 Imm = FFT_MODE_ENC | Swizzle;
9291 return true;
9292}
9293
9294bool AMDGPUAsmParser::parseSwizzleRotate(int64_t &Imm) {
9295 using namespace llvm::AMDGPU::Swizzle;
9296
9297 if (!AMDGPU::isGFX9Plus(getSTI())) {
9298 Error(getLoc(), "Rotate mode swizzle not supported on this GPU");
9299 return false;
9300 }
9301
9302 SMLoc Loc;
9303 int64_t Direction;
9304
9305 if (!parseSwizzleOperand(Direction, 0, 1,
9306 "direction must be 0 (left) or 1 (right)", Loc))
9307 return false;
9308
9309 int64_t RotateSize;
9310 if (!parseSwizzleOperand(
9311 RotateSize, 0, ROTATE_MAX_SIZE,
9312 "number of threads to rotate must be in the interval [0," +
9313 Twine(ROTATE_MAX_SIZE) + Twine(']'),
9314 Loc))
9315 return false;
9316
9318 (RotateSize << ROTATE_SIZE_SHIFT);
9319 return true;
9320}
9321
9322bool AMDGPUAsmParser::parseSwizzleOffset(int64_t &Imm) {
9323
9324 SMLoc OffsetLoc = getLoc();
9325
9326 if (!parseExpr(Imm, "a swizzle macro")) {
9327 return false;
9328 }
9329 if (!isUInt<16>(Imm)) {
9330 Error(OffsetLoc, "expected a 16-bit offset");
9331 return false;
9332 }
9333 return true;
9334}
9335
9336bool AMDGPUAsmParser::parseSwizzleMacro(int64_t &Imm) {
9337 using namespace llvm::AMDGPU::Swizzle;
9338
9339 if (skipToken(AsmToken::LParen, "expected a left parentheses")) {
9340
9341 SMLoc ModeLoc = getLoc();
9342 bool Ok = false;
9343
9344 if (trySkipId(IdSymbolic[ID_QUAD_PERM])) {
9345 Ok = parseSwizzleQuadPerm(Imm);
9346 } else if (trySkipId(IdSymbolic[ID_BITMASK_PERM])) {
9347 Ok = parseSwizzleBitmaskPerm(Imm);
9348 } else if (trySkipId(IdSymbolic[ID_BROADCAST])) {
9349 Ok = parseSwizzleBroadcast(Imm);
9350 } else if (trySkipId(IdSymbolic[ID_SWAP])) {
9351 Ok = parseSwizzleSwap(Imm);
9352 } else if (trySkipId(IdSymbolic[ID_REVERSE])) {
9353 Ok = parseSwizzleReverse(Imm);
9354 } else if (trySkipId(IdSymbolic[ID_FFT])) {
9355 Ok = parseSwizzleFFT(Imm);
9356 } else if (trySkipId(IdSymbolic[ID_ROTATE])) {
9357 Ok = parseSwizzleRotate(Imm);
9358 } else {
9359 Error(ModeLoc, "expected a swizzle mode");
9360 }
9361
9362 return Ok && skipToken(AsmToken::RParen, "expected a closing parentheses");
9363 }
9364
9365 return false;
9366}
9367
9368ParseStatus AMDGPUAsmParser::parseSwizzle(OperandVector &Operands) {
9369 SMLoc S = getLoc();
9370 int64_t Imm = 0;
9371
9372 if (trySkipId("offset")) {
9373
9374 bool Ok = false;
9375 if (skipToken(AsmToken::Colon, "expected a colon")) {
9376 if (trySkipId("swizzle")) {
9377 Ok = parseSwizzleMacro(Imm);
9378 } else {
9379 Ok = parseSwizzleOffset(Imm);
9380 }
9381 }
9382
9383 Operands.push_back(
9384 AMDGPUOperand::CreateImm(this, Imm, S, AMDGPUOperand::ImmTySwizzle));
9385
9387 }
9388 return ParseStatus::NoMatch;
9389}
9390
9391bool AMDGPUOperand::isSwizzle() const { return isImmTy(ImmTySwizzle); }
9392
9393//===----------------------------------------------------------------------===//
9394// VGPR Index Mode
9395//===----------------------------------------------------------------------===//
9396
9397int64_t AMDGPUAsmParser::parseGPRIdxMacro() {
9398
9399 using namespace llvm::AMDGPU::VGPRIndexMode;
9400
9401 if (trySkipToken(AsmToken::RParen)) {
9402 return OFF;
9403 }
9404
9405 int64_t Imm = 0;
9406
9407 while (true) {
9408 unsigned Mode = 0;
9409 SMLoc S = getLoc();
9410
9411 for (unsigned ModeId = ID_MIN; ModeId <= ID_MAX; ++ModeId) {
9412 if (trySkipId(IdSymbolic[ModeId])) {
9413 Mode = 1 << ModeId;
9414 break;
9415 }
9416 }
9417
9418 if (Mode == 0) {
9419 Error(S, (Imm == 0)
9420 ? "expected a VGPR index mode or a closing parenthesis"
9421 : "expected a VGPR index mode");
9422 return UNDEF;
9423 }
9424
9425 if (Imm & Mode) {
9426 Error(S, "duplicate VGPR index mode");
9427 return UNDEF;
9428 }
9429 Imm |= Mode;
9430
9431 if (trySkipToken(AsmToken::RParen))
9432 break;
9433 if (!skipToken(AsmToken::Comma,
9434 "expected a comma or a closing parenthesis"))
9435 return UNDEF;
9436 }
9437
9438 return Imm;
9439}
9440
9441ParseStatus AMDGPUAsmParser::parseGPRIdxMode(OperandVector &Operands) {
9442
9443 using namespace llvm::AMDGPU::VGPRIndexMode;
9444
9445 int64_t Imm = 0;
9446 SMLoc S = getLoc();
9447
9448 if (trySkipId("gpr_idx", AsmToken::LParen)) {
9449 Imm = parseGPRIdxMacro();
9450 if (Imm == UNDEF)
9451 return ParseStatus::Failure;
9452 } else {
9453 if (getParser().parseAbsoluteExpression(Imm))
9454 return ParseStatus::Failure;
9455 if (Imm < 0 || !isUInt<4>(Imm))
9456 return Error(S, "invalid immediate: only 4-bit values are legal");
9457 }
9458
9459 Operands.push_back(
9460 AMDGPUOperand::CreateImm(this, Imm, S, AMDGPUOperand::ImmTyGprIdxMode));
9461 return ParseStatus::Success;
9462}
9463
9464bool AMDGPUOperand::isGPRIdxMode() const { return isImmTy(ImmTyGprIdxMode); }
9465
9466//===----------------------------------------------------------------------===//
9467// sopp branch targets
9468//===----------------------------------------------------------------------===//
9469
9470ParseStatus AMDGPUAsmParser::parseSOPPBrTarget(OperandVector &Operands) {
9471
9472 // Make sure we are not parsing something
9473 // that looks like a label or an expression but is not.
9474 // This will improve error messages.
9475 if (isRegister() || isModifier())
9476 return ParseStatus::NoMatch;
9477
9478 if (!parseExpr(Operands))
9479 return ParseStatus::Failure;
9480
9481 AMDGPUOperand &Opr = ((AMDGPUOperand &)*Operands[Operands.size() - 1]);
9482 assert(Opr.isImm() || Opr.isExpr());
9483 SMLoc Loc = Opr.getStartLoc();
9484
9485 // Currently we do not support arbitrary expressions as branch targets.
9486 // Only labels and absolute expressions are accepted.
9487 if (Opr.isExpr() && !Opr.isSymbolRefExpr()) {
9488 Error(Loc, "expected an absolute expression or a label");
9489 } else if (Opr.isImm() && !Opr.isS16Imm()) {
9490 Error(Loc, "expected a 16-bit signed jump offset");
9491 }
9492
9493 return ParseStatus::Success;
9494}
9495
9496//===----------------------------------------------------------------------===//
9497// Boolean holding registers
9498//===----------------------------------------------------------------------===//
9499
9500ParseStatus AMDGPUAsmParser::parseBoolReg(OperandVector &Operands) {
9501 return parseReg(Operands);
9502}
9503
9504//===----------------------------------------------------------------------===//
9505// mubuf
9506//===----------------------------------------------------------------------===//
9507
9508void AMDGPUAsmParser::cvtMubufImpl(MCInst &Inst, const OperandVector &Operands,
9509 bool IsAtomic) {
9510 OptionalImmIndexMap OptionalIdx;
9511 unsigned FirstOperandIdx = 1;
9512 bool IsAtomicReturn = false;
9513
9514 if (IsAtomic) {
9515 IsAtomicReturn = SIInstrFlags::isAtomicRet(MII, Inst);
9516 }
9517
9518 for (unsigned i = FirstOperandIdx, e = Operands.size(); i != e; ++i) {
9519 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
9520
9521 // Add the register arguments
9522 if (Op.isReg()) {
9523 Op.addRegOperands(Inst, 1);
9524 // Insert a tied src for atomic return dst.
9525 // This cannot be postponed as subsequent calls to
9526 // addImmOperands rely on correct number of MC operands.
9527 if (IsAtomicReturn && i == FirstOperandIdx)
9528 Op.addRegOperands(Inst, 1);
9529 continue;
9530 }
9531
9532 // Handle the case where soffset is an immediate
9533 if (Op.isImm() && Op.getImmTy() == AMDGPUOperand::ImmTyNone) {
9534 Op.addImmOperands(Inst, 1);
9535 continue;
9536 }
9537
9538 // Handle tokens like 'offen' which are sometimes hard-coded into the
9539 // asm string. There are no MCInst operands for these.
9540 if (Op.isToken()) {
9541 continue;
9542 }
9543 assert(Op.isImm());
9544
9545 // Handle optional arguments
9546 OptionalIdx[Op.getImmTy()] = i;
9547 }
9548
9549 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9550 AMDGPUOperand::ImmTyOffset);
9551 addOptionalImmOperand(Inst, Operands, OptionalIdx, AMDGPUOperand::ImmTyCPol,
9552 0);
9553 // Parse a dummy operand as a placeholder for the SWZ operand. This enforces
9554 // agreement between MCInstrDesc.getNumOperands and MCInst.getNumOperands.
9556 // The LDS variants carry a trailing IsAsync operand. Parse a dummy the same
9557 // way as the SWZ operand.
9558 if (AMDGPU::hasNamedOperand(Inst.getOpcode(), AMDGPU::OpName::IsAsync))
9560}
9561
9562//===----------------------------------------------------------------------===//
9563// smrd
9564//===----------------------------------------------------------------------===//
9565
9566bool AMDGPUOperand::isSMRDOffset8() const {
9567 return isImmLiteral() && isUInt<8>(getImm());
9568}
9569
9570bool AMDGPUOperand::isSMEMOffset() const {
9571 // Offset range is checked later by validator.
9572 return isImmLiteral();
9573}
9574
9575bool AMDGPUOperand::isSMRDLiteralOffset() const {
9576 // 32-bit literals are only supported on CI and we only want to use them
9577 // when the offset is > 8-bits.
9578 return isImmLiteral() && !isUInt<8>(getImm()) && isUInt<32>(getImm());
9579}
9580
9581//===----------------------------------------------------------------------===//
9582// vop3
9583//===----------------------------------------------------------------------===//
9584
9585static bool ConvertOmodMul(int64_t &Mul) {
9586 if (Mul != 1 && Mul != 2 && Mul != 4)
9587 return false;
9588
9589 Mul >>= 1;
9590 return true;
9591}
9592
9593static bool ConvertOmodDiv(int64_t &Div) {
9594 if (Div == 1) {
9595 Div = 0;
9596 return true;
9597 }
9598
9599 if (Div == 2) {
9600 Div = 3;
9601 return true;
9602 }
9603
9604 return false;
9605}
9606
9607// For pre-gfx11 targets, both bound_ctrl:0 and bound_ctrl:1 are encoded as 1.
9608// This is intentional and ensures compatibility with sp3.
9609// See bug 35397 for details.
9610bool AMDGPUAsmParser::convertDppBoundCtrl(int64_t &BoundCtrl) {
9611 if (BoundCtrl == 0 || BoundCtrl == 1) {
9612 if (!isGFX11Plus())
9613 BoundCtrl = 1;
9614 return true;
9615 }
9616 return false;
9617}
9618
9619void AMDGPUAsmParser::onBeginOfFile() {
9620 if (!getParser().getStreamer().getTargetStreamer())
9621 return;
9622
9623 if (!getTargetStreamer().getTargetID())
9624 getTargetStreamer().initializeTargetID(getSTI(),
9625 /*ApplyFeatureString=*/true);
9626}
9627
9628void AMDGPUAsmParser::emitTargetDirective() {
9629 if (TargetDirectiveEmitted)
9630 return;
9631 TargetDirectiveEmitted = true;
9632
9633 if (!getParser().getStreamer().getTargetStreamer() ||
9634 getSTI().getTargetTriple().getArch() == Triple::r600)
9635 return;
9636
9637 if (isHsaAbi(getSTI()))
9638 getTargetStreamer().EmitDirectiveAMDGCNTarget();
9639}
9640
9641/// Parse AMDGPU specific expressions.
9642///
9643/// expr ::= or(expr, ...) |
9644/// max(expr, ...) |
9645/// min(expr, ...)
9646///
9647bool AMDGPUAsmParser::parsePrimaryExpr(const MCExpr *&Res, SMLoc &EndLoc) {
9648 using AGVK = AMDGPUMCExpr::VariantKind;
9649
9650 if (isToken(AsmToken::Identifier)) {
9651 StringRef TokenId = getTokenStr();
9652 AGVK VK = StringSwitch<AGVK>(TokenId)
9653 .Case("max", AGVK::AGVK_Max)
9654 .Case("min", AGVK::AGVK_Min)
9655 .Case("or", AGVK::AGVK_Or)
9656 .Case("extrasgprs", AGVK::AGVK_ExtraSGPRs)
9657 .Case("totalnumvgprs", AGVK::AGVK_TotalNumVGPRs)
9658 .Case("alignto", AGVK::AGVK_AlignTo)
9659 .Case("occupancy", AGVK::AGVK_Occupancy)
9660 .Case("instprefsize", AGVK::AGVK_InstPrefSize)
9661 .Default(AGVK::AGVK_None);
9662
9663 if (VK != AGVK::AGVK_None && peekToken().is(AsmToken::LParen)) {
9665 uint64_t CommaCount = 0;
9666 lex(); // Eat Arg ('or', 'max', 'occupancy', etc.)
9667 lex(); // Eat '('
9668 while (true) {
9669 if (trySkipToken(AsmToken::RParen)) {
9670 if (Exprs.empty()) {
9671 Error(getToken().getLoc(),
9672 "empty " + Twine(TokenId) + " expression");
9673 return true;
9674 }
9675 if (CommaCount + 1 != Exprs.size()) {
9676 Error(getToken().getLoc(),
9677 "mismatch of commas in " + Twine(TokenId) + " expression");
9678 return true;
9679 }
9680 if (unsigned Expected = AMDGPUMCExpr::getNumExpectedArgs(VK);
9681 Expected && Exprs.size() != Expected) {
9682 Error(getToken().getLoc(), Twine(TokenId) + " expression expects " +
9683 Twine(Expected) + " operands");
9684 return true;
9685 }
9686 Res = AMDGPUMCExpr::create(VK, Exprs, getContext());
9687 return false;
9688 }
9689 const MCExpr *Expr;
9690 if (getParser().parseExpression(Expr, EndLoc))
9691 return true;
9692 Exprs.push_back(Expr);
9693 bool LastTokenWasComma = trySkipToken(AsmToken::Comma);
9694 if (LastTokenWasComma)
9695 CommaCount++;
9696 if (!LastTokenWasComma && !isToken(AsmToken::RParen)) {
9697 Error(getToken().getLoc(),
9698 "unexpected token in " + Twine(TokenId) + " expression");
9699 return true;
9700 }
9701 }
9702 }
9703 }
9704 return getParser().parsePrimaryExpr(Res, EndLoc, nullptr);
9705}
9706
9707ParseStatus AMDGPUAsmParser::parseOModSI(OperandVector &Operands) {
9708 StringRef Name = getTokenStr();
9709 if (Name == "mul") {
9710 return parseIntWithPrefix("mul", Operands, AMDGPUOperand::ImmTyOModSI,
9712 }
9713
9714 if (Name == "div") {
9715 return parseIntWithPrefix("div", Operands, AMDGPUOperand::ImmTyOModSI,
9717 }
9718
9719 return ParseStatus::NoMatch;
9720}
9721
9722// Determines which bit DST_OP_SEL occupies in the op_sel operand according to
9723// the number of src operands present, then copies that bit into src0_modifiers.
9724static void cvtVOP3DstOpSelOnly(MCInst &Inst, const MCRegisterInfo &MRI) {
9725 int Opc = Inst.getOpcode();
9726 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
9727 if (OpSelIdx == -1)
9728 return;
9729
9730 int SrcNum;
9731 const AMDGPU::OpName Ops[] = {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
9732 AMDGPU::OpName::src2};
9733 for (SrcNum = 0; SrcNum < 3 && AMDGPU::hasNamedOperand(Opc, Ops[SrcNum]);
9734 ++SrcNum)
9735 ;
9736 assert(SrcNum > 0);
9737
9738 unsigned OpSel = Inst.getOperand(OpSelIdx).getImm();
9739
9740 int DstIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst);
9741 if (DstIdx == -1)
9742 return;
9743
9744 const MCOperand &DstOp = Inst.getOperand(DstIdx);
9745 int ModIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0_modifiers);
9746 uint32_t ModVal = Inst.getOperand(ModIdx).getImm();
9747 if (DstOp.isReg() &&
9748 MRI.getRegClass(AMDGPU::VGPR_16RegClassID).contains(DstOp.getReg())) {
9749 if (AMDGPU::isHi16Reg(DstOp.getReg(), MRI))
9750 ModVal |= SISrcMods::DST_OP_SEL;
9751 } else {
9752 if ((OpSel & (1 << SrcNum)) != 0)
9753 ModVal |= SISrcMods::DST_OP_SEL;
9754 }
9755 Inst.getOperand(ModIdx).setImm(ModVal);
9756}
9757
9758void AMDGPUAsmParser::cvtVOP3OpSel(MCInst &Inst,
9759 const OperandVector &Operands) {
9760 cvtVOP3P(Inst, Operands);
9761 cvtVOP3DstOpSelOnly(Inst, *getMRI());
9762}
9763
9764void AMDGPUAsmParser::cvtVOP3OpSel(MCInst &Inst, const OperandVector &Operands,
9765 OptionalImmIndexMap &OptionalIdx) {
9766 cvtVOP3P(Inst, Operands, OptionalIdx);
9767 cvtVOP3DstOpSelOnly(Inst, *getMRI());
9768}
9769
9770static bool isRegOrImmWithInputMods(const MCInstrDesc &Desc, unsigned OpNum) {
9771 return
9772 // 1. This operand is input modifiers
9773 Desc.operands()[OpNum].OperandType == AMDGPU::OPERAND_INPUT_MODS
9774 // 2. This is not last operand
9775 && Desc.NumOperands > (OpNum + 1)
9776 // 3. Next operand is register class
9777 && Desc.operands()[OpNum + 1].RegClass != -1
9778 // 4. Next register is not tied to any other operand
9779 && Desc.getOperandConstraint(OpNum + 1,
9781}
9782
9783void AMDGPUAsmParser::cvtOpSelHelper(MCInst &Inst, unsigned OpSel) {
9784 unsigned Opc = Inst.getOpcode();
9785 constexpr AMDGPU::OpName Ops[] = {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
9786 AMDGPU::OpName::src2};
9787 constexpr AMDGPU::OpName ModOps[] = {AMDGPU::OpName::src0_modifiers,
9788 AMDGPU::OpName::src1_modifiers,
9789 AMDGPU::OpName::src2_modifiers};
9790 for (int J = 0; J < 3; ++J) {
9791 int OpIdx = AMDGPU::getNamedOperandIdx(Opc, Ops[J]);
9792 if (OpIdx == -1)
9793 // Some instructions, e.g. v_interp_p2_f16 in GFX9, have src0, src2, but
9794 // no src1. So continue instead of break.
9795 continue;
9796
9797 int ModIdx = AMDGPU::getNamedOperandIdx(Opc, ModOps[J]);
9798 uint32_t ModVal = Inst.getOperand(ModIdx).getImm();
9799
9800 if ((OpSel & (1 << J)) != 0)
9801 ModVal |= SISrcMods::OP_SEL_0;
9802 // op_sel[3] is encoded in src0_modifiers.
9803 if (ModOps[J] == AMDGPU::OpName::src0_modifiers && (OpSel & (1 << 3)) != 0)
9804 ModVal |= SISrcMods::DST_OP_SEL;
9805
9806 Inst.getOperand(ModIdx).setImm(ModVal);
9807 }
9808}
9809
9810void AMDGPUAsmParser::cvtVOP3Interp(MCInst &Inst,
9811 const OperandVector &Operands) {
9812 OptionalImmIndexMap OptionalIdx;
9813 unsigned Opc = Inst.getOpcode();
9814
9815 unsigned I = 1;
9816 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
9817 for (unsigned J = 0; J < Desc.getNumDefs(); ++J) {
9818 ((AMDGPUOperand &)*Operands[I++]).addRegOperands(Inst, 1);
9819 }
9820
9821 for (unsigned E = Operands.size(); I != E; ++I) {
9822 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[I]);
9824 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9825 } else if (Op.isInterpSlot() || Op.isInterpAttr() ||
9826 Op.isInterpAttrChan()) {
9827 Inst.addOperand(MCOperand::createImm(Op.getImm()));
9828 } else if (Op.isImmModifier()) {
9829 OptionalIdx[Op.getImmTy()] = I;
9830 } else {
9831 llvm_unreachable("unhandled operand type");
9832 }
9833 }
9834
9835 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::high))
9836 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9837 AMDGPUOperand::ImmTyHigh);
9838
9839 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::clamp))
9840 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9841 AMDGPUOperand::ImmTyClamp);
9842
9843 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::omod))
9844 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9845 AMDGPUOperand::ImmTyOModSI);
9846
9847 // Some v_interp instructions use op_sel[3] for dst.
9848 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::op_sel)) {
9849 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9850 AMDGPUOperand::ImmTyOpSel);
9851 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
9852 unsigned OpSel = Inst.getOperand(OpSelIdx).getImm();
9853
9854 cvtOpSelHelper(Inst, OpSel);
9855 }
9856}
9857
9858void AMDGPUAsmParser::cvtVINTERP(MCInst &Inst, const OperandVector &Operands) {
9859 OptionalImmIndexMap OptionalIdx;
9860 unsigned Opc = Inst.getOpcode();
9861
9862 unsigned I = 1;
9863 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
9864 for (unsigned J = 0; J < Desc.getNumDefs(); ++J) {
9865 ((AMDGPUOperand &)*Operands[I++]).addRegOperands(Inst, 1);
9866 }
9867
9868 for (unsigned E = Operands.size(); I != E; ++I) {
9869 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[I]);
9871 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9872 } else if (Op.isImmModifier()) {
9873 OptionalIdx[Op.getImmTy()] = I;
9874 } else {
9875 llvm_unreachable("unhandled operand type");
9876 }
9877 }
9878
9879 addOptionalImmOperand(Inst, Operands, OptionalIdx, AMDGPUOperand::ImmTyClamp);
9880
9881 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
9882 if (OpSelIdx != -1)
9883 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9884 AMDGPUOperand::ImmTyOpSel);
9885
9886 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9887 AMDGPUOperand::ImmTyWaitEXP);
9888
9889 if (OpSelIdx == -1)
9890 return;
9891
9892 unsigned OpSel = Inst.getOperand(OpSelIdx).getImm();
9893 cvtOpSelHelper(Inst, OpSel);
9894}
9895
9896void AMDGPUAsmParser::cvtScaledMFMA(MCInst &Inst,
9897 const OperandVector &Operands) {
9898 OptionalImmIndexMap OptionalIdx;
9899 unsigned Opc = Inst.getOpcode();
9900 unsigned I = 1;
9901 int CbszOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::cbsz);
9902
9903 const MCInstrDesc &Desc = MII.get(Opc);
9904
9905 for (unsigned J = 0; J < Desc.getNumDefs(); ++J)
9906 static_cast<AMDGPUOperand &>(*Operands[I++]).addRegOperands(Inst, 1);
9907
9908 for (unsigned E = Operands.size(); I != E; ++I) {
9909 AMDGPUOperand &Op = static_cast<AMDGPUOperand &>(*Operands[I]);
9910 int NumOperands = Inst.getNumOperands();
9911 // The order of operands in MCInst and parsed operands are different.
9912 // Adding dummy cbsz and blgp operands at corresponding MCInst operand
9913 // indices for parsing scale values correctly.
9914 if (NumOperands == CbszOpIdx) {
9917 }
9918 if (isRegOrImmWithInputMods(Desc, NumOperands)) {
9919 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9920 } else if (Op.isImmModifier()) {
9921 OptionalIdx[Op.getImmTy()] = I;
9922 } else {
9923 Op.addRegOrImmOperands(Inst, 1);
9924 }
9925 }
9926
9927 // Insert CBSZ and BLGP operands for F8F6F4 variants
9928 auto CbszIdx = OptionalIdx.find(AMDGPUOperand::ImmTyCBSZ);
9929 if (CbszIdx != OptionalIdx.end()) {
9930 int CbszVal = ((AMDGPUOperand &)*Operands[CbszIdx->second]).getImm();
9931 Inst.getOperand(CbszOpIdx).setImm(CbszVal);
9932 }
9933
9934 int BlgpOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::blgp);
9935 auto BlgpIdx = OptionalIdx.find(AMDGPUOperand::ImmTyBLGP);
9936 if (BlgpIdx != OptionalIdx.end()) {
9937 int BlgpVal = ((AMDGPUOperand &)*Operands[BlgpIdx->second]).getImm();
9938 Inst.getOperand(BlgpOpIdx).setImm(BlgpVal);
9939 }
9940
9941 // Add dummy src_modifiers
9944
9945 // Handle op_sel fields
9946
9947 unsigned OpSel = 0;
9948 auto OpselIdx = OptionalIdx.find(AMDGPUOperand::ImmTyOpSel);
9949 if (OpselIdx != OptionalIdx.end()) {
9950 OpSel = static_cast<const AMDGPUOperand &>(*Operands[OpselIdx->second])
9951 .getImm();
9952 }
9953
9954 unsigned OpSelHi = 0;
9955 auto OpselHiIdx = OptionalIdx.find(AMDGPUOperand::ImmTyOpSelHi);
9956 if (OpselHiIdx != OptionalIdx.end()) {
9957 OpSelHi = static_cast<const AMDGPUOperand &>(*Operands[OpselHiIdx->second])
9958 .getImm();
9959 }
9960 const AMDGPU::OpName ModOps[] = {AMDGPU::OpName::src0_modifiers,
9961 AMDGPU::OpName::src1_modifiers};
9962
9963 for (unsigned J = 0; J < 2; ++J) {
9964 unsigned ModVal = 0;
9965 if (OpSel & (1 << J))
9966 ModVal |= SISrcMods::OP_SEL_0;
9967 if (OpSelHi & (1 << J))
9968 ModVal |= SISrcMods::OP_SEL_1;
9969
9970 const int ModIdx = AMDGPU::getNamedOperandIdx(Opc, ModOps[J]);
9971 Inst.getOperand(ModIdx).setImm(ModVal);
9972 }
9973}
9974
9975void AMDGPUAsmParser::cvtVOP3(MCInst &Inst, const OperandVector &Operands,
9976 OptionalImmIndexMap &OptionalIdx) {
9977 unsigned Opc = Inst.getOpcode();
9978
9979 unsigned I = 1;
9980 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
9981 for (unsigned J = 0; J < Desc.getNumDefs(); ++J) {
9982 ((AMDGPUOperand &)*Operands[I++]).addRegOperands(Inst, 1);
9983 }
9984
9985 for (unsigned E = Operands.size(); I != E; ++I) {
9986 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[I]);
9988 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9989 } else if (Op.isImmModifier()) {
9990 OptionalIdx[Op.getImmTy()] = I;
9991 } else {
9992 Op.addRegOrImmOperands(Inst, 1);
9993 }
9994 }
9995
9996 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::scale_sel))
9997 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9998 AMDGPUOperand::ImmTyScaleSel);
9999
10000 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::clamp))
10001 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10002 AMDGPUOperand::ImmTyClamp);
10003
10004 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::byte_sel)) {
10005 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::vdst_in))
10006 Inst.addOperand(Inst.getOperand(0));
10007 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10008 AMDGPUOperand::ImmTyByteSel);
10009 }
10010
10011 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::omod))
10012 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10013 AMDGPUOperand::ImmTyOModSI);
10014
10015 // Special case v_mac_{f16, f32} and v_fmac_{f16, f32} (gfx906/gfx10+):
10016 // it has src2 register operand that is tied to dst operand
10017 // we don't allow modifiers for this operand in assembler so src2_modifiers
10018 // should be 0.
10019 if (isMAC(Opc)) {
10020 auto *it = Inst.begin();
10021 std::advance(
10022 it, AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2_modifiers));
10023 it = Inst.insert(it, MCOperand::createImm(0)); // no modifiers for src2
10024 ++it;
10025 // Copy the operand to ensure it's not invalidated when Inst grows.
10026 Inst.insert(it, MCOperand(Inst.getOperand(0))); // src2 = dst
10027 }
10028}
10029
10030void AMDGPUAsmParser::cvtVOP3(MCInst &Inst, const OperandVector &Operands) {
10031 OptionalImmIndexMap OptionalIdx;
10032 cvtVOP3(Inst, Operands, OptionalIdx);
10033}
10034
10035void AMDGPUAsmParser::cvtVOP3P(MCInst &Inst, const OperandVector &Operands,
10036 OptionalImmIndexMap &OptIdx) {
10037 const int Opc = Inst.getOpcode();
10038
10039 const bool IsPacked = SIInstrFlags::isPacked(MII, Inst);
10040
10041 if (Opc == AMDGPU::V_CVT_SCALEF32_PK_FP4_F16_vi ||
10042 Opc == AMDGPU::V_CVT_SCALEF32_PK_FP4_BF16_vi ||
10043 Opc == AMDGPU::V_CVT_SR_BF8_F32_vi ||
10044 Opc == AMDGPU::V_CVT_SR_FP8_F32_vi ||
10045 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_gfx11 ||
10046 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_gfx11 ||
10047 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_gfx12 ||
10048 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_gfx12 ||
10049 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_gfx13 ||
10050 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_gfx13) {
10051 Inst.addOperand(MCOperand::createImm(0)); // Placeholder for src2_mods
10052 Inst.addOperand(Inst.getOperand(0));
10053 }
10054
10055 // Append vdst_in only if a previous converter (cvtVOP3DPP for DPP variants,
10056 // cvtVOP3 for byte_sel variants) hasn't already placed it. Use the position
10057 // of the named operand to detect that, the same way cvtVOP3DPP does
10058 // internally.
10059 int VdstInIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst_in);
10060 if (VdstInIdx != -1 && VdstInIdx == static_cast<int>(Inst.getNumOperands()))
10061 Inst.addOperand(Inst.getOperand(0));
10062
10063 int BitOp3Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::bitop3);
10064 if (BitOp3Idx != -1) {
10065 addOptionalImmOperand(Inst, Operands, OptIdx, AMDGPUOperand::ImmTyBitOp3);
10066 }
10067
10068 // FIXME: This is messy. Parse the modifiers as if it was a normal VOP3
10069 // instruction, and then figure out where to actually put the modifiers
10070
10071 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
10072 if (OpSelIdx != -1) {
10073 addOptionalImmOperand(Inst, Operands, OptIdx, AMDGPUOperand::ImmTyOpSel);
10074 }
10075
10076 int OpSelHiIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel_hi);
10077 if (OpSelHiIdx != -1) {
10078 int DefaultVal = IsPacked ? -1 : 0;
10079 addOptionalImmOperand(Inst, Operands, OptIdx, AMDGPUOperand::ImmTyOpSelHi,
10080 DefaultVal);
10081 }
10082
10083 int MatrixAFMTIdx =
10084 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_a_fmt);
10085 if (MatrixAFMTIdx != -1) {
10086 addOptionalImmOperand(Inst, Operands, OptIdx,
10087 AMDGPUOperand::ImmTyMatrixAFMT, 0);
10088 }
10089
10090 int MatrixBFMTIdx =
10091 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_b_fmt);
10092 if (MatrixBFMTIdx != -1) {
10093 addOptionalImmOperand(Inst, Operands, OptIdx,
10094 AMDGPUOperand::ImmTyMatrixBFMT, 0);
10095 }
10096
10097 int MatrixAScaleIdx =
10098 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_a_scale);
10099 if (MatrixAScaleIdx != -1) {
10100 addOptionalImmOperand(Inst, Operands, OptIdx,
10101 AMDGPUOperand::ImmTyMatrixAScale, 0);
10102 }
10103
10104 int MatrixBScaleIdx =
10105 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_b_scale);
10106 if (MatrixBScaleIdx != -1) {
10107 addOptionalImmOperand(Inst, Operands, OptIdx,
10108 AMDGPUOperand::ImmTyMatrixBScale, 0);
10109 }
10110
10111 int MatrixAScaleFmtIdx =
10112 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_a_scale_fmt);
10113 if (MatrixAScaleFmtIdx != -1) {
10114 addOptionalImmOperand(Inst, Operands, OptIdx,
10115 AMDGPUOperand::ImmTyMatrixAScaleFmt, 0);
10116 }
10117
10118 int MatrixBScaleFmtIdx =
10119 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_b_scale_fmt);
10120 if (MatrixBScaleFmtIdx != -1) {
10121 addOptionalImmOperand(Inst, Operands, OptIdx,
10122 AMDGPUOperand::ImmTyMatrixBScaleFmt, 0);
10123 }
10124
10125 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::matrix_a_reuse))
10126 addOptionalImmOperand(Inst, Operands, OptIdx,
10127 AMDGPUOperand::ImmTyMatrixAReuse, 0);
10128
10129 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::matrix_b_reuse))
10130 addOptionalImmOperand(Inst, Operands, OptIdx,
10131 AMDGPUOperand::ImmTyMatrixBReuse, 0);
10132
10133 int NegLoIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::neg_lo);
10134 if (NegLoIdx != -1)
10135 addOptionalImmOperand(Inst, Operands, OptIdx, AMDGPUOperand::ImmTyNegLo);
10136
10137 int NegHiIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::neg_hi);
10138 if (NegHiIdx != -1)
10139 addOptionalImmOperand(Inst, Operands, OptIdx, AMDGPUOperand::ImmTyNegHi);
10140
10141 const AMDGPU::OpName Ops[] = {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
10142 AMDGPU::OpName::src2};
10143 const AMDGPU::OpName ModOps[] = {AMDGPU::OpName::src0_modifiers,
10144 AMDGPU::OpName::src1_modifiers,
10145 AMDGPU::OpName::src2_modifiers};
10146
10147 unsigned OpSel = 0;
10148 unsigned OpSelHi = 0;
10149 unsigned NegLo = 0;
10150 unsigned NegHi = 0;
10151
10152 if (OpSelIdx != -1)
10153 OpSel = Inst.getOperand(OpSelIdx).getImm();
10154
10155 if (OpSelHiIdx != -1)
10156 OpSelHi = Inst.getOperand(OpSelHiIdx).getImm();
10157
10158 if (NegLoIdx != -1)
10159 NegLo = Inst.getOperand(NegLoIdx).getImm();
10160
10161 if (NegHiIdx != -1)
10162 NegHi = Inst.getOperand(NegHiIdx).getImm();
10163
10164 for (int J = 0; J < 3; ++J) {
10165 int OpIdx = AMDGPU::getNamedOperandIdx(Opc, Ops[J]);
10166 if (OpIdx == -1)
10167 break;
10168
10169 int ModIdx = AMDGPU::getNamedOperandIdx(Opc, ModOps[J]);
10170
10171 if (ModIdx == -1)
10172 continue;
10173
10174 // For MAC instructions, src2 is tied to vdst and its op_sel bit
10175 // is not encoded.
10176 if (AMDGPU::isMAC(Opc) && ModOps[J] == AMDGPU::OpName::src2_modifiers)
10177 continue;
10178
10179 uint32_t ModVal = 0;
10180
10181 const MCOperand &SrcOp = Inst.getOperand(OpIdx);
10182 if (SrcOp.isReg() && getMRI()
10183 ->getRegClass(AMDGPU::VGPR_16RegClassID)
10184 .contains(SrcOp.getReg())) {
10185 bool VGPRSuffixIsHi = AMDGPU::isHi16Reg(SrcOp.getReg(), *getMRI());
10186 if (VGPRSuffixIsHi)
10187 ModVal |= SISrcMods::OP_SEL_0;
10188 } else {
10189 if ((OpSel & (1 << J)) != 0)
10190 ModVal |= SISrcMods::OP_SEL_0;
10191 }
10192
10193 if ((OpSelHi & (1 << J)) != 0)
10194 ModVal |= SISrcMods::OP_SEL_1;
10195
10196 if ((NegLo & (1 << J)) != 0)
10197 ModVal |= SISrcMods::NEG;
10198
10199 if ((NegHi & (1 << J)) != 0)
10200 ModVal |= SISrcMods::NEG_HI;
10201
10202 Inst.getOperand(ModIdx).setImm(Inst.getOperand(ModIdx).getImm() | ModVal);
10203 }
10204}
10205
10206void AMDGPUAsmParser::cvtVOP3P(MCInst &Inst, const OperandVector &Operands) {
10207 OptionalImmIndexMap OptIdx;
10208 cvtVOP3(Inst, Operands, OptIdx);
10209 cvtVOP3P(Inst, Operands, OptIdx);
10210}
10211
10213 unsigned i, unsigned Opc,
10214 AMDGPU::OpName OpName) {
10215 if (AMDGPU::getNamedOperandIdx(Opc, OpName) != -1)
10216 ((AMDGPUOperand &)*Operands[i]).addRegOrImmWithFPInputModsOperands(Inst, 2);
10217 else
10218 ((AMDGPUOperand &)*Operands[i]).addRegOperands(Inst, 1);
10219}
10220
10221void AMDGPUAsmParser::cvtSWMMAC(MCInst &Inst, const OperandVector &Operands) {
10222 unsigned Opc = Inst.getOpcode();
10223
10224 ((AMDGPUOperand &)*Operands[1]).addRegOperands(Inst, 1);
10225 addSrcModifiersAndSrc(Inst, Operands, 2, Opc, AMDGPU::OpName::src0_modifiers);
10226 addSrcModifiersAndSrc(Inst, Operands, 3, Opc, AMDGPU::OpName::src1_modifiers);
10227 ((AMDGPUOperand &)*Operands[1]).addRegOperands(Inst, 1); // srcTiedDef
10228 ((AMDGPUOperand &)*Operands[4]).addRegOperands(Inst, 1); // src2
10229
10230 OptionalImmIndexMap OptIdx;
10231 for (unsigned i = 5; i < Operands.size(); ++i) {
10232 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
10233 OptIdx[Op.getImmTy()] = i;
10234 }
10235
10236 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::index_key_8bit))
10237 addOptionalImmOperand(Inst, Operands, OptIdx,
10238 AMDGPUOperand::ImmTyIndexKey8bit);
10239
10240 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::index_key_16bit))
10241 addOptionalImmOperand(Inst, Operands, OptIdx,
10242 AMDGPUOperand::ImmTyIndexKey16bit);
10243
10244 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::index_key_32bit))
10245 addOptionalImmOperand(Inst, Operands, OptIdx,
10246 AMDGPUOperand::ImmTyIndexKey32bit);
10247
10248 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::clamp))
10249 addOptionalImmOperand(Inst, Operands, OptIdx, AMDGPUOperand::ImmTyClamp);
10250
10251 cvtVOP3P(Inst, Operands, OptIdx);
10252}
10253
10254//===----------------------------------------------------------------------===//
10255// VOPD
10256//===----------------------------------------------------------------------===//
10257
10258ParseStatus AMDGPUAsmParser::parseVOPD(OperandVector &Operands) {
10259 if (!hasVOPD(getSTI()))
10260 return ParseStatus::NoMatch;
10261
10262 if (isToken(AsmToken::Colon) && peekToken(false).is(AsmToken::Colon)) {
10263 SMLoc S = getLoc();
10264 lex();
10265 lex();
10266 Operands.push_back(AMDGPUOperand::CreateToken(this, "::", S));
10267 SMLoc OpYLoc = getLoc();
10268 StringRef OpYName;
10269 if (isToken(AsmToken::Identifier) && !Parser.parseIdentifier(OpYName)) {
10270 Operands.push_back(AMDGPUOperand::CreateToken(this, OpYName, OpYLoc));
10271 return ParseStatus::Success;
10272 }
10273 return Error(OpYLoc, "expected a VOPDY instruction after ::");
10274 }
10275 return ParseStatus::NoMatch;
10276}
10277
10278// Create VOPD MCInst operands using parsed assembler operands.
10279void AMDGPUAsmParser::cvtVOPD(MCInst &Inst, const OperandVector &Operands) {
10280 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
10281
10282 auto addOp = [&](uint16_t ParsedOprIdx) { // NOLINT:function pointer
10283 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[ParsedOprIdx]);
10285 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
10286 return;
10287 }
10288 if (Op.isReg()) {
10289 Op.addRegOperands(Inst, 1);
10290 return;
10291 }
10292 if (Op.isImm()) {
10293 Op.addImmOperands(Inst, 1);
10294 return;
10295 }
10296 llvm_unreachable("Unhandled operand type in cvtVOPD");
10297 };
10298
10299 const auto &InstInfo = getVOPDInstInfo(Inst.getOpcode(), &MII);
10300
10301 // MCInst operands are ordered as follows:
10302 // dstX, dstY, src0X [, other OpX operands], src0Y [, other OpY operands]
10303
10304 for (auto CompIdx : VOPD::COMPONENTS) {
10305 addOp(InstInfo[CompIdx].getIndexOfDstInParsedOperands());
10306 }
10307
10308 for (auto CompIdx : VOPD::COMPONENTS) {
10309 const auto &CInfo = InstInfo[CompIdx];
10310 auto CompSrcOperandsNum = InstInfo[CompIdx].getCompParsedSrcOperandsNum();
10311 for (unsigned CompSrcIdx = 0; CompSrcIdx < CompSrcOperandsNum; ++CompSrcIdx)
10312 addOp(CInfo.getIndexOfSrcInParsedOperands(CompSrcIdx));
10313 if (CInfo.hasSrc2Acc())
10314 addOp(CInfo.getIndexOfDstInParsedOperands());
10315 }
10316
10317 int BitOp3Idx =
10318 AMDGPU::getNamedOperandIdx(Inst.getOpcode(), AMDGPU::OpName::bitop3);
10319 if (BitOp3Idx != -1) {
10320 OptionalImmIndexMap OptIdx;
10321 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands.back());
10322 if (Op.isImm())
10323 OptIdx[Op.getImmTy()] = Operands.size() - 1;
10324
10325 addOptionalImmOperand(Inst, Operands, OptIdx, AMDGPUOperand::ImmTyBitOp3);
10326 }
10327}
10328
10329//===----------------------------------------------------------------------===//
10330// dpp
10331//===----------------------------------------------------------------------===//
10332
10333bool AMDGPUOperand::isDPP8() const { return isImmTy(ImmTyDPP8); }
10334
10335bool AMDGPUOperand::isDPPCtrl() const {
10336 using namespace AMDGPU::DPP;
10337
10338 bool result = isImm() && getImmTy() == ImmTyDppCtrl && isUInt<9>(getImm());
10339 if (result) {
10340 int64_t Imm = getImm();
10341 return (Imm >= DppCtrl::QUAD_PERM_FIRST &&
10342 Imm <= DppCtrl::QUAD_PERM_LAST) ||
10343 (Imm >= DppCtrl::ROW_SHL_FIRST && Imm <= DppCtrl::ROW_SHL_LAST) ||
10344 (Imm >= DppCtrl::ROW_SHR_FIRST && Imm <= DppCtrl::ROW_SHR_LAST) ||
10345 (Imm >= DppCtrl::ROW_ROR_FIRST && Imm <= DppCtrl::ROW_ROR_LAST) ||
10346 (Imm == DppCtrl::WAVE_SHL1) || (Imm == DppCtrl::WAVE_ROL1) ||
10347 (Imm == DppCtrl::WAVE_SHR1) || (Imm == DppCtrl::WAVE_ROR1) ||
10348 (Imm == DppCtrl::ROW_MIRROR) || (Imm == DppCtrl::ROW_HALF_MIRROR) ||
10349 (Imm == DppCtrl::BCAST15) || (Imm == DppCtrl::BCAST31) ||
10350 (Imm >= DppCtrl::ROW_SHARE_FIRST &&
10351 Imm <= DppCtrl::ROW_SHARE_LAST) ||
10352 (Imm >= DppCtrl::ROW_XMASK_FIRST && Imm <= DppCtrl::ROW_XMASK_LAST);
10353 }
10354 return false;
10355}
10356
10357//===----------------------------------------------------------------------===//
10358// mAI
10359//===----------------------------------------------------------------------===//
10360
10361bool AMDGPUOperand::isBLGP() const {
10362 return isImm() && getImmTy() == ImmTyBLGP && isUInt<3>(getImm());
10363}
10364
10365bool AMDGPUOperand::isS16Imm() const {
10366 return isImmLiteral() && (isInt<16>(getImm()) || isUInt<16>(getImm()));
10367}
10368
10369bool AMDGPUOperand::isU16Imm() const {
10370 return isImmLiteral() && isUInt<16>(getImm());
10371}
10372
10373//===----------------------------------------------------------------------===//
10374// dim
10375//===----------------------------------------------------------------------===//
10376
10377bool AMDGPUAsmParser::parseDimId(unsigned &Encoding) {
10378 // We want to allow "dim:1D" etc.,
10379 // but the initial 1 is tokenized as an integer.
10380 std::string Token;
10381 if (isToken(AsmToken::Integer)) {
10382 SMLoc Loc = getToken().getEndLoc();
10383 Token = std::string(getTokenStr());
10384 lex();
10385 if (getLoc() != Loc)
10386 return false;
10387 }
10388
10389 StringRef Suffix;
10390 if (!parseId(Suffix))
10391 return false;
10392 Token += Suffix;
10393
10394 StringRef DimId = Token;
10395 DimId.consume_front("SQ_RSRC_IMG_");
10396
10397 const AMDGPU::MIMGDimInfo *DimInfo = AMDGPU::getMIMGDimInfoByAsmSuffix(DimId);
10398 if (!DimInfo)
10399 return false;
10400
10401 Encoding = DimInfo->Encoding;
10402 return true;
10403}
10404
10405ParseStatus AMDGPUAsmParser::parseDim(OperandVector &Operands) {
10406 if (!isGFX10Plus())
10407 return ParseStatus::NoMatch;
10408
10409 SMLoc S = getLoc();
10410
10411 if (!trySkipId("dim", AsmToken::Colon))
10412 return ParseStatus::NoMatch;
10413
10414 unsigned Encoding;
10415 SMLoc Loc = getLoc();
10416 if (!parseDimId(Encoding))
10417 return Error(Loc, "invalid dim value");
10418
10419 Operands.push_back(
10420 AMDGPUOperand::CreateImm(this, Encoding, S, AMDGPUOperand::ImmTyDim));
10421 return ParseStatus::Success;
10422}
10423
10424//===----------------------------------------------------------------------===//
10425// dpp
10426//===----------------------------------------------------------------------===//
10427
10428ParseStatus AMDGPUAsmParser::parseDPP8(OperandVector &Operands) {
10429 SMLoc S = getLoc();
10430
10431 if (!isGFX10Plus() || !trySkipId("dpp8", AsmToken::Colon))
10432 return ParseStatus::NoMatch;
10433
10434 // dpp8:[%d,%d,%d,%d,%d,%d,%d,%d]
10435
10436 int64_t Sels[8];
10437
10438 if (!skipToken(AsmToken::LBrac, "expected an opening square bracket"))
10439 return ParseStatus::Failure;
10440
10441 for (size_t i = 0; i < 8; ++i) {
10442 if (i > 0 && !skipToken(AsmToken::Comma, "expected a comma"))
10443 return ParseStatus::Failure;
10444
10445 SMLoc Loc = getLoc();
10446 if (getParser().parseAbsoluteExpression(Sels[i]))
10447 return ParseStatus::Failure;
10448 if (0 > Sels[i] || 7 < Sels[i])
10449 return Error(Loc, "expected a 3-bit value");
10450 }
10451
10452 if (!skipToken(AsmToken::RBrac, "expected a closing square bracket"))
10453 return ParseStatus::Failure;
10454
10455 unsigned DPP8 = 0;
10456 for (size_t i = 0; i < 8; ++i)
10457 DPP8 |= (Sels[i] << (i * 3));
10458
10459 Operands.push_back(
10460 AMDGPUOperand::CreateImm(this, DPP8, S, AMDGPUOperand::ImmTyDPP8));
10461 return ParseStatus::Success;
10462}
10463
10464bool AMDGPUAsmParser::isSupportedDPPCtrl(StringRef Ctrl,
10465 const OperandVector &Operands) {
10466 if (Ctrl == "row_newbcast")
10467 return isGFX90A();
10468
10469 if (Ctrl == "row_share" || Ctrl == "row_xmask")
10470 return isGFX10Plus();
10471
10472 if (Ctrl == "wave_shl" || Ctrl == "wave_shr" || Ctrl == "wave_rol" ||
10473 Ctrl == "wave_ror" || Ctrl == "row_bcast")
10474 return isVI() || isGFX9();
10475
10476 return Ctrl == "row_mirror" || Ctrl == "row_half_mirror" ||
10477 Ctrl == "quad_perm" || Ctrl == "row_shl" || Ctrl == "row_shr" ||
10478 Ctrl == "row_ror";
10479}
10480
10481int64_t AMDGPUAsmParser::parseDPPCtrlPerm() {
10482 // quad_perm:[%d,%d,%d,%d]
10483
10484 if (!skipToken(AsmToken::LBrac, "expected an opening square bracket"))
10485 return -1;
10486
10487 int64_t Val = 0;
10488 for (int i = 0; i < 4; ++i) {
10489 if (i > 0 && !skipToken(AsmToken::Comma, "expected a comma"))
10490 return -1;
10491
10492 int64_t Temp;
10493 SMLoc Loc = getLoc();
10494 if (getParser().parseAbsoluteExpression(Temp))
10495 return -1;
10496 if (Temp < 0 || Temp > 3) {
10497 Error(Loc, "expected a 2-bit value");
10498 return -1;
10499 }
10500
10501 Val += (Temp << i * 2);
10502 }
10503
10504 if (!skipToken(AsmToken::RBrac, "expected a closing square bracket"))
10505 return -1;
10506
10507 return Val;
10508}
10509
10510int64_t AMDGPUAsmParser::parseDPPCtrlSel(StringRef Ctrl) {
10511 using namespace AMDGPU::DPP;
10512
10513 // sel:%d
10514
10515 int64_t Val;
10516 SMLoc Loc = getLoc();
10517
10518 if (getParser().parseAbsoluteExpression(Val))
10519 return -1;
10520
10521 struct DppCtrlCheck {
10522 int64_t Ctrl;
10523 int Lo;
10524 int Hi;
10525 };
10526
10527 DppCtrlCheck Check =
10528 StringSwitch<DppCtrlCheck>(Ctrl)
10529 .Case("wave_shl", {DppCtrl::WAVE_SHL1, 1, 1})
10530 .Case("wave_rol", {DppCtrl::WAVE_ROL1, 1, 1})
10531 .Case("wave_shr", {DppCtrl::WAVE_SHR1, 1, 1})
10532 .Case("wave_ror", {DppCtrl::WAVE_ROR1, 1, 1})
10533 .Case("row_shl", {DppCtrl::ROW_SHL0, 1, 15})
10534 .Case("row_shr", {DppCtrl::ROW_SHR0, 1, 15})
10535 .Case("row_ror", {DppCtrl::ROW_ROR0, 1, 15})
10536 .Case("row_share", {DppCtrl::ROW_SHARE_FIRST, 0, 15})
10537 .Case("row_xmask", {DppCtrl::ROW_XMASK_FIRST, 0, 15})
10538 .Case("row_newbcast", {DppCtrl::ROW_NEWBCAST_FIRST, 0, 15})
10539 .Default({-1, 0, 0});
10540
10541 bool Valid;
10542 if (Check.Ctrl == -1) {
10543 Valid = (Ctrl == "row_bcast" && (Val == 15 || Val == 31));
10544 Val = (Val == 15) ? DppCtrl::BCAST15 : DppCtrl::BCAST31;
10545 } else {
10546 Valid = Check.Lo <= Val && Val <= Check.Hi;
10547 Val = (Check.Lo == Check.Hi) ? Check.Ctrl : (Check.Ctrl | Val);
10548 }
10549
10550 if (!Valid) {
10551 Error(Loc, Twine("invalid ", Ctrl) + Twine(" value"));
10552 return -1;
10553 }
10554
10555 return Val;
10556}
10557
10558ParseStatus AMDGPUAsmParser::parseDPPCtrl(OperandVector &Operands) {
10559 using namespace AMDGPU::DPP;
10560
10561 if (!isToken(AsmToken::Identifier) ||
10562 !isSupportedDPPCtrl(getTokenStr(), Operands))
10563 return ParseStatus::NoMatch;
10564
10565 SMLoc S = getLoc();
10566 int64_t Val = -1;
10567 StringRef Ctrl;
10568
10569 parseId(Ctrl);
10570
10571 if (Ctrl == "row_mirror") {
10572 Val = DppCtrl::ROW_MIRROR;
10573 } else if (Ctrl == "row_half_mirror") {
10574 Val = DppCtrl::ROW_HALF_MIRROR;
10575 } else {
10576 if (skipToken(AsmToken::Colon, "expected a colon")) {
10577 if (Ctrl == "quad_perm") {
10578 Val = parseDPPCtrlPerm();
10579 } else {
10580 Val = parseDPPCtrlSel(Ctrl);
10581 }
10582 }
10583 }
10584
10585 if (Val == -1)
10586 return ParseStatus::Failure;
10587
10588 Operands.push_back(
10589 AMDGPUOperand::CreateImm(this, Val, S, AMDGPUOperand::ImmTyDppCtrl));
10590 return ParseStatus::Success;
10591}
10592
10593void AMDGPUAsmParser::cvtVOP3DPP(MCInst &Inst, const OperandVector &Operands,
10594 bool IsDPP8) {
10595 OptionalImmIndexMap OptionalIdx;
10596 unsigned Opc = Inst.getOpcode();
10597 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
10598
10599 // MAC instructions are special because they have 'old'
10600 // operand which is not tied to dst (but assumed to be).
10601 // They also have dummy unused src2_modifiers.
10602 int OldIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::old);
10603 int Src2ModIdx =
10604 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2_modifiers);
10605 bool IsMAC = OldIdx != -1 && Src2ModIdx != -1 &&
10606 Desc.getOperandConstraint(OldIdx, MCOI::TIED_TO) == -1;
10607
10608 unsigned I = 1;
10609 for (unsigned J = 0; J < Desc.getNumDefs(); ++J) {
10610 ((AMDGPUOperand &)*Operands[I++]).addRegOperands(Inst, 1);
10611 }
10612
10613 int Fi = 0;
10614 int VdstInIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst_in);
10615 bool IsVOP3CvtSrDpp = Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp8_gfx12 ||
10616 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp8_gfx13 ||
10617 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp8_gfx12 ||
10618 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp8_gfx13 ||
10619 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp_gfx12 ||
10620 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp_gfx13 ||
10621 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp_gfx12 ||
10622 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp_gfx13;
10623
10624 for (unsigned E = Operands.size(); I != E; ++I) {
10625
10626 if (IsMAC) {
10627 int NumOperands = Inst.getNumOperands();
10628 if (OldIdx == NumOperands) {
10629 // Handle old operand
10630 constexpr int DST_IDX = 0;
10631 Inst.addOperand(Inst.getOperand(DST_IDX));
10632 } else if (Src2ModIdx == NumOperands) {
10633 // Add unused dummy src2_modifiers
10635 }
10636 }
10637
10638 if (VdstInIdx == static_cast<int>(Inst.getNumOperands())) {
10639 Inst.addOperand(Inst.getOperand(0));
10640 }
10641
10642 if (IsVOP3CvtSrDpp) {
10643 if (Src2ModIdx == static_cast<int>(Inst.getNumOperands())) {
10645 Inst.addOperand(MCOperand::createReg(MCRegister()));
10646 }
10647 }
10648
10649 auto TiedTo =
10650 Desc.getOperandConstraint(Inst.getNumOperands(), MCOI::TIED_TO);
10651 if (TiedTo != -1) {
10652 assert((unsigned)TiedTo < Inst.getNumOperands());
10653 // handle tied old or src2 for MAC instructions
10654 Inst.addOperand(Inst.getOperand(TiedTo));
10655 }
10656 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[I]);
10657 // Add the register arguments
10658 if (IsDPP8 && Op.isDppFI()) {
10659 Fi = Op.getImm();
10660 } else if (isRegOrImmWithInputMods(Desc, Inst.getNumOperands())) {
10661 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
10662 } else if (Op.isReg()) {
10663 Op.addRegOperands(Inst, 1);
10664 } else if (Op.isImm() &&
10665 Desc.operands()[Inst.getNumOperands()].RegClass != -1) {
10666 Op.addImmOperands(Inst, 1);
10667 } else if (Op.isImm()) {
10668 OptionalIdx[Op.getImmTy()] = I;
10669 } else {
10670 llvm_unreachable("unhandled operand type");
10671 }
10672 }
10673
10674 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::clamp) && !IsVOP3CvtSrDpp)
10675 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10676 AMDGPUOperand::ImmTyClamp);
10677
10678 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::byte_sel)) {
10679 if (VdstInIdx == static_cast<int>(Inst.getNumOperands()))
10680 Inst.addOperand(Inst.getOperand(0));
10681 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10682 AMDGPUOperand::ImmTyByteSel);
10683 }
10684
10685 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::omod))
10686 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10687 AMDGPUOperand::ImmTyOModSI);
10688
10690 cvtVOP3P(Inst, Operands, OptionalIdx);
10691 else if (SIInstrFlags::isVOP3(Desc))
10692 cvtVOP3OpSel(Inst, Operands, OptionalIdx);
10693 else if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::op_sel)) {
10694 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10695 AMDGPUOperand::ImmTyOpSel);
10696 }
10697
10698 if (IsDPP8) {
10699 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10700 AMDGPUOperand::ImmTyDPP8);
10701 using namespace llvm::AMDGPU::DPP;
10702 Inst.addOperand(MCOperand::createImm(Fi ? DPP8_FI_1 : DPP8_FI_0));
10703 } else {
10704 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10705 AMDGPUOperand::ImmTyDppCtrl, 0xe4);
10706 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10707 AMDGPUOperand::ImmTyDppRowMask, 0xf);
10708 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10709 AMDGPUOperand::ImmTyDppBankMask, 0xf);
10710 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10711 AMDGPUOperand::ImmTyDppBoundCtrl);
10712
10713 if (AMDGPU::hasNamedOperand(Inst.getOpcode(), AMDGPU::OpName::fi))
10714 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10715 AMDGPUOperand::ImmTyDppFI);
10716 }
10717}
10718
10719void AMDGPUAsmParser::cvtDPP(MCInst &Inst, const OperandVector &Operands,
10720 bool IsDPP8) {
10721 OptionalImmIndexMap OptionalIdx;
10722
10723 unsigned I = 1;
10724 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
10725 for (unsigned J = 0; J < Desc.getNumDefs(); ++J) {
10726 ((AMDGPUOperand &)*Operands[I++]).addRegOperands(Inst, 1);
10727 }
10728
10729 int Fi = 0;
10730 for (unsigned E = Operands.size(); I != E; ++I) {
10731 auto TiedTo =
10732 Desc.getOperandConstraint(Inst.getNumOperands(), MCOI::TIED_TO);
10733 if (TiedTo != -1) {
10734 assert((unsigned)TiedTo < Inst.getNumOperands());
10735 // handle tied old or src2 for MAC instructions
10736 Inst.addOperand(Inst.getOperand(TiedTo));
10737 }
10738 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[I]);
10739 // Add the register arguments
10740 if (Op.isReg() && validateVccOperand(Op.getReg())) {
10741 // VOP2b (v_add_u32, v_sub_u32 ...) dpp use "vcc" token.
10742 // Skip it.
10743 continue;
10744 }
10745
10746 if (IsDPP8) {
10747 if (Op.isDPP8()) {
10748 Op.addImmOperands(Inst, 1);
10749 } else if (isRegOrImmWithInputMods(Desc, Inst.getNumOperands())) {
10750 Op.addRegWithFPInputModsOperands(Inst, 2);
10751 } else if (Op.isDppFI()) {
10752 Fi = Op.getImm();
10753 } else if (Op.isReg()) {
10754 Op.addRegOperands(Inst, 1);
10755 } else {
10756 llvm_unreachable("Invalid operand type");
10757 }
10758 } else {
10760 Op.addRegWithFPInputModsOperands(Inst, 2);
10761 } else if (Op.isReg()) {
10762 Op.addRegOperands(Inst, 1);
10763 } else if (Op.isDPPCtrl()) {
10764 Op.addImmOperands(Inst, 1);
10765 } else if (Op.isImm()) {
10766 // Handle optional arguments
10767 OptionalIdx[Op.getImmTy()] = I;
10768 } else {
10769 llvm_unreachable("Invalid operand type");
10770 }
10771 }
10772 }
10773
10774 if (IsDPP8) {
10775 using namespace llvm::AMDGPU::DPP;
10776 Inst.addOperand(MCOperand::createImm(Fi ? DPP8_FI_1 : DPP8_FI_0));
10777 } else {
10778 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10779 AMDGPUOperand::ImmTyDppRowMask, 0xf);
10780 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10781 AMDGPUOperand::ImmTyDppBankMask, 0xf);
10782 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10783 AMDGPUOperand::ImmTyDppBoundCtrl);
10784 if (AMDGPU::hasNamedOperand(Inst.getOpcode(), AMDGPU::OpName::fi)) {
10785 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10786 AMDGPUOperand::ImmTyDppFI);
10787 }
10788 }
10789}
10790
10791//===----------------------------------------------------------------------===//
10792// sdwa
10793//===----------------------------------------------------------------------===//
10794
10795ParseStatus AMDGPUAsmParser::parseSDWASel(OperandVector &Operands,
10796 StringRef Prefix,
10797 AMDGPUOperand::ImmTy Type) {
10798 return parseStringOrIntWithPrefix(
10799 Operands, Prefix,
10800 {"BYTE_0", "BYTE_1", "BYTE_2", "BYTE_3", "WORD_0", "WORD_1", "DWORD"},
10801 Type);
10802}
10803
10804ParseStatus AMDGPUAsmParser::parseSDWADstUnused(OperandVector &Operands) {
10805 return parseStringOrIntWithPrefix(
10806 Operands, "dst_unused", {"UNUSED_PAD", "UNUSED_SEXT", "UNUSED_PRESERVE"},
10807 AMDGPUOperand::ImmTySDWADstUnused);
10808}
10809
10810void AMDGPUAsmParser::cvtSdwaVOP1(MCInst &Inst, const OperandVector &Operands) {
10811 cvtSDWA(Inst, Operands, SDWAInstType::VOP1);
10812}
10813
10814void AMDGPUAsmParser::cvtSdwaVOP2(MCInst &Inst, const OperandVector &Operands) {
10815 cvtSDWA(Inst, Operands, SDWAInstType::VOP2);
10816}
10817
10818void AMDGPUAsmParser::cvtSdwaVOP2b(MCInst &Inst,
10819 const OperandVector &Operands) {
10820 cvtSDWA(Inst, Operands, SDWAInstType::VOP2, true, true);
10821}
10822
10823void AMDGPUAsmParser::cvtSdwaVOP2e(MCInst &Inst,
10824 const OperandVector &Operands) {
10825 cvtSDWA(Inst, Operands, SDWAInstType::VOP2, false, true);
10826}
10827
10828void AMDGPUAsmParser::cvtSdwaVOPC(MCInst &Inst, const OperandVector &Operands) {
10829 cvtSDWA(Inst, Operands, SDWAInstType::VOPC, isVI());
10830}
10831
10832void AMDGPUAsmParser::cvtSDWA(MCInst &Inst, const OperandVector &Operands,
10833 SDWAInstType BasicInstType, bool SkipDstVcc,
10834 bool SkipSrcVcc) {
10835 using namespace llvm::AMDGPU::SDWA;
10836
10837 OptionalImmIndexMap OptionalIdx;
10838 bool SkipVcc = SkipDstVcc || SkipSrcVcc;
10839 bool SkippedVcc = false;
10840
10841 unsigned I = 1;
10842 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
10843 for (unsigned J = 0; J < Desc.getNumDefs(); ++J) {
10844 ((AMDGPUOperand &)*Operands[I++]).addRegOperands(Inst, 1);
10845 }
10846
10847 for (unsigned E = Operands.size(); I != E; ++I) {
10848 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[I]);
10849 if (SkipVcc && !SkippedVcc && Op.isReg() &&
10850 (Op.getReg() == AMDGPU::VCC || Op.getReg() == AMDGPU::VCC_LO)) {
10851 // VOP2b (v_add_u32, v_sub_u32 ...) sdwa use "vcc" token as dst.
10852 // Skip it if it's 2nd (e.g. v_add_i32_sdwa v1, vcc, v2, v3)
10853 // or 4th (v_addc_u32_sdwa v1, vcc, v2, v3, vcc) operand.
10854 // Skip VCC only if we didn't skip it on previous iteration.
10855 // Note that src0 and src1 occupy 2 slots each because of modifiers.
10856 if (BasicInstType == SDWAInstType::VOP2 &&
10857 ((SkipDstVcc && Inst.getNumOperands() == 1) ||
10858 (SkipSrcVcc && Inst.getNumOperands() == 5))) {
10859 SkippedVcc = true;
10860 continue;
10861 }
10862 if (BasicInstType == SDWAInstType::VOPC && Inst.getNumOperands() == 0) {
10863 SkippedVcc = true;
10864 continue;
10865 }
10866 }
10868 Op.addRegOrImmWithInputModsOperands(Inst, 2);
10869 } else if (Op.isImm()) {
10870 // Handle optional arguments
10871 OptionalIdx[Op.getImmTy()] = I;
10872 } else {
10873 llvm_unreachable("Invalid operand type");
10874 }
10875 SkippedVcc = false;
10876 }
10877
10878 const unsigned Opc = Inst.getOpcode();
10879 if (Opc != AMDGPU::V_NOP_sdwa_gfx10 && Opc != AMDGPU::V_NOP_sdwa_gfx9 &&
10880 Opc != AMDGPU::V_NOP_sdwa_vi) {
10881 // v_nop_sdwa_sdwa_vi/gfx9 has no optional sdwa arguments
10882 switch (BasicInstType) {
10883 case SDWAInstType::VOP1:
10884 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::clamp))
10885 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10886 AMDGPUOperand::ImmTyClamp, 0);
10887
10888 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::omod))
10889 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10890 AMDGPUOperand::ImmTyOModSI, 0);
10891
10892 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::dst_sel))
10893 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10894 AMDGPUOperand::ImmTySDWADstSel, SdwaSel::DWORD);
10895
10896 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::dst_unused))
10897 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10898 AMDGPUOperand::ImmTySDWADstUnused,
10899 DstUnused::UNUSED_PRESERVE);
10900
10901 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10902 AMDGPUOperand::ImmTySDWASrc0Sel, SdwaSel::DWORD);
10903 break;
10904
10905 case SDWAInstType::VOP2:
10906 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10907 AMDGPUOperand::ImmTyClamp, 0);
10908
10909 if (AMDGPU::hasNamedOperand(Inst.getOpcode(), AMDGPU::OpName::omod))
10910 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10911 AMDGPUOperand::ImmTyOModSI, 0);
10912
10913 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10914 AMDGPUOperand::ImmTySDWADstSel, SdwaSel::DWORD);
10915 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10916 AMDGPUOperand::ImmTySDWADstUnused,
10917 DstUnused::UNUSED_PRESERVE);
10918 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10919 AMDGPUOperand::ImmTySDWASrc0Sel, SdwaSel::DWORD);
10920 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10921 AMDGPUOperand::ImmTySDWASrc1Sel, SdwaSel::DWORD);
10922 break;
10923
10924 case SDWAInstType::VOPC:
10925 if (AMDGPU::hasNamedOperand(Inst.getOpcode(), AMDGPU::OpName::clamp))
10926 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10927 AMDGPUOperand::ImmTyClamp, 0);
10928 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10929 AMDGPUOperand::ImmTySDWASrc0Sel, SdwaSel::DWORD);
10930 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10931 AMDGPUOperand::ImmTySDWASrc1Sel, SdwaSel::DWORD);
10932 break;
10933 }
10934 }
10935
10936 // special case v_mac_{f16, f32}:
10937 // it has src2 register operand that is tied to dst operand
10938 if (Inst.getOpcode() == AMDGPU::V_MAC_F32_sdwa_vi ||
10939 Inst.getOpcode() == AMDGPU::V_MAC_F16_sdwa_vi) {
10940 auto *it = Inst.begin();
10941 std::advance(
10942 it, AMDGPU::getNamedOperandIdx(Inst.getOpcode(), AMDGPU::OpName::src2));
10943 Inst.insert(it, Inst.getOperand(0)); // src2 = dst
10944 }
10945}
10946
10947/// Force static initialization.
10948extern "C" LLVM_ABI LLVM_EXTERNAL_VISIBILITY void
10954
10955#define GET_MATCHER_IMPLEMENTATION
10956#define GET_MNEMONIC_SPELL_CHECKER
10957#define GET_MNEMONIC_CHECKER
10958#include "AMDGPUGenAsmMatcher.inc"
10959
10960ParseStatus AMDGPUAsmParser::parseCustomOperand(OperandVector &Operands,
10961 unsigned MCK) {
10962 switch (MCK) {
10963 case MCK_addr64:
10964 return parseTokenOp("addr64", Operands);
10965 case MCK_done:
10966 return parseNamedBit("done", Operands, AMDGPUOperand::ImmTyDone, true);
10967 case MCK_idxen:
10968 return parseTokenOp("idxen", Operands);
10969 case MCK_lds:
10970 return parseNamedBit("lds", Operands, AMDGPUOperand::ImmTyLDS,
10971 /*IgnoreNegative=*/true);
10972 case MCK_offen:
10973 return parseTokenOp("offen", Operands);
10974 case MCK_off:
10975 return parseTokenOp("off", Operands);
10976 case MCK_row_95_en:
10977 return parseNamedBit("row_en", Operands, AMDGPUOperand::ImmTyRowEn, true);
10978 case MCK_gds:
10979 return parseNamedBit("gds", Operands, AMDGPUOperand::ImmTyGDS);
10980 case MCK_tfe:
10981 return parseNamedBit("tfe", Operands, AMDGPUOperand::ImmTyTFE);
10982 }
10983 return tryCustomParseOperand(Operands, MCK);
10984}
10985
10986// This function should be defined after auto-generated include so that we have
10987// MatchClassKind enum defined
10988unsigned AMDGPUAsmParser::validateTargetOperandClass(MCParsedAsmOperand &Op,
10989 unsigned Kind) {
10990 // Tokens like "glc" would be parsed as immediate operands in ParseOperand().
10991 // But MatchInstructionImpl() expects to meet token and fails to validate
10992 // operand. This method checks if we are given immediate operand but expect to
10993 // get corresponding token.
10994 AMDGPUOperand &Operand = (AMDGPUOperand &)Op;
10995 switch (Kind) {
10996 case MCK_addr64:
10997 return Operand.isAddr64() ? Match_Success : Match_InvalidOperand;
10998 case MCK_gds:
10999 return Operand.isGDS() ? Match_Success : Match_InvalidOperand;
11000 case MCK_lds:
11001 return Operand.isLDS() ? Match_Success : Match_InvalidOperand;
11002 case MCK_idxen:
11003 return Operand.isIdxen() ? Match_Success : Match_InvalidOperand;
11004 case MCK_offen:
11005 return Operand.isOffen() ? Match_Success : Match_InvalidOperand;
11006 case MCK_tfe:
11007 return Operand.isTFE() ? Match_Success : Match_InvalidOperand;
11008 case MCK_done:
11009 return Operand.isDone() ? Match_Success : Match_InvalidOperand;
11010 case MCK_row_95_en:
11011 return Operand.isRowEn() ? Match_Success : Match_InvalidOperand;
11012 case MCK_SSrc_b32:
11013 // When operands have expression values, they will return true for isToken,
11014 // because it is not possible to distinguish between a token and an
11015 // expression at parse time. MatchInstructionImpl() will always try to
11016 // match an operand as a token, when isToken returns true, and when the
11017 // name of the expression is not a valid token, the match will fail,
11018 // so we need to handle it here.
11019 return Operand.isSSrc_b32() ? Match_Success : Match_InvalidOperand;
11020 case MCK_SSrc_f32:
11021 return Operand.isSSrc_f32() ? Match_Success : Match_InvalidOperand;
11022 case MCK_SOPPBrTarget:
11023 return Operand.isSOPPBrTarget() ? Match_Success : Match_InvalidOperand;
11024 case MCK_VReg32OrOff:
11025 return Operand.isVReg32OrOff() ? Match_Success : Match_InvalidOperand;
11026 case MCK_InterpSlot:
11027 return Operand.isInterpSlot() ? Match_Success : Match_InvalidOperand;
11028 case MCK_InterpAttr:
11029 return Operand.isInterpAttr() ? Match_Success : Match_InvalidOperand;
11030 case MCK_InterpAttrChan:
11031 return Operand.isInterpAttrChan() ? Match_Success : Match_InvalidOperand;
11032 case MCK_SReg_64:
11033 case MCK_SReg_64_XEXEC:
11034 // Null is defined as a 32-bit register but
11035 // it should also be enabled with 64-bit operands or larger.
11036 // The following code enables it for SReg_64 and larger operands
11037 // used as source and destination. Remaining source
11038 // operands are handled in isInlinableImm.
11039 case MCK_SReg_96:
11040 case MCK_SReg_128:
11041 case MCK_SReg_256:
11042 case MCK_SReg_512:
11043 return Operand.isNull() ? Match_Success : Match_InvalidOperand;
11044 default:
11045 return Match_InvalidOperand;
11046 }
11047}
11048
11049//===----------------------------------------------------------------------===//
11050// endpgm
11051//===----------------------------------------------------------------------===//
11052
11053ParseStatus AMDGPUAsmParser::parseEndpgm(OperandVector &Operands) {
11054 SMLoc S = getLoc();
11055 int64_t Imm = 0;
11056
11057 if (!parseExpr(Imm)) {
11058 // The operand is optional, if not present default to 0
11059 Imm = 0;
11060 }
11061
11062 if (!isUInt<16>(Imm))
11063 return Error(S, "expected a 16-bit value");
11064
11065 Operands.push_back(
11066 AMDGPUOperand::CreateImm(this, Imm, S, AMDGPUOperand::ImmTyEndpgm));
11067 return ParseStatus::Success;
11068}
11069
11070bool AMDGPUOperand::isEndpgm() const { return isImmTy(ImmTyEndpgm); }
11071
11072//===----------------------------------------------------------------------===//
11073// Split Barrier
11074//===----------------------------------------------------------------------===//
11075
11076bool AMDGPUOperand::isSplitBarrier() const {
11077 if (!isImm())
11078 return false;
11079
11080 int64_t Imm = getImm();
11083}
#define Success
static const TargetRegisterClass * getRegClass(const MachineInstr &MI, Register Reg)
unsigned RegSize
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
SmallVector< int16_t, MAX_SRC_OPERANDS_NUM > OperandIndices
static bool checkWriteLane(const MCInst &Inst)
static bool getRegNum(StringRef Str, unsigned &Num)
static void addSrcModifiersAndSrc(MCInst &Inst, const OperandVector &Operands, unsigned i, unsigned Opc, AMDGPU::OpName OpName)
static constexpr RegInfo RegularRegisters[]
static const RegInfo * getRegularRegInfo(StringRef Str)
static ArrayRef< unsigned > getAllVariants()
static OperandIndices getSrcOperandIndices(unsigned Opcode, bool AddMandatoryLiterals=false)
static int IsAGPROperand(const MCInst &Inst, AMDGPU::OpName Name, const MCRegisterInfo *MRI)
static bool IsMovrelsSDWAOpcode(const unsigned Opcode)
static const fltSemantics * getFltSemantics(unsigned Size)
static bool isRegularReg(RegisterKind Kind)
LLVM_ABI LLVM_EXTERNAL_VISIBILITY void LLVMInitializeAMDGPUAsmParser()
Force static initialization.
static bool ConvertOmodMul(int64_t &Mul)
#define PARSE_BITS_ENTRY(FIELD, ENTRY, VALUE, RANGE)
static bool isInlineableLiteralOp16(int64_t Val, MVT VT, bool HasInv2Pi)
static bool canLosslesslyConvertToFPType(APFloat &FPLiteral, MVT VT)
static bool AMDGPUCheckMnemonic(StringRef Mnemonic, const FeatureBitset &AvailableFeatures, unsigned VariantID)
static void applyMnemonicAliases(StringRef &Mnemonic, const FeatureBitset &Features, unsigned VariantID)
constexpr unsigned MAX_SRC_OPERANDS_NUM
#define EXPR_RESOLVE_OR_ERROR(RESOLVED)
static bool ConvertOmodDiv(int64_t &Div)
static bool IsRevOpcode(const unsigned Opcode)
static bool encodeCnt(const AMDGPU::IsaVersion ISA, int64_t &IntVal, int64_t CntVal, bool Saturate, unsigned(*encode)(const IsaVersion &Version, unsigned, unsigned), unsigned(*decode)(const IsaVersion &Version, unsigned))
static MCRegister getSpecialRegForName(StringRef RegName)
static void addOptionalImmOperand(MCInst &Inst, const OperandVector &Operands, AMDGPUAsmParser::OptionalImmIndexMap &OptionalIdx, AMDGPUOperand::ImmTy ImmT, int64_t Default=0, std::optional< unsigned > InsertAt=std::nullopt)
static void cvtVOP3DstOpSelOnly(MCInst &Inst, const MCRegisterInfo &MRI)
static bool isRegOrImmWithInputMods(const MCInstrDesc &Desc, unsigned OpNum)
static const fltSemantics * getOpFltSemantics(uint8_t OperandType)
static bool isInvalidVOPDY(const OperandVector &Operands, uint64_t InvalidOprIdx)
static std::string AMDGPUMnemonicSpellCheck(StringRef S, const FeatureBitset &FBS, unsigned VariantID=0)
static LLVM_READNONE unsigned encodeBitmaskPerm(const unsigned AndMask, const unsigned OrMask, const unsigned XorMask)
static bool isSafeTruncation(int64_t Val, unsigned Size)
unsigned uint64_t
AMDHSA kernel descriptor MCExpr struct for use in MC layer.
Provides AMDGPU specific target descriptions.
AMDGPU metadata definitions and in-memory representations.
Enums shared between the AMDGPU backend (LLVM) and the ELF linker (LLD) for the .amdgpu....
AMDHSA kernel descriptor definitions.
static bool parseExpr(MCAsmParser &MCParser, const MCExpr *&Value, raw_ostream &Err)
MC layer struct for AMDGPUMCKernelCodeT, provides MCExpr functionality where required.
@ AMD_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32
This file declares a class to represent arbitrary precision floating point values and provide a varie...
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
#define LLVM_READNONE
Definition Compiler.h:331
#define LLVM_ABI
Definition Compiler.h:215
#define LLVM_EXTERNAL_VISIBILITY
Definition Compiler.h:132
@ Default
#define Check(C,...)
static llvm::Expected< InlineInfo > decode(GsymDataExtractor &Data, uint64_t &Offset, uint64_t BaseAddr)
Decode an InlineInfo in Data at the specified offset.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define RegName(no)
Loop::LoopBounds::Direction Direction
Definition LoopInfo.cpp:253
static bool hasFeature(StringRef Feature, const FeatureBitset &FeatureBits, ArrayRef< SubtargetFeatureKV > ProcFeatures)
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
static bool isReg(const MCInst &MI, unsigned OpNo)
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t IntrinsicInst * II
#define P(N)
if(PassOpts->AAPipeline)
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
SI Fold Operands
Interface definition for SIInstrInfo.
Func getContext().diagnose(DiagnosticInfoUnsupported(Func
const char * Msg
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
This file implements the SmallBitVector class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static void initialize(TargetLibraryInfoImpl &TLI, const Triple &T, const llvm::StringTable &StandardNames, VectorLibrary VecLib)
Initialize the set of available library functions based on the specified target triple.
BinaryOperator * Mul
static const char * getRegisterName(MCRegister Reg)
static const AMDGPUMCExpr * createMax(ArrayRef< const MCExpr * > Args, MCContext &Ctx)
static unsigned getNumExpectedArgs(VariantKind Kind)
static const AMDGPUMCExpr * createLit(LitModifier Lit, int64_t Value, MCContext &Ctx)
static const AMDGPUMCExpr * create(VariantKind Kind, ArrayRef< const MCExpr * > Args, MCContext &Ctx)
static const AMDGPUMCExpr * createExtraSGPRs(const MCExpr *VCCUsed, const MCExpr *FlatScrUsed, bool XNACKUsed, MCContext &Ctx)
Allow delayed MCExpr resolve of ExtraSGPRs (in case VCCUsed or FlatScrUsed are unresolvable but neede...
static const AMDGPUMCExpr * createAlignTo(const MCExpr *Value, const MCExpr *Align, MCContext &Ctx)
static std::optional< TargetID > parseTargetIDString(StringRef TargetIDDirective)
Parse and validate a TargetID from a full "<triple>-<processor>:<features>" directive string.
TargetIDSetting getXnackSetting() const
StringRef getTargetTripleString() const
std::string toString() const
TargetIDSetting getSramEccSetting() const
static const fltSemantics & IEEEsingle()
Definition APFloat.h:304
static const fltSemantics & BFloat()
Definition APFloat.h:303
static const fltSemantics & IEEEdouble()
Definition APFloat.h:305
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:361
static const fltSemantics & IEEEhalf()
Definition APFloat.h:302
opStatus
IEEE-754R 7: Default exception handling.
Definition APFloat.h:377
LLVM_ABI opStatus convert(const fltSemantics &ToSemantics, roundingMode RM, bool *losesInfo)
Definition APFloat.cpp:6034
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
ArrayRef< T > take_front(size_t N=1) const
Return a copy of *this with only the first N elements.
Definition ArrayRef.h:218
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
Definition ArrayRef.h:194
const T & front() const
Get the first element.
Definition ArrayRef.h:144
iterator end() const
Definition ArrayRef.h:130
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
StringRef getString() const
Get the string for the current token, this includes all characters (for example, the quotes on string...
Definition MCAsmMacro.h:103
bool is(TokenKind K) const
Definition MCAsmMacro.h:75
Register getReg() const
Container class for subtarget features.
constexpr bool test(unsigned I) const
constexpr FeatureBitset & flip(unsigned I)
void printExpr(raw_ostream &, const MCExpr &) const
virtual void Initialize(MCAsmParser &Parser)
Initialize the extension for parsing using the given Parser.
static const MCBinaryExpr * createAdd(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx, SMLoc Loc=SMLoc())
Definition MCExpr.h:342
static const MCBinaryExpr * createDiv(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:352
static const MCBinaryExpr * createSub(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:427
static LLVM_ABI const MCConstantExpr * create(int64_t Value, MCContext &Ctx, bool PrintInHex=false, unsigned SizeInBytes=0)
Definition MCExpr.cpp:212
Context object for machine code objects.
Definition MCContext.h:83
LLVM_ABI MCSymbol * getOrCreateSymbol(const Twine &Name)
Lookup the symbol inside with the specified Name.
Instances of this class represent a single low-level machine instruction.
Definition MCInst.h:188
unsigned getNumOperands() const
Definition MCInst.h:212
SMLoc getLoc() const
Definition MCInst.h:208
void setLoc(SMLoc loc)
Definition MCInst.h:207
unsigned getOpcode() const
Definition MCInst.h:202
iterator insert(iterator I, const MCOperand &Op)
Definition MCInst.h:232
void addOperand(const MCOperand Op)
Definition MCInst.h:215
iterator begin()
Definition MCInst.h:227
size_t size() const
Definition MCInst.h:226
const MCOperand & getOperand(unsigned i) const
Definition MCInst.h:210
Describe properties that are true of each instruction in the target description file.
const MCInstrDesc & get(unsigned Opcode) const
Return the machine instruction descriptor that corresponds to the specified instruction opcode.
Definition MCInstrInfo.h:89
const int16_t * getRegClassByHwModeTable(unsigned ModeId) const
Definition MCInstrInfo.h:70
int16_t getOpRegClassID(const MCOperandInfo &OpInfo, unsigned HwModeId) const
Return the ID of the register class to use for OpInfo, for the active HwMode HwModeId.
Definition MCInstrInfo.h:79
Instances of this class represent operands of the MCInst class.
Definition MCInst.h:40
void setImm(int64_t Val)
Definition MCInst.h:89
static MCOperand createExpr(const MCExpr *Val)
Definition MCInst.h:166
int64_t getImm() const
Definition MCInst.h:84
static MCOperand createReg(MCRegister Reg)
Definition MCInst.h:138
static MCOperand createImm(int64_t Val)
Definition MCInst.h:145
bool isImm() const
Definition MCInst.h:66
void setReg(MCRegister Reg)
Set the register number.
Definition MCInst.h:79
bool isReg() const
Definition MCInst.h:65
MCRegister getReg() const
Returns the register number.
Definition MCInst.h:73
const MCExpr * getExpr() const
Definition MCInst.h:118
bool isExpr() const
Definition MCInst.h:69
MCParsedAsmOperand - This abstract class represents a source-level assembly instruction operand.
MCRegisterClass - Base class of TargetRegisterClass.
MCRegister getRegister(unsigned i) const
getRegister - Return the specified register in the class.
unsigned getNumRegs() const
getNumRegs - Return the number of registers in this class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
MCRegisterInfo base class - We assume that the target defines a static array of MCRegisterDesc object...
bool regsOverlap(MCRegister RegA, MCRegister RegB) const
Returns true if the two registers are equal or alias each other.
const MCRegisterClass & getRegClass(unsigned i) const
Returns the register class associated with the enumeration value.
MCRegister getSubReg(MCRegister Reg, unsigned Idx) const
Returns the physical register number of sub-register "Index" for physical register RegNo.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
constexpr bool isValid() const
Definition MCRegister.h:84
Generic base class for all target subtargets.
MCSymbol - Instances of this class represent a symbol name in the MC file, and MCSymbols are created ...
Definition MCSymbol.h:42
StringRef getName() const
getName - Get the symbol name.
Definition MCSymbol.h:188
bool isVariable() const
isVariable - Check if this is a variable symbol.
Definition MCSymbol.h:267
LLVM_ABI void setVariableValue(const MCExpr *Value)
Definition MCSymbol.cpp:50
void setRedefinable(bool Value)
Mark this symbol as redefinable.
Definition MCSymbol.h:210
const MCExpr * getVariableValue() const
Get the expression of the variable symbol.
Definition MCSymbol.h:270
MCTargetAsmParser - Generic interface to target specific assembly parsers.
Machine Value Type.
SimpleValueType SimpleTy
uint64_t getScalarSizeInBits() const
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
Ternary parse status returned by various parse* methods.
constexpr bool isFailure() const
static constexpr StatusTy Failure
constexpr bool isSuccess() const
static constexpr StatusTy Success
static constexpr StatusTy NoMatch
constexpr bool isNoMatch() const
constexpr unsigned id() const
Definition Register.h:100
Represents a location in source code.
Definition SMLoc.h:22
static SMLoc getFromPointer(const char *Ptr)
Definition SMLoc.h:35
constexpr const char * getPointer() const
Definition SMLoc.h:33
constexpr bool isValid() const
Definition SMLoc.h:28
SMLoc Start
Definition SMLoc.h:49
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
A wrapper around a string literal that serves as a proxy for constructing global tables of StringRefs...
Definition StringRef.h:888
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
bool consume_back(StringRef Suffix)
Returns true if this StringRef has the given suffix and removes that suffix.
Definition StringRef.h:691
constexpr StringRef substr(size_t Start, size_t N=npos) const
Return a reference to the substring from [Start, Start + N).
Definition StringRef.h:597
bool starts_with(StringRef Prefix) const
Check if this string starts with the given Prefix.
Definition StringRef.h:258
constexpr bool empty() const
Check if the string is empty.
Definition StringRef.h:141
StringRef drop_front(size_t N=1) const
Return a StringRef equal to 'this' but with the first N elements dropped.
Definition StringRef.h:635
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
constexpr const char * data() const
Get a pointer to the start of the string (which may not be null terminated).
Definition StringRef.h:138
bool ends_with(StringRef Suffix) const
Check if this string ends with the given Suffix.
Definition StringRef.h:270
bool consume_front(char Prefix)
Returns true if this StringRef has the given prefix and removes that prefix.
Definition StringRef.h:661
bool contains(StringRef key) const
Check if the set contains the given key.
Definition StringSet.h:60
std::pair< typename Base::iterator, bool > insert(StringRef key)
Definition StringSet.h:39
A switch()-like statement whose cases are string literals.
StringSwitch & Case(StringLiteral S, T Value)
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
LLVM_ABI std::string str() const
Return the twine contents as a std::string.
Definition Twine.cpp:17
std::pair< iterator, bool > insert(const ValueT &V)
Definition DenseSet.h:209
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
CallInst * Call
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
int encodeDepCtr(const StringRef Name, int64_t Val, unsigned &UsedOprMask, const MCSubtargetInfo &STI)
int getDefaultDepCtrEncoding(const MCSubtargetInfo &STI)
bool isSupportedTgtId(unsigned Id, const MCSubtargetInfo &STI)
unsigned getTgtId(const StringRef Name)
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char NumSGPRs[]
Key for Kernel::CodeProps::Metadata::mNumSGPRs.
constexpr char SymbolName[]
Key for Kernel::Metadata::mSymbolName.
constexpr char AssemblerDirectiveBegin[]
HSA metadata beginning assembler directive.
constexpr char AssemblerDirectiveEnd[]
HSA metadata ending assembler directive.
constexpr char AssemblerDirectiveBegin[]
Old HSA metadata beginning assembler directive for V2.
int64_t getHwregId(StringRef Name, const MCSubtargetInfo &STI)
unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI, std::optional< bool > EnableWavefrontSize32)
unsigned getSGPREncodingGranule(const MCSubtargetInfo &STI)
bool targetIDSettingsConflict(TargetIDSetting Lhs, TargetIDSetting Rhs)
Returns true if Lhs and Rhs are incompatible (both specific but different).
unsigned getLocalMemorySize(const MCSubtargetInfo &STI)
unsigned getDefaultFormatEncoding(const MCSubtargetInfo &STI)
int64_t convertDfmtNfmt2Ufmt(unsigned Dfmt, unsigned Nfmt, const MCSubtargetInfo &STI)
int64_t encodeDfmtNfmt(unsigned Dfmt, unsigned Nfmt)
int64_t getUnifiedFormat(const StringRef Name, const MCSubtargetInfo &STI)
bool isValidFormatEncoding(unsigned Val, const MCSubtargetInfo &STI)
int64_t getNfmt(const StringRef Name, const MCSubtargetInfo &STI)
int64_t getDfmt(const StringRef Name)
constexpr char AssemblerDirective[]
PAL metadata (old linear format) assembler directive.
constexpr char AssemblerDirectiveBegin[]
PAL metadata (new MsgPack format) beginning assembler directive.
constexpr char AssemblerDirectiveEnd[]
PAL metadata (new MsgPack format) ending assembler directive.
int64_t getMsgOpId(int64_t MsgId, StringRef Name, const MCSubtargetInfo &STI)
Map from a symbolic name for a sendmsg operation to the operation portion of the immediate encoding.
int64_t getMsgId(StringRef Name, const MCSubtargetInfo &STI)
Map from a symbolic name for a msg_id to the message portion of the immediate encoding.
uint64_t encodeMsg(uint64_t MsgId, uint64_t OpId, uint64_t StreamId)
bool msgSupportsStream(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI)
bool isValidMsgId(int64_t MsgId, const MCSubtargetInfo &STI)
bool isValidMsgStream(int64_t MsgId, int64_t OpId, int64_t StreamId, const MCSubtargetInfo &STI, bool Strict)
bool msgRequiresOp(int64_t MsgId, const MCSubtargetInfo &STI)
bool isValidMsgOp(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI, bool Strict)
ArrayRef< GFXVersion > getGFXVersions()
constexpr unsigned COMPONENTS[]
constexpr const char *const ModMatrixFmt[]
constexpr const char *const ModMatrixScaleFmt[]
constexpr const char *const ModMatrixScale[]
bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi)
bool isGFX10_BEncoding(const MCSubtargetInfo &STI)
bool isInlineValue(MCRegister Reg)
bool isPKFMACF16InlineConstant(uint32_t Literal, bool IsGFX11Plus)
LLVM_READONLY const MIMGInfo * getMIMGInfo(unsigned Opc)
bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi)
bool isSGPR(MCRegister Reg, const MCRegisterInfo *TRI)
Is Reg - scalar register.
MCRegister getMCReg(MCRegister Reg, const MCSubtargetInfo &STI)
If Reg is a pseudo reg, return the correct hardware register given STI otherwise return Reg.
FuncInfoFlags
Per-function flags packed into INFO_FLAGS entries.
uint8_t wmmaScaleF8F6F4FormatToNumRegs(unsigned Fmt)
const int OPR_ID_UNSUPPORTED
bool isInlinableLiteralV2I16(uint32_t Literal)
bool isHi16Reg(MCRegister Reg, const MCRegisterInfo &MRI)
unsigned getTemporalHintType(const MCInstrDesc TID)
bool isGFX10(const MCSubtargetInfo &STI)
LLVM_READONLY bool isLitExpr(const MCExpr *Expr)
bool isInlinableLiteralV2BF16(uint32_t Literal)
LLVM_ABI bool isCPUValidForSubArch(Triple::SubArchType SubArch, GPUKind AK)
Return true if the GPU AK is usable with the triple subarch SubArch.
unsigned getMaxNumUserSGPRs(const MCSubtargetInfo &STI)
unsigned getNumFlatOffsetBits(const MCSubtargetInfo &ST)
For pre-GFX12 FLAT instructions the offset must be positive; MSB is ignored and forced to zero.
bool hasA16(const MCSubtargetInfo &STI)
bool isLegalSMRDEncodedSignedOffset(const MCSubtargetInfo &ST, int64_t EncodedOffset, bool IsBuffer)
bool isGFX12Plus(const MCSubtargetInfo &STI)
unsigned getNSAMaxSize(const MCSubtargetInfo &STI, bool HasSampler)
bool hasPackedD16(const MCSubtargetInfo &STI)
bool isGFX940(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2F16(uint32_t Literal)
bool isHsaAbi(const MCSubtargetInfo &STI)
bool isGFX11(const MCSubtargetInfo &STI)
const int OPR_VAL_INVALID
bool getSMEMIsBuffer(unsigned Opc)
bool isPackedSingleSGPRFP32Inst(unsigned Opc)
The opcode is a packed fp32 instruction which only reads low 32 bits of a scalar operand and propagat...
bool isGFX13(const MCSubtargetInfo &STI)
LLVM_ABI unsigned getAddressableNumSGPRs(GPUKind AK)
uint8_t mfmaScaleF8F6F4FormatToNumRegs(unsigned EncodingVal)
LLVM_ABI IsaVersion getIsaVersion(StringRef GPU)
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
bool isValid32BitLiteral(uint64_t Val, bool IsFP64)
LLVM_ABI unsigned getTotalNumVGPRs(GPUKind AK, bool IsWave32)
CanBeVOPD getCanBeVOPD(unsigned Opc, unsigned EncodingFamily, bool VOPD3)
LLVM_READNONE bool isLegalDPALU_DPPControl(const MCSubtargetInfo &ST, unsigned DC)
bool isSI(const MCSubtargetInfo &STI)
bool hasPrivateApertureRegs(const MCSubtargetInfo &STI)
unsigned decodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt)
unsigned getWaitcntBitMask(const IsaVersion &Version)
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
bool isGFX9(const MCSubtargetInfo &STI)
unsigned getVOPDEncodingFamily(const MCSubtargetInfo &ST)
bool isKImmOperand(const MCInstrDesc &Desc, unsigned OpNo)
Is this a KImm operand?
GPUKind
GPU kinds supported by the AMDGPU target.
bool isGFX90A(const MCSubtargetInfo &STI)
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfoByEncoding(uint8_t DimEnc)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
bool isGFX12(const MCSubtargetInfo &STI)
unsigned encodeExpcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Expcnt)
bool hasMAIInsts(const MCSubtargetInfo &STI)
constexpr bool isSISrcOperand(const MCOperandInfo &OpInfo)
Is this an AMDGPU specific source operand?
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfoByAsmSuffix(StringRef AsmSuffix)
bool hasMIMG_R128(const MCSubtargetInfo &STI)
LLVM_ABI GPUKind parseArchAMDGCN(StringRef CPU)
bool hasG16(const MCSubtargetInfo &STI)
unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode, const MIMGDimInfo *Dim, bool IsA16, bool IsG16Supported)
bool isGFX13Plus(const MCSubtargetInfo &STI)
bool hasArchitectedFlatScratch(const MCSubtargetInfo &STI)
LLVM_READONLY int64_t getLitValue(const MCExpr *Expr)
bool isGFX11Plus(const MCSubtargetInfo &STI)
bool isSISrcFPOperand(const MCInstrDesc &Desc, unsigned OpNo)
Is this floating-point operand?
bool isGFX10Plus(const MCSubtargetInfo &STI)
AMDGPU::TargetID TargetID
int64_t encode32BitLiteral(int64_t Imm, OperandType Type, bool IsLit)
bool isValidWMMAScaleFmtCombination(unsigned AFmt, unsigned AScale, unsigned BFmt, unsigned BScale)
@ OPERAND_REG_IMM_V2FP64
Definition SIDefines.h:441
@ OPERAND_KIMM32
Operand with 32-bit immediate that uses the constant bus.
Definition SIDefines.h:459
@ OPERAND_REG_IMM_INT64
Definition SIDefines.h:426
@ OPERAND_REG_IMM_V2FP16
Definition SIDefines.h:434
@ OPERAND_REG_INLINE_C_FP64
Definition SIDefines.h:450
@ OPERAND_REG_IMM_NOINLINE_FP16
Definition SIDefines.h:432
@ OPERAND_REG_INLINE_C_BF16
Definition SIDefines.h:447
@ OPERAND_REG_INLINE_C_V2BF16
Definition SIDefines.h:452
@ OPERAND_REG_IMM_V2INT64
Definition SIDefines.h:437
@ OPERAND_REG_IMM_V2INT16
Definition SIDefines.h:436
@ OPERAND_REG_IMM_BF16
Definition SIDefines.h:430
@ OPERAND_REG_IMM_INT32
Operands with register, 32-bit, or 64-bit immediate.
Definition SIDefines.h:425
@ OPERAND_REG_IMM_V2BF16
Definition SIDefines.h:433
@ OPERAND_REG_IMM_FP16
Definition SIDefines.h:431
@ OPERAND_REG_IMM_V2FP16_SPLAT
Definition SIDefines.h:435
@ OPERAND_REG_INLINE_C_INT64
Definition SIDefines.h:446
@ OPERAND_REG_INLINE_C_INT16
Operands with register or inline constant.
Definition SIDefines.h:444
@ OPERAND_REG_IMM_NOINLINE_V2FP16
Definition SIDefines.h:438
@ OPERAND_REG_IMM_FP64
Definition SIDefines.h:429
@ OPERAND_REG_INLINE_C_V2FP16
Definition SIDefines.h:453
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
Definition SIDefines.h:464
@ OPERAND_REG_INLINE_AC_FP32
Definition SIDefines.h:465
@ OPERAND_REG_IMM_V2INT32
Definition SIDefines.h:439
@ OPERAND_REG_IMM_FP32
Definition SIDefines.h:428
@ OPERAND_REG_INLINE_C_FP32
Definition SIDefines.h:449
@ OPERAND_REG_INLINE_C_INT32
Definition SIDefines.h:445
@ OPERAND_REG_INLINE_C_V2INT16
Definition SIDefines.h:451
@ OPERAND_REG_IMM_V2FP32
Definition SIDefines.h:440
@ OPERAND_REG_INLINE_AC_FP64
Definition SIDefines.h:466
@ OPERAND_REG_INLINE_C_FP16
Definition SIDefines.h:448
@ OPERAND_REG_IMM_INT16
Definition SIDefines.h:427
@ OPERAND_INLINE_SPLIT_BARRIER_INT32
Definition SIDefines.h:456
constexpr bool isBF16SrcOperand(const MCOperandInfo &OpInfo)
Is this a scalar (i.e. not packed) bf16 source operand?
bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII, const MCSubtargetInfo &ST)
LLVM_ABI StringRef getArchNameAMDGCN(GPUKind AK)
bool hasGDS(const MCSubtargetInfo &STI)
bool isLegalSMRDEncodedUnsignedOffset(const MCSubtargetInfo &ST, int64_t EncodedOffset)
bool isGFX9Plus(const MCSubtargetInfo &STI)
bool hasDPPSrc1SGPR(const MCSubtargetInfo &STI)
const int OPR_ID_DUPLICATE
bool isVOPD(unsigned Opc)
VOPD::InstInfo getVOPDInstInfo(const MCInstrDesc &OpX, const MCInstrDesc &OpY)
unsigned encodeVmcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Vmcnt)
unsigned decodeExpcnt(const IsaVersion &Version, unsigned Waitcnt)
bool isGFX1250(const MCSubtargetInfo &STI)
const MIMGBaseOpcodeInfo * getMIMGBaseOpcode(unsigned Opc)
bool isVI(const MCSubtargetInfo &STI)
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
MCRegister mc2PseudoReg(MCRegister Reg)
Convert hardware register Reg to a pseudo register.
unsigned hasKernargPreload(const MCSubtargetInfo &STI)
bool supportsWGP(const MCSubtargetInfo &STI)
bool isMAC(unsigned Opc)
LLVM_READNONE unsigned getOperandSize(const MCOperandInfo &OpInfo)
bool isCI(const MCSubtargetInfo &STI)
unsigned encodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Lgkmcnt)
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
const int OPR_ID_UNKNOWN
bool isGFX1250Plus(const MCSubtargetInfo &STI)
bool hasPopsExitingWaveID(const MCSubtargetInfo &STI)
unsigned decodeVmcnt(const IsaVersion &Version, unsigned Waitcnt)
bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi)
bool hasVOPD(const MCSubtargetInfo &STI)
bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi)
Is this literal inlinable.
bool isPermlane16(unsigned Opc)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ STT_AMDGPU_HSA_KERNEL
Definition ELF.h:1447
@ UNDEF
UNDEF - An undefined node.
Definition ISDOpcodes.h:235
@ OPERAND_IMMEDIATE
Definition MCInstrDesc.h:61
Predicate getPredicate(unsigned Condition, unsigned Hint)
Return predicate consisting of specified condition and hint bits.
void validate(const Triple &TT, const FeatureBitset &FeatureBits)
constexpr bool isAtomicRet(const T &...O)
Definition SIDefines.h:361
constexpr bool isVOPC(const T &...O)
Definition SIDefines.h:233
constexpr bool isVOP3(const T &...O)
Definition SIDefines.h:236
constexpr bool isVOP1(const T &...O)
Definition SIDefines.h:227
constexpr bool usesTENSOR_CNT(const T &...O)
Definition SIDefines.h:307
constexpr bool isMAI(const T &...O)
Definition SIDefines.h:349
constexpr bool isVOP2(const T &...O)
Definition SIDefines.h:230
constexpr bool isSWMMAC(const T &...O)
Definition SIDefines.h:376
constexpr bool isSOP2(const T &...O)
Definition SIDefines.h:215
constexpr bool isFLAT(const T &...O)
Definition SIDefines.h:283
constexpr bool isVOP3P(const T &...O)
Definition SIDefines.h:239
constexpr bool isBuffer(const T &...O)
Definition SIDefines.h:264
constexpr bool hasIntClamp(const T &...O)
Definition SIDefines.h:328
constexpr bool isAtomicNoRet(const T &...O)
Definition SIDefines.h:358
constexpr bool isSMRD(const T &...O)
Definition SIDefines.h:268
constexpr bool isVOP3Like(const T &...O)
Definition SIDefines.h:242
constexpr bool isMIMG(const T &...O)
Definition SIDefines.h:271
constexpr bool isVMEM(const T &...O)
Definition SIDefines.h:401
constexpr bool isImage(const T &...O)
Definition SIDefines.h:397
constexpr bool isWMMA(const T &...O)
Definition SIDefines.h:364
constexpr bool isVOPD3(const T &...O)
Definition SIDefines.h:379
constexpr bool isGWS(const T &...O)
Definition SIDefines.h:373
constexpr bool isMUBUF(const T &...O)
Definition SIDefines.h:258
constexpr bool isSDWA(const T &...O)
Definition SIDefines.h:249
constexpr bool isSOPC(const T &...O)
Definition SIDefines.h:218
constexpr bool isDOT(const T &...O)
Definition SIDefines.h:352
constexpr bool isVSAMPLE(const T &...O)
Definition SIDefines.h:277
constexpr bool isDS(const T &...O)
Definition SIDefines.h:286
constexpr bool isAtomic(const T &...O)
Definition SIDefines.h:390
constexpr bool isGather4(const T &...O)
Definition SIDefines.h:304
constexpr bool isPacked(const T &...O)
Definition SIDefines.h:337
constexpr bool isDPP(const T &...O)
Definition SIDefines.h:252
constexpr bool isSegmentSpecificFLAT(const T &...O)
Definition SIDefines.h:393
@ Valid
The data is already valid.
EnumSet< Modifier > Modifiers
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
bool isNull(StringRef S)
Definition YAMLTraits.h:571
This is an optimization pass for GlobalISel generic memory operations.
bool errorToBool(Error Err)
Helper for converting an Error to a bool.
Definition Error.h:1129
@ Offset
Definition DWP.cpp:577
StringMapEntry< Value * > ValueName
Definition Value.h:56
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
Printable print(const GCNRegPressure &RP, const GCNSubtarget *ST=nullptr, unsigned DynamicVGPRBlockSize=0)
unsigned encode(MaybeAlign A)
Returns a representation of the alignment that encodes undefined as 0.
Definition Alignment.h:206
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
static bool isMem(const MachineInstr &MI, unsigned Op)
LLVM_ABI std::pair< StringRef, StringRef > getToken(StringRef Source, StringRef Delimiters=" \t\n\v\f\r")
getToken - This function extracts one token from source, ignoring any leading characters that appear ...
static StringRef getCPU(StringRef CPU)
Processes a CPU name.
testing::Matcher< const detail::ErrorHolder & > Failed()
Definition Error.h:198
LLVM_ABI void PrintError(const Twine &Msg)
Definition Error.cpp:104
constexpr bool isUIntN(unsigned N, uint64_t x)
Checks if an unsigned integer fits into the given (dynamic) bit width.
Definition MathExtras.h:244
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
T bit_ceil(T Value)
Returns the smallest integral power of two no smaller than Value if Value is nonzero.
Definition bit.h:362
Op::Description Desc
Target & getTheR600Target()
The target for R600 GPUs.
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
Definition bit.h:156
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
SmallVectorImpl< std::unique_ptr< MCParsedAsmOperand > > OperandVector
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
MutableArrayRef(T &OneElt) -> MutableArrayRef< T >
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
Definition ModRef.h:74
Target & getTheGCNTarget()
The target for GCN GPUs.
@ Sub
Subtraction of integers.
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
Definition MCRegister.h:21
DWARFExpression::Operation Op
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
unsigned M0(unsigned Val)
Definition VE.h:376
ArrayRef(const T &OneElt) -> ArrayRef< T >
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
constexpr bool isIntN(unsigned N, int64_t x)
Checks if an signed integer fits into the given (dynamic) bit width.
Definition MathExtras.h:249
Target & getTheGCNLegacyTarget()
The target for GCN GPUs, registered under the legacy "amdgcn" architecture name for use with -march.
@ Enabled
Convert any .debug_str_offsets tables to DWARF64 if needed.
Definition DWP.h:31
@ Default
The result value is uniform if and only if all operands are uniform.
Definition Uniformity.h:20
#define N
RegisterKind Kind
StringLiteral Name
void initDefault(const MCSubtargetInfo &STI, MCContext &Ctx, bool InitMCExpr=true)
void validate(const MCSubtargetInfo *STI, MCContext &Ctx)
SmallVector< std::pair< MCSymbol *, std::string >, 4 > IndirectCalls
SmallVector< std::pair< MCSymbol *, MCSymbol * >, 8 > Calls
SmallVector< FuncInfo, 8 > Funcs
SmallVector< std::pair< MCSymbol *, std::string >, 4 > TypeIds
SmallVector< std::pair< MCSymbol *, MCSymbol * >, 4 > Uses
Instruction set architecture version.
static void bits_set(const MCExpr *&Dst, const MCExpr *Value, uint32_t Shift, uint32_t Mask, MCContext &Ctx)
static MCKernelDescriptor getDefaultAmdhsaKernelDescriptor(const MCSubtargetInfo *STI, MCContext &Ctx)
RegisterMCAsmParser - Helper template for registering a target specific assembly parser,...