70 enum KindTy { Token, Immediate, Register, Expression } Kind;
72 SMLoc StartLoc, EndLoc;
73 const AMDGPUAsmParser *AsmParser;
76 AMDGPUOperand(KindTy Kind_,
const AMDGPUAsmParser *AsmParser_)
77 : Kind(Kind_), AsmParser(AsmParser_) {}
79 using Ptr = std::unique_ptr<AMDGPUOperand>;
87 bool hasFPModifiers()
const {
return Abs || Neg; }
88 bool hasIntModifiers()
const {
return Sext; }
89 bool hasModifiers()
const {
return hasFPModifiers() || hasIntModifiers(); }
90 bool isForcedLit()
const {
return Lit == LitModifier::Lit; }
91 bool isForcedLit64()
const {
return Lit == LitModifier::Lit64; }
93 int64_t getFPModifiersOperand()
const {
100 int64_t getIntModifiersOperand()
const {
106 int64_t getModifiersOperand()
const {
107 assert(!(hasFPModifiers() && hasIntModifiers()) &&
108 "fp and int modifiers should not be used simultaneously");
109 if (hasFPModifiers())
110 return getFPModifiersOperand();
111 if (hasIntModifiers())
112 return getIntModifiersOperand();
116 friend raw_ostream &
operator<<(raw_ostream &OS,
117 AMDGPUOperand::Modifiers Mods);
191 ImmTyMatrixAScaleFmt,
192 ImmTyMatrixBScaleFmt,
225 mutable int MCOpIdx = -1;
228 bool isToken()
const override {
return Kind == Token; }
230 bool isSymbolRefExpr()
const {
234 bool isImm()
const override {
return Kind == Immediate; }
236 bool isInlinableImm(MVT type)
const;
237 bool isLiteralImm(MVT type)
const;
239 bool isRegKind()
const {
return Kind == Register; }
241 bool isReg()
const override {
return isRegKind() && !hasModifiers(); }
243 bool isRegOrInline(
unsigned RCID, MVT type)
const {
244 return isRegClass(RCID) || isInlinableImm(type);
247 bool isRegOrInlineTarget(
unsigned TargetRCIdx, MVT type)
const {
248 return isRegClassTarget(TargetRCIdx) || isInlinableImm(type);
252 return isRegOrInline(RCID, type) || isLiteralImm(type);
255 bool isRegOrImmWithInputModsTarget(
unsigned TargetRCIdx, MVT type)
const {
256 return isRegOrInlineTarget(TargetRCIdx, type) || isLiteralImm(type);
259 bool isRegOrImmWithInt16InputMods()
const {
263 template <
bool IsFake16>
bool isRegOrImmWithIntT16InputMods()
const {
265 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::i16);
268 bool isRegOrImmWithInt32InputMods()
const {
272 bool isRegOrInlineImmWithInt16InputMods()
const {
273 return isRegOrInline(AMDGPU::VS_32RegClassID, MVT::i16);
276 template <
bool IsFake16>
bool isRegOrInlineImmWithIntT16InputMods()
const {
277 return isRegOrInline(
278 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::i16);
281 bool isRegOrInlineImmWithInt32InputMods()
const {
282 return isRegOrInline(AMDGPU::VS_32RegClassID, MVT::i32);
285 bool isRegOrImmWithInt64InputMods()
const {
286 return isRegOrImmWithInputModsTarget(AMDGPU::VS_64_AlignTarget, MVT::i64);
289 bool isRegOrImmWithFP16InputMods()
const {
293 template <
bool IsFake16>
bool isRegOrImmWithFPT16InputMods()
const {
295 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::f16);
298 bool isRegOrImmWithFP32InputMods()
const {
302 bool isRegOrImmWithFP64InputMods()
const {
303 return isRegOrImmWithInputModsTarget(AMDGPU::VS_64_AlignTarget, MVT::f64);
306 template <
bool IsFake16>
bool isRegOrInlineImmWithFP16InputMods()
const {
307 return isRegOrInline(
308 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::f16);
311 bool isRegOrInlineImmWithFP32InputMods()
const {
312 return isRegOrInline(AMDGPU::VS_32RegClassID, MVT::f32);
315 bool isRegOrInlineImmWithFP64InputMods()
const {
316 return isRegOrInlineTarget(AMDGPU::VS_64_AlignTarget, MVT::f64);
319 bool isVRegWithInputMods(
unsigned RCID)
const {
return isRegClass(RCID); }
321 bool isVRegWithFP32InputMods()
const {
322 return isVRegWithInputMods(AMDGPU::VGPR_32RegClassID);
325 bool isVRegWithFP64InputMods()
const {
326 return isRegClassTarget(AMDGPU::VReg_64_AlignTarget);
329 bool isPackedFP16InputMods()
const {
333 bool isPackedVGPRFP32InputMods()
const {
337 bool isVReg()
const {
338 return isRegClass(AMDGPU::VGPR_32RegClassID) ||
339 isRegClass(AMDGPU::VReg_64RegClassID) ||
340 isRegClass(AMDGPU::VReg_96RegClassID) ||
341 isRegClass(AMDGPU::VReg_128RegClassID) ||
342 isRegClass(AMDGPU::VReg_160RegClassID) ||
343 isRegClass(AMDGPU::VReg_192RegClassID) ||
344 isRegClass(AMDGPU::VReg_256RegClassID) ||
345 isRegClass(AMDGPU::VReg_512RegClassID) ||
346 isRegClass(AMDGPU::VReg_1024RegClassID);
349 bool isVReg32()
const {
return isRegClass(AMDGPU::VGPR_32RegClassID); }
351 bool isVReg32OrOff()
const {
return isOff() || isVReg32(); }
353 bool isRsrcReg32()
const {
return isRegClass(AMDGPU::RsrcReg32RegClassID); }
355 bool isNull()
const {
return isRegKind() &&
getReg() == AMDGPU::SGPR_NULL; }
357 bool isAV_LdSt_32_Align2_RegOp()
const {
358 return isRegClass(AMDGPU::VGPR_32RegClassID) ||
359 isRegClass(AMDGPU::AGPR_32RegClassID);
362 bool isVRegWithInputMods()
const;
363 template <
bool IsFake16>
bool isT16_Lo128VRegWithInputMods()
const;
364 template <
bool IsFake16>
bool isT16VRegWithInputMods()
const;
366 bool isSDWAOperand(MVT type)
const;
367 bool isSDWAFP16Operand()
const;
368 bool isSDWAFP32Operand()
const;
369 bool isSDWAInt16Operand()
const;
370 bool isSDWAInt32Operand()
const;
372 bool isImmTy(ImmTy ImmT)
const {
return isImm() &&
Imm.Type == ImmT; }
374 template <ImmTy Ty>
bool isImmTy()
const {
return isImmTy(Ty); }
376 bool isImmLiteral()
const {
return isImmTy(ImmTyNone); }
378 bool isImmModifier()
const {
return isImm() &&
Imm.Type != ImmTyNone; }
380 bool isOModSI()
const {
return isImmTy(ImmTyOModSI); }
381 bool isDim()
const {
return isImmTy(ImmTyDim); }
382 bool isR128A16()
const {
return isImmTy(ImmTyR128A16); }
383 bool isOff()
const {
return isImmTy(ImmTyOff); }
384 bool isExpTgt()
const {
return isImmTy(ImmTyExpTgt); }
385 bool isOffen()
const {
return isImmTy(ImmTyOffen); }
386 bool isIdxen()
const {
return isImmTy(ImmTyIdxen); }
387 bool isAddr64()
const {
return isImmTy(ImmTyAddr64); }
388 bool isSMEMOffsetMod()
const {
return isImmTy(ImmTySMEMOffsetMod); }
389 bool isFlatOffset()
const {
390 return isImmTy(ImmTyOffset) || isImmTy(ImmTyInstOffset);
392 bool isGDS()
const {
return isImmTy(ImmTyGDS); }
393 bool isLDS()
const {
return isImmTy(ImmTyLDS); }
394 bool isCPol()
const {
return isImmTy(ImmTyCPol); }
395 bool isIndexKey8bit()
const {
return isImmTy(ImmTyIndexKey8bit); }
396 bool isIndexKey16bit()
const {
return isImmTy(ImmTyIndexKey16bit); }
397 bool isIndexKey32bit()
const {
return isImmTy(ImmTyIndexKey32bit); }
398 bool isMatrixAFMT()
const {
return isImmTy(ImmTyMatrixAFMT); }
399 bool isMatrixBFMT()
const {
return isImmTy(ImmTyMatrixBFMT); }
400 bool isMatrixAScale()
const {
return isImmTy(ImmTyMatrixAScale); }
401 bool isMatrixBScale()
const {
return isImmTy(ImmTyMatrixBScale); }
402 bool isMatrixAScaleFmt()
const {
return isImmTy(ImmTyMatrixAScaleFmt); }
403 bool isMatrixBScaleFmt()
const {
return isImmTy(ImmTyMatrixBScaleFmt); }
404 bool isMatrixAReuse()
const {
return isImmTy(ImmTyMatrixAReuse); }
405 bool isMatrixBReuse()
const {
return isImmTy(ImmTyMatrixBReuse); }
406 bool isTFE()
const {
return isImmTy(ImmTyTFE); }
407 bool isFORMAT()
const {
return isImmTy(ImmTyFORMAT) &&
isUInt<7>(
getImm()); }
408 bool isDppFI()
const {
return isImmTy(ImmTyDppFI); }
409 bool isSDWADstSel()
const {
return isImmTy(ImmTySDWADstSel); }
410 bool isSDWASrc0Sel()
const {
return isImmTy(ImmTySDWASrc0Sel); }
411 bool isSDWASrc1Sel()
const {
return isImmTy(ImmTySDWASrc1Sel); }
412 bool isSDWADstUnused()
const {
return isImmTy(ImmTySDWADstUnused); }
413 bool isInterpSlot()
const {
return isImmTy(ImmTyInterpSlot); }
414 bool isInterpAttr()
const {
return isImmTy(ImmTyInterpAttr); }
415 bool isInterpAttrChan()
const {
return isImmTy(ImmTyInterpAttrChan); }
416 bool isOpSel()
const {
return isImmTy(ImmTyOpSel); }
417 bool isOpSelHi()
const {
return isImmTy(ImmTyOpSelHi); }
418 bool isNegLo()
const {
return isImmTy(ImmTyNegLo); }
419 bool isNegHi()
const {
return isImmTy(ImmTyNegHi); }
420 bool isBitOp3()
const {
return isImmTy(ImmTyBitOp3) &&
isUInt<8>(
getImm()); }
421 bool isDone()
const {
return isImmTy(ImmTyDone); }
422 bool isRowEn()
const {
return isImmTy(ImmTyRowEn); }
424 bool isRegOrImm()
const {
return isReg() || isImm(); }
426 bool isRegClass(
unsigned RCID)
const;
429 bool isRegClassTarget(
unsigned TargetRCIdx)
const;
433 bool isRegOrInlineNoMods(
unsigned RCID, MVT type)
const {
434 return isRegOrInline(RCID, type) && !hasModifiers();
437 bool isRegOrInlineNoModsTarget(
unsigned TargetRCIdx, MVT type)
const {
438 return isRegOrInlineTarget(TargetRCIdx, type) && !hasModifiers();
441 bool isSCSrcB16()
const {
442 return isRegOrInlineNoMods(AMDGPU::SReg_32RegClassID, MVT::i16);
445 bool isSCSrcV2B16()
const {
return isSCSrcB16(); }
447 bool isSCSrc_b32()
const {
448 return isRegOrInlineNoMods(AMDGPU::SReg_32RegClassID, MVT::i32);
451 bool isSCSrc_b64()
const {
452 return isRegOrInlineNoMods(AMDGPU::SReg_64RegClassID, MVT::i64);
455 bool isBoolReg()
const;
457 bool isSCSrcF16()
const {
458 return isRegOrInlineNoMods(AMDGPU::SReg_32RegClassID, MVT::f16);
461 bool isSCSrcV2F16()
const {
return isSCSrcF16(); }
463 bool isSCSrcF32()
const {
464 return isRegOrInlineNoMods(AMDGPU::SReg_32RegClassID, MVT::f32);
467 bool isSCSrcF64()
const {
468 return isRegOrInlineNoMods(AMDGPU::SReg_64RegClassID, MVT::f64);
471 bool isSSrc_b32()
const {
472 return isSCSrc_b32() || isLiteralImm(MVT::i32) || isExpr();
475 bool isSSrc_b16()
const {
return isSCSrcB16() || isLiteralImm(MVT::i16); }
477 bool isSSrcV2B16()
const {
482 bool isSSrc_b64()
const {
485 return isSCSrc_b64() || isLiteralImm(MVT::i64) ||
486 (((
const MCTargetAsmParser *)AsmParser)
487 ->getAvailableFeatures()[AMDGPU::Feature64BitLiterals] &&
491 bool isSSrc_f32()
const {
492 return isSCSrc_b32() || isLiteralImm(MVT::f32) || isExpr();
495 bool isSSrcF64()
const {
return isSCSrc_b64() || isLiteralImm(MVT::f64); }
497 bool isSSrc_bf16()
const {
return isSCSrcB16() || isLiteralImm(MVT::bf16); }
499 bool isSSrc_f16()
const {
return isSCSrcB16() || isLiteralImm(MVT::f16); }
501 bool isSSrc_NoInline_f16()
const {
return isSSrc_f16(); }
503 bool isSSrcV2F16()
const {
508 bool isSSrcV2FP32()
const {
513 bool isSCSrcV2FP32()
const {
518 bool isSSrcV2INT32()
const {
523 bool isSCSrcV2INT32()
const {
525 return isSCSrc_b32();
528 bool isSSrcOrLds_b32()
const {
529 return isRegOrInlineNoMods(AMDGPU::SRegOrLds_32RegClassID, MVT::i32) ||
530 isLiteralImm(MVT::i32) || isExpr();
533 bool isVCSrc_b32()
const {
534 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::i32);
537 bool isVCSrc_b32_Lo256()
const {
538 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo256RegClassID, MVT::i32);
541 bool isVCSrc_b64_Lo256()
const {
542 return isRegOrInlineNoMods(AMDGPU::VS_64_Lo256RegClassID, MVT::i64);
545 bool isVCSrc_b64()
const {
546 return isRegOrInlineNoModsTarget(AMDGPU::VS_64_AlignTarget, MVT::i64);
549 bool isVCSrcT_b16()
const {
550 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::i16);
553 bool isVCSrcTB16_Lo128()
const {
554 return isRegOrInlineNoMods(AMDGPU::VS_16_Lo128RegClassID, MVT::i16);
557 bool isVCSrcFake16B16_Lo128()
const {
558 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo128RegClassID, MVT::i16);
561 bool isVCSrc_b16()
const {
562 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::i16);
565 bool isVCSrc_v2b16()
const {
return isVCSrc_b16(); }
567 bool isVCSrc_f32()
const {
568 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::f32);
571 bool isVCSrc_f64()
const {
572 return isRegOrInlineNoModsTarget(AMDGPU::VS_64_AlignTarget, MVT::f64);
575 bool isVCSrcTBF16()
const {
576 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::bf16);
579 bool isVCSrcT_f16()
const {
580 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::f16);
583 bool isVCSrcT_bf16()
const {
584 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::f16);
587 bool isVCSrcTBF16_Lo128()
const {
588 return isRegOrInlineNoMods(AMDGPU::VS_16_Lo128RegClassID, MVT::bf16);
591 bool isVCSrcTF16_Lo128()
const {
592 return isRegOrInlineNoMods(AMDGPU::VS_16_Lo128RegClassID, MVT::f16);
595 bool isVCSrcFake16BF16_Lo128()
const {
596 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo128RegClassID, MVT::bf16);
599 bool isVCSrcFake16F16_Lo128()
const {
600 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo128RegClassID, MVT::f16);
603 bool isVCSrc_bf16()
const {
604 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::bf16);
607 bool isVCSrc_f16()
const {
608 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::f16);
611 bool isVCSrc_v2bf16()
const {
return isVCSrc_bf16(); }
613 bool isVCSrc_v2f16()
const {
return isVCSrc_f16(); }
615 bool isVSrc_b32()
const {
616 return isVCSrc_f32() || isLiteralImm(MVT::i32) || isExpr();
619 bool isVSrc_b64()
const {
return isVCSrc_f64() || isLiteralImm(MVT::i64); }
621 bool isVSrc_v2b64()
const {
622 return isRegOrInlineNoMods(AMDGPU::VS_128RegClassID, MVT::i64) ||
623 isLiteralImm(MVT::i64);
626 bool isVSrc_v2f64()
const {
627 return isRegOrInlineNoMods(AMDGPU::VS_128RegClassID, MVT::f64) ||
628 isLiteralImm(MVT::f64);
631 bool isVSrcT_b16()
const {
return isVCSrcT_b16() || isLiteralImm(MVT::i16); }
633 bool isVSrcT_b16_Lo128()
const {
634 return isVCSrcTB16_Lo128() || isLiteralImm(MVT::i16);
637 bool isVSrcFake16_b16_Lo128()
const {
638 return isVCSrcFake16B16_Lo128() || isLiteralImm(MVT::i16);
641 bool isVSrc_b16()
const {
return isVCSrc_b16() || isLiteralImm(MVT::i16); }
643 bool isVSrc_v2b16()
const {
return isVSrc_b16() || isLiteralImm(MVT::v2i16); }
645 bool isVCSrcV2FP32()
const {
return isVCSrc_f64(); }
647 bool isVSrc_v2f32()
const {
return isVSrc_f64() || isLiteralImm(MVT::v2f32); }
649 bool isVCSrc_v2b32()
const {
return isVCSrc_b64(); }
651 bool isVSrc_v2b32()
const {
return isVSrc_b64() || isLiteralImm(MVT::v2i32); }
653 bool isVSrc_f32()
const {
654 return isVCSrc_f32() || isLiteralImm(MVT::f32) || isExpr();
657 bool isVSrc_f64()
const {
658 return isRegOrInlineNoModsTarget(AMDGPU::VS_64_AlignTarget, MVT::f64) ||
659 isLiteralImm(MVT::f64);
662 bool isVSrcT_bf16()
const {
663 return isVCSrcTBF16() || isLiteralImm(MVT::bf16);
666 bool isVSrcT_f16()
const {
return isVCSrcT_f16() || isLiteralImm(MVT::f16); }
668 bool isVSrcT_bf16_Lo128()
const {
669 return isVCSrcTBF16_Lo128() || isLiteralImm(MVT::bf16);
672 bool isVSrcT_f16_Lo128()
const {
673 return isVCSrcTF16_Lo128() || isLiteralImm(MVT::f16);
676 bool isVSrcFake16_bf16_Lo128()
const {
677 return isVCSrcFake16BF16_Lo128() || isLiteralImm(MVT::bf16);
680 bool isVSrcFake16_f16_Lo128()
const {
681 return isVCSrcFake16F16_Lo128() || isLiteralImm(MVT::f16);
684 bool isVSrc_bf16()
const {
return isVCSrc_bf16() || isLiteralImm(MVT::bf16); }
686 bool isVSrc_f16()
const {
return isVCSrc_f16() || isLiteralImm(MVT::f16); }
688 bool isVSrc_v2bf16()
const {
689 return isVSrc_bf16() || isLiteralImm(MVT::v2bf16);
692 bool isVSrc_v2f16()
const {
return isVSrc_f16() || isLiteralImm(MVT::v2f16); }
694 bool isVSrc_v2f16_splat()
const {
return isVSrc_v2f16(); }
696 bool isVSrc_NoInline_v2f16()
const {
return isVSrc_v2f16(); }
698 bool isVISrcB32()
const {
699 return isRegOrInlineNoMods(AMDGPU::VGPR_32RegClassID, MVT::i32);
702 bool isVISrcB16()
const {
703 return isRegOrInlineNoMods(AMDGPU::VGPR_32RegClassID, MVT::i16);
706 bool isVISrcV2B16()
const {
return isVISrcB16(); }
708 bool isVISrcF32()
const {
709 return isRegOrInlineNoMods(AMDGPU::VGPR_32RegClassID, MVT::f32);
712 bool isVISrcF16()
const {
713 return isRegOrInlineNoMods(AMDGPU::VGPR_32RegClassID, MVT::f16);
716 bool isVISrcV2F16()
const {
return isVISrcF16() || isVISrcB32(); }
718 bool isVISrc_64_bf16()
const {
719 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::bf16);
722 bool isVISrc_64_f16()
const {
723 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::f16);
726 bool isVISrc_64_b32()
const {
727 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::i32);
730 bool isVISrc_64B64()
const {
731 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::i64);
734 bool isVISrc_64_f64()
const {
735 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::f64);
738 bool isVISrc_64V2FP32()
const {
739 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::f32);
742 bool isVISrc_64V2INT32()
const {
743 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::i32);
746 bool isVISrc_256_b32()
const {
747 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::i32);
750 bool isVISrc_256_f32()
const {
751 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::f32);
754 bool isVISrc_256B64()
const {
755 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::i64);
758 bool isVISrc_256_f64()
const {
759 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::f64);
762 bool isVISrc_512_f64()
const {
763 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::f64);
766 bool isVISrc_128B16()
const {
767 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::i16);
770 bool isVISrc_128V2B16()
const {
return isVISrc_128B16(); }
772 bool isVISrc_128_b32()
const {
773 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::i32);
776 bool isVISrc_128_f32()
const {
777 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::f32);
780 bool isVISrc_256V2FP32()
const {
781 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::f32);
784 bool isVISrc_256V2INT32()
const {
785 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::i32);
788 bool isVISrc_512_b32()
const {
789 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::i32);
792 bool isVISrc_512B16()
const {
793 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::i16);
796 bool isVISrc_512V2B16()
const {
return isVISrc_512B16(); }
798 bool isVISrc_512_f32()
const {
799 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::f32);
802 bool isVISrc_512F16()
const {
803 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::f16);
806 bool isVISrc_512V2F16()
const {
807 return isVISrc_512F16() || isVISrc_512_b32();
810 bool isVISrc_1024_b32()
const {
811 return isRegOrInlineNoModsTarget(AMDGPU::VReg_1024_AlignTarget, MVT::i32);
814 bool isVISrc_1024B16()
const {
815 return isRegOrInlineNoModsTarget(AMDGPU::VReg_1024_AlignTarget, MVT::i16);
818 bool isVISrc_1024V2B16()
const {
return isVISrc_1024B16(); }
820 bool isVISrc_1024_f32()
const {
821 return isRegOrInlineNoModsTarget(AMDGPU::VReg_1024_AlignTarget, MVT::f32);
824 bool isVISrc_1024F16()
const {
825 return isRegOrInlineNoModsTarget(AMDGPU::VReg_1024_AlignTarget, MVT::f16);
828 bool isVISrc_1024V2F16()
const {
829 return isVISrc_1024F16() || isVISrc_1024_b32();
832 bool isAISrcB32()
const {
833 return isRegOrInlineNoMods(AMDGPU::AGPR_32RegClassID, MVT::i32);
836 bool isAISrcB16()
const {
837 return isRegOrInlineNoMods(AMDGPU::AGPR_32RegClassID, MVT::i16);
840 bool isAISrcV2B16()
const {
return isAISrcB16(); }
842 bool isAISrcF32()
const {
843 return isRegOrInlineNoMods(AMDGPU::AGPR_32RegClassID, MVT::f32);
846 bool isAISrcF16()
const {
847 return isRegOrInlineNoMods(AMDGPU::AGPR_32RegClassID, MVT::f16);
850 bool isAISrcV2F16()
const {
return isAISrcF16() || isAISrcB32(); }
852 bool isAISrc_64B64()
const {
853 return isRegOrInlineNoModsTarget(AMDGPU::AReg_64_AlignTarget, MVT::i64);
856 bool isAISrc_64_f64()
const {
857 return isRegOrInlineNoModsTarget(AMDGPU::AReg_64_AlignTarget, MVT::f64);
860 bool isAISrc_128_b32()
const {
861 return isRegOrInlineNoModsTarget(AMDGPU::AReg_128_AlignTarget, MVT::i32);
864 bool isAISrc_128B16()
const {
865 return isRegOrInlineNoModsTarget(AMDGPU::AReg_128_AlignTarget, MVT::i16);
868 bool isAISrc_128V2B16()
const {
return isAISrc_128B16(); }
870 bool isAISrc_128_f32()
const {
871 return isRegOrInlineNoModsTarget(AMDGPU::AReg_128_AlignTarget, MVT::f32);
874 bool isAISrc_128F16()
const {
875 return isRegOrInlineNoModsTarget(AMDGPU::AReg_128_AlignTarget, MVT::f16);
878 bool isAISrc_128V2F16()
const {
879 return isAISrc_128F16() || isAISrc_128_b32();
882 bool isVISrc_128_bf16()
const {
883 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::bf16);
886 bool isVISrc_128_f16()
const {
887 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::f16);
890 bool isVISrc_128V2F16()
const {
891 return isVISrc_128_f16() || isVISrc_128_b32();
894 bool isAISrc_256B64()
const {
895 return isRegOrInlineNoModsTarget(AMDGPU::AReg_256_AlignTarget, MVT::i64);
898 bool isAISrc_256_f64()
const {
899 return isRegOrInlineNoModsTarget(AMDGPU::AReg_256_AlignTarget, MVT::f64);
902 bool isAISrc_512_b32()
const {
903 return isRegOrInlineNoModsTarget(AMDGPU::AReg_512_AlignTarget, MVT::i32);
906 bool isAISrc_512B16()
const {
907 return isRegOrInlineNoModsTarget(AMDGPU::AReg_512_AlignTarget, MVT::i16);
910 bool isAISrc_512V2B16()
const {
return isAISrc_512B16(); }
912 bool isAISrc_512_f32()
const {
913 return isRegOrInlineNoModsTarget(AMDGPU::AReg_512_AlignTarget, MVT::f32);
916 bool isAISrc_512F16()
const {
917 return isRegOrInlineNoModsTarget(AMDGPU::AReg_512_AlignTarget, MVT::f16);
920 bool isAISrc_512V2F16()
const {
921 return isAISrc_512F16() || isAISrc_512_b32();
924 bool isAISrc_1024_b32()
const {
925 return isRegOrInlineNoModsTarget(AMDGPU::AReg_1024_AlignTarget, MVT::i32);
928 bool isAISrc_1024B16()
const {
929 return isRegOrInlineNoModsTarget(AMDGPU::AReg_1024_AlignTarget, MVT::i16);
932 bool isAISrc_1024V2B16()
const {
return isAISrc_1024B16(); }
934 bool isAISrc_1024_f32()
const {
935 return isRegOrInlineNoModsTarget(AMDGPU::AReg_1024_AlignTarget, MVT::f32);
938 bool isAISrc_1024F16()
const {
939 return isRegOrInlineNoModsTarget(AMDGPU::AReg_1024_AlignTarget, MVT::f16);
942 bool isAISrc_1024V2F16()
const {
943 return isAISrc_1024F16() || isAISrc_1024_b32();
946 bool isKImmFP32()
const {
return isLiteralImm(MVT::f32); }
948 bool isKImmFP16()
const {
return isLiteralImm(MVT::f16); }
950 bool isKImmFP64()
const {
return isLiteralImm(MVT::f64); }
952 bool isMem()
const override {
return false; }
954 bool isExpr()
const {
return Kind == Expression; }
956 bool isSOPPBrTarget()
const {
return isExpr() || isImm(); }
958 bool isSWaitCnt()
const;
959 bool isDepCtr()
const;
960 bool isSDelayALU()
const;
961 bool isHwreg()
const;
962 bool isSendMsg()
const;
963 bool isWaitEvent()
const;
964 bool isSplitBarrier()
const;
965 bool isSwizzle()
const;
966 bool isSMRDOffset8()
const;
967 bool isSMEMOffset()
const;
968 bool isSMRDLiteralOffset()
const;
970 bool isDPPCtrl()
const;
972 bool isGPRIdxMode()
const;
973 bool isS16Imm()
const;
974 bool isU16Imm()
const;
975 bool isEndpgm()
const;
977 auto getPredicate(std::function<
bool(
const AMDGPUOperand &
Op)>
P)
const {
978 return [
this,
P]() {
return P(*
this); };
983 return StringRef(Tok.Data, Tok.Length);
991 void setImm(int64_t Val) {
996 ImmTy getImmTy()
const {
1001 MCRegister
getReg()
const override {
1006 SMLoc getStartLoc()
const override {
return StartLoc; }
1008 SMLoc getEndLoc()
const override {
return EndLoc; }
1010 SMRange getLocRange()
const {
return SMRange(StartLoc, EndLoc); }
1012 int getMCOpIdx()
const {
return MCOpIdx; }
1014 Modifiers getModifiers()
const {
1015 assert(isRegKind() || isImmTy(ImmTyNone));
1016 return isRegKind() ?
Reg.Mods :
Imm.Mods;
1019 void setModifiers(Modifiers Mods) {
1020 assert(isRegKind() || isImmTy(ImmTyNone));
1027 bool hasModifiers()
const {
return getModifiers().hasModifiers(); }
1029 bool hasFPModifiers()
const {
return getModifiers().hasFPModifiers(); }
1031 bool hasIntModifiers()
const {
return getModifiers().hasIntModifiers(); }
1033 bool isForcedLit()
const {
1034 return isImmLiteral() && getModifiers().isForcedLit();
1037 bool isForcedLit64()
const {
1038 return isImmLiteral() && getModifiers().isForcedLit64();
1043 void addImmOperands(MCInst &Inst,
unsigned N,
1044 bool ApplyModifiers =
true)
const;
1046 void addLiteralImmOperand(MCInst &Inst, int64_t Val,
1047 bool ApplyModifiers)
const;
1049 void addRegOperands(MCInst &Inst,
unsigned N)
const;
1051 void addRegOrImmOperands(MCInst &Inst,
unsigned N)
const {
1053 addRegOperands(Inst,
N);
1055 addImmOperands(Inst,
N);
1058 void addRegOrImmWithInputModsOperands(MCInst &Inst,
unsigned N)
const {
1059 Modifiers Mods = getModifiers();
1062 addRegOperands(Inst,
N);
1064 addImmOperands(Inst,
N,
false);
1068 void addRegOrImmWithFPInputModsOperands(MCInst &Inst,
unsigned N)
const {
1069 assert(!hasIntModifiers());
1070 addRegOrImmWithInputModsOperands(Inst,
N);
1073 void addRegOrImmWithIntInputModsOperands(MCInst &Inst,
unsigned N)
const {
1074 assert(!hasFPModifiers());
1075 addRegOrImmWithInputModsOperands(Inst,
N);
1078 void addRegWithInputModsOperands(MCInst &Inst,
unsigned N)
const {
1079 Modifiers Mods = getModifiers();
1082 addRegOperands(Inst,
N);
1085 void addRegWithFPInputModsOperands(MCInst &Inst,
unsigned N)
const {
1086 assert(!hasIntModifiers());
1087 addRegWithInputModsOperands(Inst,
N);
1090 void addRegWithIntInputModsOperands(MCInst &Inst,
unsigned N)
const {
1091 assert(!hasFPModifiers());
1092 addRegWithInputModsOperands(Inst,
N);
1095 static void printImmTy(raw_ostream &OS, ImmTy
Type) {
1098 case ImmTyNone: OS <<
"None";
break;
1099 case ImmTyGDS: OS <<
"GDS";
break;
1100 case ImmTyLDS: OS <<
"LDS";
break;
1101 case ImmTyOffen: OS <<
"Offen";
break;
1102 case ImmTyIdxen: OS <<
"Idxen";
break;
1103 case ImmTyAddr64: OS <<
"Addr64";
break;
1104 case ImmTyOffset: OS <<
"Offset";
break;
1105 case ImmTyInstOffset: OS <<
"InstOffset";
break;
1106 case ImmTyOffset0: OS <<
"Offset0";
break;
1107 case ImmTyOffset1: OS <<
"Offset1";
break;
1108 case ImmTySMEMOffsetMod: OS <<
"SMEMOffsetMod";
break;
1109 case ImmTyCPol: OS <<
"CPol";
break;
1110 case ImmTyIndexKey8bit: OS <<
"index_key";
break;
1111 case ImmTyIndexKey16bit: OS <<
"index_key";
break;
1112 case ImmTyIndexKey32bit: OS <<
"index_key";
break;
1113 case ImmTyTFE: OS <<
"TFE";
break;
1114 case ImmTyIsAsync: OS <<
"IsAsync";
break;
1115 case ImmTyD16: OS <<
"D16";
break;
1116 case ImmTyFORMAT: OS <<
"FORMAT";
break;
1117 case ImmTyClamp: OS <<
"Clamp";
break;
1118 case ImmTyOModSI: OS <<
"OModSI";
break;
1119 case ImmTyDPP8: OS <<
"DPP8";
break;
1120 case ImmTyDppCtrl: OS <<
"DppCtrl";
break;
1121 case ImmTyDppRowMask: OS <<
"DppRowMask";
break;
1122 case ImmTyDppBankMask: OS <<
"DppBankMask";
break;
1123 case ImmTyDppBoundCtrl: OS <<
"DppBoundCtrl";
break;
1124 case ImmTyDppFI: OS <<
"DppFI";
break;
1125 case ImmTySDWADstSel: OS <<
"SDWADstSel";
break;
1126 case ImmTySDWASrc0Sel: OS <<
"SDWASrc0Sel";
break;
1127 case ImmTySDWASrc1Sel: OS <<
"SDWASrc1Sel";
break;
1128 case ImmTySDWADstUnused: OS <<
"SDWADstUnused";
break;
1129 case ImmTyDMask: OS <<
"DMask";
break;
1130 case ImmTyDim: OS <<
"Dim";
break;
1131 case ImmTyUNorm: OS <<
"UNorm";
break;
1132 case ImmTyDA: OS <<
"DA";
break;
1133 case ImmTyR128A16: OS <<
"R128A16";
break;
1134 case ImmTyA16: OS <<
"A16";
break;
1135 case ImmTyLWE: OS <<
"LWE";
break;
1136 case ImmTyOff: OS <<
"Off";
break;
1137 case ImmTyExpTgt: OS <<
"ExpTgt";
break;
1138 case ImmTyExpCompr: OS <<
"ExpCompr";
break;
1139 case ImmTyExpVM: OS <<
"ExpVM";
break;
1140 case ImmTyDone: OS <<
"Done";
break;
1141 case ImmTyRowEn: OS <<
"RowEn";
break;
1142 case ImmTyHwreg: OS <<
"Hwreg";
break;
1143 case ImmTySendMsg: OS <<
"SendMsg";
break;
1144 case ImmTyWaitEvent: OS <<
"WaitEvent";
break;
1145 case ImmTyInterpSlot: OS <<
"InterpSlot";
break;
1146 case ImmTyInterpAttr: OS <<
"InterpAttr";
break;
1147 case ImmTyInterpAttrChan: OS <<
"InterpAttrChan";
break;
1148 case ImmTyOpSel: OS <<
"OpSel";
break;
1149 case ImmTyOpSelHi: OS <<
"OpSelHi";
break;
1150 case ImmTyNegLo: OS <<
"NegLo";
break;
1151 case ImmTyNegHi: OS <<
"NegHi";
break;
1152 case ImmTySwizzle: OS <<
"Swizzle";
break;
1153 case ImmTyGprIdxMode: OS <<
"GprIdxMode";
break;
1154 case ImmTyHigh: OS <<
"High";
break;
1155 case ImmTyBLGP: OS <<
"BLGP";
break;
1156 case ImmTyCBSZ: OS <<
"CBSZ";
break;
1157 case ImmTyABID: OS <<
"ABID";
break;
1158 case ImmTyEndpgm: OS <<
"Endpgm";
break;
1159 case ImmTyWaitVDST: OS <<
"WaitVDST";
break;
1160 case ImmTyWaitEXP: OS <<
"WaitEXP";
break;
1161 case ImmTyWaitVAVDst: OS <<
"WaitVAVDst";
break;
1162 case ImmTyWaitVMVSrc: OS <<
"WaitVMVSrc";
break;
1163 case ImmTyBitOp3: OS <<
"BitOp3";
break;
1164 case ImmTyMatrixAFMT: OS <<
"ImmTyMatrixAFMT";
break;
1165 case ImmTyMatrixBFMT: OS <<
"ImmTyMatrixBFMT";
break;
1166 case ImmTyMatrixAScale: OS <<
"ImmTyMatrixAScale";
break;
1167 case ImmTyMatrixBScale: OS <<
"ImmTyMatrixBScale";
break;
1168 case ImmTyMatrixAScaleFmt: OS <<
"ImmTyMatrixAScaleFmt";
break;
1169 case ImmTyMatrixBScaleFmt: OS <<
"ImmTyMatrixBScaleFmt";
break;
1170 case ImmTyMatrixAReuse: OS <<
"ImmTyMatrixAReuse";
break;
1171 case ImmTyMatrixBReuse: OS <<
"ImmTyMatrixBReuse";
break;
1172 case ImmTyScaleSel: OS <<
"ScaleSel" ;
break;
1173 case ImmTyByteSel: OS <<
"ByteSel" ;
break;
1178 void print(raw_ostream &OS,
const MCAsmInfo &MAI)
const override {
1182 <<
" mods: " <<
Reg.Mods <<
'>';
1186 if (getImmTy() != ImmTyNone) {
1188 printImmTy(OS, getImmTy());
1190 OS <<
" mods: " <<
Imm.Mods <<
'>';
1203 static AMDGPUOperand::Ptr CreateImm(
const AMDGPUAsmParser *AsmParser,
1204 int64_t Val, SMLoc Loc,
1205 ImmTy
Type = ImmTyNone,
1206 bool IsFPImm =
false) {
1207 auto Op = std::make_unique<AMDGPUOperand>(Immediate, AsmParser);
1209 Op->Imm.IsFPImm = IsFPImm;
1211 Op->Imm.Mods = Modifiers();
1217 static AMDGPUOperand::Ptr CreateToken(
const AMDGPUAsmParser *AsmParser,
1218 StringRef Str, SMLoc Loc,
1219 bool HasExplicitEncodingSize =
true) {
1220 auto Res = std::make_unique<AMDGPUOperand>(Token, AsmParser);
1221 Res->Tok.Data = Str.data();
1222 Res->Tok.Length = Str.size();
1223 Res->StartLoc = Loc;
1228 static AMDGPUOperand::Ptr CreateReg(
const AMDGPUAsmParser *AsmParser,
1229 MCRegister
Reg, SMLoc S, SMLoc
E) {
1230 auto Op = std::make_unique<AMDGPUOperand>(Register, AsmParser);
1231 Op->Reg.RegNo =
Reg;
1232 Op->Reg.Mods = Modifiers();
1238 static AMDGPUOperand::Ptr CreateExpr(
const AMDGPUAsmParser *AsmParser,
1239 const class MCExpr *Expr, SMLoc S) {
1240 auto Op = std::make_unique<AMDGPUOperand>(Expression, AsmParser);
1249 OS <<
"abs:" << Mods.Abs <<
" neg: " << Mods.Neg <<
" sext:" << Mods.Sext;
1258#define GET_REGISTER_MATCHER
1259#include "AMDGPUGenAsmMatcher.inc"
1260#undef GET_REGISTER_MATCHER
1261#undef GET_SUBTARGET_FEATURE_NAME
1266class KernelScopeInfo {
1267 int SgprIndexUnusedMin = -1;
1268 int VgprIndexUnusedMin = -1;
1269 int AgprIndexUnusedMin = -1;
1273 void usesSgprAt(
int i) {
1274 if (i >= SgprIndexUnusedMin) {
1275 SgprIndexUnusedMin = ++i;
1278 Ctx->getOrCreateSymbol(
Twine(
".kernel.sgpr_count"));
1284 void usesVgprAt(
int i) {
1285 if (i >= VgprIndexUnusedMin) {
1286 VgprIndexUnusedMin = ++i;
1289 Ctx->getOrCreateSymbol(
Twine(
".kernel.vgpr_count"));
1291 VgprIndexUnusedMin);
1297 void usesAgprAt(
int i) {
1302 if (i >= AgprIndexUnusedMin) {
1303 AgprIndexUnusedMin = ++i;
1306 Ctx->getOrCreateSymbol(
Twine(
".kernel.agpr_count"));
1311 Ctx->getOrCreateSymbol(
Twine(
".kernel.vgpr_count"));
1313 VgprIndexUnusedMin);
1320 KernelScopeInfo() =
default;
1324 MSTI = Ctx->getSubtargetInfo();
1326 usesSgprAt(SgprIndexUnusedMin = -1);
1327 usesVgprAt(VgprIndexUnusedMin = -1);
1329 usesAgprAt(AgprIndexUnusedMin = -1);
1333 void usesRegister(RegisterKind RegKind,
unsigned DwordRegIndex,
1334 unsigned RegWidth) {
1337 usesSgprAt(DwordRegIndex +
divideCeil(RegWidth, 32) - 1);
1340 usesAgprAt(DwordRegIndex +
divideCeil(RegWidth, 32) - 1);
1343 usesVgprAt(DwordRegIndex +
divideCeil(RegWidth, 32) - 1);
1352 MCAsmParser &Parser;
1354 unsigned ForcedEncodingSize = 0;
1355 bool ForcedDPP =
false;
1356 bool ForcedSDWA =
false;
1357 KernelScopeInfo KernelScope;
1358 const unsigned HwMode;
1360 const AMDGPU::IsaVersion ISA;
1365#define GET_ASSEMBLER_HEADER
1366#include "AMDGPUGenAsmMatcher.inc"
1371 unsigned getRegOperandSize(
const MCInstrDesc &
Desc,
unsigned OpNo)
const {
1373 int16_t RCID = MII.getOpRegClassID(
Desc.operands()[OpNo], HwMode);
1377 std::optional<AMDGPU::InfoSectionData> InfoData;
1384 bool TargetDirectiveEmitted =
false;
1393 SmallVector<unsigned> OpcodeStream;
1395 OpcodeStreamSymbols;
1396 SmallPtrSet<const MCSymbol *, 8> AMDHSAKernelSymbols;
1399 void checkKernelPrologues();
1402 void createConstantSymbol(StringRef Id, int64_t Val);
1404 bool ParseAsAbsoluteExpression(uint32_t &Ret);
1405 bool OutOfRangeError(SMRange
Range);
1421 bool calculateGPRBlocks(
const FeatureBitset &Features,
const MCExpr *VCCUsed,
1422 const MCExpr *FlatScrUsed,
bool XNACKUsed,
1423 std::optional<bool> EnableWavefrontSize32,
1424 const MCExpr *NextFreeVGPR, SMRange VGPRRange,
1425 const MCExpr *NextFreeSGPR, SMRange SGPRRange,
1426 const MCExpr *&VGPRBlocks,
const MCExpr *&SGPRBlocks);
1427 bool ParseDirectiveAMDGCNTarget();
1428 bool ParseDirectiveAMDHSACodeObjectVersion();
1429 bool ParseDirectiveAMDHSAKernel();
1430 bool ParseAMDKernelCodeTValue(StringRef ID, AMDGPUMCKernelCodeT &Header);
1431 bool ParseDirectiveAMDKernelCodeT();
1433 bool subtargetHasRegister(
const MCRegisterInfo &MRI, MCRegister
Reg);
1434 bool ParseDirectiveAMDGPUHsaKernel();
1436 bool ParseDirectiveISAVersion();
1437 bool ParseDirectiveHSAMetadata();
1438 bool ParseDirectivePALMetadataBegin();
1439 bool ParseDirectivePALMetadata();
1440 bool ParseDirectiveAMDGPULDS();
1441 bool ParseDirectiveAMDGPUInfo();
1445 bool ParseToEndDirective(
const char *AssemblerDirectiveBegin,
1446 const char *AssemblerDirectiveEnd,
1447 std::string &CollectString);
1449 bool AddNextRegisterToList(MCRegister &
Reg,
unsigned &RegWidth,
1450 RegisterKind RegKind, MCRegister Reg1,
1451 RegisterKind RegKind1, SMLoc Loc);
1452 bool ParseAMDGPURegister(RegisterKind &RegKind, MCRegister &
Reg,
1453 unsigned &RegNum,
unsigned &RegWidth,
1454 bool RestoreOnFailure =
false);
1455 bool ParseAMDGPURegister(RegisterKind &RegKind, MCRegister &
Reg,
1456 unsigned &RegNum,
unsigned &RegWidth,
1457 SmallVectorImpl<AsmToken> &Tokens);
1458 MCRegister ParseRegularReg(RegisterKind &RegKind,
unsigned &RegNum,
1460 SmallVectorImpl<AsmToken> &Tokens);
1461 MCRegister ParseSpecialReg(RegisterKind &RegKind,
unsigned &RegNum,
1463 SmallVectorImpl<AsmToken> &Tokens);
1464 MCRegister ParseRegList(RegisterKind &RegKind,
unsigned &RegNum,
1466 SmallVectorImpl<AsmToken> &Tokens);
1467 bool ParseRegRange(
unsigned &Num,
unsigned &Width,
unsigned &SubReg);
1468 MCRegister getRegularReg(RegisterKind RegKind,
unsigned RegNum,
1469 unsigned SubReg,
unsigned RegWidth, SMLoc Loc);
1472 bool isRegister(
const AsmToken &Token,
const AsmToken &NextToken)
const;
1473 std::optional<StringRef> getGprCountSymbolName(RegisterKind RegKind);
1474 void initializeGprCountSymbol(RegisterKind RegKind);
1475 bool updateGprCountSymbols(RegisterKind RegKind,
unsigned DwordRegIndex,
1481 OperandMode_Default,
1485 using OptionalImmIndexMap = std::map<AMDGPUOperand::ImmTy, unsigned>;
1487 AMDGPUAsmParser(
const MCSubtargetInfo &STI, MCAsmParser &_Parser,
1488 const MCInstrInfo &MII)
1489 : MCTargetAsmParser(STI, MII), Parser(_Parser),
1490 HwMode(STI.getHwMode(MCSubtargetInfo::HwMode_RegInfo)),
1495 setAvailableFeatures(ComputeAvailableFeatures(getFeatureBits()));
1497 if (ISA.Major >= 6 &&
isHsaAbi(getSTI())) {
1498 createConstantSymbol(
".amdgcn.gfx_generation_number", ISA.Major);
1499 createConstantSymbol(
".amdgcn.gfx_generation_minor", ISA.Minor);
1500 createConstantSymbol(
".amdgcn.gfx_generation_stepping", ISA.Stepping);
1502 createConstantSymbol(
".option.machine_version_major", ISA.Major);
1503 createConstantSymbol(
".option.machine_version_minor", ISA.Minor);
1504 createConstantSymbol(
".option.machine_version_stepping", ISA.Stepping);
1506 if (ISA.Major >= 6 &&
isHsaAbi(getSTI())) {
1507 initializeGprCountSymbol(IS_VGPR);
1508 initializeGprCountSymbol(IS_SGPR);
1513 createConstantSymbol(Symbol, Code);
1515 createConstantSymbol(
"UC_VERSION_W64_BIT", 0x2000);
1516 createConstantSymbol(
"UC_VERSION_W32_BIT", 0x4000);
1517 createConstantSymbol(
"UC_VERSION_MDP_BIT", 0x8000);
1565 bool hasBVHRayTracingInsts()
const {
1566 return getFeatureBits()[AMDGPU::FeatureBVHRayTracingInsts];
1571 bool isWave32()
const {
return getAvailableFeatures()[Feature_isWave32Bit]; }
1573 bool isWave64()
const {
return getAvailableFeatures()[Feature_isWave64Bit]; }
1575 bool hasInv2PiInlineImm()
const {
1576 return getFeatureBits()[AMDGPU::FeatureInv2PiInlineImm];
1579 bool has64BitLiterals()
const {
1580 return getFeatureBits()[AMDGPU::Feature64BitLiterals];
1583 bool hasFlatOffsets()
const {
1584 return getFeatureBits()[AMDGPU::FeatureFlatInstOffsets];
1587 bool hasTrue16Insts()
const {
1588 return getFeatureBits()[AMDGPU::FeatureTrue16BitInsts];
1592 return getFeatureBits()[AMDGPU::FeatureArchitectedFlatScratch];
1595 bool hasSGPR102_SGPR103()
const {
return !
isVI() && !
isGFX9(); }
1597 bool hasSGPR104_SGPR105()
const {
return isGFX10Plus(); }
1599 bool hasIntClamp()
const {
return getFeatureBits()[AMDGPU::FeatureIntClamp]; }
1601 bool hasPartialNSAEncoding()
const {
1602 return getFeatureBits()[AMDGPU::FeaturePartialNSAEncoding];
1605 bool hasGloballyAddressableScratch()
const {
1606 return getFeatureBits()[AMDGPU::FeatureGloballyAddressableScratch];
1619 AMDGPUTargetStreamer &getTargetStreamer() {
1620 MCTargetStreamer &TS = *getParser().getStreamer().getTargetStreamer();
1621 return static_cast<AMDGPUTargetStreamer &
>(TS);
1627 return const_cast<AMDGPUAsmParser *
>(
this)->MCTargetAsmParser::getContext();
1630 const MCRegisterInfo *getMRI()
const {
1634 const MCInstrInfo *getMII()
const {
return &MII; }
1638 int16_t getTargetRegClass(
unsigned TargetRCIdx)
const {
1644 const FeatureBitset &getFeatureBits()
const {
1645 return getSTI().getFeatureBits();
1648 void setForcedEncodingSize(
unsigned Size) { ForcedEncodingSize =
Size; }
1649 void setForcedDPP(
bool ForceDPP_) { ForcedDPP = ForceDPP_; }
1650 void setForcedSDWA(
bool ForceSDWA_) { ForcedSDWA = ForceSDWA_; }
1652 unsigned getForcedEncodingSize()
const {
return ForcedEncodingSize; }
1653 bool isForcedVOP3()
const {
return ForcedEncodingSize == 64; }
1654 bool isForcedDPP()
const {
return ForcedDPP; }
1655 bool isForcedSDWA()
const {
return ForcedSDWA; }
1656 ArrayRef<unsigned> getMatchedVariants()
const;
1657 StringRef getMatchedVariantName()
const;
1659 std::unique_ptr<AMDGPUOperand> parseRegister(
bool RestoreOnFailure =
false);
1660 bool ParseRegister(MCRegister &RegNo, SMLoc &StartLoc, SMLoc &EndLoc,
1661 bool RestoreOnFailure);
1662 bool parseRegister(MCRegister &
Reg, SMLoc &StartLoc, SMLoc &EndLoc)
override;
1663 ParseStatus tryParseRegister(MCRegister &
Reg, SMLoc &StartLoc,
1664 SMLoc &EndLoc)
override;
1665 unsigned checkTargetMatchPredicate(MCInst &Inst)
override;
1666 unsigned validateTargetOperandClass(MCParsedAsmOperand &
Op,
1667 unsigned Kind)
override;
1668 bool matchAndEmitInstruction(SMLoc IDLoc,
unsigned &Opcode,
1671 bool MatchingInlineAsm)
override;
1672 bool ParseDirective(AsmToken DirectiveID)
override;
1673 void doBeforeLabelEmit(MCSymbol *Symbol, SMLoc IDLoc)
override;
1674 void onEndOfFile()
override;
1676 OperandMode
Mode = OperandMode_Default);
1677 StringRef parseMnemonicSuffix(StringRef Name);
1678 bool parseInstruction(ParseInstructionInfo &Info, StringRef Name,
1684 ParseStatus parseIntWithPrefix(
const char *Prefix, int64_t &
Int);
1688 AMDGPUOperand::ImmTy ImmTy = AMDGPUOperand::ImmTyNone,
1689 std::function<
bool(int64_t &)> ConvertResult =
nullptr);
1691 ParseStatus parseOperandArrayWithPrefix(
1693 AMDGPUOperand::ImmTy ImmTy = AMDGPUOperand::ImmTyNone,
1694 bool (*ConvertResult)(int64_t &) =
nullptr);
1698 AMDGPUOperand::ImmTy ImmTy = AMDGPUOperand::ImmTyNone,
1699 bool IgnoreNegative =
false);
1700 unsigned getCPolKind(StringRef Id, StringRef Mnemo,
bool &Disabling)
const;
1704 ParseStatus parseStringWithPrefix(StringRef Prefix, StringRef &
Value,
1708 ArrayRef<const char *> Ids,
1712 ArrayRef<const char *> Ids,
1713 AMDGPUOperand::ImmTy
Type);
1716 bool isOperandModifier(
const AsmToken &Token,
1717 const AsmToken &NextToken)
const;
1718 bool isRegOrOperandModifier(
const AsmToken &Token,
1719 const AsmToken &NextToken)
const;
1720 bool isNamedOperandModifier(
const AsmToken &Token,
1721 const AsmToken &NextToken)
const;
1722 bool isOpcodeModifierWithVal(
const AsmToken &Token,
1723 const AsmToken &NextToken)
const;
1724 bool parseSP3NegModifier();
1731 bool AllowImm =
true);
1733 bool AllowImm =
true);
1739 AMDGPUOperand::ImmTy ImmTy);
1744 AMDGPUOperand::ImmTy
Type);
1748 AMDGPUOperand::ImmTy
Type);
1752 AMDGPUOperand::ImmTy
Type);
1756 ParseStatus parseDfmtNfmt(int64_t &
Format);
1757 ParseStatus parseUfmt(int64_t &
Format);
1758 ParseStatus parseSymbolicSplitFormat(StringRef FormatStr, SMLoc Loc,
1760 ParseStatus parseSymbolicUnifiedFormat(StringRef FormatStr, SMLoc Loc,
1763 ParseStatus parseSymbolicOrNumericFormat(int64_t &
Format);
1764 ParseStatus parseNumericFormat(int64_t &
Format);
1768 bool tryParseFmt(
const char *Pref, int64_t MaxVal, int64_t &Val);
1769 bool matchDfmtNfmt(int64_t &Dfmt, int64_t &Nfmt, StringRef FormatStr,
1774 bool parseCnt(int64_t &IntVal);
1777 bool parseDepCtr(int64_t &IntVal,
unsigned &Mask);
1778 void depCtrError(SMLoc Loc,
int ErrorId, StringRef DepCtrName);
1781 bool parseDelay(int64_t &Delay);
1787 struct OperandInfoTy {
1790 bool IsSymbolic =
false;
1791 bool IsDefined =
false;
1793 constexpr OperandInfoTy(int64_t Val) : Val(Val) {}
1796 struct StructuredOpField : OperandInfoTy {
1800 bool IsDefined =
false;
1802 constexpr StructuredOpField(StringLiteral Id, StringLiteral Desc,
1803 unsigned Width, int64_t
Default)
1804 : OperandInfoTy(
Default), Id(Id), Desc(Desc), Width(Width) {}
1805 virtual ~StructuredOpField() =
default;
1807 bool Error(AMDGPUAsmParser &Parser,
const Twine &Err)
const {
1808 Parser.Error(Loc,
"invalid " + Desc +
": " + Err);
1812 virtual bool validate(AMDGPUAsmParser &Parser)
const {
1814 return Error(Parser,
"not supported on this GPU");
1816 return Error(Parser,
"only " + Twine(Width) +
"-bit values are legal");
1824 bool parseSendMsgBody(OperandInfoTy &
Msg, OperandInfoTy &
Op,
1825 OperandInfoTy &Stream);
1826 bool validateSendMsg(
const OperandInfoTy &
Msg,
const OperandInfoTy &
Op,
1827 const OperandInfoTy &Stream);
1829 ParseStatus parseHwregFunc(OperandInfoTy &HwReg, OperandInfoTy &
Offset,
1830 OperandInfoTy &Width);
1835 static SMLoc getLaterLoc(SMLoc a, SMLoc b);
1842 SMLoc getOperandLoc(std::function<
bool(
const AMDGPUOperand &)>
Test,
1844 SMLoc getImmLoc(AMDGPUOperand::ImmTy
Type,
1848 bool validateInstruction(
const MCInst &Inst, SMLoc IDLoc,
1853 bool validateBF16InlineConst(
const MCInst &Inst,
1856 bool validateConstantBusLimitations(
const MCInst &Inst,
1858 std::optional<unsigned> checkVOPDRegBankConstraints(
const MCInst &Inst,
1861 bool tryVOPD(
const MCInst &Inst);
1862 bool tryVOPD3(
const MCInst &Inst);
1863 bool tryAnotherVOPDEncoding(
const MCInst &Inst);
1865 bool validateIntClampSupported(
const MCInst &Inst);
1866 bool validateMIMGAtomicDMask(
const MCInst &Inst);
1867 bool validateMIMGGatherDMask(
const MCInst &Inst);
1869 bool validateMIMGDataSize(
const MCInst &Inst, SMLoc IDLoc);
1870 bool validateMIMGAddrSize(
const MCInst &Inst, SMLoc IDLoc);
1871 bool validateMIMGD16(
const MCInst &Inst);
1873 bool validateTensorR128(
const MCInst &Inst);
1874 bool validateMIMGMSAA(
const MCInst &Inst);
1875 bool validateOpSel(
const MCInst &Inst);
1876 bool validateTrue16OpSel(
const MCInst &Inst);
1877 bool validateNeg(
const MCInst &Inst, AMDGPU::OpName OpName);
1879 bool validateVccOperand(MCRegister
Reg)
const;
1884 bool validateAGPRLdSt(
const MCInst &Inst)
const;
1885 bool validateVGPRAlign(
const MCInst &Inst)
const;
1889 bool validateDivScale(
const MCInst &Inst);
1894 const unsigned CPol);
1899 bool validateClusterBarrierIsFirst(
const MCInst &Inst,
1902 unsigned getConstantBusLimit(
unsigned Opcode)
const;
1903 bool usesConstantBus(
const MCInst &Inst,
unsigned OpIdx);
1904 bool isInlineConstant(
const MCInst &Inst,
unsigned OpIdx)
const;
1905 MCRegister findImplicitSGPRReadInVOP(
const MCInst &Inst)
const;
1907 bool isSupportedMnemo(StringRef Mnemo,
const FeatureBitset &FBS);
1908 bool isSupportedMnemo(StringRef Mnemo,
const FeatureBitset &FBS,
1909 ArrayRef<unsigned> Variants);
1910 bool checkUnsupportedInstruction(StringRef Name, SMLoc IDLoc);
1912 bool isId(
const StringRef Id)
const;
1913 bool isId(
const AsmToken &Token,
const StringRef Id)
const;
1915 StringRef getId()
const;
1916 bool trySkipId(
const StringRef Id);
1917 bool trySkipId(
const StringRef Pref,
const StringRef Id);
1921 bool parseString(StringRef &Val,
1922 const StringRef ErrMsg =
"expected a string");
1923 bool parseId(StringRef &Val,
const StringRef ErrMsg =
"");
1929 StringRef getTokenStr()
const;
1930 AsmToken peekToken(
bool ShouldSkipSpace =
true);
1932 SMLoc getLoc()
const;
1936 void onBeginOfFile()
override;
1940 void emitTargetDirective();
1941 bool parsePrimaryExpr(
const MCExpr *&Res, SMLoc &EndLoc)
override;
1953 bool parseSwizzleOperand(int64_t &
Op,
const unsigned MinVal,
1954 const unsigned MaxVal,
const Twine &ErrMsg,
1956 bool parseSwizzleOperands(
const unsigned OpNum, int64_t *
Op,
1957 const unsigned MinVal,
const unsigned MaxVal,
1958 const StringRef ErrMsg);
1960 bool parseSwizzleOffset(int64_t &
Imm);
1961 bool parseSwizzleMacro(int64_t &
Imm);
1962 bool parseSwizzleQuadPerm(int64_t &
Imm);
1963 bool parseSwizzleBitmaskPerm(int64_t &
Imm);
1964 bool parseSwizzleBroadcast(int64_t &
Imm);
1965 bool parseSwizzleSwap(int64_t &
Imm);
1966 bool parseSwizzleReverse(int64_t &
Imm);
1967 bool parseSwizzleFFT(int64_t &
Imm);
1968 bool parseSwizzleRotate(int64_t &
Imm);
1971 int64_t parseGPRIdxMacro();
1974 cvtMubufImpl(Inst,
Operands,
false);
1977 cvtMubufImpl(Inst,
Operands,
true);
1983 OptionalImmIndexMap &OptionalIdx);
1992 OptionalImmIndexMap &OptionalIdx);
1994 OptionalImmIndexMap &OptionalIdx);
1998 void cvtOpSelHelper(MCInst &Inst,
unsigned OpSel);
2000 bool parseDimId(
unsigned &Encoding);
2002 bool convertDppBoundCtrl(int64_t &BoundCtrl);
2006 int64_t parseDPPCtrlSel(StringRef Ctrl);
2007 int64_t parseDPPCtrlPerm();
2013 bool IsDPP8 =
false);
2019 AMDGPUOperand::ImmTy
Type);
2027 enum class SDWAInstType :
unsigned {
VOP1 = 0,
VOP2 = 1,
VOPC = 2 };
2030 SDWAInstType BasicInstType,
bool SkipDstVcc =
false,
2031 bool SkipSrcVcc =
false);
2141bool AMDGPUOperand::isInlinableImm(
MVT type)
const {
2151 if (!isImmTy(ImmTyNone)) {
2156 if (getModifiers().
Lit != LitModifier::None)
2166 if (type == MVT::f64 || type == MVT::i64) {
2168 AsmParser->hasInv2PiInlineImm());
2171 APFloat FPLiteral(APFloat::IEEEdouble(), APInt(64,
Imm.Val));
2190 APFloat::rmNearestTiesToEven, &Lost);
2197 uint32_t ImmVal = FPLiteral.bitcastToAPInt().getZExtValue();
2199 AsmParser->hasInv2PiInlineImm());
2204 static_cast<int32_t
>(FPLiteral.bitcastToAPInt().getZExtValue()),
2205 AsmParser->hasInv2PiInlineImm());
2209 if (type == MVT::f64 || type == MVT::i64) {
2211 AsmParser->hasInv2PiInlineImm());
2220 static_cast<int16_t
>(
Literal.getLoBits(16).getSExtValue()), type,
2221 AsmParser->hasInv2PiInlineImm());
2225 static_cast<int32_t
>(
Literal.getLoBits(32).getZExtValue()),
2226 AsmParser->hasInv2PiInlineImm());
2229bool AMDGPUOperand::isLiteralImm(MVT type)
const {
2231 if (!isImmTy(ImmTyNone)) {
2236 (type == MVT::i64 || type == MVT::f64) && AsmParser->has64BitLiterals();
2241 if (type == MVT::f64 && hasFPModifiers()) {
2261 if (type == MVT::f64) {
2266 if (type == MVT::i64) {
2279 MVT ExpectedType = (type == MVT::v2f16) ? MVT::f16
2280 : (type == MVT::v2i16) ? MVT::f32
2281 : (type == MVT::v2f32) ? MVT::f32
2284 APFloat FPLiteral(APFloat::IEEEdouble(), APInt(64,
Imm.Val));
2288bool AMDGPUOperand::isRegClassTarget(
unsigned TargetRCIdx)
const {
2291 int16_t RCID = AsmParser->getTargetRegClass(TargetRCIdx);
2292 return RCID >= 0 && isRegClass(RCID);
2295bool AMDGPUOperand::isRegClass(
unsigned RCID)
const {
2296 return isRegKind() &&
2297 AsmParser->getMRI()->getRegClass(RCID).contains(
getReg());
2300bool AMDGPUOperand::isVRegWithInputMods()
const {
2301 return isRegClass(AMDGPU::VGPR_32RegClassID) ||
2303 (AsmParser->getFeatureBits()[AMDGPU::FeatureDPALU_DPP] &&
2304 isRegClassTarget(AMDGPU::VReg_64_AlignTarget));
2307template <
bool IsFake16>
2308bool AMDGPUOperand::isT16_Lo128VRegWithInputMods()
const {
2309 return isRegClass(IsFake16 ? AMDGPU::VGPR_32_Lo128RegClassID
2310 : AMDGPU::VGPR_16_Lo128RegClassID);
2313template <
bool IsFake16>
bool AMDGPUOperand::isT16VRegWithInputMods()
const {
2314 return isRegClass(IsFake16 ? AMDGPU::VGPR_32RegClassID
2315 : AMDGPU::VGPR_16RegClassID);
2318bool AMDGPUOperand::isSDWAOperand(MVT type)
const {
2319 if (AsmParser->isVI())
2321 if (AsmParser->isGFX9Plus())
2322 return isRegClass(AMDGPU::VS_32RegClassID) || isInlinableImm(type);
2326bool AMDGPUOperand::isSDWAFP16Operand()
const {
2327 return isSDWAOperand(MVT::f16);
2330bool AMDGPUOperand::isSDWAFP32Operand()
const {
2331 return isSDWAOperand(MVT::f32);
2334bool AMDGPUOperand::isSDWAInt16Operand()
const {
2335 return isSDWAOperand(MVT::i16);
2338bool AMDGPUOperand::isSDWAInt32Operand()
const {
2339 return isSDWAOperand(MVT::i32);
2342bool AMDGPUOperand::isBoolReg()
const {
2343 return isReg() && ((AsmParser->isWave64() && isSCSrc_b64()) ||
2344 (AsmParser->isWave32() && isSCSrc_b32()));
2348 unsigned Size)
const {
2349 assert(isImmTy(ImmTyNone) &&
Imm.Mods.hasFPModifiers());
2364void AMDGPUOperand::addImmOperands(MCInst &Inst,
unsigned N,
2365 bool ApplyModifiers)
const {
2375 addLiteralImmOperand(Inst,
Imm.Val,
2376 ApplyModifiers & isImmTy(ImmTyNone) &&
2377 Imm.Mods.hasFPModifiers());
2379 assert(!isImmTy(ImmTyNone) || !hasModifiers());
2384void AMDGPUOperand::addLiteralImmOperand(MCInst &Inst, int64_t Val,
2385 bool ApplyModifiers)
const {
2386 const auto &InstDesc = AsmParser->getMII()->get(Inst.
getOpcode());
2391 if (ApplyModifiers) {
2393 const unsigned Size =
2395 Val = applyInputFPModifiers(Val,
Size);
2399 uint8_t OpTy = InstDesc.operands()[OpNum].OperandType;
2401 bool CanUse64BitLiterals =
2404 MCContext &Ctx = AsmParser->getContext();
2415 if (
Lit == LitModifier::None &&
2417 AsmParser->hasInv2PiInlineImm())) {
2425 bool HasMandatoryLiteral =
2428 if (
Literal.getLoBits(32) != 0 &&
2429 (InstDesc.getSize() != 4 || !AsmParser->has64BitLiterals()) &&
2430 !HasMandatoryLiteral) {
2431 const_cast<AMDGPUAsmParser *
>(AsmParser)->
Warning(
2433 "Can't encode literal as exact 64-bit floating-point operand. "
2434 "Low 32-bits will be set to zero");
2435 Val &= 0xffffffff00000000u;
2441 if (CanUse64BitLiterals &&
Lit == LitModifier::None &&
2447 Lit = LitModifier::Lit64;
2448 }
else if (
Lit == LitModifier::Lit) {
2462 if (CanUse64BitLiterals &&
Lit == LitModifier::None &&
2464 Lit = LitModifier::Lit64;
2471 if (
Lit == LitModifier::None && AsmParser->hasInv2PiInlineImm() &&
2472 Literal == 0x3fc45f306725feed) {
2512 Val = FPLiteral.bitcastToAPInt().getZExtValue();
2519 if (
Lit != LitModifier::None) {
2550 if (
Lit == LitModifier::None &&
2560 if (!AsmParser->has64BitLiterals() ||
Lit == LitModifier::Lit)
2568 if (
Lit == LitModifier::None &&
2576 if (!AsmParser->has64BitLiterals()) {
2577 Val =
static_cast<uint64_t>(Val) << 32;
2584 if (
Lit == LitModifier::Lit ||
2586 Val =
static_cast<uint64_t>(Val) << 32;
2590 if (
Lit == LitModifier::Lit)
2617 if (
Lit != LitModifier::None) {
2625void AMDGPUOperand::addRegOperands(MCInst &Inst,
unsigned N)
const {
2631bool AMDGPUOperand::isInlineValue()
const {
2639void AMDGPUAsmParser::createConstantSymbol(StringRef Id, int64_t Val) {
2650 if (Is == IS_VGPR) {
2655 return AMDGPU::VGPR_32RegClassID;
2657 return AMDGPU::VReg_64RegClassID;
2659 return AMDGPU::VReg_96RegClassID;
2661 return AMDGPU::VReg_128RegClassID;
2663 return AMDGPU::VReg_160RegClassID;
2665 return AMDGPU::VReg_192RegClassID;
2667 return AMDGPU::VReg_224RegClassID;
2669 return AMDGPU::VReg_256RegClassID;
2671 return AMDGPU::VReg_288RegClassID;
2673 return AMDGPU::VReg_320RegClassID;
2675 return AMDGPU::VReg_352RegClassID;
2677 return AMDGPU::VReg_384RegClassID;
2679 return AMDGPU::VReg_512RegClassID;
2681 return AMDGPU::VReg_1024RegClassID;
2683 }
else if (Is == IS_TTMP) {
2688 return AMDGPU::TTMP_32RegClassID;
2690 return AMDGPU::TTMP_64RegClassID;
2692 return AMDGPU::TTMP_128RegClassID;
2694 return AMDGPU::TTMP_256RegClassID;
2696 return AMDGPU::TTMP_512RegClassID;
2698 }
else if (Is == IS_SGPR) {
2703 return AMDGPU::SGPR_32RegClassID;
2705 return AMDGPU::SGPR_64RegClassID;
2707 return AMDGPU::SGPR_96RegClassID;
2709 return AMDGPU::SGPR_128RegClassID;
2711 return AMDGPU::SGPR_160RegClassID;
2713 return AMDGPU::SGPR_192RegClassID;
2715 return AMDGPU::SGPR_224RegClassID;
2717 return AMDGPU::SGPR_256RegClassID;
2719 return AMDGPU::SGPR_288RegClassID;
2721 return AMDGPU::SGPR_320RegClassID;
2723 return AMDGPU::SGPR_352RegClassID;
2725 return AMDGPU::SGPR_384RegClassID;
2727 return AMDGPU::SGPR_512RegClassID;
2729 }
else if (Is == IS_AGPR) {
2734 return AMDGPU::AGPR_32RegClassID;
2736 return AMDGPU::AReg_64RegClassID;
2738 return AMDGPU::AReg_96RegClassID;
2740 return AMDGPU::AReg_128RegClassID;
2742 return AMDGPU::AReg_160RegClassID;
2744 return AMDGPU::AReg_192RegClassID;
2746 return AMDGPU::AReg_224RegClassID;
2748 return AMDGPU::AReg_256RegClassID;
2750 return AMDGPU::AReg_288RegClassID;
2752 return AMDGPU::AReg_320RegClassID;
2754 return AMDGPU::AReg_352RegClassID;
2756 return AMDGPU::AReg_384RegClassID;
2758 return AMDGPU::AReg_512RegClassID;
2760 return AMDGPU::AReg_1024RegClassID;
2768 .
Case(
"exec", AMDGPU::EXEC)
2769 .
Case(
"vcc", AMDGPU::VCC)
2770 .
Case(
"flat_scratch", AMDGPU::FLAT_SCR)
2771 .
Case(
"xnack_mask", AMDGPU::XNACK_MASK)
2772 .
Case(
"shared_base", AMDGPU::SRC_SHARED_BASE)
2773 .
Case(
"src_shared_base", AMDGPU::SRC_SHARED_BASE)
2774 .
Case(
"shared_limit", AMDGPU::SRC_SHARED_LIMIT)
2775 .
Case(
"src_shared_limit", AMDGPU::SRC_SHARED_LIMIT)
2776 .
Case(
"private_base", AMDGPU::SRC_PRIVATE_BASE)
2777 .
Case(
"src_private_base", AMDGPU::SRC_PRIVATE_BASE)
2778 .
Case(
"private_limit", AMDGPU::SRC_PRIVATE_LIMIT)
2779 .
Case(
"src_private_limit", AMDGPU::SRC_PRIVATE_LIMIT)
2780 .
Case(
"src_flat_scratch_base_lo", AMDGPU::SRC_FLAT_SCRATCH_BASE_LO)
2781 .
Case(
"src_flat_scratch_base_hi", AMDGPU::SRC_FLAT_SCRATCH_BASE_HI)
2782 .
Case(
"pops_exiting_wave_id", AMDGPU::SRC_POPS_EXITING_WAVE_ID)
2783 .
Case(
"src_pops_exiting_wave_id", AMDGPU::SRC_POPS_EXITING_WAVE_ID)
2784 .
Case(
"lds_direct", AMDGPU::LDS_DIRECT)
2785 .
Case(
"src_lds_direct", AMDGPU::LDS_DIRECT)
2786 .
Case(
"m0", AMDGPU::M0)
2787 .
Case(
"vccz", AMDGPU::SRC_VCCZ)
2788 .
Case(
"src_vccz", AMDGPU::SRC_VCCZ)
2789 .
Case(
"execz", AMDGPU::SRC_EXECZ)
2790 .
Case(
"src_execz", AMDGPU::SRC_EXECZ)
2791 .
Case(
"scc", AMDGPU::SRC_SCC)
2792 .
Case(
"src_scc", AMDGPU::SRC_SCC)
2793 .
Case(
"tba", AMDGPU::TBA)
2794 .
Case(
"tma", AMDGPU::TMA)
2795 .
Case(
"flat_scratch_lo", AMDGPU::FLAT_SCR_LO)
2796 .
Case(
"flat_scratch_hi", AMDGPU::FLAT_SCR_HI)
2797 .
Case(
"xnack_mask_lo", AMDGPU::XNACK_MASK_LO)
2798 .
Case(
"xnack_mask_hi", AMDGPU::XNACK_MASK_HI)
2799 .
Case(
"vcc_lo", AMDGPU::VCC_LO)
2800 .
Case(
"vcc_hi", AMDGPU::VCC_HI)
2801 .
Case(
"exec_lo", AMDGPU::EXEC_LO)
2802 .
Case(
"exec_hi", AMDGPU::EXEC_HI)
2803 .
Case(
"tma_lo", AMDGPU::TMA_LO)
2804 .
Case(
"tma_hi", AMDGPU::TMA_HI)
2805 .
Case(
"tba_lo", AMDGPU::TBA_LO)
2806 .
Case(
"tba_hi", AMDGPU::TBA_HI)
2807 .
Case(
"pc", AMDGPU::PC_REG)
2808 .
Case(
"null", AMDGPU::SGPR_NULL)
2812bool AMDGPUAsmParser::ParseRegister(MCRegister &RegNo, SMLoc &StartLoc,
2813 SMLoc &EndLoc,
bool RestoreOnFailure) {
2814 auto R = parseRegister();
2818 RegNo =
R->getReg();
2819 StartLoc =
R->getStartLoc();
2820 EndLoc =
R->getEndLoc();
2824bool AMDGPUAsmParser::parseRegister(MCRegister &
Reg, SMLoc &StartLoc,
2826 return ParseRegister(
Reg, StartLoc, EndLoc,
false);
2829ParseStatus AMDGPUAsmParser::tryParseRegister(MCRegister &
Reg, SMLoc &StartLoc,
2831 bool Result = ParseRegister(
Reg, StartLoc, EndLoc,
true);
2832 bool PendingErrors = getParser().hasPendingError();
2833 getParser().clearPendingErrors();
2841bool AMDGPUAsmParser::AddNextRegisterToList(MCRegister &
Reg,
unsigned &RegWidth,
2842 RegisterKind RegKind,
2844 RegisterKind RegKind1, SMLoc Loc) {
2846 if (RegKind == IS_SGPR) {
2847 unsigned RegIdx = (
Reg - AMDGPU::SGPR0) + RegWidth / 32;
2848 if ((RegIdx == 106 && Reg1 == AMDGPU::VCC_LO) ||
2849 (RegIdx == 107 && Reg1 == AMDGPU::VCC_HI)) {
2855 if (RegKind != RegKind1) {
2856 Error(Loc,
"registers in a list must be of the same kind");
2862 if (
Reg == AMDGPU::EXEC_LO && Reg1 == AMDGPU::EXEC_HI) {
2867 if (
Reg == AMDGPU::FLAT_SCR_LO && Reg1 == AMDGPU::FLAT_SCR_HI) {
2868 Reg = AMDGPU::FLAT_SCR;
2872 if (
Reg == AMDGPU::XNACK_MASK_LO && Reg1 == AMDGPU::XNACK_MASK_HI) {
2873 Reg = AMDGPU::XNACK_MASK;
2877 if (
Reg == AMDGPU::VCC_LO && Reg1 == AMDGPU::VCC_HI) {
2882 if (
Reg == AMDGPU::TBA_LO && Reg1 == AMDGPU::TBA_HI) {
2887 if (
Reg == AMDGPU::TMA_LO && Reg1 == AMDGPU::TMA_HI) {
2892 Error(Loc,
"register does not fit in the list");
2898 if (Reg1 !=
Reg + RegWidth / 32) {
2899 Error(Loc,
"registers in a list must have consecutive indices");
2915 {{
"v"}, IS_VGPR}, {{
"s"}, IS_SGPR}, {{
"ttmp"}, IS_TTMP},
2916 {{
"acc"}, IS_AGPR}, {{
"a"}, IS_AGPR},
2920 return Kind == IS_VGPR || Kind == IS_SGPR || Kind == IS_TTMP ||
2926 if (Str.starts_with(
Reg.Name))
2932 return !Str.getAsInteger(10, Num);
2935bool AMDGPUAsmParser::isRegister(
const AsmToken &Token,
2936 const AsmToken &NextToken)
const {
2951 StringRef RegSuffix = Str.substr(
RegName.size());
2952 if (!RegSuffix.
empty()) {
2969bool AMDGPUAsmParser::isRegister() {
2970 return isRegister(
getToken(), peekToken());
2973MCRegister AMDGPUAsmParser::getRegularReg(RegisterKind RegKind,
unsigned RegNum,
2974 unsigned SubReg,
unsigned RegWidth,
2978 unsigned AlignSize = 1;
2979 if (RegKind == IS_SGPR || RegKind == IS_TTMP) {
2985 if (RegNum % AlignSize != 0) {
2986 Error(Loc,
"invalid register alignment");
2987 return MCRegister();
2990 unsigned RegIdx = RegNum / AlignSize;
2993 Error(Loc,
"invalid or unsupported register size");
2994 return MCRegister();
2998 const MCRegisterClass &RC =
TRI->getRegClass(RCID);
2999 if (RegIdx >= RC.
getNumRegs() || (RegKind == IS_VGPR && RegIdx > 255)) {
3000 Error(Loc,
"register index is out of range");
3001 return AMDGPU::NoRegister;
3004 if (RegKind == IS_VGPR && !
isGFX1250Plus() && RegIdx + RegWidth / 32 > 256) {
3005 Error(Loc,
"register index is out of range");
3006 return MCRegister();
3015 Error(Loc,
"invalid subregister");
3021bool AMDGPUAsmParser::ParseRegRange(
unsigned &Num,
unsigned &RegWidth,
3023 int64_t RegLo, RegHi;
3027 SMLoc FirstIdxLoc = getLoc();
3034 SecondIdxLoc = getLoc();
3045 Error(FirstIdxLoc,
"invalid register index");
3050 Error(SecondIdxLoc,
"invalid register index");
3054 if (RegLo > RegHi) {
3055 Error(FirstIdxLoc,
"first register index should not exceed second index");
3059 if (RegHi == RegLo) {
3060 StringRef RegSuffix = getTokenStr();
3061 if (RegSuffix ==
".l") {
3062 SubReg = AMDGPU::lo16;
3064 }
else if (RegSuffix ==
".h") {
3065 SubReg = AMDGPU::hi16;
3070 Num =
static_cast<unsigned>(RegLo);
3071 RegWidth = 32 * ((RegHi - RegLo) + 1);
3076MCRegister AMDGPUAsmParser::ParseSpecialReg(RegisterKind &RegKind,
3079 SmallVectorImpl<AsmToken> &Tokens) {
3085 RegKind = IS_SPECIAL;
3092MCRegister AMDGPUAsmParser::ParseRegularReg(RegisterKind &RegKind,
3095 SmallVectorImpl<AsmToken> &Tokens) {
3097 StringRef
RegName = getTokenStr();
3098 auto Loc = getLoc();
3102 Error(Loc,
"invalid register name");
3103 return MCRegister();
3111 unsigned SubReg = NoSubRegister;
3112 bool IsRange =
false;
3113 if (!RegSuffix.
empty()) {
3115 SubReg = AMDGPU::lo16;
3117 SubReg = AMDGPU::hi16;
3121 Error(Loc,
"invalid register index");
3122 return MCRegister();
3128 if (!ParseRegRange(RegNum, RegWidth, SubReg))
3129 return MCRegister();
3133 MCRegister
Reg = getRegularReg(RegKind, RegNum, SubReg, RegWidth, Loc);
3134 const MCRegisterInfo &
TRI = *
getContext().getRegisterInfo();
3135 if (RegKind == IS_SGPR && IsRange
3136 ? (
TRI.isSubRegister(
Reg, VCC_LO) ||
TRI.isSubRegister(
Reg, VCC_HI))
3137 : (
Reg == VCC_LO ||
Reg == VCC_HI)) {
3138 Error(Loc,
"register index is out of range");
3139 return MCRegister();
3145MCRegister AMDGPUAsmParser::ParseRegList(RegisterKind &RegKind,
3146 unsigned &RegNum,
unsigned &RegWidth,
3147 SmallVectorImpl<AsmToken> &Tokens) {
3149 auto ListLoc = getLoc();
3152 "expected a register or a list of registers")) {
3153 return MCRegister();
3158 auto Loc = getLoc();
3159 if (!ParseAMDGPURegister(RegKind,
Reg, RegNum, RegWidth))
3160 return MCRegister();
3161 if (RegWidth != 32) {
3162 Error(Loc,
"expected a single 32-bit register");
3163 return MCRegister();
3167 RegisterKind NextRegKind;
3169 unsigned NextRegNum, NextRegWidth;
3172 if (!ParseAMDGPURegister(NextRegKind, NextReg, NextRegNum, NextRegWidth,
3174 return MCRegister();
3176 if (NextRegWidth != 32) {
3177 Error(Loc,
"expected a single 32-bit register");
3178 return MCRegister();
3180 if (!AddNextRegisterToList(
Reg, RegWidth, RegKind, NextReg, NextRegKind,
3182 return MCRegister();
3186 "expected a comma or a closing square bracket")) {
3187 return MCRegister();
3191 Reg = getRegularReg(RegKind, RegNum, NoSubRegister, RegWidth, ListLoc);
3196bool AMDGPUAsmParser::ParseAMDGPURegister(RegisterKind &RegKind,
3197 MCRegister &
Reg,
unsigned &RegNum,
3199 SmallVectorImpl<AsmToken> &Tokens) {
3200 auto Loc = getLoc();
3204 Reg = ParseSpecialReg(RegKind, RegNum, RegWidth, Tokens);
3206 Reg = ParseRegularReg(RegKind, RegNum, RegWidth, Tokens);
3208 Reg = ParseRegList(RegKind, RegNum, RegWidth, Tokens);
3213 assert(Parser.hasPendingError());
3217 if (!subtargetHasRegister(*
TRI,
Reg)) {
3218 if (
Reg == AMDGPU::SGPR_NULL) {
3219 Error(Loc,
"'null' operand is not supported on this GPU");
3222 " register not available on this GPU");
3230bool AMDGPUAsmParser::ParseAMDGPURegister(RegisterKind &RegKind,
3231 MCRegister &
Reg,
unsigned &RegNum,
3233 bool RestoreOnFailure ) {
3237 if (ParseAMDGPURegister(RegKind,
Reg, RegNum, RegWidth, Tokens)) {
3238 if (RestoreOnFailure) {
3239 while (!Tokens.
empty()) {
3248std::optional<StringRef>
3249AMDGPUAsmParser::getGprCountSymbolName(RegisterKind RegKind) {
3252 return StringRef(
".amdgcn.next_free_vgpr");
3254 return StringRef(
".amdgcn.next_free_sgpr");
3256 return std::nullopt;
3260void AMDGPUAsmParser::initializeGprCountSymbol(RegisterKind RegKind) {
3261 auto SymbolName = getGprCountSymbolName(RegKind);
3262 assert(SymbolName &&
"initializing invalid register kind");
3268bool AMDGPUAsmParser::updateGprCountSymbols(RegisterKind RegKind,
3269 unsigned DwordRegIndex,
3270 unsigned RegWidth) {
3275 auto SymbolName = getGprCountSymbolName(RegKind);
3280 int64_t NewMax = DwordRegIndex +
divideCeil(RegWidth, 32) - 1;
3284 return !
Error(getLoc(),
3285 ".amdgcn.next_free_{v,s}gpr symbols must be variable");
3289 ".amdgcn.next_free_{v,s}gpr symbols must be absolute expressions");
3291 if (OldCount <= NewMax)
3297std::unique_ptr<AMDGPUOperand>
3298AMDGPUAsmParser::parseRegister(
bool RestoreOnFailure) {
3300 SMLoc StartLoc = Tok.getLoc();
3301 SMLoc EndLoc = Tok.getEndLoc();
3302 RegisterKind RegKind;
3304 unsigned RegNum, RegWidth;
3306 if (!ParseAMDGPURegister(RegKind,
Reg, RegNum, RegWidth)) {
3310 if (!updateGprCountSymbols(RegKind, RegNum, RegWidth))
3313 KernelScope.usesRegister(RegKind, RegNum, RegWidth);
3314 return AMDGPUOperand::CreateReg(
this,
Reg, StartLoc, EndLoc);
3321 if (isRegister() || isModifier())
3324 if (
Lit == LitModifier::None) {
3325 if (trySkipId(
"lit"))
3326 Lit = LitModifier::Lit;
3327 else if (trySkipId(
"lit64"))
3328 Lit = LitModifier::Lit64;
3330 if (
Lit != LitModifier::None) {
3333 ParseStatus S = parseImm(
Operands, HasSP3AbsModifier,
Lit);
3342 const auto &NextTok = peekToken();
3345 bool Negate =
false;
3353 AMDGPUOperand::Modifiers Mods;
3361 StringRef Num = getTokenStr();
3364 APFloat RealVal(APFloat::IEEEdouble());
3365 auto roundMode = APFloat::rmNearestTiesToEven;
3366 if (
errorToBool(RealVal.convertFromString(Num, roundMode).takeError()))
3369 RealVal.changeSign();
3372 AMDGPUOperand::CreateImm(
this, RealVal.bitcastToAPInt().getZExtValue(),
3373 S, AMDGPUOperand::ImmTyNone,
true));
3374 AMDGPUOperand &
Op =
static_cast<AMDGPUOperand &
>(*
Operands.back());
3375 Op.setModifiers(Mods);
3384 if (HasSP3AbsModifier) {
3393 if (getParser().parsePrimaryExpr(Expr, EndLoc,
nullptr))
3396 if (Parser.parseExpression(Expr))
3400 if (Expr->evaluateAsAbsolute(IntVal)) {
3402 return Error(S,
"literal value out of range");
3403 Operands.push_back(AMDGPUOperand::CreateImm(
this, IntVal, S));
3404 AMDGPUOperand &
Op =
static_cast<AMDGPUOperand &
>(*
Operands.back());
3405 Op.setModifiers(Mods);
3407 if (
Lit != LitModifier::None)
3409 Operands.push_back(AMDGPUOperand::CreateExpr(
this, Expr, S));
3422 if (
auto R = parseRegister()) {
3432 ParseStatus Res = parseReg(
Operands);
3440bool AMDGPUAsmParser::isNamedOperandModifier(
const AsmToken &Token,
3441 const AsmToken &NextToken)
const {
3444 return str ==
"abs" || str ==
"neg" || str ==
"sext";
3449bool AMDGPUAsmParser::isOpcodeModifierWithVal(
const AsmToken &Token,
3450 const AsmToken &NextToken)
const {
3454bool AMDGPUAsmParser::isOperandModifier(
const AsmToken &Token,
3455 const AsmToken &NextToken)
const {
3456 return isNamedOperandModifier(Token, NextToken) || Token.
is(
AsmToken::Pipe);
3459bool AMDGPUAsmParser::isRegOrOperandModifier(
const AsmToken &Token,
3460 const AsmToken &NextToken)
const {
3461 return isRegister(Token, NextToken) || isOperandModifier(Token, NextToken);
3477bool AMDGPUAsmParser::isModifier() {
3480 AsmToken NextToken[2];
3481 peekTokens(NextToken);
3483 return isOperandModifier(Tok, NextToken[0]) ||
3485 isRegOrOperandModifier(NextToken[0], NextToken[1])) ||
3486 isOpcodeModifierWithVal(Tok, NextToken[0]);
3511bool AMDGPUAsmParser::parseSP3NegModifier() {
3513 AsmToken NextToken[2];
3514 peekTokens(NextToken);
3517 (isRegister(NextToken[0], NextToken[1]) ||
3535 return Error(getLoc(),
"invalid syntax, expected 'neg' modifier");
3537 SP3Neg = parseSP3NegModifier();
3540 Neg = trySkipId(
"neg");
3542 return Error(Loc,
"expected register or immediate");
3546 Abs = trySkipId(
"abs");
3551 if (trySkipId(
"lit")) {
3552 Lit = LitModifier::Lit;
3555 }
else if (trySkipId(
"lit64")) {
3556 Lit = LitModifier::Lit64;
3559 if (!has64BitLiterals())
3560 return Error(Loc,
"lit64 is not supported on this GPU");
3566 return Error(Loc,
"expected register or immediate");
3575 return (SP3Neg || Neg || SP3Abs || Abs ||
Lit != LitModifier::None)
3579 if (
Lit != LitModifier::None && !
Operands.back()->isImm())
3580 Error(Loc,
"expected immediate with lit modifier");
3582 if (SP3Abs && !skipToken(
AsmToken::Pipe,
"expected vertical bar"))
3588 if (
Lit != LitModifier::None &&
3592 AMDGPUOperand::Modifiers Mods;
3593 Mods.Abs = Abs || SP3Abs;
3594 Mods.Neg = Neg || SP3Neg;
3597 if (Mods.hasFPModifiers() ||
Lit != LitModifier::None) {
3598 AMDGPUOperand &
Op =
static_cast<AMDGPUOperand &
>(*
Operands.back());
3600 return Error(
Op.getStartLoc(),
"expected an absolute expression");
3601 Op.setModifiers(Mods);
3609 bool Sext = trySkipId(
"sext");
3610 if (Sext && !skipToken(
AsmToken::LParen,
"expected left paren after sext"))
3625 AMDGPUOperand::Modifiers Mods;
3628 if (Mods.hasIntModifiers()) {
3629 AMDGPUOperand &
Op =
static_cast<AMDGPUOperand &
>(*
Operands.back());
3631 return Error(
Op.getStartLoc(),
"expected an absolute expression");
3632 Op.setModifiers(Mods);
3639 return parseRegOrImmWithFPInputMods(
Operands,
false);
3643 return parseRegOrImmWithIntInputMods(
Operands,
false);
3650 if (!trySkipId(
"rsrcidx"))
3656 SMLoc RegLoc = getLoc();
3657 std::unique_ptr<AMDGPUOperand>
Reg = parseRegister();
3665 if (!
Reg->isRsrcReg32())
3666 return Error(RegLoc,
"rsrcidx operand must be a 32-bit SGPR or VGPR");
3676 auto Loc = getLoc();
3677 if (trySkipId(
"off")) {
3679 AMDGPUOperand::CreateImm(
this, 0, Loc, AMDGPUOperand::ImmTyOff,
false));
3686 std::unique_ptr<AMDGPUOperand>
Reg = parseRegister();
3695unsigned AMDGPUAsmParser::checkTargetMatchPredicate(MCInst &Inst) {
3700 return Match_InvalidOperand;
3702 if (Inst.
getOpcode() == AMDGPU::V_MAC_F32_sdwa_vi ||
3703 Inst.
getOpcode() == AMDGPU::V_MAC_F16_sdwa_vi) {
3706 AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), AMDGPU::OpName::dst_sel);
3708 if (!
Op.isImm() ||
Op.getImm() != AMDGPU::SDWA::SdwaSel::DWORD) {
3709 return Match_InvalidOperand;
3717 if (tryAnotherVOPDEncoding(Inst))
3718 return Match_InvalidOperand;
3720 return Match_Success;
3724 static const unsigned Variants[] = {
3733ArrayRef<unsigned> AMDGPUAsmParser::getMatchedVariants()
const {
3734 if (isForcedDPP() && isForcedVOP3()) {
3738 if (getForcedEncodingSize() == 32) {
3743 if (isForcedVOP3()) {
3748 if (isForcedSDWA()) {
3754 if (isForcedDPP()) {
3762StringRef AMDGPUAsmParser::getMatchedVariantName()
const {
3763 if (isForcedDPP() && isForcedVOP3())
3766 if (getForcedEncodingSize() == 32)
3782AMDGPUAsmParser::findImplicitSGPRReadInVOP(
const MCInst &Inst)
const {
3786 case AMDGPU::FLAT_SCR:
3788 case AMDGPU::VCC_LO:
3789 case AMDGPU::VCC_HI:
3796 return MCRegister();
3803bool AMDGPUAsmParser::isInlineConstant(
const MCInst &Inst,
3804 unsigned OpIdx)
const {
3812 const MCOperand &MO = Inst.
getOperand(OpIdx);
3862unsigned AMDGPUAsmParser::getConstantBusLimit(
unsigned Opcode)
const {
3868 case AMDGPU::V_LSHLREV_B64_e64:
3869 case AMDGPU::V_LSHLREV_B64_gfx10:
3870 case AMDGPU::V_LSHLREV_B64_e64_gfx11:
3871 case AMDGPU::V_LSHLREV_B64_e32_gfx12:
3872 case AMDGPU::V_LSHLREV_B64_e64_gfx12:
3873 case AMDGPU::V_LSHRREV_B64_e64:
3874 case AMDGPU::V_LSHRREV_B64_gfx10:
3875 case AMDGPU::V_LSHRREV_B64_e64_gfx11:
3876 case AMDGPU::V_LSHRREV_B64_e64_gfx12:
3877 case AMDGPU::V_ASHRREV_I64_e64:
3878 case AMDGPU::V_ASHRREV_I64_gfx10:
3879 case AMDGPU::V_ASHRREV_I64_e64_gfx11:
3880 case AMDGPU::V_ASHRREV_I64_e64_gfx12:
3881 case AMDGPU::V_LSHL_B64_e64:
3882 case AMDGPU::V_LSHR_B64_e64:
3883 case AMDGPU::V_ASHR_I64_e64:
3896 bool AddMandatoryLiterals =
false) {
3899 AddMandatoryLiterals ? getNamedOperandIdx(Opcode, OpName::imm) : -1;
3903 AddMandatoryLiterals ? getNamedOperandIdx(Opcode, OpName::immX) : -1;
3905 return {getNamedOperandIdx(Opcode, OpName::src0X),
3906 getNamedOperandIdx(Opcode, OpName::vsrc1X),
3907 getNamedOperandIdx(Opcode, OpName::vsrc2X),
3908 getNamedOperandIdx(Opcode, OpName::src0Y),
3909 getNamedOperandIdx(Opcode, OpName::vsrc1Y),
3910 getNamedOperandIdx(Opcode, OpName::vsrc2Y),
3915 return {getNamedOperandIdx(Opcode, OpName::src0),
3916 getNamedOperandIdx(Opcode, OpName::src1),
3917 getNamedOperandIdx(Opcode, OpName::src2), ImmIdx};
3920bool AMDGPUAsmParser::usesConstantBus(
const MCInst &Inst,
unsigned OpIdx) {
3921 const MCOperand &MO = Inst.
getOperand(OpIdx);
3923 return !isInlineConstant(Inst, OpIdx);
3930 return isSGPR(PReg,
TRI) && PReg != SGPR_NULL;
3941 const unsigned Opcode = Inst.
getOpcode();
3942 if (Opcode != V_WRITELANE_B32_gfx6_gfx7 && Opcode != V_WRITELANE_B32_vi)
3945 if (!LaneSelOp.
isReg())
3948 return LaneSelReg ==
M0 || LaneSelReg == M0_gfxpre11;
3951bool AMDGPUAsmParser::validateConstantBusLimitations(
3953 const unsigned Opcode = Inst.
getOpcode();
3954 const MCInstrDesc &
Desc = MII.
get(Opcode);
3955 MCRegister LastSGPR;
3956 unsigned ConstantBusUseCount = 0;
3957 unsigned NumLiterals = 0;
3958 unsigned LiteralSize;
3974 SmallDenseSet<MCRegister> SGPRsUsed;
3975 MCRegister SGPRUsed = findImplicitSGPRReadInVOP(Inst);
3977 SGPRsUsed.
insert(SGPRUsed);
3978 ++ConstantBusUseCount;
3983 unsigned ConstantBusLimit = getConstantBusLimit(Opcode);
3985 for (
int OpIdx : OpIndices) {
3989 const MCOperand &MO = Inst.
getOperand(OpIdx);
3990 if (usesConstantBus(Inst, OpIdx)) {
3999 if (SGPRsUsed.
insert(LastSGPR).second) {
4000 ++ConstantBusUseCount;
4020 if (NumLiterals == 0) {
4023 }
else if (LiteralSize !=
Size) {
4029 if (ConstantBusUseCount + NumLiterals > ConstantBusLimit) {
4031 "invalid operand (violates constant bus restrictions)");
4038std::optional<unsigned>
4039AMDGPUAsmParser::checkVOPDRegBankConstraints(
const MCInst &Inst,
bool AsVOPD3) {
4041 const unsigned Opcode = Inst.
getOpcode();
4047 auto getVRegIdx = [&](unsigned,
unsigned OperandIdx) {
4048 const MCOperand &Opr = Inst.
getOperand(OperandIdx);
4057 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx1170 ||
4058 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx12 ||
4059 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx1250 ||
4060 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx13 ||
4061 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_e96_gfx1250 ||
4062 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_e96_gfx13;
4066 for (
auto OpName : {OpName::src0X, OpName::src0Y}) {
4067 int I = getNamedOperandIdx(Opcode, OpName);
4071 int64_t
Imm =
Op.getImm();
4077 for (
auto OpName : {OpName::vsrc1X, OpName::vsrc1Y, OpName::vsrc2X,
4078 OpName::vsrc2Y, OpName::imm}) {
4079 int I = getNamedOperandIdx(Opcode, OpName);
4089 auto InvalidCompOprIdx = InstInfo.getInvalidCompOperandIndex(
4090 getVRegIdx, *
TRI, SkipSrc, AllowSameVGPR, AsVOPD3);
4092 return InvalidCompOprIdx;
4095bool AMDGPUAsmParser::validateVOPD(
const MCInst &Inst,
4102 for (
const std::unique_ptr<MCParsedAsmOperand> &Operand :
Operands) {
4103 AMDGPUOperand &
Op = (AMDGPUOperand &)*Operand;
4104 if ((
Op.isRegKind() ||
Op.isImmTy(AMDGPUOperand::ImmTyNone)) &&
4106 Error(
Op.getStartLoc(),
"ABS not allowed in VOPD3 instructions");
4110 auto InvalidCompOprIdx = checkVOPDRegBankConstraints(Inst, AsVOPD3);
4111 if (!InvalidCompOprIdx.has_value())
4114 auto CompOprIdx = *InvalidCompOprIdx;
4117 std::max(InstInfo[
VOPD::X].getIndexInParsedOperands(CompOprIdx),
4118 InstInfo[
VOPD::Y].getIndexInParsedOperands(CompOprIdx));
4121 auto Loc = ((AMDGPUOperand &)*
Operands[ParsedIdx]).getStartLoc();
4122 if (CompOprIdx == VOPD::Component::DST) {
4124 Error(Loc,
"dst registers must be distinct");
4126 Error(Loc,
"one dst register must be even and the other odd");
4128 auto CompSrcIdx = CompOprIdx - VOPD::Component::DST_NUM;
4129 Error(Loc, Twine(
"src") + Twine(CompSrcIdx) +
4130 " operands must use different VGPR banks");
4138bool AMDGPUAsmParser::tryVOPD3(
const MCInst &Inst) {
4140 auto InvalidCompOprIdx = checkVOPDRegBankConstraints(Inst,
false);
4141 if (!InvalidCompOprIdx.has_value())
4145 InvalidCompOprIdx = checkVOPDRegBankConstraints(Inst,
true);
4146 if (InvalidCompOprIdx.has_value()) {
4151 if (*InvalidCompOprIdx == VOPD::Component::DST)
4164bool AMDGPUAsmParser::tryVOPD(
const MCInst &Inst) {
4165 const unsigned Opcode = Inst.
getOpcode();
4180 for (
auto OpName : {OpName::src0X_modifiers, OpName::src0Y_modifiers,
4181 OpName::vsrc1X_modifiers, OpName::vsrc1Y_modifiers,
4182 OpName::vsrc2X_modifiers, OpName::vsrc2Y_modifiers}) {
4183 int I = getNamedOperandIdx(Opcode, OpName);
4190 return !tryVOPD3(Inst);
4195bool AMDGPUAsmParser::tryAnotherVOPDEncoding(
const MCInst &Inst) {
4200 return tryVOPD(Inst);
4201 return tryVOPD3(Inst);
4204bool AMDGPUAsmParser::validateIntClampSupported(
const MCInst &Inst) {
4209 int ClampIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::clamp);
4217bool AMDGPUAsmParser::validateMIMGDataSize(
const MCInst &Inst, SMLoc IDLoc) {
4225 int VDataIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdata);
4226 int DMaskIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::dmask);
4227 int TFEIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::tfe);
4232 if ((DMaskIdx == -1 || TFEIdx == -1) &&
4233 hasBVHRayTracingInsts())
4236 unsigned VDataSize = getRegOperandSize(
Desc, VDataIdx);
4237 unsigned TFESize = (TFEIdx != -1 && Inst.
getOperand(TFEIdx).
getImm()) ? 1 : 0;
4242 bool IsPackedD16 =
false;
4245 int D16Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::d16);
4246 IsPackedD16 = D16Idx >= 0;
4248 DataSize = (DataSize + 1) / 2;
4251 if ((VDataSize / 4) == DataSize + TFESize)
4256 Modifiers = IsPackedD16 ?
"dmask and d16" :
"dmask";
4258 Modifiers = IsPackedD16 ?
"dmask, d16 and tfe" :
"dmask and tfe";
4260 Error(IDLoc, Twine(
"image data size does not match ") + Modifiers);
4264bool AMDGPUAsmParser::validateMIMGAddrSize(
const MCInst &Inst, SMLoc IDLoc) {
4273 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode =
4275 int VAddr0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr0);
4276 AMDGPU::OpName RSrcOpName =
4278 int SrsrcIdx = AMDGPU::getNamedOperandIdx(
Opc, RSrcOpName);
4279 int DimIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::dim);
4280 int A16Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::a16);
4284 assert(SrsrcIdx > VAddr0Idx);
4287 if (BaseOpcode->
BVH) {
4288 if (IsA16 == BaseOpcode->
A16)
4290 Error(IDLoc,
"image address size does not match a16");
4296 bool IsNSA = SrsrcIdx - VAddr0Idx > 1;
4297 unsigned ActualAddrSize =
4298 IsNSA ? SrsrcIdx - VAddr0Idx : getRegOperandSize(
Desc, VAddr0Idx) / 4;
4300 unsigned ExpectedAddrSize =
4304 if (hasPartialNSAEncoding() &&
4306 int VAddrLastIdx = SrsrcIdx - 1;
4307 unsigned VAddrLastSize = getRegOperandSize(
Desc, VAddrLastIdx) / 4;
4309 ActualAddrSize = VAddrLastIdx - VAddr0Idx + VAddrLastSize;
4312 if (ExpectedAddrSize > 12)
4313 ExpectedAddrSize = 16;
4318 if (ActualAddrSize == 8 && (ExpectedAddrSize >= 5 && ExpectedAddrSize <= 7))
4322 if (ActualAddrSize == ExpectedAddrSize)
4325 Error(IDLoc,
"image address size does not match dim and a16");
4329bool AMDGPUAsmParser::validateMIMGAtomicDMask(
const MCInst &Inst) {
4336 if (!
Desc.mayLoad() || !
Desc.mayStore())
4339 int DMaskIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::dmask);
4346 return DMask == 0x1 || DMask == 0x3 || DMask == 0xf;
4349bool AMDGPUAsmParser::validateMIMGGatherDMask(
const MCInst &Inst) {
4356 int DMaskIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::dmask);
4364 return DMask == 0x1 || DMask == 0x2 || DMask == 0x4 || DMask == 0x8;
4367bool AMDGPUAsmParser::validateMIMGDim(
const MCInst &Inst,
4381 for (
unsigned i = 1, e =
Operands.size(); i != e; ++i) {
4382 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
4389bool AMDGPUAsmParser::validateMIMGMSAA(
const MCInst &Inst) {
4396 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode =
4399 if (!BaseOpcode->
MSAA)
4402 int DimIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::dim);
4408 return DimInfo->
MSAA;
4413 case AMDGPU::V_MOVRELS_B32_sdwa_gfx10:
4414 case AMDGPU::V_MOVRELSD_B32_sdwa_gfx10:
4415 case AMDGPU::V_MOVRELSD_2_B32_sdwa_gfx10:
4425bool AMDGPUAsmParser::validateMovrels(
const MCInst &Inst,
4433 const int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
4436 const MCOperand &Src0 = Inst.
getOperand(Src0Idx);
4444 Error(getOperandLoc(
Operands, Src0Idx),
"source operand must be a VGPR");
4448bool AMDGPUAsmParser::validateMAIAccWrite(
const MCInst &Inst,
4453 if (
Opc != AMDGPU::V_ACCVGPR_WRITE_B32_vi)
4456 const int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
4459 const MCOperand &Src0 = Inst.
getOperand(Src0Idx);
4467 "source operand must be either a VGPR or an inline constant");
4474bool AMDGPUAsmParser::validateMAISrc2(
const MCInst &Inst,
4479 !getFeatureBits()[FeatureMFMAInlineLiteralBug])
4482 const int Src2Idx = getNamedOperandIdx(Opcode, OpName::src2);
4486 if (Inst.
getOperand(Src2Idx).
isImm() && isInlineConstant(Inst, Src2Idx)) {
4488 "inline constants are not allowed for this operand");
4495bool AMDGPUAsmParser::validateMFMA(
const MCInst &Inst,
4503 int BlgpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::blgp);
4504 if (BlgpIdx != -1) {
4505 if (
const MFMA_F8F6F4_Info *Info = AMDGPU::isMFMA_F8F6F4(
Opc)) {
4506 int CbszIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::cbsz);
4516 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
4518 "wrong register tuple size for cbsz value " + Twine(CBSZ));
4523 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
4525 "wrong register tuple size for blgp value " + Twine(BLGP));
4533 const int Src2Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2);
4537 const MCOperand &Src2 = Inst.
getOperand(Src2Idx);
4541 MCRegister Src2Reg = Src2.
getReg();
4543 if (Src2Reg == DstReg)
4548 .getSizeInBits() <= 128)
4551 if (
TRI->regsOverlap(Src2Reg, DstReg)) {
4553 "source 2 operand must not partially overlap with dst");
4560bool AMDGPUAsmParser::validateDivScale(
const MCInst &Inst) {
4564 case V_DIV_SCALE_F32_gfx6_gfx7:
4565 case V_DIV_SCALE_F32_vi:
4566 case V_DIV_SCALE_F32_gfx10:
4567 case V_DIV_SCALE_F64_gfx6_gfx7:
4568 case V_DIV_SCALE_F64_vi:
4569 case V_DIV_SCALE_F64_gfx10:
4576 {AMDGPU::OpName::src0_modifiers, AMDGPU::OpName::src2_modifiers,
4577 AMDGPU::OpName::src2_modifiers}) {
4588bool AMDGPUAsmParser::validateMIMGD16(
const MCInst &Inst) {
4595 int D16Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::d16);
4604bool AMDGPUAsmParser::validateTensorR128(
const MCInst &Inst) {
4610 int R128Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::r128);
4617 case AMDGPU::V_SUBREV_F32_e32:
4618 case AMDGPU::V_SUBREV_F32_e64:
4619 case AMDGPU::V_SUBREV_F32_e32_gfx10:
4620 case AMDGPU::V_SUBREV_F32_e32_gfx6_gfx7:
4621 case AMDGPU::V_SUBREV_F32_e32_vi:
4622 case AMDGPU::V_SUBREV_F32_e64_gfx10:
4623 case AMDGPU::V_SUBREV_F32_e64_gfx6_gfx7:
4624 case AMDGPU::V_SUBREV_F32_e64_vi:
4626 case AMDGPU::V_SUBREV_CO_U32_e32:
4627 case AMDGPU::V_SUBREV_CO_U32_e64:
4628 case AMDGPU::V_SUBREV_I32_e32_gfx6_gfx7:
4629 case AMDGPU::V_SUBREV_I32_e64_gfx6_gfx7:
4631 case AMDGPU::V_SUBBREV_U32_e32:
4632 case AMDGPU::V_SUBBREV_U32_e64:
4633 case AMDGPU::V_SUBBREV_U32_e32_gfx6_gfx7:
4634 case AMDGPU::V_SUBBREV_U32_e32_vi:
4635 case AMDGPU::V_SUBBREV_U32_e64_gfx6_gfx7:
4636 case AMDGPU::V_SUBBREV_U32_e64_vi:
4638 case AMDGPU::V_SUBREV_U32_e32:
4639 case AMDGPU::V_SUBREV_U32_e64:
4640 case AMDGPU::V_SUBREV_U32_e32_gfx9:
4641 case AMDGPU::V_SUBREV_U32_e32_vi:
4642 case AMDGPU::V_SUBREV_U32_e64_gfx9:
4643 case AMDGPU::V_SUBREV_U32_e64_vi:
4645 case AMDGPU::V_SUBREV_F16_e32:
4646 case AMDGPU::V_SUBREV_F16_e64:
4647 case AMDGPU::V_SUBREV_F16_e32_gfx10:
4648 case AMDGPU::V_SUBREV_F16_e32_vi:
4649 case AMDGPU::V_SUBREV_F16_e64_gfx10:
4650 case AMDGPU::V_SUBREV_F16_e64_vi:
4652 case AMDGPU::V_SUBREV_U16_e32:
4653 case AMDGPU::V_SUBREV_U16_e64:
4654 case AMDGPU::V_SUBREV_U16_e32_vi:
4655 case AMDGPU::V_SUBREV_U16_e64_vi:
4657 case AMDGPU::V_SUBREV_CO_U32_e32_gfx9:
4658 case AMDGPU::V_SUBREV_CO_U32_e64_gfx10:
4659 case AMDGPU::V_SUBREV_CO_U32_e64_gfx9:
4661 case AMDGPU::V_SUBBREV_CO_U32_e32_gfx9:
4662 case AMDGPU::V_SUBBREV_CO_U32_e64_gfx9:
4664 case AMDGPU::V_SUBREV_NC_U32_e32_gfx10:
4665 case AMDGPU::V_SUBREV_NC_U32_e64_gfx10:
4667 case AMDGPU::V_SUBREV_CO_CI_U32_e32_gfx10:
4668 case AMDGPU::V_SUBREV_CO_CI_U32_e64_gfx10:
4670 case AMDGPU::V_LSHRREV_B32_e32:
4671 case AMDGPU::V_LSHRREV_B32_e64:
4672 case AMDGPU::V_LSHRREV_B32_e32_gfx6_gfx7:
4673 case AMDGPU::V_LSHRREV_B32_e64_gfx6_gfx7:
4674 case AMDGPU::V_LSHRREV_B32_e32_vi:
4675 case AMDGPU::V_LSHRREV_B32_e64_vi:
4676 case AMDGPU::V_LSHRREV_B32_e32_gfx10:
4677 case AMDGPU::V_LSHRREV_B32_e64_gfx10:
4679 case AMDGPU::V_ASHRREV_I32_e32:
4680 case AMDGPU::V_ASHRREV_I32_e64:
4681 case AMDGPU::V_ASHRREV_I32_e32_gfx10:
4682 case AMDGPU::V_ASHRREV_I32_e32_gfx6_gfx7:
4683 case AMDGPU::V_ASHRREV_I32_e32_vi:
4684 case AMDGPU::V_ASHRREV_I32_e64_gfx10:
4685 case AMDGPU::V_ASHRREV_I32_e64_gfx6_gfx7:
4686 case AMDGPU::V_ASHRREV_I32_e64_vi:
4688 case AMDGPU::V_LSHLREV_B32_e32:
4689 case AMDGPU::V_LSHLREV_B32_e64:
4690 case AMDGPU::V_LSHLREV_B32_e32_gfx10:
4691 case AMDGPU::V_LSHLREV_B32_e32_gfx6_gfx7:
4692 case AMDGPU::V_LSHLREV_B32_e32_vi:
4693 case AMDGPU::V_LSHLREV_B32_e64_gfx10:
4694 case AMDGPU::V_LSHLREV_B32_e64_gfx6_gfx7:
4695 case AMDGPU::V_LSHLREV_B32_e64_vi:
4697 case AMDGPU::V_LSHLREV_B16_e32:
4698 case AMDGPU::V_LSHLREV_B16_e64:
4699 case AMDGPU::V_LSHLREV_B16_e32_vi:
4700 case AMDGPU::V_LSHLREV_B16_e64_vi:
4701 case AMDGPU::V_LSHLREV_B16_gfx10:
4703 case AMDGPU::V_LSHRREV_B16_e32:
4704 case AMDGPU::V_LSHRREV_B16_e64:
4705 case AMDGPU::V_LSHRREV_B16_e32_vi:
4706 case AMDGPU::V_LSHRREV_B16_e64_vi:
4707 case AMDGPU::V_LSHRREV_B16_gfx10:
4709 case AMDGPU::V_ASHRREV_I16_e32:
4710 case AMDGPU::V_ASHRREV_I16_e64:
4711 case AMDGPU::V_ASHRREV_I16_e32_vi:
4712 case AMDGPU::V_ASHRREV_I16_e64_vi:
4713 case AMDGPU::V_ASHRREV_I16_gfx10:
4715 case AMDGPU::V_LSHLREV_B64_e64:
4716 case AMDGPU::V_LSHLREV_B64_gfx10:
4717 case AMDGPU::V_LSHLREV_B64_vi:
4719 case AMDGPU::V_LSHRREV_B64_e64:
4720 case AMDGPU::V_LSHRREV_B64_gfx10:
4721 case AMDGPU::V_LSHRREV_B64_vi:
4723 case AMDGPU::V_ASHRREV_I64_e64:
4724 case AMDGPU::V_ASHRREV_I64_gfx10:
4725 case AMDGPU::V_ASHRREV_I64_vi:
4727 case AMDGPU::V_PK_LSHLREV_B16:
4728 case AMDGPU::V_PK_LSHLREV_B16_gfx10:
4729 case AMDGPU::V_PK_LSHLREV_B16_vi:
4731 case AMDGPU::V_PK_LSHRREV_B16:
4732 case AMDGPU::V_PK_LSHRREV_B16_gfx10:
4733 case AMDGPU::V_PK_LSHRREV_B16_vi:
4734 case AMDGPU::V_PK_ASHRREV_I16:
4735 case AMDGPU::V_PK_ASHRREV_I16_gfx10:
4736 case AMDGPU::V_PK_ASHRREV_I16_vi:
4743bool AMDGPUAsmParser::validateLdsDirect(
const MCInst &Inst,
4745 const unsigned Opcode = Inst.
getOpcode();
4754 for (
auto SrcName : {OpName::src0, OpName::src1, OpName::src2}) {
4755 auto SrcIdx = getNamedOperandIdx(Opcode, SrcName);
4759 if (Src.isReg() && Src.getReg() == LDS_DIRECT) {
4763 "lds_direct is not supported on this GPU");
4769 "lds_direct cannot be used with this instruction");
4773 if (SrcName != OpName::src0) {
4775 "lds_direct may be used as src0 only");
4785 for (
unsigned i = 1, e =
Operands.size(); i != e; ++i) {
4786 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
4787 if (
Op.isFlatOffset())
4788 return Op.getStartLoc();
4793bool AMDGPUAsmParser::validateOffset(
const MCInst &Inst,
4796 auto OpNum = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::offset);
4801 return validateFlatOffset(Inst,
Operands);
4804 return validateSMEMOffset(Inst,
Operands);
4809 const unsigned OffsetSize = 24;
4810 if (!
isUIntN(OffsetSize - 1,
Op.getImm())) {
4812 Twine(
"expected a ") + Twine(OffsetSize - 1) +
4813 "-bit unsigned offset for buffer ops");
4817 const unsigned OffsetSize = 16;
4818 if (!
isUIntN(OffsetSize,
Op.getImm())) {
4820 Twine(
"expected a ") + Twine(OffsetSize) +
"-bit unsigned offset");
4827bool AMDGPUAsmParser::validateFlatOffset(
const MCInst &Inst,
4833 auto OpNum = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::offset);
4837 if (!hasFlatOffsets() &&
Op.getImm() != 0) {
4839 "flat offset modifier is not supported on this GPU");
4846 bool AllowNegative =
4848 if (!
isIntN(OffsetSize,
Op.getImm()) || (!AllowNegative &&
Op.getImm() < 0)) {
4850 Twine(
"expected a ") +
4851 (AllowNegative ? Twine(OffsetSize) +
"-bit signed offset"
4852 : Twine(OffsetSize - 1) +
"-bit unsigned offset"));
4861 for (
unsigned i = 2, e =
Operands.size(); i != e; ++i) {
4862 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
4863 if (
Op.isSMEMOffset() ||
Op.isSMEMOffsetMod())
4864 return Op.getStartLoc();
4869bool AMDGPUAsmParser::validateSMEMOffset(
const MCInst &Inst,
4878 auto OpNum = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::offset);
4894 ?
"expected a 23-bit unsigned offset for buffer ops"
4895 :
isGFX12Plus() ?
"expected a 24-bit signed offset"
4896 : (
isVI() || IsBuffer) ?
"expected a 20-bit unsigned offset"
4897 :
"expected a 21-bit signed offset");
4907bool AMDGPUAsmParser::validateBF16InlineConst(
const MCInst &Inst,
4909 if (!getFeatureBits()[AMDGPU::FeatureBF16InlineConstFromUpperFP32])
4924 const int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
4928 const MCOperandInfo &Src0Info =
Desc.operands()[Src0Idx];
4932 const MCOperand &Src0 = Inst.
getOperand(Src0Idx);
4933 if (!Src0.
isImm() ||
4935 hasInv2PiInlineImm()))
4940 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0_modifiers);
4941 if (ModsIdx != -1 &&
4947 "bf16 inline constant is read from the high half of the fp32 inline "
4948 "constant on this GPU; use the e64 encoding with op_sel:[1,0]");
4952bool AMDGPUAsmParser::validateSOPLiteral(
const MCInst &Inst,
4955 const MCInstrDesc &
Desc = MII.
get(Opcode);
4959 const int Src0Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0);
4960 const int Src1Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src1);
4962 const int OpIndices[] = {Src0Idx, Src1Idx};
4964 unsigned NumExprs = 0;
4965 unsigned NumLiterals = 0;
4968 for (
int OpIdx : OpIndices) {
4972 const MCOperand &MO = Inst.
getOperand(OpIdx);
4976 std::optional<int64_t>
Imm;
4979 }
else if (MO.
isExpr()) {
4988 if (!
Imm.has_value()) {
4990 }
else if (!isInlineConstant(Inst, OpIdx)) {
4994 if (NumLiterals == 0 || LiteralValue !=
Value) {
5002 if (NumLiterals + NumExprs <= 1)
5006 "only one unique literal operand is allowed");
5010bool AMDGPUAsmParser::validateOpSel(
const MCInst &Inst) {
5013 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
5021 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
5022 if (OpSelIdx != -1) {
5026 int OpSelHiIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel_hi);
5027 if (OpSelHiIdx != -1) {
5036 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
5046 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
5047 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
5048 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
5049 int OpSelHiIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel_hi);
5051 const MCOperand &Src0 = Inst.
getOperand(Src0Idx);
5052 const MCOperand &Src1 = Inst.
getOperand(Src1Idx);
5058 auto VerifyOneSGPR = [
OpSel, OpSelHi](
unsigned Index) ->
bool {
5060 return ((OpSel & Mask) == 0) && ((OpSelHi &
Mask) == 0);
5070 int Src2Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2);
5071 if (Src2Idx != -1) {
5072 const MCOperand &Src2 = Inst.
getOperand(Src2Idx);
5082bool AMDGPUAsmParser::validateTrue16OpSel(
const MCInst &Inst) {
5083 if (!hasTrue16Insts())
5085 const MCRegisterInfo *MRI = getMRI();
5087 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
5093 if (OpSelOpValue == 0)
5095 unsigned OpCount = 0;
5096 for (AMDGPU::OpName OpName : {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
5097 AMDGPU::OpName::src2, AMDGPU::OpName::vdst}) {
5098 int OpIdx = AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), OpName);
5105 bool OpSelOpIsHi = ((OpSelOpValue & (1 << OpCount)) != 0);
5106 if (OpSelOpIsHi != VGPRSuffixIsHi)
5115bool AMDGPUAsmParser::validateNeg(
const MCInst &Inst, AMDGPU::OpName OpName) {
5116 assert(OpName == AMDGPU::OpName::neg_lo || OpName == AMDGPU::OpName::neg_hi);
5128 int NegIdx = AMDGPU::getNamedOperandIdx(
Opc, OpName);
5139 const AMDGPU::OpName SrcMods[3] = {AMDGPU::OpName::src0_modifiers,
5140 AMDGPU::OpName::src1_modifiers,
5141 AMDGPU::OpName::src2_modifiers};
5143 for (
unsigned i = 0; i < 3; ++i) {
5153bool AMDGPUAsmParser::validateDPP(
const MCInst &Inst,
5156 int DppCtrlIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::dpp_ctrl);
5157 if (DppCtrlIdx >= 0) {
5161 getSTI().
hasFeature(AMDGPU::FeatureDPALU_DPP) &&
5165 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyDppCtrl,
Operands);
5166 Error(S,
isGFX12() ?
"DP ALU dpp only supports row_share"
5167 :
"DP ALU dpp only supports row_newbcast");
5172 int Dpp8Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::dpp8);
5173 bool IsDPP = DppCtrlIdx >= 0 || Dpp8Idx >= 0;
5176 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
5178 const MCOperand &Src1 = Inst.
getOperand(Src1Idx);
5182 "invalid operand for instruction");
5187 "src1 immediate operand invalid for instruction");
5197bool AMDGPUAsmParser::validateVccOperand(MCRegister
Reg)
const {
5198 return (
Reg == AMDGPU::VCC && isWave64()) ||
5199 (
Reg == AMDGPU::VCC_LO && isWave32());
5203bool AMDGPUAsmParser::validateVOPLiteral(
const MCInst &Inst,
5206 const MCInstrDesc &
Desc = MII.
get(Opcode);
5207 bool HasMandatoryLiteral = getNamedOperandIdx(Opcode, OpName::imm) != -1;
5214 std::optional<unsigned> LiteralOpIdx;
5217 for (
int OpIdx : OpIndices) {
5221 const MCOperand &MO = Inst.
getOperand(OpIdx);
5227 std::optional<int64_t>
Imm;
5233 bool IsAnotherLiteral =
false;
5234 bool IsForcedLit = findMCOperand(
Operands, OpIdx).isForcedLit();
5235 bool IsForcedLit64 = findMCOperand(
Operands, OpIdx).isForcedLit64();
5236 if (!
Imm.has_value()) {
5238 IsAnotherLiteral =
true;
5239 }
else if (IsForcedLit || IsForcedLit64 || !isInlineConstant(Inst, OpIdx)) {
5244 HasMandatoryLiteral);
5256 (IsForcedLit64 && !HasMandatoryLiteral)) &&
5257 (!has64BitLiterals() ||
Desc.getSize() != 4)) {
5259 "invalid operand for instruction");
5264 if (!IsForcedFP64 && (IsForcedLit64 || !IsValid32Op) &&
5265 OpIdx != getNamedOperandIdx(Opcode, OpName::src0)) {
5267 "invalid operand for instruction");
5272 if (IsValid32Op && !IsForcedFP64 && !IsForcedLit64) {
5273 Value =
static_cast<uint32_t
>(
5281 if (IsAnotherLiteral && !HasMandatoryLiteral &&
5282 !getFeatureBits()[FeatureVOP3Literal]) {
5284 "literal operands are not supported");
5288 if (LiteralOpIdx && IsAnotherLiteral) {
5290 getOperandLoc(
Operands, *LiteralOpIdx)),
5291 "only one unique literal operand is allowed");
5295 if (IsAnotherLiteral)
5296 LiteralOpIdx = OpIdx;
5305 int OpIdx = AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), Name);
5319bool AMDGPUAsmParser::validateAGPRLdSt(
const MCInst &Inst)
const {
5325 ? AMDGPU::OpName::data0
5326 : AMDGPU::OpName::vdata;
5328 const MCRegisterInfo *MRI = getMRI();
5329 int DstAreg =
IsAGPROperand(Inst, AMDGPU::OpName::vdst, MRI);
5333 int Data2Areg =
IsAGPROperand(Inst, AMDGPU::OpName::data1, MRI);
5334 if (Data2Areg >= 0 && Data2Areg != DataAreg)
5338 auto FB = getFeatureBits();
5339 if (FB[AMDGPU::FeatureGFX90AInsts]) {
5340 if (DataAreg < 0 || DstAreg < 0)
5342 return DstAreg == DataAreg;
5345 return DstAreg < 1 && DataAreg < 1;
5348bool AMDGPUAsmParser::validateVGPRAlign(
const MCInst &Inst)
const {
5349 auto FB = getFeatureBits();
5350 if (!FB[AMDGPU::FeatureRequiresAlignedVGPRs])
5354 const MCRegisterInfo *MRI = getMRI();
5357 if (FB[AMDGPU::FeatureGFX90AInsts] &&
Opc == AMDGPU::DS_READ_B96_TR_B6_vi)
5360 if (FB[AMDGPU::FeatureGFX1250Insts]) {
5364 case AMDGPU::DS_LOAD_TR6_B96:
5365 case AMDGPU::DS_LOAD_TR6_B96_gfx12:
5369 case AMDGPU::GLOBAL_LOAD_TR6_B96:
5370 case AMDGPU::GLOBAL_LOAD_TR6_B96_gfx1250: {
5374 int VAddrIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr);
5375 if (VAddrIdx != -1) {
5378 if ((
Sub - AMDGPU::VGPR0) & 1)
5383 case AMDGPU::GLOBAL_LOAD_TR6_B96_SADDR:
5384 case AMDGPU::GLOBAL_LOAD_TR6_B96_SADDR_gfx1250:
5389 const MCRegisterClass &VGPR32 = MRI->
getRegClass(AMDGPU::VGPR_32RegClassID);
5390 const MCRegisterClass &AGPR32 = MRI->
getRegClass(AMDGPU::AGPR_32RegClassID);
5410 for (
unsigned i = 1, e =
Operands.size(); i != e; ++i) {
5411 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
5413 return Op.getStartLoc();
5418bool AMDGPUAsmParser::validateBLGP(
const MCInst &Inst,
5421 int BlgpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::blgp);
5424 SMLoc BLGPLoc = getBLGPLoc(
Operands);
5427 bool IsNeg = StringRef(BLGPLoc.
getPointer()).starts_with(
"neg:");
5428 auto FB = getFeatureBits();
5429 bool UsesNeg =
false;
5430 if (FB[AMDGPU::FeatureGFX940Insts]) {
5432 case AMDGPU::V_MFMA_F64_16X16X4F64_gfx940_acd:
5433 case AMDGPU::V_MFMA_F64_16X16X4F64_gfx940_vcd:
5434 case AMDGPU::V_MFMA_F64_4X4X4F64_gfx940_acd:
5435 case AMDGPU::V_MFMA_F64_4X4X4F64_gfx940_vcd:
5440 if (IsNeg == UsesNeg)
5443 Error(BLGPLoc, UsesNeg ?
"invalid modifier: blgp is not supported"
5444 :
"invalid modifier: neg is not supported");
5449bool AMDGPUAsmParser::validateWaitCnt(
const MCInst &Inst,
5455 if (
Opc != AMDGPU::S_WAITCNT_EXPCNT_gfx11 &&
5456 Opc != AMDGPU::S_WAITCNT_LGKMCNT_gfx11 &&
5457 Opc != AMDGPU::S_WAITCNT_VMCNT_gfx11 &&
5458 Opc != AMDGPU::S_WAITCNT_VSCNT_gfx11)
5461 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::sdst);
5464 if (
Reg == AMDGPU::SGPR_NULL)
5467 Error(getOperandLoc(
Operands, Src0Idx),
"src0 must be null");
5471bool AMDGPUAsmParser::validateDS(
const MCInst &Inst,
5476 return validateGWS(Inst,
Operands);
5481 AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), AMDGPU::OpName::gds);
5486 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyGDS,
Operands);
5487 Error(S,
"gds modifier is not supported on this GPU");
5495bool AMDGPUAsmParser::validateGWS(
const MCInst &Inst,
5497 if (!getFeatureBits()[AMDGPU::FeatureGFX90AInsts])
5501 if (
Opc != AMDGPU::DS_GWS_INIT_vi &&
Opc != AMDGPU::DS_GWS_BARRIER_vi &&
5502 Opc != AMDGPU::DS_GWS_SEMA_BR_vi)
5505 const MCRegisterInfo *MRI = getMRI();
5506 const MCRegisterClass &VGPR32 = MRI->
getRegClass(AMDGPU::VGPR_32RegClassID);
5508 AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), AMDGPU::OpName::data0);
5511 auto RegIdx =
Reg - (VGPR32.
contains(
Reg) ? AMDGPU::VGPR0 : AMDGPU::AGPR0);
5513 Error(getOperandLoc(
Operands, Data0Pos),
"vgpr must be even aligned");
5520bool AMDGPUAsmParser::validateCoherencyBits(
const MCInst &Inst,
5524 AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), AMDGPU::OpName::cpol);
5532 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol,
Operands);
5535 Error(S,
"scale_offset is not supported on this GPU");
5538 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol,
Operands);
5541 Error(S,
"nv is not supported on this GPU");
5546 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol,
Operands);
5549 Error(S,
"scale_offset is not supported for this instruction");
5553 return validateTHAndScopeBits(Inst,
Operands, CPol);
5557 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol,
Operands);
5558 Error(S,
"cache policy is not supported for SMRD instructions");
5562 Error(IDLoc,
"invalid cache policy for SMEM instruction");
5569 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol,
Operands);
5573 "scc modifier is not supported for this instruction on this GPU");
5584 :
"instruction must use glc");
5589 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol,
Operands);
5592 &CStr.data()[CStr.find(
isGFX940() ?
"sc0" :
"glc")]);
5594 :
"instruction must not use glc");
5602bool AMDGPUAsmParser::validateTHAndScopeBits(
const MCInst &Inst,
5604 const unsigned CPol) {
5609 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol,
Operands);
5616 return PrintError(
"th:TH_ATOMIC_RETURN requires a destination operand");
5621 return PrintError(
"instruction must use th:TH_ATOMIC_RETURN");
5629 return PrintError(
"invalid th value for SMEM instruction");
5636 return PrintError(
"scope and th combination is not valid");
5642 return PrintError(
"invalid th value for atomic instructions");
5645 return PrintError(
"invalid th value for store instructions");
5648 return PrintError(
"invalid th value for load instructions");
5654bool AMDGPUAsmParser::validateTFE(
const MCInst &Inst,
5658 SMLoc Loc = getImmLoc(AMDGPUOperand::ImmTyTFE,
Operands);
5660 Error(Loc,
"TFE modifier has no meaning for store instructions");
5668bool AMDGPUAsmParser::validateWMMA(
const MCInst &Inst,
5674 int AFmtIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_a_fmt);
5678 int BFmtIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_b_fmt);
5681 auto validateFmt = [&](
unsigned Fmt, AMDGPU::OpName SrcOp) ->
bool {
5682 int SrcIdx = AMDGPU::getNamedOperandIdx(
Opc, SrcOp);
5691 "wrong register tuple size for " +
5696 if (!validateFmt(AFmt, AMDGPU::OpName::src0) ||
5697 !validateFmt(BFmt, AMDGPU::OpName::src1))
5701 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_a_scale_fmt);
5702 if (AScaleIdx == -1)
5706 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_b_scale_fmt);
5710 "invalid matrix and scale format combination");
5717bool AMDGPUAsmParser::validateMonitorSleep(
const MCInst &Inst,
5720 if (
Opc != AMDGPU::S_MONITOR_SLEEP_gfx12 ||
5721 !getSTI().
hasFeature(AMDGPU::FeatureNoSleepForever))
5724 int ImmIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::simm16);
5727 "sleep forever is unsuported on the target");
5734bool AMDGPUAsmParser::validateClusterBarrierIsFirst(
5737 if (
Opc != AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM_gfx12 &&
5738 Opc != AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM_gfx13)
5741 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
5748 "s_barrier_signal_isfirst does not support user_cluster_barrier_id (-3)");
5752bool AMDGPUAsmParser::validateScaleSel(
const MCInst &Inst,
5755 int ScaleSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::scale_sel);
5756 if (ScaleSelIdx == -1)
5760 case AMDGPU::V_CVT_SCALE_PK16_F16_FP6_e64_gfx1250:
5761 case AMDGPU::V_CVT_SCALE_PK16_BF16_FP6_e64_gfx1250:
5762 case AMDGPU::V_CVT_SCALE_PK16_F16_BF6_e64_gfx1250:
5763 case AMDGPU::V_CVT_SCALE_PK16_BF16_BF6_e64_gfx1250:
5764 case AMDGPU::V_CVT_SCALE_PK16_F32_FP6_e64_gfx1250:
5765 case AMDGPU::V_CVT_SCALE_PK16_F32_BF6_e64_gfx1250:
5766 case AMDGPU::V_CVT_SCALE_PK8_F16_FP4_e64_gfx1250:
5767 case AMDGPU::V_CVT_SCALE_PK8_BF16_FP4_e64_gfx1250:
5768 case AMDGPU::V_CVT_SCALE_PK8_F32_FP4_e64_gfx1250:
5771 case AMDGPU::V_CVT_SCALE_PK8_F16_FP8_e64_gfx1250:
5772 case AMDGPU::V_CVT_SCALE_PK8_BF16_FP8_e64_gfx1250:
5773 case AMDGPU::V_CVT_SCALE_PK8_F16_BF8_e64_gfx1250:
5774 case AMDGPU::V_CVT_SCALE_PK8_BF16_BF8_e64_gfx1250:
5775 case AMDGPU::V_CVT_SCALE_PK8_F32_FP8_e64_gfx1250:
5776 case AMDGPU::V_CVT_SCALE_PK8_F32_BF8_e64_gfx1250:
5783 if (getSTI().
hasFeature(AMDGPU::FeatureBlock16ConversionScaleInsts))
5787 if (ScaleSel < MaxSel)
5791 "scale_sel maximum supported value is " + Twine(MaxSel - 1));
5795bool AMDGPUAsmParser::validateInstruction(
const MCInst &Inst, SMLoc IDLoc,
5797 if (!validateLdsDirect(Inst,
Operands))
5799 if (!validateTrue16OpSel(Inst)) {
5801 "op_sel operand conflicts with 16-bit operand suffix");
5804 if (!validateSOPLiteral(Inst,
Operands))
5806 if (!validateVOPLiteral(Inst,
Operands)) {
5809 if (!validateConstantBusLimitations(Inst,
Operands)) {
5812 if (!validateVOPD(Inst,
Operands)) {
5815 if (!validateIntClampSupported(Inst)) {
5817 "integer clamping is not supported on this GPU");
5820 if (!validateOpSel(Inst)) {
5822 "invalid op_sel operand");
5825 if (!validateNeg(Inst, AMDGPU::OpName::neg_lo)) {
5827 "invalid neg_lo operand");
5830 if (!validateNeg(Inst, AMDGPU::OpName::neg_hi)) {
5832 "invalid neg_hi operand");
5835 if (!validateDPP(Inst,
Operands)) {
5839 if (!validateMIMGD16(Inst)) {
5841 "d16 modifier is not supported on this GPU");
5844 if (!validateMIMGDim(Inst,
Operands)) {
5845 Error(IDLoc,
"missing dim operand");
5848 if (!validateTensorR128(Inst)) {
5850 "instruction must set modifier r128=0");
5853 if (!validateMIMGMSAA(Inst)) {
5855 "invalid dim; must be MSAA type");
5858 if (!validateMIMGDataSize(Inst, IDLoc)) {
5861 if (!validateMIMGAddrSize(Inst, IDLoc))
5863 if (!validateMIMGAtomicDMask(Inst)) {
5865 "invalid atomic image dmask");
5868 if (!validateMIMGGatherDMask(Inst)) {
5870 "invalid image_gather dmask: only one bit must be set");
5873 if (!validateMovrels(Inst,
Operands)) {
5876 if (!validateOffset(Inst,
Operands)) {
5879 if (!validateBF16InlineConst(Inst,
Operands)) {
5882 if (!validateMAIAccWrite(Inst,
Operands)) {
5885 if (!validateMAISrc2(Inst,
Operands)) {
5888 if (!validateMFMA(Inst,
Operands)) {
5891 if (!validateCoherencyBits(Inst,
Operands, IDLoc)) {
5895 if (!validateAGPRLdSt(Inst)) {
5898 getFeatureBits()[AMDGPU::FeatureGFX90AInsts]
5899 ?
"invalid register class: data and dst should be all VGPR or AGPR"
5900 :
"invalid register class: agpr loads and stores not supported on "
5904 if (!validateVGPRAlign(Inst)) {
5905 Error(IDLoc,
"invalid register class: vgpr tuples must be 64 bit aligned");
5912 if (!validateBLGP(Inst,
Operands)) {
5916 if (!validateDivScale(Inst)) {
5917 Error(IDLoc,
"ABS not allowed in VOP3B instructions");
5920 if (!validateWaitCnt(Inst,
Operands)) {
5923 if (!validateTFE(Inst,
Operands)) {
5926 if (!validateWMMA(Inst,
Operands)) {
5929 if (!validateMonitorSleep(Inst,
Operands)) {
5932 if (!validateClusterBarrierIsFirst(Inst,
Operands)) {
5935 if (!validateScaleSel(Inst,
Operands)) {
5944 unsigned VariantID = 0);
5948 unsigned VariantID);
5950bool AMDGPUAsmParser::isSupportedMnemo(
StringRef Mnemo,
5955bool AMDGPUAsmParser::isSupportedMnemo(StringRef Mnemo,
5956 const FeatureBitset &FBS,
5957 ArrayRef<unsigned> Variants) {
5958 for (
auto Variant : Variants) {
5966bool AMDGPUAsmParser::checkUnsupportedInstruction(StringRef Mnemo,
5968 FeatureBitset FBS = ComputeAvailableFeatures(getFeatureBits());
5971 if (isSupportedMnemo(Mnemo, FBS, getMatchedVariants()))
5976 getParser().clearPendingErrors();
5980 StringRef VariantName = getMatchedVariantName();
5981 if (!VariantName.
empty() && isSupportedMnemo(Mnemo, FBS)) {
5982 return Error(IDLoc, Twine(VariantName,
5983 " variant of this instruction is not supported"));
5987 if (
isGFX10Plus() && getFeatureBits()[AMDGPU::FeatureWavefrontSize64] &&
5988 !getFeatureBits()[AMDGPU::FeatureWavefrontSize32]) {
5990 FeatureBitset FeaturesWS32 = getFeatureBits();
5991 FeaturesWS32.
flip(AMDGPU::FeatureWavefrontSize64)
5992 .
flip(AMDGPU::FeatureWavefrontSize32);
5993 FeatureBitset AvailableFeaturesWS32 =
5994 ComputeAvailableFeatures(FeaturesWS32);
5996 if (isSupportedMnemo(Mnemo, AvailableFeaturesWS32, getMatchedVariants()))
5997 return Error(IDLoc,
"instruction requires wavesize=32");
6001 if (isSupportedMnemo(Mnemo, FeatureBitset().set())) {
6002 return Error(IDLoc,
"instruction not supported on this GPU (" +
6003 getSTI().
getCPU() +
")" +
": " + Mnemo);
6008 return Error(IDLoc,
"invalid instruction" + Suggestion);
6014 const auto &
Op = ((AMDGPUOperand &)*
Operands[InvalidOprIdx]);
6015 if (
Op.isToken() && InvalidOprIdx > 1) {
6016 const auto &PrevOp = ((AMDGPUOperand &)*
Operands[InvalidOprIdx - 1]);
6017 return PrevOp.isToken() && PrevOp.getToken() ==
"::";
6022bool AMDGPUAsmParser::matchAndEmitInstruction(SMLoc IDLoc,
unsigned &Opcode,
6026 bool MatchingInlineAsm) {
6029 unsigned Result = Match_Success;
6034 auto atLeastAsSpecific = [](
unsigned New,
unsigned Cur) {
6035 auto rank = [](
unsigned M) {
6036 return M == Match_MnemonicFail ? 1
6037 :
M == Match_InvalidOperand ? 2
6038 :
M == Match_MissingFeature ? 3
6041 return rank(New) >= rank(Cur);
6044 for (
auto Variant : getMatchedVariants()) {
6047 MatchInstructionImpl(
Operands, Inst, EI, MatchingInlineAsm, Variant);
6048 if (R == Match_Success || atLeastAsSpecific(R, Result)) {
6052 if (R == Match_Success)
6056 if (Result == Match_Success) {
6057 if (!validateInstruction(Inst, IDLoc,
Operands)) {
6060 emitTargetDirective();
6061 Out.emitInstruction(Inst, getSTI());
6068 if (checkUnsupportedInstruction(Mnemo, IDLoc)) {
6075 case Match_MissingFeature:
6079 return Error(IDLoc,
"operands are not valid for this GPU or mode");
6081 case Match_InvalidOperand: {
6082 SMLoc ErrorLoc = IDLoc;
6083 if (ErrorInfo != ~0ULL) {
6084 if (ErrorInfo >=
Operands.size()) {
6085 return Error(IDLoc,
"too few operands for instruction");
6087 AMDGPUOperand &ErrorOp = (AMDGPUOperand &)*
Operands[ErrorInfo];
6088 ErrorLoc = ErrorOp.getStartLoc();
6089 if (ErrorLoc == SMLoc())
6093 return Error(ErrorLoc,
"invalid VOPDY instruction");
6095 return Error(ErrorLoc,
"invalid operand for instruction");
6098 case Match_MnemonicFail:
6104bool AMDGPUAsmParser::ParseAsAbsoluteExpression(uint32_t &Ret) {
6109 if (getParser().parseAbsoluteExpression(Tmp)) {
6112 Ret =
static_cast<uint32_t
>(Tmp);
6116bool AMDGPUAsmParser::ParseDirectiveAMDGCNTarget() {
6117 if (!getSTI().getTargetTriple().isAMDGCN())
6118 return TokError(
"directive only supported for amdgcn architecture");
6120 std::string TargetIDDirective;
6121 SMLoc TargetStart = getTok().getLoc();
6122 if (getParser().parseEscapedString(TargetIDDirective))
6125 std::optional<AMDGPU::TargetID> MaybeParsed =
6128 return getParser().Error(TargetStart,
6129 "malformed target id '" + TargetIDDirective +
"'");
6132 const Triple &
TT = getSTI().getTargetTriple();
6138 return getParser().Error(
6139 TargetStart,
"target id '" + TargetIDDirective +
6140 "' specifies a processor that is not valid for "
6142 TT.getArchName() +
"'");
6145 const std::optional<AMDGPU::TargetID> &CurrentTargetID =
6146 getTargetStreamer().getTargetID();
6149 const Triple &STITriple = getSTI().getTargetTriple();
6150 if (!DirectiveTriple.isCompatibleWith(STITriple)) {
6151 return getParser().Error(
6152 TargetStart,
".amdgcn_target " + Twine(ParsedTargetID.
toString()) +
6153 " is incompatible with " +
6154 Twine(CurrentTargetID->toString()));
6158 StringRef DirectiveProcessor =
6161 if (DirectiveISA != ISA) {
6162 return getParser().Error(TargetStart,
6163 ".amdgcn_target directive processor " +
6164 Twine(DirectiveProcessor) +
6165 " does not match the specified processor " +
6166 Twine(getSTI().
getCPU()));
6172 CurrentTargetID->getXnackSetting())) {
6174 ".amdgcn_target directive has conflicting xnack settings");
6178 CurrentTargetID->getSramEccSetting())) {
6180 ".amdgcn_target directive has conflicting sramecc settings");
6186 getTargetStreamer().getTargetID()->setXnackSetting(
6188 getTargetStreamer().getTargetID()->setSramEccSetting(
6194bool AMDGPUAsmParser::OutOfRangeError(SMRange
Range) {
6198bool AMDGPUAsmParser::calculateGPRBlocks(
6199 const FeatureBitset &Features,
const MCExpr *VCCUsed,
6200 const MCExpr *FlatScrUsed,
bool XNACKUsed,
6201 std::optional<bool> EnableWavefrontSize32,
const MCExpr *NextFreeVGPR,
6202 SMRange VGPRRange,
const MCExpr *NextFreeSGPR, SMRange SGPRRange,
6203 const MCExpr *&VGPRBlocks,
const MCExpr *&SGPRBlocks) {
6208 const MCExpr *
NumSGPRs = NextFreeSGPR;
6209 int64_t EvaluatedSGPRs;
6211 if (
ISA.Major >= 10)
6216 if (
NumSGPRs->evaluateAsAbsolute(EvaluatedSGPRs) &&
ISA.Major >= 8 &&
6217 !Features.
test(FeatureSGPRInitBug) &&
6218 static_cast<uint64_t>(EvaluatedSGPRs) > MaxAddressableNumSGPRs)
6219 return OutOfRangeError(SGPRRange);
6221 const MCExpr *ExtraSGPRs =
6225 if (
NumSGPRs->evaluateAsAbsolute(EvaluatedSGPRs) &&
6226 (
ISA.Major <= 7 || Features.
test(FeatureSGPRInitBug)) &&
6227 static_cast<uint64_t>(EvaluatedSGPRs) > MaxAddressableNumSGPRs)
6228 return OutOfRangeError(SGPRRange);
6230 if (Features.
test(FeatureSGPRInitBug))
6237 auto GetNumGPRBlocks = [&Ctx](
const MCExpr *NumGPR,
6238 unsigned Granule) ->
const MCExpr * {
6242 const MCExpr *AlignToGPR =
6244 const MCExpr *DivGPR =
6250 VGPRBlocks = GetNumGPRBlocks(
6259bool AMDGPUAsmParser::ParseDirectiveAMDHSAKernel() {
6260 if (!getSTI().getTargetTriple().isAMDGCN())
6261 return TokError(
"directive only supported for amdgcn architecture");
6264 return TokError(
"directive only supported for amdhsa OS");
6266 StringRef KernelName;
6267 if (getParser().parseIdentifier(KernelName))
6274 AMDGPU::MCKernelDescriptor KD =
6284 const MCExpr *NextFreeVGPR = ZeroExpr;
6286 const MCExpr *NamedBarCnt = ZeroExpr;
6291 const MCExpr *NextFreeSGPR = ZeroExpr;
6294 unsigned ImpliedUserSGPRCount = 0;
6298 std::optional<unsigned> ExplicitUserSGPRCount;
6299 const MCExpr *ReserveVCC = OneExpr;
6300 const MCExpr *ReserveFlatScr = OneExpr;
6301 std::optional<bool> EnableWavefrontSize32;
6308 SMRange IDRange = getTok().getLocRange();
6309 if (!parseId(ID,
"expected .amdhsa_ directive or .end_amdhsa_kernel"))
6312 if (ID ==
".end_amdhsa_kernel")
6315 if (!Seen.
insert(ID).second)
6316 return TokError(
".amdhsa_ directives cannot be repeated");
6318 SMLoc ValStart = getLoc();
6319 const MCExpr *ExprVal;
6320 if (getParser().parseExpression(ExprVal))
6322 SMLoc ValEnd = getLoc();
6323 SMRange ValRange = SMRange(ValStart, ValEnd);
6327 bool EvaluatableExpr;
6328 if ((EvaluatableExpr = ExprVal->evaluateAsAbsolute(IVal))) {
6330 return OutOfRangeError(ValRange);
6334#define PARSE_BITS_ENTRY(FIELD, ENTRY, VALUE, RANGE) \
6335 if (!isUInt<ENTRY##_WIDTH>(Val)) \
6336 return OutOfRangeError(RANGE); \
6337 AMDGPU::MCKernelDescriptor::bits_set(FIELD, VALUE, ENTRY##_SHIFT, ENTRY, \
6342#define EXPR_RESOLVE_OR_ERROR(RESOLVED) \
6344 return Error(IDRange.Start, "directive should have resolvable expression", \
6347 if (ID ==
".amdhsa_group_segment_fixed_size") {
6350 return OutOfRangeError(ValRange);
6352 }
else if (ID ==
".amdhsa_private_segment_fixed_size") {
6355 return OutOfRangeError(ValRange);
6357 }
else if (ID ==
".amdhsa_kernarg_size") {
6359 return OutOfRangeError(ValRange);
6361 }
else if (ID ==
".amdhsa_user_sgpr_count") {
6363 ExplicitUserSGPRCount = Val;
6364 }
else if (ID ==
".amdhsa_user_sgpr_private_segment_buffer") {
6368 "directive is not supported with architected flat scratch",
6371 KERNEL_CODE_PROPERTY_ENABLE_SGPR_PRIVATE_SEGMENT_BUFFER,
6374 ImpliedUserSGPRCount += 4;
6375 }
else if (ID ==
".amdhsa_user_sgpr_kernarg_preload_length") {
6378 return Error(IDRange.
Start,
"directive requires gfx90a+", IDRange);
6381 return OutOfRangeError(ValRange);
6385 ImpliedUserSGPRCount += Val;
6386 PreloadLength = Val;
6388 }
else if (ID ==
".amdhsa_user_sgpr_kernarg_preload_offset") {
6391 return Error(IDRange.
Start,
"directive requires gfx90a+", IDRange);
6394 return OutOfRangeError(ValRange);
6398 PreloadOffset = Val;
6399 }
else if (ID ==
".amdhsa_user_sgpr_dispatch_ptr") {
6402 KERNEL_CODE_PROPERTY_ENABLE_SGPR_DISPATCH_PTR, ExprVal,
6405 ImpliedUserSGPRCount += 2;
6406 }
else if (ID ==
".amdhsa_user_sgpr_queue_ptr") {
6409 KERNEL_CODE_PROPERTY_ENABLE_SGPR_QUEUE_PTR, ExprVal,
6412 ImpliedUserSGPRCount += 2;
6413 }
else if (ID ==
".amdhsa_user_sgpr_kernarg_segment_ptr") {
6416 KERNEL_CODE_PROPERTY_ENABLE_SGPR_KERNARG_SEGMENT_PTR,
6419 ImpliedUserSGPRCount += 2;
6420 }
else if (ID ==
".amdhsa_user_sgpr_dispatch_id") {
6423 KERNEL_CODE_PROPERTY_ENABLE_SGPR_DISPATCH_ID, ExprVal,
6426 ImpliedUserSGPRCount += 2;
6427 }
else if (ID ==
".amdhsa_user_sgpr_flat_scratch_init") {
6430 "directive is not supported with architected flat scratch",
6434 KERNEL_CODE_PROPERTY_ENABLE_SGPR_FLAT_SCRATCH_INIT,
6437 ImpliedUserSGPRCount += 2;
6438 }
else if (ID ==
".amdhsa_user_sgpr_private_segment_size") {
6441 KERNEL_CODE_PROPERTY_ENABLE_SGPR_PRIVATE_SEGMENT_SIZE,
6444 ImpliedUserSGPRCount += 1;
6445 }
else if (ID ==
".amdhsa_wavefront_size32") {
6448 return Error(IDRange.
Start,
"directive requires gfx10+", IDRange);
6449 EnableWavefrontSize32 = Val;
6451 KERNEL_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32, ExprVal,
6453 }
else if (ID ==
".amdhsa_uses_dynamic_stack") {
6455 KERNEL_CODE_PROPERTY_USES_DYNAMIC_STACK, ExprVal,
6457 }
else if (ID ==
".amdhsa_system_sgpr_private_segment_wavefront_offset") {
6460 "directive is not supported with architected flat scratch",
6463 COMPUTE_PGM_RSRC2_ENABLE_PRIVATE_SEGMENT, ExprVal,
6465 }
else if (ID ==
".amdhsa_enable_private_segment") {
6469 "directive is not supported without architected flat scratch",
6472 COMPUTE_PGM_RSRC2_ENABLE_PRIVATE_SEGMENT, ExprVal,
6474 }
else if (ID ==
".amdhsa_system_sgpr_workgroup_id_x") {
6476 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_ID_X, ExprVal,
6478 }
else if (ID ==
".amdhsa_system_sgpr_workgroup_id_y") {
6480 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_ID_Y, ExprVal,
6482 }
else if (ID ==
".amdhsa_system_sgpr_workgroup_id_z") {
6484 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_ID_Z, ExprVal,
6486 }
else if (ID ==
".amdhsa_system_sgpr_workgroup_info") {
6488 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_INFO, ExprVal,
6490 }
else if (ID ==
".amdhsa_system_vgpr_workitem_id") {
6492 COMPUTE_PGM_RSRC2_ENABLE_VGPR_WORKITEM_ID, ExprVal,
6494 }
else if (ID ==
".amdhsa_next_free_vgpr") {
6495 VGPRRange = ValRange;
6496 NextFreeVGPR = ExprVal;
6497 }
else if (ID ==
".amdhsa_next_free_sgpr") {
6498 SGPRRange = ValRange;
6499 NextFreeSGPR = ExprVal;
6500 }
else if (ID ==
".amdhsa_accum_offset") {
6502 return Error(IDRange.
Start,
"directive requires gfx90a+", IDRange);
6503 AccumOffset = ExprVal;
6504 }
else if (ID ==
".amdhsa_named_barrier_count") {
6506 return Error(IDRange.
Start,
"directive requires gfx1250+", IDRange);
6507 NamedBarCnt = ExprVal;
6508 }
else if (ID ==
".amdhsa_reserve_vcc") {
6510 return OutOfRangeError(ValRange);
6511 ReserveVCC = ExprVal;
6512 }
else if (ID ==
".amdhsa_reserve_flat_scratch") {
6514 return Error(IDRange.
Start,
"directive requires gfx7+", IDRange);
6517 "directive is not supported with architected flat scratch",
6520 return OutOfRangeError(ValRange);
6521 ReserveFlatScr = ExprVal;
6522 }
else if (ID ==
".amdhsa_reserve_xnack_mask") {
6524 return Error(IDRange.
Start,
"directive requires gfx8+", IDRange);
6526 return OutOfRangeError(ValRange);
6527 bool XnackOn = getTargetStreamer().getTargetID()->isXnackOnOrAny();
6528 if (Val != XnackOn) {
6529 return getParser().Error(
6531 ".amdhsa_reserve_xnack_mask does not match target id", IDRange);
6533 }
else if (ID ==
".amdhsa_float_round_mode_32") {
6535 COMPUTE_PGM_RSRC1_FLOAT_ROUND_MODE_32, ExprVal,
6537 }
else if (ID ==
".amdhsa_float_round_mode_16_64") {
6539 COMPUTE_PGM_RSRC1_FLOAT_ROUND_MODE_16_64, ExprVal,
6541 }
else if (ID ==
".amdhsa_float_denorm_mode_32") {
6543 COMPUTE_PGM_RSRC1_FLOAT_DENORM_MODE_32, ExprVal,
6545 }
else if (ID ==
".amdhsa_float_denorm_mode_16_64") {
6547 COMPUTE_PGM_RSRC1_FLOAT_DENORM_MODE_16_64, ExprVal,
6549 }
else if (ID ==
".amdhsa_dx10_clamp") {
6550 if (!getSTI().
hasFeature(AMDGPU::FeatureDX10ClampAndIEEEMode))
6551 return Error(IDRange.
Start,
"directive unsupported on gfx1170+",
6554 COMPUTE_PGM_RSRC1_GFX6_GFX11_ENABLE_DX10_CLAMP, ExprVal,
6556 }
else if (ID ==
".amdhsa_ieee_mode") {
6557 if (!getSTI().
hasFeature(AMDGPU::FeatureDX10ClampAndIEEEMode))
6558 return Error(IDRange.
Start,
"directive unsupported on gfx1170+",
6561 COMPUTE_PGM_RSRC1_GFX6_GFX11_ENABLE_IEEE_MODE, ExprVal,
6563 }
else if (ID ==
".amdhsa_fp16_overflow") {
6565 return Error(IDRange.
Start,
"directive requires gfx9+", IDRange);
6567 COMPUTE_PGM_RSRC1_GFX9_PLUS_FP16_OVFL, ExprVal,
6569 }
else if (ID ==
".amdhsa_tg_split") {
6571 return Error(IDRange.
Start,
"directive requires gfx90a+", IDRange);
6574 }
else if (ID ==
".amdhsa_workgroup_processor_mode") {
6577 "directive unsupported on " + getSTI().
getCPU(), IDRange);
6579 COMPUTE_PGM_RSRC1_GFX10_PLUS_WGP_MODE, ExprVal,
6581 }
else if (ID ==
".amdhsa_memory_ordered") {
6583 return Error(IDRange.
Start,
"directive requires gfx10+", IDRange);
6585 COMPUTE_PGM_RSRC1_GFX10_PLUS_MEM_ORDERED, ExprVal,
6587 }
else if (ID ==
".amdhsa_forward_progress") {
6589 return Error(IDRange.
Start,
"directive requires gfx10+", IDRange);
6591 COMPUTE_PGM_RSRC1_GFX10_PLUS_FWD_PROGRESS, ExprVal,
6593 }
else if (ID ==
".amdhsa_shared_vgpr_count") {
6595 if (
ISA.Major < 10 ||
ISA.Major >= 12)
6596 return Error(IDRange.
Start,
"directive requires gfx10 or gfx11",
6598 SharedVGPRCount = Val;
6600 COMPUTE_PGM_RSRC3_GFX10_GFX11_SHARED_VGPR_COUNT, ExprVal,
6602 }
else if (ID ==
".amdhsa_inst_pref_size") {
6604 return Error(IDRange.
Start,
"directive requires gfx11+", IDRange);
6605 if (
ISA.Major == 11) {
6607 COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE, ExprVal,
6611 COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE, ExprVal,
6614 }
else if (ID ==
".amdhsa_exception_fp_ieee_invalid_op") {
6617 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_INVALID_OPERATION,
6619 }
else if (ID ==
".amdhsa_exception_fp_denorm_src") {
6621 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_FP_DENORMAL_SOURCE,
6623 }
else if (ID ==
".amdhsa_exception_fp_ieee_div_zero") {
6626 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_DIVISION_BY_ZERO,
6628 }
else if (ID ==
".amdhsa_exception_fp_ieee_overflow") {
6630 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_OVERFLOW,
6632 }
else if (ID ==
".amdhsa_exception_fp_ieee_underflow") {
6634 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_UNDERFLOW,
6636 }
else if (ID ==
".amdhsa_exception_fp_ieee_inexact") {
6638 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_INEXACT,
6640 }
else if (ID ==
".amdhsa_exception_int_div_zero") {
6642 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_INT_DIVIDE_BY_ZERO,
6644 }
else if (ID ==
".amdhsa_round_robin_scheduling") {
6646 return Error(IDRange.
Start,
"directive requires gfx12+", IDRange);
6648 COMPUTE_PGM_RSRC1_GFX12_PLUS_ENABLE_WG_RR_EN, ExprVal,
6651 return Error(IDRange.
Start,
"unknown .amdhsa_kernel directive", IDRange);
6654#undef PARSE_BITS_ENTRY
6657 if (!Seen.
contains(
".amdhsa_next_free_vgpr"))
6658 return TokError(
".amdhsa_next_free_vgpr directive is required");
6660 if (!Seen.
contains(
".amdhsa_next_free_sgpr"))
6661 return TokError(
".amdhsa_next_free_sgpr directive is required");
6663 unsigned UserSGPRCount = ExplicitUserSGPRCount.value_or(ImpliedUserSGPRCount);
6665 return TokError(
"too many user SGPRs enabled, found " +
6666 Twine(UserSGPRCount) +
", but only " +
6672 if (PreloadLength) {
6678 const MCExpr *VGPRBlocks;
6679 const MCExpr *SGPRBlocks;
6680 if (calculateGPRBlocks(getFeatureBits(), ReserveVCC, ReserveFlatScr,
6681 getTargetStreamer().getTargetID()->isXnackOnOrAny(),
6682 EnableWavefrontSize32, NextFreeVGPR, VGPRRange,
6683 NextFreeSGPR, SGPRRange, VGPRBlocks, SGPRBlocks))
6686 int64_t EvaluatedVGPRBlocks;
6687 bool VGPRBlocksEvaluatable =
6688 VGPRBlocks->evaluateAsAbsolute(EvaluatedVGPRBlocks);
6689 if (VGPRBlocksEvaluatable &&
6691 static_cast<uint64_t>(EvaluatedVGPRBlocks))) {
6692 return OutOfRangeError(VGPRRange);
6696 COMPUTE_PGM_RSRC1_GRANULATED_WORKITEM_VGPR_COUNT_SHIFT,
6697 COMPUTE_PGM_RSRC1_GRANULATED_WORKITEM_VGPR_COUNT,
getContext());
6699 int64_t EvaluatedSGPRBlocks;
6700 if (SGPRBlocks->evaluateAsAbsolute(EvaluatedSGPRBlocks) &&
6702 static_cast<uint64_t>(EvaluatedSGPRBlocks)))
6703 return OutOfRangeError(SGPRRange);
6706 COMPUTE_PGM_RSRC1_GRANULATED_WAVEFRONT_SGPR_COUNT_SHIFT,
6707 COMPUTE_PGM_RSRC1_GRANULATED_WAVEFRONT_SGPR_COUNT,
getContext());
6709 if (ExplicitUserSGPRCount && ImpliedUserSGPRCount > *ExplicitUserSGPRCount)
6710 return TokError(
"amdgpu_user_sgpr_count smaller than implied by "
6711 "enabled user SGPRs");
6717 COMPUTE_PGM_RSRC2_GFX125_USER_SGPR_COUNT_SHIFT,
6718 COMPUTE_PGM_RSRC2_GFX125_USER_SGPR_COUNT,
getContext());
6723 COMPUTE_PGM_RSRC2_GFX6_GFX120_USER_SGPR_COUNT_SHIFT,
6724 COMPUTE_PGM_RSRC2_GFX6_GFX120_USER_SGPR_COUNT,
getContext());
6729 return TokError(
"Kernarg size should be resolvable");
6731 if (PreloadLength && kernarg_size &&
6732 (PreloadLength * 4 + PreloadOffset * 4 > kernarg_size))
6733 return TokError(
"Kernarg preload length + offset is larger than the "
6734 "kernarg segment size");
6737 if (!Seen.
contains(
".amdhsa_accum_offset"))
6738 return TokError(
".amdhsa_accum_offset directive is required");
6739 int64_t EvaluatedAccum;
6740 bool AccumEvaluatable = AccumOffset->evaluateAsAbsolute(EvaluatedAccum);
6741 uint64_t UEvaluatedAccum = EvaluatedAccum;
6742 if (AccumEvaluatable &&
6743 (UEvaluatedAccum < 4 || UEvaluatedAccum > 256 || (UEvaluatedAccum & 3)))
6744 return TokError(
"accum_offset should be in range [4..256] in "
6747 int64_t EvaluatedNumVGPR;
6748 if (NextFreeVGPR->evaluateAsAbsolute(EvaluatedNumVGPR) &&
6752 return TokError(
"accum_offset exceeds total VGPR allocation");
6758 COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET_SHIFT,
6759 COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET,
6765 COMPUTE_PGM_RSRC3_GFX125_NAMED_BAR_CNT_SHIFT,
6766 COMPUTE_PGM_RSRC3_GFX125_NAMED_BAR_CNT,
6769 if (
ISA.Major >= 10 &&
ISA.Major < 12) {
6771 if (SharedVGPRCount && EnableWavefrontSize32 && *EnableWavefrontSize32) {
6772 return TokError(
"shared_vgpr_count directive not valid on "
6773 "wavefront size 32");
6776 if (VGPRBlocksEvaluatable &&
6777 (SharedVGPRCount * 2 +
static_cast<uint64_t>(EvaluatedVGPRBlocks) >
6779 return TokError(
"shared_vgpr_count*2 + "
6780 "compute_pgm_rsrc1.GRANULATED_WORKITEM_VGPR_COUNT cannot "
6785 emitTargetDirective();
6786 getTargetStreamer().EmitAmdhsaKernelDescriptor(getSTI(), KernelName, KD,
6787 NextFreeVGPR, NextFreeSGPR,
6788 ReserveVCC, ReserveFlatScr);
6792bool AMDGPUAsmParser::ParseDirectiveAMDHSACodeObjectVersion() {
6794 if (ParseAsAbsoluteExpression(
Version))
6797 getTargetStreamer().EmitDirectiveAMDHSACodeObjectVersion(
Version);
6798 emitTargetDirective();
6802bool AMDGPUAsmParser::ParseAMDKernelCodeTValue(StringRef ID,
6803 AMDGPUMCKernelCodeT &
C) {
6806 if (ID ==
"max_scratch_backing_memory_byte_size") {
6807 Parser.eatToEndOfStatement();
6811 SmallString<40> ErrStr;
6812 raw_svector_ostream Err(ErrStr);
6813 if (!
C.ParseKernelCodeT(ID, getParser(), Err)) {
6814 return TokError(Err.
str());
6818 if (ID ==
"enable_wavefront_size32") {
6821 return TokError(
"enable_wavefront_size32=1 is only allowed on GFX10+");
6823 return TokError(
"enable_wavefront_size32=1 requires +WavefrontSize32");
6826 return TokError(
"enable_wavefront_size32=0 requires +WavefrontSize64");
6830 if (ID ==
"wavefront_size") {
6831 if (
C.wavefront_size == 5) {
6833 return TokError(
"wavefront_size=5 is only allowed on GFX10+");
6835 return TokError(
"wavefront_size=5 requires +WavefrontSize32");
6836 }
else if (
C.wavefront_size == 6) {
6838 return TokError(
"wavefront_size=6 requires +WavefrontSize64");
6845bool AMDGPUAsmParser::ParseDirectiveAMDKernelCodeT() {
6846 AMDGPUMCKernelCodeT KernelCode;
6856 if (!parseId(ID,
"expected value identifier or .end_amd_kernel_code_t"))
6859 if (ID ==
".end_amd_kernel_code_t")
6862 if (ParseAMDKernelCodeTValue(ID, KernelCode))
6867 getTargetStreamer().EmitAMDKernelCodeT(KernelCode);
6872bool AMDGPUAsmParser::ParseDirectiveAMDGPUHsaKernel() {
6873 StringRef KernelName;
6874 if (!parseId(KernelName,
"expected symbol name"))
6877 getTargetStreamer().EmitAMDGPUSymbolType(KernelName,
6884bool AMDGPUAsmParser::ParseDirectiveISAVersion() {
6885 if (!getSTI().getTargetTriple().isAMDGCN()) {
6886 return Error(getLoc(),
6887 ".amd_amdgpu_isa directive is not available on non-amdgcn "
6891 StringRef TargetIDDirective = getLexer().getTok().getStringContents();
6893 std::optional<AMDGPU::TargetID> MaybeParsed =
6896 return Error(getParser().getTok().getLoc(),
6897 "malformed target id '" + TargetIDDirective +
"'");
6900 const Triple &
TT = getSTI().getTargetTriple();
6906 return Error(getParser().getTok().getLoc(),
6907 "target id '" + TargetIDDirective +
6908 "' specifies a processor that is not valid for subarch '" +
6909 TT.getArchName() +
"'");
6912 const std::optional<AMDGPU::TargetID> &CurrentTargetID =
6913 getTargetStreamer().getTargetID();
6916 const Triple &STITriple = getSTI().getTargetTriple();
6917 if (!DirectiveTriple.isCompatibleWith(STITriple)) {
6918 return Error(getParser().getTok().getLoc(),
6919 ".amd_amdgpu_isa " + Twine(ParsedTargetID.
toString()) +
6920 " is incompatible with " +
6921 Twine(CurrentTargetID->toString()));
6925 StringRef DirectiveProcessor =
6928 if (DirectiveISA != ISA) {
6929 return Error(getParser().getTok().getLoc(),
6930 ".amd_amdgpu_isa directive processor " +
6931 Twine(DirectiveProcessor) +
6932 " does not match the specified processor " +
6933 Twine(getSTI().
getCPU()));
6936 getTargetStreamer().EmitISAVersion();
6942bool AMDGPUAsmParser::ParseDirectiveHSAMetadata() {
6945 std::string HSAMetadataString;
6950 if (!getTargetStreamer().EmitHSAMetadataV3(HSAMetadataString))
6951 return Error(getLoc(),
"invalid HSA metadata");
6958bool AMDGPUAsmParser::ParseToEndDirective(
const char *AssemblerDirectiveBegin,
6959 const char *AssemblerDirectiveEnd,
6960 std::string &CollectString) {
6962 raw_string_ostream CollectStream(CollectString);
6964 getLexer().setSkipSpace(
false);
6966 bool FoundEnd =
false;
6969 CollectStream << getTokenStr();
6973 if (trySkipId(AssemblerDirectiveEnd)) {
6978 CollectStream << Parser.parseStringToEndOfStatement()
6979 <<
getContext().getAsmInfo().getSeparatorString();
6981 Parser.eatToEndOfStatement();
6984 getLexer().setSkipSpace(
true);
6987 return TokError(Twine(
"expected directive ") +
6988 Twine(AssemblerDirectiveEnd) + Twine(
" not found"));
6995bool AMDGPUAsmParser::ParseDirectivePALMetadataBegin() {
7001 auto *PALMetadata = getTargetStreamer().getPALMetadata();
7002 if (!PALMetadata->setFromString(
String))
7003 return Error(getLoc(),
"invalid PAL metadata");
7008bool AMDGPUAsmParser::ParseDirectivePALMetadata() {
7011 Twine(
" directive is "
7012 "not available on non-amdpal OSes"))
7016 auto *PALMetadata = getTargetStreamer().getPALMetadata();
7017 PALMetadata->setLegacy();
7020 if (ParseAsAbsoluteExpression(
Key)) {
7021 return TokError(Twine(
"invalid value in ") +
7025 return TokError(Twine(
"expected an even number of values in ") +
7028 if (ParseAsAbsoluteExpression(
Value)) {
7029 return TokError(Twine(
"invalid value in ") +
7032 PALMetadata->setRegister(
Key,
Value);
7041bool AMDGPUAsmParser::ParseDirectiveAMDGPULDS() {
7042 if (getParser().checkForValidSection())
7046 SMLoc NameLoc = getLoc();
7047 if (getParser().parseIdentifier(Name))
7048 return TokError(
"expected identifier in directive");
7051 if (getParser().parseComma())
7057 SMLoc SizeLoc = getLoc();
7058 if (getParser().parseAbsoluteExpression(
Size))
7061 return Error(SizeLoc,
"size must be non-negative");
7062 if (
Size > LocalMemorySize)
7063 return Error(SizeLoc,
"size is too large");
7067 SMLoc AlignLoc = getLoc();
7068 if (getParser().parseAbsoluteExpression(Alignment))
7071 return Error(AlignLoc,
"alignment must be a power of two");
7076 if (Alignment >= 1u << 31)
7077 return Error(AlignLoc,
"alignment is too large");
7083 Symbol->redefineIfPossible();
7084 if (!
Symbol->isUndefined())
7085 return Error(NameLoc,
"invalid symbol redefinition");
7087 getTargetStreamer().emitAMDGPULDS(Symbol,
Size,
Align(Alignment));
7091bool AMDGPUAsmParser::ParseDirectiveAMDGPUInfo() {
7092 if (getParser().checkForValidSection())
7096 if (getParser().parseIdentifier(FuncName))
7097 return TokError(
"expected symbol name after .amdgpu_info");
7100 AMDGPU::InfoSectionData ParsedInfoData;
7101 AMDGPU::FuncInfo FI;
7103 bool HasScalarAttrs =
false;
7110 SMLoc IDLoc = getLoc();
7111 if (!parseId(ID,
"expected directive or .end_amdgpu_info"))
7114 if (ID ==
".end_amdgpu_info")
7122 return Error(IDLoc,
"unknown .amdgpu_info directive '" + ID +
"'");
7124 if (Dir ==
"flags") {
7126 if (getParser().parseAbsoluteExpression(Val))
7129 FI.
UsesVCC = !!(
Flags & AMDGPU::FuncInfoFlags::FUNC_USES_VCC);
7131 !!(
Flags & AMDGPU::FuncInfoFlags::FUNC_USES_FLAT_SCRATCH);
7133 HasScalarAttrs =
true;
7134 }
else if (Dir ==
"num_sgpr") {
7136 if (getParser().parseAbsoluteExpression(Val))
7138 FI.
NumSGPR =
static_cast<uint32_t
>(Val);
7139 HasScalarAttrs =
true;
7140 }
else if (Dir ==
"num_vgpr") {
7142 if (getParser().parseAbsoluteExpression(Val))
7145 HasScalarAttrs =
true;
7146 }
else if (Dir ==
"num_agpr") {
7148 if (getParser().parseAbsoluteExpression(Val))
7151 HasScalarAttrs =
true;
7152 }
else if (Dir ==
"private_segment_size") {
7154 if (getParser().parseAbsoluteExpression(Val))
7157 HasScalarAttrs =
true;
7158 }
else if (Dir ==
"use") {
7160 if (getParser().parseIdentifier(ResName))
7161 return TokError(
"expected resource symbol for .amdgpu_use");
7162 ParsedInfoData.
Uses.push_back(
7163 {FuncSym,
getContext().getOrCreateSymbol(ResName)});
7164 }
else if (Dir ==
"call") {
7166 if (getParser().parseIdentifier(DstName))
7167 return TokError(
"expected callee symbol for .amdgpu_call");
7168 ParsedInfoData.
Calls.push_back(
7169 {FuncSym,
getContext().getOrCreateSymbol(DstName)});
7170 }
else if (Dir ==
"indirect_call") {
7172 if (getParser().parseEscapedString(TypeId))
7173 return TokError(
"expected type ID string for .amdgpu_indirect_call");
7174 ParsedInfoData.
IndirectCalls.push_back({FuncSym, std::move(TypeId)});
7175 }
else if (Dir ==
"typeid") {
7177 if (getParser().parseEscapedString(TypeId))
7178 return TokError(
"expected type ID string for .amdgpu_typeid");
7179 ParsedInfoData.
TypeIds.push_back({FuncSym, std::move(TypeId)});
7181 return Error(IDLoc,
"unknown .amdgpu_info directive '" + ID +
"'");
7186 ParsedInfoData.
Funcs.push_back(std::move(FI));
7188 AMDGPU::InfoSectionData &
Data = InfoData ? *InfoData : InfoData.emplace();
7189 for (AMDGPU::FuncInfo &Func : ParsedInfoData.
Funcs)
7190 Data.Funcs.push_back(std::move(Func));
7191 for (std::pair<MCSymbol *, MCSymbol *> &Use : ParsedInfoData.
Uses)
7192 Data.Uses.push_back(Use);
7193 for (std::pair<MCSymbol *, MCSymbol *> &
Call : ParsedInfoData.
Calls)
7195 for (std::pair<MCSymbol *, std::string> &
IndirectCall :
7198 for (std::pair<MCSymbol *, std::string> &TypeId : ParsedInfoData.
TypeIds)
7199 Data.TypeIds.push_back(std::move(TypeId));
7204void AMDGPUAsmParser::doBeforeLabelEmit(MCSymbol *Symbol, SMLoc IDLoc) {
7211void AMDGPUAsmParser::checkKernelPrologues() {
7212 if (getFeatureBits()[AMDGPU::FeatureRequiresInitialUnclausedVmem]) {
7213 static const unsigned Required[] = {S_MOV_B64_gfx12, V_NOP_e32_gfx12,
7214 GLOBAL_PREFETCH_B8_SADDR_gfx1250};
7215 for (
auto [Sym, Loc,
Offset] : OpcodeStreamSymbols) {
7216 if (!AMDHSAKernelSymbols.
contains(Sym))
7218 ArrayRef<unsigned> Prologue =
ArrayRef(OpcodeStream).drop_front(
Offset);
7219 if (!Prologue.
empty() && Prologue.
front() == S_SETREG_IMM32_B32_gfx12)
7223 "' does not begin with the required prologue "
7224 "sequence: s_mov_b64 followed by v_nop and "
7225 "global_prefetch_b8");
7229 OpcodeStream.
clear();
7230 OpcodeStreamSymbols.clear();
7231 AMDHSAKernelSymbols.
clear();
7234void AMDGPUAsmParser::onEndOfFile() {
7235 emitTargetDirective();
7236 checkKernelPrologues();
7238 getTargetStreamer().emitAMDGPUInfo(*InfoData);
7241bool AMDGPUAsmParser::ParseDirective(AsmToken DirectiveID) {
7242 StringRef IDVal = DirectiveID.
getString();
7245 if (IDVal ==
".amdhsa_kernel")
7246 return ParseDirectiveAMDHSAKernel();
7248 if (IDVal ==
".amdhsa_code_object_version")
7249 return ParseDirectiveAMDHSACodeObjectVersion();
7253 return ParseDirectiveHSAMetadata();
7255 if (IDVal ==
".amd_kernel_code_t")
7256 return ParseDirectiveAMDKernelCodeT();
7258 if (IDVal ==
".amdgpu_hsa_kernel")
7259 return ParseDirectiveAMDGPUHsaKernel();
7261 if (IDVal ==
".amd_amdgpu_isa")
7262 return ParseDirectiveISAVersion();
7266 Twine(
" directive is "
7267 "not available on non-amdhsa OSes"))
7272 if (IDVal ==
".amdgcn_target")
7273 return ParseDirectiveAMDGCNTarget();
7275 if (IDVal ==
".amdgpu_lds")
7276 return ParseDirectiveAMDGPULDS();
7278 if (IDVal ==
".amdgpu_info")
7279 return ParseDirectiveAMDGPUInfo();
7282 return ParseDirectivePALMetadataBegin();
7285 return ParseDirectivePALMetadata();
7290bool AMDGPUAsmParser::subtargetHasRegister(
const MCRegisterInfo &MRI,
7297 return hasSGPR104_SGPR105();
7300 case SRC_SHARED_BASE_LO:
7301 case SRC_SHARED_BASE:
7302 case SRC_SHARED_LIMIT_LO:
7303 case SRC_SHARED_LIMIT:
7305 case SRC_PRIVATE_BASE_LO:
7306 case SRC_PRIVATE_BASE:
7307 case SRC_PRIVATE_LIMIT_LO:
7308 case SRC_PRIVATE_LIMIT:
7310 case SRC_FLAT_SCRATCH_BASE_LO:
7311 case SRC_FLAT_SCRATCH_BASE_HI:
7312 return hasGloballyAddressableScratch();
7313 case SRC_POPS_EXITING_WAVE_ID:
7326 getTargetStreamer().getTargetID()->isXnackSupported();
7356 return hasSGPR102_SGPR103();
7364 ParseStatus Res = parseVOPD(
Operands);
7369 Res = MatchOperandParserImpl(
Operands, Mnemonic);
7381 SMLoc LBraceLoc = getLoc();
7386 auto Loc = getLoc();
7389 Error(Loc,
"expected a register");
7393 RBraceLoc = getLoc();
7398 "expected a comma or a closing square bracket"))
7402 if (
Operands.size() - Prefix > 1) {
7404 AMDGPUOperand::CreateToken(
this,
"[", LBraceLoc));
7405 Operands.push_back(AMDGPUOperand::CreateToken(
this,
"]", RBraceLoc));
7414StringRef AMDGPUAsmParser::parseMnemonicSuffix(StringRef Name) {
7416 setForcedEncodingSize(0);
7417 setForcedDPP(
false);
7418 setForcedSDWA(
false);
7420 if (
Name.consume_back(
"_e64_dpp")) {
7422 setForcedEncodingSize(64);
7425 if (
Name.consume_back(
"_e64")) {
7426 setForcedEncodingSize(64);
7429 if (
Name.consume_back(
"_e32")) {
7430 setForcedEncodingSize(32);
7433 if (
Name.consume_back(
"_dpp")) {
7437 if (
Name.consume_back(
"_sdwa")) {
7438 setForcedSDWA(
true);
7446 unsigned VariantID);
7452 Name = parseMnemonicSuffix(Name);
7458 Operands.push_back(AMDGPUOperand::CreateToken(
this, Name, NameLoc));
7460 bool IsMIMG = Name.starts_with(
"image_");
7463 OperandMode
Mode = OperandMode_Default;
7465 Mode = OperandMode_NSA;
7469 checkUnsupportedInstruction(Name, NameLoc);
7470 if (!Parser.hasPendingError()) {
7473 :
"not a valid operand.";
7493ParseStatus AMDGPUAsmParser::parseTokenOp(StringRef Name,
7496 if (!trySkipId(Name))
7499 Operands.push_back(AMDGPUOperand::CreateToken(
this, Name, S));
7503ParseStatus AMDGPUAsmParser::parseIntWithPrefix(
const char *Prefix,
7512ParseStatus AMDGPUAsmParser::parseIntWithPrefix(
7514 std::function<
bool(int64_t &)> ConvertResult) {
7518 ParseStatus Res = parseIntWithPrefix(Prefix,
Value);
7522 if (ConvertResult && !ConvertResult(
Value)) {
7523 Error(S,
"invalid " + StringRef(Prefix) +
" value.");
7526 Operands.push_back(AMDGPUOperand::CreateImm(
this,
Value, S, ImmTy));
7530ParseStatus AMDGPUAsmParser::parseOperandArrayWithPrefix(
7532 bool (*ConvertResult)(int64_t &)) {
7541 const unsigned MaxSize = 4;
7545 for (
int I = 0;; ++
I) {
7547 SMLoc Loc = getLoc();
7551 if (
Op != 0 &&
Op != 1)
7552 return Error(Loc,
"invalid " + StringRef(Prefix) +
" value.");
7559 if (
I + 1 == MaxSize)
7560 return Error(getLoc(),
"expected a closing square bracket");
7566 Operands.push_back(AMDGPUOperand::CreateImm(
this, Val, S, ImmTy));
7570ParseStatus AMDGPUAsmParser::parseNamedBit(StringRef Name,
7572 AMDGPUOperand::ImmTy ImmTy,
7573 bool IgnoreNegative) {
7577 if (trySkipId(Name)) {
7579 }
else if (trySkipId(
"no", Name)) {
7588 return Error(S,
"r128 modifier is not supported on this GPU");
7589 if (Name ==
"a16" && !
hasA16())
7590 return Error(S,
"a16 modifier is not supported on this GPU");
7592 if (Bit == 0 && Name ==
"gds") {
7595 return Error(S,
"nogds is not allowed");
7598 if (
isGFX9() && ImmTy == AMDGPUOperand::ImmTyA16)
7599 ImmTy = AMDGPUOperand::ImmTyR128A16;
7601 Operands.push_back(AMDGPUOperand::CreateImm(
this, Bit, S, ImmTy));
7605unsigned AMDGPUAsmParser::getCPolKind(StringRef Id, StringRef Mnemo,
7606 bool &Disabling)
const {
7607 Disabling =
Id.consume_front(
"no");
7610 return StringSwitch<unsigned>(Id)
7617 return StringSwitch<unsigned>(Id)
7627 SMLoc StringLoc = getLoc();
7629 int64_t CPolVal = 0;
7649 ResScope = parseScope(
Operands, Scope);
7662 if (trySkipId(
"nv")) {
7666 }
else if (trySkipId(
"no",
"nv")) {
7673 if (trySkipId(
"scale_offset")) {
7677 }
else if (trySkipId(
"no",
"scale_offset")) {
7690 Operands.push_back(AMDGPUOperand::CreateImm(
this, CPolVal, StringLoc,
7691 AMDGPUOperand::ImmTyCPol));
7696 SMLoc OpLoc = getLoc();
7697 unsigned Enabled = 0, Seen = 0;
7701 unsigned CPol = getCPolKind(getId(), Mnemo, Disabling);
7708 return Error(S,
"dlc modifier is not supported on this GPU");
7711 return Error(S,
"scc modifier is not supported on this GPU");
7714 return Error(S,
"duplicate cache policy modifier");
7726 AMDGPUOperand::CreateImm(
this,
Enabled, OpLoc, AMDGPUOperand::ImmTyCPol));
7735 ParseStatus Res = parseStringOrIntWithPrefix(
7736 Operands,
"scope", {
"SCOPE_CU",
"SCOPE_SE",
"SCOPE_DEV",
"SCOPE_SYS"},
7750 ParseStatus Res = parseStringWithPrefix(
"th",
Value, StringLoc);
7754 if (
Value ==
"TH_DEFAULT")
7756 else if (
Value ==
"TH_STORE_LU" ||
Value ==
"TH_LOAD_WB" ||
7757 Value ==
"TH_LOAD_NT_WB") {
7758 return Error(StringLoc,
"invalid th value");
7759 }
else if (
Value.consume_front(
"TH_ATOMIC_")) {
7761 }
else if (
Value.consume_front(
"TH_LOAD_")) {
7763 }
else if (
Value.consume_front(
"TH_STORE_")) {
7766 return Error(StringLoc,
"invalid th value");
7769 if (
Value ==
"BYPASS")
7774 TH |= StringSwitch<int64_t>(
Value)
7784 .Default(0xffffffff);
7786 TH |= StringSwitch<int64_t>(
Value)
7797 .Default(0xffffffff);
7800 if (TH == 0xffffffff)
7801 return Error(StringLoc,
"invalid th value");
7808 AMDGPUAsmParser::OptionalImmIndexMap &OptionalIdx,
7809 AMDGPUOperand::ImmTy ImmT, int64_t
Default = 0,
7810 std::optional<unsigned> InsertAt = std::nullopt) {
7811 auto i = OptionalIdx.find(ImmT);
7812 if (i != OptionalIdx.end()) {
7813 unsigned Idx = i->second;
7814 const AMDGPUOperand &
Op =
7815 static_cast<const AMDGPUOperand &
>(*
Operands[Idx]);
7819 Op.addImmOperands(Inst, 1);
7821 if (InsertAt.has_value())
7828ParseStatus AMDGPUAsmParser::parseStringWithPrefix(StringRef Prefix,
7834 StringLoc = getLoc();
7839ParseStatus AMDGPUAsmParser::parseStringOrIntWithPrefix(
7845 SMLoc StringLoc = getLoc();
7849 Value = getTokenStr();
7853 if (
Value == Ids[IntVal])
7858 if (IntVal < 0 || IntVal >= (int64_t)Ids.
size())
7859 return Error(StringLoc,
"invalid " + Twine(Name) +
" value");
7864ParseStatus AMDGPUAsmParser::parseStringOrIntWithPrefix(
7866 AMDGPUOperand::ImmTy
Type) {
7870 ParseStatus Res = parseStringOrIntWithPrefix(
Operands, Name, Ids, IntVal);
7872 Operands.push_back(AMDGPUOperand::CreateImm(
this, IntVal, S,
Type));
7881bool AMDGPUAsmParser::tryParseFmt(
const char *Pref, int64_t MaxVal,
7884 SMLoc Loc = getLoc();
7886 auto Res = parseIntWithPrefix(Pref, Val);
7892 if (Val < 0 || Val > MaxVal) {
7893 Error(Loc, Twine(
"out of range ", StringRef(Pref)));
7902 AMDGPUOperand::ImmTy ImmTy) {
7903 const char *Pref =
"index_key";
7905 SMLoc Loc = getLoc();
7906 auto Res = parseIntWithPrefix(Pref, ImmVal);
7910 if ((ImmTy == AMDGPUOperand::ImmTyIndexKey16bit ||
7911 ImmTy == AMDGPUOperand::ImmTyIndexKey32bit) &&
7912 (ImmVal < 0 || ImmVal > 1))
7913 return Error(Loc, Twine(
"out of range ", StringRef(Pref)));
7915 if (ImmTy == AMDGPUOperand::ImmTyIndexKey8bit && (ImmVal < 0 || ImmVal > 3))
7916 return Error(Loc, Twine(
"out of range ", StringRef(Pref)));
7918 Operands.push_back(AMDGPUOperand::CreateImm(
this, ImmVal, Loc, ImmTy));
7923 return tryParseIndexKey(
Operands, AMDGPUOperand::ImmTyIndexKey8bit);
7927 return tryParseIndexKey(
Operands, AMDGPUOperand::ImmTyIndexKey16bit);
7931 return tryParseIndexKey(
Operands, AMDGPUOperand::ImmTyIndexKey32bit);
7936 AMDGPUOperand::ImmTy
Type) {
7942 return tryParseMatrixFMT(
Operands,
"matrix_a_fmt",
7943 AMDGPUOperand::ImmTyMatrixAFMT);
7947 return tryParseMatrixFMT(
Operands,
"matrix_b_fmt",
7948 AMDGPUOperand::ImmTyMatrixBFMT);
7953 AMDGPUOperand::ImmTy
Type) {
7959 return tryParseMatrixScale(
Operands,
"matrix_a_scale",
7960 AMDGPUOperand::ImmTyMatrixAScale);
7964 return tryParseMatrixScale(
Operands,
"matrix_b_scale",
7965 AMDGPUOperand::ImmTyMatrixBScale);
7970 AMDGPUOperand::ImmTy
Type) {
7976 return tryParseMatrixScaleFmt(
Operands,
"matrix_a_scale_fmt",
7977 AMDGPUOperand::ImmTyMatrixAScaleFmt);
7981 return tryParseMatrixScaleFmt(
Operands,
"matrix_b_scale_fmt",
7982 AMDGPUOperand::ImmTyMatrixBScaleFmt);
7987ParseStatus AMDGPUAsmParser::parseDfmtNfmt(int64_t &
Format) {
7988 using namespace llvm::AMDGPU::MTBUFFormat;
7994 for (
int I = 0;
I < 2; ++
I) {
7995 if (Dfmt == DFMT_UNDEF && !tryParseFmt(
"dfmt", DFMT_MAX, Dfmt))
7998 if (Nfmt == NFMT_UNDEF && !tryParseFmt(
"nfmt", NFMT_MAX, Nfmt))
8003 if ((Dfmt == DFMT_UNDEF) != (Nfmt == NFMT_UNDEF) &&
8009 if (Dfmt == DFMT_UNDEF && Nfmt == NFMT_UNDEF)
8012 Dfmt = (Dfmt ==
DFMT_UNDEF) ? DFMT_DEFAULT : Dfmt;
8013 Nfmt = (Nfmt ==
NFMT_UNDEF) ? NFMT_DEFAULT : Nfmt;
8019ParseStatus AMDGPUAsmParser::parseUfmt(int64_t &
Format) {
8020 using namespace llvm::AMDGPU::MTBUFFormat;
8024 if (!tryParseFmt(
"format", UFMT_MAX, Fmt))
8027 if (Fmt == UFMT_UNDEF)
8034bool AMDGPUAsmParser::matchDfmtNfmt(int64_t &Dfmt, int64_t &Nfmt,
8035 StringRef FormatStr, SMLoc Loc) {
8036 using namespace llvm::AMDGPU::MTBUFFormat;
8040 if (
Format != DFMT_UNDEF) {
8046 if (
Format != NFMT_UNDEF) {
8051 Error(Loc,
"unsupported format");
8055ParseStatus AMDGPUAsmParser::parseSymbolicSplitFormat(StringRef FormatStr,
8058 using namespace llvm::AMDGPU::MTBUFFormat;
8062 if (!matchDfmtNfmt(Dfmt, Nfmt, FormatStr, FormatLoc))
8067 SMLoc Loc = getLoc();
8068 if (!parseId(Str,
"expected a format string") ||
8069 !matchDfmtNfmt(Dfmt, Nfmt, Str, Loc))
8071 if (Dfmt == DFMT_UNDEF)
8072 return Error(Loc,
"duplicate numeric format");
8073 if (Nfmt == NFMT_UNDEF)
8074 return Error(Loc,
"duplicate data format");
8077 Dfmt = (Dfmt ==
DFMT_UNDEF) ? DFMT_DEFAULT : Dfmt;
8078 Nfmt = (Nfmt ==
NFMT_UNDEF) ? NFMT_DEFAULT : Nfmt;
8082 if (Ufmt == UFMT_UNDEF)
8083 return Error(FormatLoc,
"unsupported format");
8092ParseStatus AMDGPUAsmParser::parseSymbolicUnifiedFormat(StringRef FormatStr,
8095 using namespace llvm::AMDGPU::MTBUFFormat;
8098 if (Id == UFMT_UNDEF)
8102 return Error(Loc,
"unified format is not supported on this GPU");
8108ParseStatus AMDGPUAsmParser::parseNumericFormat(int64_t &
Format) {
8109 using namespace llvm::AMDGPU::MTBUFFormat;
8110 SMLoc Loc = getLoc();
8115 return Error(Loc,
"out of range format");
8120ParseStatus AMDGPUAsmParser::parseSymbolicOrNumericFormat(int64_t &
Format) {
8121 using namespace llvm::AMDGPU::MTBUFFormat;
8127 StringRef FormatStr;
8128 SMLoc Loc = getLoc();
8129 if (!parseId(FormatStr,
"expected a format string"))
8132 auto Res = parseSymbolicUnifiedFormat(FormatStr, Loc,
Format);
8134 Res = parseSymbolicSplitFormat(FormatStr, Loc,
Format);
8144 return parseNumericFormat(
Format);
8148 using namespace llvm::AMDGPU::MTBUFFormat;
8152 SMLoc Loc = getLoc();
8162 AMDGPUOperand::CreateImm(
this,
Format, Loc, AMDGPUOperand::ImmTyFORMAT));
8181 Res = parseSymbolicOrNumericFormat(
Format);
8186 AMDGPUOperand &
Op =
static_cast<AMDGPUOperand &
>(*
Operands[
Size - 2]);
8187 assert(
Op.isImm() &&
Op.getImmTy() == AMDGPUOperand::ImmTyFORMAT);
8194 return Error(getLoc(),
"duplicate format");
8200 parseIntWithPrefix(
"offset",
Operands, AMDGPUOperand::ImmTyOffset);
8202 Res = parseIntWithPrefix(
"inst_offset",
Operands,
8203 AMDGPUOperand::ImmTyInstOffset);
8210 parseNamedBit(
"r128",
Operands, AMDGPUOperand::ImmTyR128A16);
8212 Res = parseNamedBit(
"a16",
Operands, AMDGPUOperand::ImmTyA16);
8218 parseIntWithPrefix(
"blgp",
Operands, AMDGPUOperand::ImmTyBLGP);
8221 parseOperandArrayWithPrefix(
"neg",
Operands, AMDGPUOperand::ImmTyBLGP);
8231 OptionalImmIndexMap OptionalIdx;
8233 unsigned OperandIdx[4];
8234 unsigned EnMask = 0;
8237 for (
unsigned i = 1, e =
Operands.size(); i != e; ++i) {
8238 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
8243 OperandIdx[SrcIdx] = Inst.
size();
8244 Op.addRegOperands(Inst, 1);
8251 OperandIdx[SrcIdx] = Inst.
size();
8257 if (
Op.isImm() &&
Op.getImmTy() == AMDGPUOperand::ImmTyExpTgt) {
8258 Op.addImmOperands(Inst, 1);
8262 if (
Op.isToken() && (
Op.getToken() ==
"done" ||
Op.getToken() ==
"row_en"))
8266 OptionalIdx[
Op.getImmTy()] = i;
8272 if (OptionalIdx.find(AMDGPUOperand::ImmTyExpCompr) != OptionalIdx.end()) {
8279 for (
auto i = 0; i < SrcIdx; ++i) {
8281 EnMask |= Compr ? (0x3 << i * 2) : (0x1 << i);
8287 AMDGPUOperand::ImmTyExpCompr);
8297 int64_t CntVal,
bool Saturate,
8303 IntVal =
encode(ISA, IntVal, CntVal);
8304 if (CntVal !=
decode(ISA, IntVal)) {
8306 IntVal =
encode(ISA, IntVal, -1);
8314bool AMDGPUAsmParser::parseCnt(int64_t &IntVal) {
8316 SMLoc CntLoc = getLoc();
8317 StringRef CntName = getTokenStr();
8324 SMLoc ValLoc = getLoc();
8331 if (CntName ==
"vmcnt" || CntName ==
"vmcnt_sat") {
8333 }
else if (CntName ==
"expcnt" || CntName ==
"expcnt_sat") {
8335 }
else if (CntName ==
"lgkmcnt" || CntName ==
"lgkmcnt_sat") {
8338 Error(CntLoc,
"invalid counter name " + CntName);
8343 Error(ValLoc,
"too large value for " + CntName);
8352 Error(getLoc(),
"expected a counter name");
8366 if (!parseCnt(Waitcnt))
8374 Operands.push_back(AMDGPUOperand::CreateImm(
this, Waitcnt, S));
8378bool AMDGPUAsmParser::parseDelay(int64_t &Delay) {
8379 SMLoc FieldLoc = getLoc();
8380 StringRef FieldName = getTokenStr();
8385 SMLoc ValueLoc = getLoc();
8392 if (FieldName ==
"instid0") {
8394 }
else if (FieldName ==
"instskip") {
8396 }
else if (FieldName ==
"instid1") {
8399 Error(FieldLoc,
"invalid field name " + FieldName);
8418 .Case(
"VALU_DEP_1", 1)
8419 .Case(
"VALU_DEP_2", 2)
8420 .Case(
"VALU_DEP_3", 3)
8421 .Case(
"VALU_DEP_4", 4)
8422 .Case(
"TRANS32_DEP_1", 5)
8423 .Case(
"TRANS32_DEP_2", 6)
8424 .Case(
"TRANS32_DEP_3", 7)
8425 .Case(
"FMA_ACCUM_CYCLE_1", 8)
8426 .Case(
"SALU_CYCLE_1", 9)
8427 .Case(
"SALU_CYCLE_2", 10)
8428 .Case(
"SALU_CYCLE_3", 11)
8436 Delay |=
Value << Shift;
8446 if (!parseDelay(Delay))
8454 Operands.push_back(AMDGPUOperand::CreateImm(
this, Delay, S));
8458bool AMDGPUOperand::isSWaitCnt()
const {
return isImm(); }
8460bool AMDGPUOperand::isSDelayALU()
const {
return isImm(); }
8466void AMDGPUAsmParser::depCtrError(SMLoc Loc,
int ErrorId,
8467 StringRef DepCtrName) {
8470 Error(Loc, Twine(
"invalid counter name ", DepCtrName));
8473 Error(Loc, Twine(DepCtrName,
" is not supported on this GPU"));
8476 Error(Loc, Twine(
"duplicate counter name ", DepCtrName));
8479 Error(Loc, Twine(
"invalid value for ", DepCtrName));
8486bool AMDGPUAsmParser::parseDepCtr(int64_t &DepCtr,
unsigned &UsedOprMask) {
8488 using namespace llvm::AMDGPU::DepCtr;
8490 SMLoc DepCtrLoc = getLoc();
8491 StringRef DepCtrName = getTokenStr();
8501 unsigned PrevOprMask = UsedOprMask;
8502 int CntVal =
encodeDepCtr(DepCtrName, ExprVal, UsedOprMask, getSTI());
8505 depCtrError(DepCtrLoc, CntVal, DepCtrName);
8514 Error(getLoc(),
"expected a counter name");
8519 int64_t CntValMask = PrevOprMask ^ UsedOprMask;
8520 DepCtr = (DepCtr & ~CntValMask) | CntVal;
8525 using namespace llvm::AMDGPU::DepCtr;
8528 SMLoc Loc = getLoc();
8531 unsigned UsedOprMask = 0;
8533 if (!parseDepCtr(DepCtr, UsedOprMask))
8541 Operands.push_back(AMDGPUOperand::CreateImm(
this, DepCtr, Loc));
8545bool AMDGPUOperand::isDepCtr()
const {
return isS16Imm(); }
8551ParseStatus AMDGPUAsmParser::parseHwregFunc(OperandInfoTy &HwReg,
8553 OperandInfoTy &Width) {
8554 using namespace llvm::AMDGPU::Hwreg;
8560 HwReg.Loc = getLoc();
8563 HwReg.IsSymbolic =
true;
8565 }
else if (!
parseExpr(HwReg.Val,
"a register name")) {
8573 if (!skipToken(
AsmToken::Comma,
"expected a comma or a closing parenthesis"))
8583 Width.Loc = getLoc();
8592 using namespace llvm::AMDGPU::Hwreg;
8595 SMLoc Loc = getLoc();
8597 StructuredOpField HwReg(
"id",
"hardware register", HwregId::Width,
8599 StructuredOpField
Offset(
"offset",
"bit offset", HwregOffset::Width,
8600 HwregOffset::Default);
8601 struct : StructuredOpField {
8602 using StructuredOpField::StructuredOpField;
8603 bool validate(AMDGPUAsmParser &Parser)
const override {
8605 return Error(Parser,
"only values from 1 to 32 are legal");
8608 } Width(
"size",
"bitfield width", HwregSize::Width, HwregSize::Default);
8609 ParseStatus Res = parseStructuredOpFields({&HwReg, &
Offset, &Width});
8612 Res = parseHwregFunc(HwReg,
Offset, Width);
8615 if (!validateStructuredOpFields({&HwReg, &
Offset, &Width}))
8617 ImmVal = HwregEncoding::encode(HwReg.Val,
Offset.Val, Width.Val);
8621 parseExpr(ImmVal,
"a hwreg macro, structured immediate"))
8628 return Error(Loc,
"invalid immediate: only 16-bit values are legal");
8630 AMDGPUOperand::CreateImm(
this, ImmVal, Loc, AMDGPUOperand::ImmTyHwreg));
8634bool AMDGPUOperand::isHwreg()
const {
return isImmTy(ImmTyHwreg); }
8640bool AMDGPUAsmParser::parseSendMsgBody(OperandInfoTy &
Msg, OperandInfoTy &
Op,
8641 OperandInfoTy &Stream) {
8642 using namespace llvm::AMDGPU::SendMsg;
8647 Msg.IsSymbolic =
true;
8654 Op.IsDefined =
true;
8660 }
else if (!
parseExpr(
Op.Val,
"an operation name")) {
8665 Stream.IsDefined =
true;
8666 Stream.Loc = getLoc();
8675bool AMDGPUAsmParser::validateSendMsg(
const OperandInfoTy &
Msg,
8676 const OperandInfoTy &
Op,
8677 const OperandInfoTy &Stream) {
8678 using namespace llvm::AMDGPU::SendMsg;
8687 Error(
Msg.Loc,
"specified message id is not supported on this GPU");
8692 Error(
Msg.Loc,
"invalid message id");
8698 Error(
Op.Loc,
"message does not support operations");
8700 Error(
Msg.Loc,
"missing message operation");
8706 Error(
Op.Loc,
"specified operation id is not supported on this GPU");
8708 Error(
Op.Loc,
"invalid operation id");
8713 Error(Stream.Loc,
"message operation does not support streams");
8717 Error(Stream.Loc,
"invalid message stream id");
8724 using namespace llvm::AMDGPU::SendMsg;
8727 SMLoc Loc = getLoc();
8731 OperandInfoTy
Op(OP_NONE_);
8732 OperandInfoTy Stream(STREAM_ID_NONE_);
8733 if (parseSendMsgBody(
Msg,
Op, Stream) && validateSendMsg(
Msg,
Op, Stream)) {
8738 }
else if (
parseExpr(ImmVal,
"a sendmsg macro")) {
8740 return Error(Loc,
"invalid immediate: only 16-bit values are legal");
8746 AMDGPUOperand::CreateImm(
this, ImmVal, Loc, AMDGPUOperand::ImmTySendMsg));
8750bool AMDGPUOperand::isSendMsg()
const {
return isImmTy(ImmTySendMsg); }
8753 using namespace llvm::AMDGPU::WaitEvent;
8755 SMLoc Loc = getLoc();
8758 StructuredOpField DontWaitExportReady(
"dont_wait_export_ready",
"bit value",
8760 StructuredOpField ExportReady(
"export_ready",
"bit value", 1, 0);
8762 StructuredOpField *TargetBitfield =
8763 isGFX11() ? &DontWaitExportReady : &ExportReady;
8765 ParseStatus Res = parseStructuredOpFields({TargetBitfield});
8769 if (!validateStructuredOpFields({TargetBitfield}))
8771 ImmVal = TargetBitfield->Val;
8778 return Error(Loc,
"invalid immediate: only 16-bit values are legal");
8780 Operands.push_back(AMDGPUOperand::CreateImm(
this, ImmVal, Loc,
8781 AMDGPUOperand::ImmTyWaitEvent));
8785bool AMDGPUOperand::isWaitEvent()
const {
return isImmTy(ImmTyWaitEvent); }
8798 int Slot = StringSwitch<int>(Str)
8805 return Error(S,
"invalid interpolation slot");
8808 AMDGPUOperand::CreateImm(
this, Slot, S, AMDGPUOperand::ImmTyInterpSlot));
8819 if (!Str.starts_with(
"attr"))
8820 return Error(S,
"invalid interpolation attribute");
8822 StringRef Chan = Str.take_back(2);
8823 int AttrChan = StringSwitch<int>(Chan)
8830 return Error(S,
"invalid or missing interpolation attribute channel");
8832 Str = Str.drop_back(2).drop_front(4);
8835 if (Str.getAsInteger(10, Attr))
8836 return Error(S,
"invalid or missing interpolation attribute number");
8839 return Error(S,
"out of bounds interpolation attribute number");
8844 AMDGPUOperand::CreateImm(
this, Attr, S, AMDGPUOperand::ImmTyInterpAttr));
8845 Operands.push_back(AMDGPUOperand::CreateImm(
8846 this, AttrChan, SChan, AMDGPUOperand::ImmTyInterpAttrChan));
8855 using namespace llvm::AMDGPU::Exp;
8865 return Error(S, (Id == ET_INVALID)
8866 ?
"invalid exp target"
8867 :
"exp target is not supported on this GPU");
8870 AMDGPUOperand::CreateImm(
this, Id, S, AMDGPUOperand::ImmTyExpTgt));
8878bool AMDGPUAsmParser::isId(
const AsmToken &Token,
const StringRef Id)
const {
8882bool AMDGPUAsmParser::isId(
const StringRef Id)
const {
8887 return getTokenKind() ==
Kind;
8890StringRef AMDGPUAsmParser::getId()
const {
8894bool AMDGPUAsmParser::trySkipId(
const StringRef Id) {
8902bool AMDGPUAsmParser::trySkipId(
const StringRef Pref,
const StringRef Id) {
8904 StringRef Tok = getTokenStr();
8913bool AMDGPUAsmParser::trySkipId(
const StringRef Id,
8915 if (isId(Id) && peekToken().is(Kind)) {
8924 if (isToken(Kind)) {
8932 const StringRef ErrMsg) {
8933 if (!trySkipToken(Kind)) {
8934 Error(getLoc(), ErrMsg);
8940bool AMDGPUAsmParser::parseExpr(int64_t &
Imm, StringRef Expected) {
8944 if (Parser.parseExpression(Expr))
8947 if (Expr->evaluateAsAbsolute(
Imm))
8950 if (Expected.empty()) {
8951 Error(S,
"expected absolute expression");
8954 Twine(
"expected ", Expected) + Twine(
" or an absolute expression"));
8963 if (Parser.parseExpression(Expr))
8967 if (Expr->evaluateAsAbsolute(IntVal)) {
8968 Operands.push_back(AMDGPUOperand::CreateImm(
this, IntVal, S));
8970 Operands.push_back(AMDGPUOperand::CreateExpr(
this, Expr, S));
8975bool AMDGPUAsmParser::parseString(StringRef &Val,
const StringRef ErrMsg) {
8977 Val =
getToken().getStringContents();
8981 Error(getLoc(), ErrMsg);
8985bool AMDGPUAsmParser::parseId(StringRef &Val,
const StringRef ErrMsg) {
8987 Val = getTokenStr();
8991 if (!ErrMsg.
empty())
8992 Error(getLoc(), ErrMsg);
8996AsmToken AMDGPUAsmParser::getToken()
const {
return Parser.getTok(); }
8998AsmToken AMDGPUAsmParser::peekToken(
bool ShouldSkipSpace) {
9001 : getLexer().peekTok(ShouldSkipSpace);
9005 auto TokCount = getLexer().peekTokens(Tokens);
9007 for (
auto Idx = TokCount; Idx < Tokens.
size(); ++Idx)
9012 return getLexer().getKind();
9015SMLoc AMDGPUAsmParser::getLoc()
const {
return getToken().getLoc(); }
9017StringRef AMDGPUAsmParser::getTokenStr()
const {
9021void AMDGPUAsmParser::lex() { Parser.Lex(); }
9023const AMDGPUOperand &
9025 int MCOpIdx)
const {
9027 const AMDGPUOperand &TargetOp =
static_cast<AMDGPUOperand &
>(*Op);
9028 if (TargetOp.getMCOpIdx() == MCOpIdx)
9035 return ((AMDGPUOperand &)*
Operands[0]).getStartLoc();
9039SMLoc AMDGPUAsmParser::getLaterLoc(SMLoc a, SMLoc b) {
9044 int MCOpIdx)
const {
9045 return findMCOperand(
Operands, MCOpIdx).getStartLoc();
9048SMLoc AMDGPUAsmParser::getOperandLoc(
9049 std::function<
bool(
const AMDGPUOperand &)>
Test,
9051 for (
unsigned i =
Operands.size() - 1; i > 0; --i) {
9052 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
9054 return Op.getStartLoc();
9059SMLoc AMDGPUAsmParser::getImmLoc(AMDGPUOperand::ImmTy
Type,
9061 auto Test = [=](
const AMDGPUOperand &
Op) {
return Op.isImmTy(
Type); };
9076 StringRef
Id = getTokenStr();
9077 SMLoc IdLoc = getLoc();
9083 find_if(Fields, [Id](StructuredOpField *
F) {
return F->Id ==
Id; });
9084 if (
I == Fields.
end())
9085 return Error(IdLoc,
"unknown field");
9086 if ((*I)->IsDefined)
9087 return Error(IdLoc,
"duplicate field");
9090 (*I)->Loc = getLoc();
9093 (*I)->IsDefined =
true;
9100bool AMDGPUAsmParser::validateStructuredOpFields(
9102 return all_of(Fields, [
this](
const StructuredOpField *
F) {
9103 return F->validate(*
this);
9113 const unsigned XorMask) {
9120bool AMDGPUAsmParser::parseSwizzleOperand(int64_t &
Op,
const unsigned MinVal,
9121 const unsigned MaxVal,
9122 const Twine &ErrMsg, SMLoc &Loc) {
9138bool AMDGPUAsmParser::parseSwizzleOperands(
const unsigned OpNum, int64_t *
Op,
9139 const unsigned MinVal,
9140 const unsigned MaxVal,
9141 const StringRef ErrMsg) {
9143 for (
unsigned i = 0; i < OpNum; ++i) {
9144 if (!parseSwizzleOperand(
Op[i], MinVal, MaxVal, ErrMsg, Loc))
9151bool AMDGPUAsmParser::parseSwizzleQuadPerm(int64_t &
Imm) {
9152 using namespace llvm::AMDGPU::Swizzle;
9155 if (parseSwizzleOperands(LANE_NUM, Lane, 0, LANE_MAX,
9156 "expected a 2-bit lane id")) {
9166bool AMDGPUAsmParser::parseSwizzleBroadcast(int64_t &
Imm) {
9167 using namespace llvm::AMDGPU::Swizzle;
9173 if (!parseSwizzleOperand(GroupSize, 2, 32,
9174 "group size must be in the interval [2,32]", Loc)) {
9178 Error(Loc,
"group size must be a power of two");
9181 if (parseSwizzleOperand(LaneIdx, 0, GroupSize - 1,
9182 "lane id must be in the interval [0,group size - 1]",
9190bool AMDGPUAsmParser::parseSwizzleReverse(int64_t &
Imm) {
9191 using namespace llvm::AMDGPU::Swizzle;
9196 if (!parseSwizzleOperand(GroupSize, 2, 32,
9197 "group size must be in the interval [2,32]", Loc)) {
9201 Error(Loc,
"group size must be a power of two");
9209bool AMDGPUAsmParser::parseSwizzleSwap(int64_t &
Imm) {
9210 using namespace llvm::AMDGPU::Swizzle;
9215 if (!parseSwizzleOperand(GroupSize, 1, 16,
9216 "group size must be in the interval [1,16]", Loc)) {
9220 Error(Loc,
"group size must be a power of two");
9228bool AMDGPUAsmParser::parseSwizzleBitmaskPerm(int64_t &
Imm) {
9229 using namespace llvm::AMDGPU::Swizzle;
9236 SMLoc StrLoc = getLoc();
9237 if (!parseString(Ctl)) {
9240 if (Ctl.
size() != BITMASK_WIDTH) {
9241 Error(StrLoc,
"expected a 5-character mask");
9245 unsigned AndMask = 0;
9246 unsigned OrMask = 0;
9247 unsigned XorMask = 0;
9249 for (
size_t i = 0; i < Ctl.
size(); ++i) {
9253 Error(StrLoc,
"invalid mask");
9274bool AMDGPUAsmParser::parseSwizzleFFT(int64_t &
Imm) {
9275 using namespace llvm::AMDGPU::Swizzle;
9278 Error(getLoc(),
"FFT mode swizzle not supported on this GPU");
9284 if (!parseSwizzleOperand(Swizzle, 0, FFT_SWIZZLE_MAX,
9285 "FFT swizzle must be in the interval [0," +
9286 Twine(FFT_SWIZZLE_MAX) + Twine(
']'),
9294bool AMDGPUAsmParser::parseSwizzleRotate(int64_t &
Imm) {
9295 using namespace llvm::AMDGPU::Swizzle;
9298 Error(getLoc(),
"Rotate mode swizzle not supported on this GPU");
9305 if (!parseSwizzleOperand(
Direction, 0, 1,
9306 "direction must be 0 (left) or 1 (right)", Loc))
9310 if (!parseSwizzleOperand(
9311 RotateSize, 0, ROTATE_MAX_SIZE,
9312 "number of threads to rotate must be in the interval [0," +
9313 Twine(ROTATE_MAX_SIZE) + Twine(
']'),
9318 (RotateSize << ROTATE_SIZE_SHIFT);
9322bool AMDGPUAsmParser::parseSwizzleOffset(int64_t &
Imm) {
9324 SMLoc OffsetLoc = getLoc();
9330 Error(OffsetLoc,
"expected a 16-bit offset");
9336bool AMDGPUAsmParser::parseSwizzleMacro(int64_t &
Imm) {
9337 using namespace llvm::AMDGPU::Swizzle;
9341 SMLoc ModeLoc = getLoc();
9344 if (trySkipId(IdSymbolic[ID_QUAD_PERM])) {
9345 Ok = parseSwizzleQuadPerm(
Imm);
9346 }
else if (trySkipId(IdSymbolic[ID_BITMASK_PERM])) {
9347 Ok = parseSwizzleBitmaskPerm(
Imm);
9348 }
else if (trySkipId(IdSymbolic[ID_BROADCAST])) {
9349 Ok = parseSwizzleBroadcast(
Imm);
9350 }
else if (trySkipId(IdSymbolic[ID_SWAP])) {
9351 Ok = parseSwizzleSwap(
Imm);
9352 }
else if (trySkipId(IdSymbolic[ID_REVERSE])) {
9353 Ok = parseSwizzleReverse(
Imm);
9354 }
else if (trySkipId(IdSymbolic[ID_FFT])) {
9355 Ok = parseSwizzleFFT(
Imm);
9356 }
else if (trySkipId(IdSymbolic[ID_ROTATE])) {
9357 Ok = parseSwizzleRotate(
Imm);
9359 Error(ModeLoc,
"expected a swizzle mode");
9362 return Ok && skipToken(
AsmToken::RParen,
"expected a closing parentheses");
9372 if (trySkipId(
"offset")) {
9376 if (trySkipId(
"swizzle")) {
9377 Ok = parseSwizzleMacro(
Imm);
9379 Ok = parseSwizzleOffset(
Imm);
9384 AMDGPUOperand::CreateImm(
this,
Imm, S, AMDGPUOperand::ImmTySwizzle));
9391bool AMDGPUOperand::isSwizzle()
const {
return isImmTy(ImmTySwizzle); }
9397int64_t AMDGPUAsmParser::parseGPRIdxMacro() {
9399 using namespace llvm::AMDGPU::VGPRIndexMode;
9411 for (
unsigned ModeId = ID_MIN; ModeId <=
ID_MAX; ++ModeId) {
9412 if (trySkipId(IdSymbolic[ModeId])) {
9420 ?
"expected a VGPR index mode or a closing parenthesis"
9421 :
"expected a VGPR index mode");
9426 Error(S,
"duplicate VGPR index mode");
9434 "expected a comma or a closing parenthesis"))
9443 using namespace llvm::AMDGPU::VGPRIndexMode;
9449 Imm = parseGPRIdxMacro();
9453 if (getParser().parseAbsoluteExpression(
Imm))
9456 return Error(S,
"invalid immediate: only 4-bit values are legal");
9460 AMDGPUOperand::CreateImm(
this,
Imm, S, AMDGPUOperand::ImmTyGprIdxMode));
9464bool AMDGPUOperand::isGPRIdxMode()
const {
return isImmTy(ImmTyGprIdxMode); }
9475 if (isRegister() || isModifier())
9482 assert(Opr.isImm() || Opr.isExpr());
9483 SMLoc Loc = Opr.getStartLoc();
9487 if (Opr.isExpr() && !Opr.isSymbolRefExpr()) {
9488 Error(Loc,
"expected an absolute expression or a label");
9489 }
else if (Opr.isImm() && !Opr.isS16Imm()) {
9490 Error(Loc,
"expected a 16-bit signed jump offset");
9510 OptionalImmIndexMap OptionalIdx;
9511 unsigned FirstOperandIdx = 1;
9512 bool IsAtomicReturn =
false;
9518 for (
unsigned i = FirstOperandIdx, e =
Operands.size(); i != e; ++i) {
9519 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
9523 Op.addRegOperands(Inst, 1);
9527 if (IsAtomicReturn && i == FirstOperandIdx)
9528 Op.addRegOperands(Inst, 1);
9533 if (
Op.isImm() &&
Op.getImmTy() == AMDGPUOperand::ImmTyNone) {
9534 Op.addImmOperands(Inst, 1);
9546 OptionalIdx[
Op.getImmTy()] = i;
9550 AMDGPUOperand::ImmTyOffset);
9566bool AMDGPUOperand::isSMRDOffset8()
const {
9570bool AMDGPUOperand::isSMEMOffset()
const {
9572 return isImmLiteral();
9575bool AMDGPUOperand::isSMRDLiteralOffset()
const {
9610bool AMDGPUAsmParser::convertDppBoundCtrl(int64_t &BoundCtrl) {
9611 if (BoundCtrl == 0 || BoundCtrl == 1) {
9619void AMDGPUAsmParser::onBeginOfFile() {
9620 if (!getParser().getStreamer().getTargetStreamer())
9623 if (!getTargetStreamer().getTargetID())
9624 getTargetStreamer().initializeTargetID(getSTI(),
9628void AMDGPUAsmParser::emitTargetDirective() {
9629 if (TargetDirectiveEmitted)
9631 TargetDirectiveEmitted =
true;
9633 if (!getParser().getStreamer().getTargetStreamer() ||
9638 getTargetStreamer().EmitDirectiveAMDGCNTarget();
9647bool AMDGPUAsmParser::parsePrimaryExpr(
const MCExpr *&Res, SMLoc &EndLoc) {
9651 StringRef TokenId = getTokenStr();
9652 AGVK VK = StringSwitch<AGVK>(TokenId)
9653 .Case(
"max", AGVK::AGVK_Max)
9654 .Case(
"min", AGVK::AGVK_Min)
9655 .Case(
"or", AGVK::AGVK_Or)
9656 .Case(
"extrasgprs", AGVK::AGVK_ExtraSGPRs)
9657 .Case(
"totalnumvgprs", AGVK::AGVK_TotalNumVGPRs)
9658 .Case(
"alignto", AGVK::AGVK_AlignTo)
9659 .Case(
"occupancy", AGVK::AGVK_Occupancy)
9660 .Case(
"instprefsize", AGVK::AGVK_InstPrefSize)
9661 .Default(AGVK::AGVK_None);
9670 if (Exprs.
empty()) {
9672 "empty " + Twine(TokenId) +
" expression");
9675 if (CommaCount + 1 != Exprs.
size()) {
9677 "mismatch of commas in " + Twine(TokenId) +
" expression");
9681 Expected && Exprs.
size() != Expected) {
9682 Error(
getToken().getLoc(), Twine(TokenId) +
" expression expects " +
9683 Twine(Expected) +
" operands");
9690 if (getParser().parseExpression(Expr, EndLoc))
9694 if (LastTokenWasComma)
9698 "unexpected token in " + Twine(TokenId) +
" expression");
9704 return getParser().parsePrimaryExpr(Res, EndLoc,
nullptr);
9708 StringRef
Name = getTokenStr();
9709 if (Name ==
"mul") {
9710 return parseIntWithPrefix(
"mul",
Operands, AMDGPUOperand::ImmTyOModSI,
9714 if (Name ==
"div") {
9715 return parseIntWithPrefix(
"div",
Operands, AMDGPUOperand::ImmTyOModSI,
9726 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
9731 const AMDGPU::OpName
Ops[] = {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
9732 AMDGPU::OpName::src2};
9740 int DstIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst);
9745 int ModIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0_modifiers);
9747 if (
DstOp.isReg() &&
9752 if ((OpSel & (1 << SrcNum)) != 0)
9758void AMDGPUAsmParser::cvtVOP3OpSel(MCInst &Inst,
9765 OptionalImmIndexMap &OptionalIdx) {
9766 cvtVOP3P(Inst,
Operands, OptionalIdx);
9775 &&
Desc.NumOperands > (OpNum + 1)
9777 &&
Desc.operands()[OpNum + 1].RegClass != -1
9779 &&
Desc.getOperandConstraint(OpNum + 1,
9783void AMDGPUAsmParser::cvtOpSelHelper(MCInst &Inst,
unsigned OpSel) {
9785 constexpr AMDGPU::OpName
Ops[] = {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
9786 AMDGPU::OpName::src2};
9787 constexpr AMDGPU::OpName ModOps[] = {AMDGPU::OpName::src0_modifiers,
9788 AMDGPU::OpName::src1_modifiers,
9789 AMDGPU::OpName::src2_modifiers};
9790 for (
int J = 0; J < 3; ++J) {
9791 int OpIdx = AMDGPU::getNamedOperandIdx(
Opc,
Ops[J]);
9797 int ModIdx = AMDGPU::getNamedOperandIdx(
Opc, ModOps[J]);
9800 if ((OpSel & (1 << J)) != 0)
9803 if (ModOps[J] == AMDGPU::OpName::src0_modifiers && (OpSel & (1 << 3)) != 0)
9810void AMDGPUAsmParser::cvtVOP3Interp(MCInst &Inst,
9812 OptionalImmIndexMap OptionalIdx;
9817 for (
unsigned J = 0; J <
Desc.getNumDefs(); ++J) {
9818 ((AMDGPUOperand &)*
Operands[
I++]).addRegOperands(Inst, 1);
9822 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[
I]);
9824 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9825 }
else if (
Op.isInterpSlot() ||
Op.isInterpAttr() ||
9826 Op.isInterpAttrChan()) {
9828 }
else if (
Op.isImmModifier()) {
9829 OptionalIdx[
Op.getImmTy()] =
I;
9837 AMDGPUOperand::ImmTyHigh);
9841 AMDGPUOperand::ImmTyClamp);
9845 AMDGPUOperand::ImmTyOModSI);
9850 AMDGPUOperand::ImmTyOpSel);
9851 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
9854 cvtOpSelHelper(Inst, OpSel);
9859 OptionalImmIndexMap OptionalIdx;
9864 for (
unsigned J = 0; J <
Desc.getNumDefs(); ++J) {
9865 ((AMDGPUOperand &)*
Operands[
I++]).addRegOperands(Inst, 1);
9869 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[
I]);
9871 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9872 }
else if (
Op.isImmModifier()) {
9873 OptionalIdx[
Op.getImmTy()] =
I;
9881 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
9884 AMDGPUOperand::ImmTyOpSel);
9887 AMDGPUOperand::ImmTyWaitEXP);
9893 cvtOpSelHelper(Inst, OpSel);
9896void AMDGPUAsmParser::cvtScaledMFMA(MCInst &Inst,
9898 OptionalImmIndexMap OptionalIdx;
9901 int CbszOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::cbsz);
9905 for (
unsigned J = 0; J <
Desc.getNumDefs(); ++J)
9906 static_cast<AMDGPUOperand &
>(*
Operands[
I++]).addRegOperands(Inst, 1);
9909 AMDGPUOperand &
Op =
static_cast<AMDGPUOperand &
>(*
Operands[
I]);
9914 if (NumOperands == CbszOpIdx) {
9919 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9920 }
else if (
Op.isImmModifier()) {
9921 OptionalIdx[
Op.getImmTy()] =
I;
9923 Op.addRegOrImmOperands(Inst, 1);
9928 auto CbszIdx = OptionalIdx.find(AMDGPUOperand::ImmTyCBSZ);
9929 if (CbszIdx != OptionalIdx.end()) {
9930 int CbszVal = ((AMDGPUOperand &)*
Operands[CbszIdx->second]).
getImm();
9934 int BlgpOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::blgp);
9935 auto BlgpIdx = OptionalIdx.find(AMDGPUOperand::ImmTyBLGP);
9936 if (BlgpIdx != OptionalIdx.end()) {
9937 int BlgpVal = ((AMDGPUOperand &)*
Operands[BlgpIdx->second]).
getImm();
9948 auto OpselIdx = OptionalIdx.find(AMDGPUOperand::ImmTyOpSel);
9949 if (OpselIdx != OptionalIdx.end()) {
9950 OpSel =
static_cast<const AMDGPUOperand &
>(*
Operands[OpselIdx->second])
9954 unsigned OpSelHi = 0;
9955 auto OpselHiIdx = OptionalIdx.find(AMDGPUOperand::ImmTyOpSelHi);
9956 if (OpselHiIdx != OptionalIdx.end()) {
9957 OpSelHi =
static_cast<const AMDGPUOperand &
>(*
Operands[OpselHiIdx->second])
9960 const AMDGPU::OpName ModOps[] = {AMDGPU::OpName::src0_modifiers,
9961 AMDGPU::OpName::src1_modifiers};
9963 for (
unsigned J = 0; J < 2; ++J) {
9964 unsigned ModVal = 0;
9965 if (OpSel & (1 << J))
9967 if (OpSelHi & (1 << J))
9970 const int ModIdx = AMDGPU::getNamedOperandIdx(
Opc, ModOps[J]);
9976 OptionalImmIndexMap &OptionalIdx) {
9981 for (
unsigned J = 0; J <
Desc.getNumDefs(); ++J) {
9982 ((AMDGPUOperand &)*
Operands[
I++]).addRegOperands(Inst, 1);
9986 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[
I]);
9988 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9989 }
else if (
Op.isImmModifier()) {
9990 OptionalIdx[
Op.getImmTy()] =
I;
9992 Op.addRegOrImmOperands(Inst, 1);
9998 AMDGPUOperand::ImmTyScaleSel);
10002 AMDGPUOperand::ImmTyClamp);
10008 AMDGPUOperand::ImmTyByteSel);
10013 AMDGPUOperand::ImmTyOModSI);
10020 auto *it = Inst.
begin();
10022 it, AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2_modifiers));
10031 OptionalImmIndexMap OptionalIdx;
10032 cvtVOP3(Inst,
Operands, OptionalIdx);
10036 OptionalImmIndexMap &OptIdx) {
10041 if (
Opc == AMDGPU::V_CVT_SCALEF32_PK_FP4_F16_vi ||
10042 Opc == AMDGPU::V_CVT_SCALEF32_PK_FP4_BF16_vi ||
10043 Opc == AMDGPU::V_CVT_SR_BF8_F32_vi ||
10044 Opc == AMDGPU::V_CVT_SR_FP8_F32_vi ||
10045 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_gfx11 ||
10046 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_gfx11 ||
10047 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_gfx12 ||
10048 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_gfx12 ||
10049 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_gfx13 ||
10050 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_gfx13) {
10059 int VdstInIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst_in);
10060 if (VdstInIdx != -1 && VdstInIdx ==
static_cast<int>(Inst.
getNumOperands()))
10063 int BitOp3Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::bitop3);
10064 if (BitOp3Idx != -1) {
10071 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
10072 if (OpSelIdx != -1) {
10076 int OpSelHiIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel_hi);
10077 if (OpSelHiIdx != -1) {
10078 int DefaultVal =
IsPacked ? -1 : 0;
10083 int MatrixAFMTIdx =
10084 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_a_fmt);
10085 if (MatrixAFMTIdx != -1) {
10087 AMDGPUOperand::ImmTyMatrixAFMT, 0);
10090 int MatrixBFMTIdx =
10091 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_b_fmt);
10092 if (MatrixBFMTIdx != -1) {
10094 AMDGPUOperand::ImmTyMatrixBFMT, 0);
10097 int MatrixAScaleIdx =
10098 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_a_scale);
10099 if (MatrixAScaleIdx != -1) {
10101 AMDGPUOperand::ImmTyMatrixAScale, 0);
10104 int MatrixBScaleIdx =
10105 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_b_scale);
10106 if (MatrixBScaleIdx != -1) {
10108 AMDGPUOperand::ImmTyMatrixBScale, 0);
10111 int MatrixAScaleFmtIdx =
10112 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_a_scale_fmt);
10113 if (MatrixAScaleFmtIdx != -1) {
10115 AMDGPUOperand::ImmTyMatrixAScaleFmt, 0);
10118 int MatrixBScaleFmtIdx =
10119 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_b_scale_fmt);
10120 if (MatrixBScaleFmtIdx != -1) {
10122 AMDGPUOperand::ImmTyMatrixBScaleFmt, 0);
10127 AMDGPUOperand::ImmTyMatrixAReuse, 0);
10131 AMDGPUOperand::ImmTyMatrixBReuse, 0);
10133 int NegLoIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::neg_lo);
10134 if (NegLoIdx != -1)
10137 int NegHiIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::neg_hi);
10138 if (NegHiIdx != -1)
10141 const AMDGPU::OpName
Ops[] = {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
10142 AMDGPU::OpName::src2};
10143 const AMDGPU::OpName ModOps[] = {AMDGPU::OpName::src0_modifiers,
10144 AMDGPU::OpName::src1_modifiers,
10145 AMDGPU::OpName::src2_modifiers};
10147 unsigned OpSel = 0;
10148 unsigned OpSelHi = 0;
10149 unsigned NegLo = 0;
10150 unsigned NegHi = 0;
10152 if (OpSelIdx != -1)
10155 if (OpSelHiIdx != -1)
10158 if (NegLoIdx != -1)
10161 if (NegHiIdx != -1)
10164 for (
int J = 0; J < 3; ++J) {
10165 int OpIdx = AMDGPU::getNamedOperandIdx(
Opc,
Ops[J]);
10169 int ModIdx = AMDGPU::getNamedOperandIdx(
Opc, ModOps[J]);
10179 uint32_t ModVal = 0;
10181 const MCOperand &SrcOp = Inst.
getOperand(OpIdx);
10182 if (SrcOp.
isReg() && getMRI()
10186 if (VGPRSuffixIsHi)
10189 if ((OpSel & (1 << J)) != 0)
10193 if ((OpSelHi & (1 << J)) != 0)
10196 if ((NegLo & (1 << J)) != 0)
10199 if ((NegHi & (1 << J)) != 0)
10207 OptionalImmIndexMap OptIdx;
10213 unsigned i,
unsigned Opc,
10214 AMDGPU::OpName
OpName) {
10215 if (AMDGPU::getNamedOperandIdx(
Opc,
OpName) != -1)
10216 ((AMDGPUOperand &)*
Operands[i]).addRegOrImmWithFPInputModsOperands(Inst, 2);
10218 ((AMDGPUOperand &)*
Operands[i]).addRegOperands(Inst, 1);
10224 ((AMDGPUOperand &)*
Operands[1]).addRegOperands(Inst, 1);
10227 ((AMDGPUOperand &)*
Operands[1]).addRegOperands(Inst, 1);
10228 ((AMDGPUOperand &)*
Operands[4]).addRegOperands(Inst, 1);
10230 OptionalImmIndexMap OptIdx;
10231 for (
unsigned i = 5; i <
Operands.size(); ++i) {
10232 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
10233 OptIdx[
Op.getImmTy()] = i;
10238 AMDGPUOperand::ImmTyIndexKey8bit);
10242 AMDGPUOperand::ImmTyIndexKey16bit);
10246 AMDGPUOperand::ImmTyIndexKey32bit);
10263 SMLoc S = getLoc();
10266 Operands.push_back(AMDGPUOperand::CreateToken(
this,
"::", S));
10267 SMLoc OpYLoc = getLoc();
10270 Operands.push_back(AMDGPUOperand::CreateToken(
this, OpYName, OpYLoc));
10273 return Error(OpYLoc,
"expected a VOPDY instruction after ::");
10282 auto addOp = [&](uint16_t ParsedOprIdx) {
10283 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[ParsedOprIdx]);
10285 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
10289 Op.addRegOperands(Inst, 1);
10293 Op.addImmOperands(Inst, 1);
10305 addOp(InstInfo[CompIdx].getIndexOfDstInParsedOperands());
10309 const auto &CInfo = InstInfo[CompIdx];
10310 auto CompSrcOperandsNum = InstInfo[CompIdx].getCompParsedSrcOperandsNum();
10311 for (
unsigned CompSrcIdx = 0; CompSrcIdx < CompSrcOperandsNum; ++CompSrcIdx)
10312 addOp(CInfo.getIndexOfSrcInParsedOperands(CompSrcIdx));
10313 if (CInfo.hasSrc2Acc())
10314 addOp(CInfo.getIndexOfDstInParsedOperands());
10318 AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), AMDGPU::OpName::bitop3);
10319 if (BitOp3Idx != -1) {
10320 OptionalImmIndexMap OptIdx;
10321 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands.back());
10323 OptIdx[
Op.getImmTy()] =
Operands.size() - 1;
10333bool AMDGPUOperand::isDPP8()
const {
return isImmTy(ImmTyDPP8); }
10335bool AMDGPUOperand::isDPPCtrl()
const {
10336 using namespace AMDGPU::DPP;
10338 bool result = isImm() && getImmTy() == ImmTyDppCtrl &&
isUInt<9>(
getImm());
10341 return (
Imm >= DppCtrl::QUAD_PERM_FIRST &&
10342 Imm <= DppCtrl::QUAD_PERM_LAST) ||
10343 (
Imm >= DppCtrl::ROW_SHL_FIRST &&
Imm <= DppCtrl::ROW_SHL_LAST) ||
10344 (
Imm >= DppCtrl::ROW_SHR_FIRST &&
Imm <= DppCtrl::ROW_SHR_LAST) ||
10345 (
Imm >= DppCtrl::ROW_ROR_FIRST &&
Imm <= DppCtrl::ROW_ROR_LAST) ||
10346 (
Imm == DppCtrl::WAVE_SHL1) || (
Imm == DppCtrl::WAVE_ROL1) ||
10347 (
Imm == DppCtrl::WAVE_SHR1) || (
Imm == DppCtrl::WAVE_ROR1) ||
10348 (
Imm == DppCtrl::ROW_MIRROR) || (
Imm == DppCtrl::ROW_HALF_MIRROR) ||
10349 (
Imm == DppCtrl::BCAST15) || (
Imm == DppCtrl::BCAST31) ||
10350 (
Imm >= DppCtrl::ROW_SHARE_FIRST &&
10351 Imm <= DppCtrl::ROW_SHARE_LAST) ||
10352 (
Imm >= DppCtrl::ROW_XMASK_FIRST &&
Imm <= DppCtrl::ROW_XMASK_LAST);
10361bool AMDGPUOperand::isBLGP()
const {
10365bool AMDGPUOperand::isS16Imm()
const {
10369bool AMDGPUOperand::isU16Imm()
const {
10377bool AMDGPUAsmParser::parseDimId(
unsigned &Encoding) {
10382 SMLoc Loc =
getToken().getEndLoc();
10383 Token = std::string(getTokenStr());
10385 if (getLoc() != Loc)
10390 if (!parseId(Suffix))
10394 StringRef DimId = Token;
10409 SMLoc S = getLoc();
10415 SMLoc Loc = getLoc();
10416 if (!parseDimId(Encoding))
10417 return Error(Loc,
"invalid dim value");
10420 AMDGPUOperand::CreateImm(
this, Encoding, S, AMDGPUOperand::ImmTyDim));
10429 SMLoc S = getLoc();
10438 if (!skipToken(
AsmToken::LBrac,
"expected an opening square bracket"))
10441 for (
size_t i = 0; i < 8; ++i) {
10445 SMLoc Loc = getLoc();
10446 if (getParser().parseAbsoluteExpression(Sels[i]))
10448 if (0 > Sels[i] || 7 < Sels[i])
10449 return Error(Loc,
"expected a 3-bit value");
10452 if (!skipToken(
AsmToken::RBrac,
"expected a closing square bracket"))
10456 for (
size_t i = 0; i < 8; ++i)
10457 DPP8 |= (Sels[i] << (i * 3));
10460 AMDGPUOperand::CreateImm(
this, DPP8, S, AMDGPUOperand::ImmTyDPP8));
10464bool AMDGPUAsmParser::isSupportedDPPCtrl(StringRef Ctrl,
10466 if (Ctrl ==
"row_newbcast")
10469 if (Ctrl ==
"row_share" || Ctrl ==
"row_xmask")
10472 if (Ctrl ==
"wave_shl" || Ctrl ==
"wave_shr" || Ctrl ==
"wave_rol" ||
10473 Ctrl ==
"wave_ror" || Ctrl ==
"row_bcast")
10476 return Ctrl ==
"row_mirror" ||
Ctrl ==
"row_half_mirror" ||
10477 Ctrl ==
"quad_perm" ||
Ctrl ==
"row_shl" ||
Ctrl ==
"row_shr" ||
10481int64_t AMDGPUAsmParser::parseDPPCtrlPerm() {
10484 if (!skipToken(
AsmToken::LBrac,
"expected an opening square bracket"))
10488 for (
int i = 0; i < 4; ++i) {
10493 SMLoc Loc = getLoc();
10494 if (getParser().parseAbsoluteExpression(Temp))
10496 if (Temp < 0 || Temp > 3) {
10497 Error(Loc,
"expected a 2-bit value");
10501 Val += (Temp << i * 2);
10504 if (!skipToken(
AsmToken::RBrac,
"expected a closing square bracket"))
10510int64_t AMDGPUAsmParser::parseDPPCtrlSel(StringRef Ctrl) {
10511 using namespace AMDGPU::DPP;
10516 SMLoc Loc = getLoc();
10518 if (getParser().parseAbsoluteExpression(Val))
10521 struct DppCtrlCheck {
10527 DppCtrlCheck
Check =
10528 StringSwitch<DppCtrlCheck>(Ctrl)
10529 .Case(
"wave_shl", {DppCtrl::WAVE_SHL1, 1, 1})
10530 .Case(
"wave_rol", {DppCtrl::WAVE_ROL1, 1, 1})
10531 .Case(
"wave_shr", {DppCtrl::WAVE_SHR1, 1, 1})
10532 .Case(
"wave_ror", {DppCtrl::WAVE_ROR1, 1, 1})
10533 .Case(
"row_shl", {DppCtrl::ROW_SHL0, 1, 15})
10534 .Case(
"row_shr", {DppCtrl::ROW_SHR0, 1, 15})
10535 .Case(
"row_ror", {DppCtrl::ROW_ROR0, 1, 15})
10536 .Case(
"row_share", {DppCtrl::ROW_SHARE_FIRST, 0, 15})
10537 .Case(
"row_xmask", {DppCtrl::ROW_XMASK_FIRST, 0, 15})
10538 .Case(
"row_newbcast", {DppCtrl::ROW_NEWBCAST_FIRST, 0, 15})
10542 if (
Check.Ctrl == -1) {
10543 Valid = (
Ctrl ==
"row_bcast" && (Val == 15 || Val == 31));
10551 Error(Loc, Twine(
"invalid ", Ctrl) + Twine(
" value"));
10559 using namespace AMDGPU::DPP;
10562 !isSupportedDPPCtrl(getTokenStr(),
Operands))
10565 SMLoc S = getLoc();
10571 if (Ctrl ==
"row_mirror") {
10572 Val = DppCtrl::ROW_MIRROR;
10573 }
else if (Ctrl ==
"row_half_mirror") {
10574 Val = DppCtrl::ROW_HALF_MIRROR;
10577 if (Ctrl ==
"quad_perm") {
10578 Val = parseDPPCtrlPerm();
10580 Val = parseDPPCtrlSel(Ctrl);
10589 AMDGPUOperand::CreateImm(
this, Val, S, AMDGPUOperand::ImmTyDppCtrl));
10595 OptionalImmIndexMap OptionalIdx;
10602 int OldIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::old);
10604 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2_modifiers);
10605 bool IsMAC = OldIdx != -1 && Src2ModIdx != -1 &&
10609 for (
unsigned J = 0; J <
Desc.getNumDefs(); ++J) {
10610 ((AMDGPUOperand &)*
Operands[
I++]).addRegOperands(Inst, 1);
10614 int VdstInIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst_in);
10615 bool IsVOP3CvtSrDpp =
Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp8_gfx12 ||
10616 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp8_gfx13 ||
10617 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp8_gfx12 ||
10618 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp8_gfx13 ||
10619 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp_gfx12 ||
10620 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp_gfx13 ||
10621 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp_gfx12 ||
10622 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp_gfx13;
10628 if (OldIdx == NumOperands) {
10630 constexpr int DST_IDX = 0;
10632 }
else if (Src2ModIdx == NumOperands) {
10642 if (IsVOP3CvtSrDpp) {
10651 if (TiedTo != -1) {
10656 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[
I]);
10658 if (IsDPP8 &&
Op.isDppFI()) {
10661 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
10662 }
else if (
Op.isReg()) {
10663 Op.addRegOperands(Inst, 1);
10664 }
else if (
Op.isImm() &&
10666 Op.addImmOperands(Inst, 1);
10667 }
else if (
Op.isImm()) {
10668 OptionalIdx[
Op.getImmTy()] =
I;
10676 AMDGPUOperand::ImmTyClamp);
10682 AMDGPUOperand::ImmTyByteSel);
10687 AMDGPUOperand::ImmTyOModSI);
10690 cvtVOP3P(Inst,
Operands, OptionalIdx);
10692 cvtVOP3OpSel(Inst,
Operands, OptionalIdx);
10695 AMDGPUOperand::ImmTyOpSel);
10700 AMDGPUOperand::ImmTyDPP8);
10701 using namespace llvm::AMDGPU::DPP;
10705 AMDGPUOperand::ImmTyDppCtrl, 0xe4);
10707 AMDGPUOperand::ImmTyDppRowMask, 0xf);
10709 AMDGPUOperand::ImmTyDppBankMask, 0xf);
10711 AMDGPUOperand::ImmTyDppBoundCtrl);
10715 AMDGPUOperand::ImmTyDppFI);
10721 OptionalImmIndexMap OptionalIdx;
10725 for (
unsigned J = 0; J <
Desc.getNumDefs(); ++J) {
10726 ((AMDGPUOperand &)*
Operands[
I++]).addRegOperands(Inst, 1);
10733 if (TiedTo != -1) {
10738 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[
I]);
10740 if (
Op.isReg() && validateVccOperand(
Op.getReg())) {
10748 Op.addImmOperands(Inst, 1);
10750 Op.addRegWithFPInputModsOperands(Inst, 2);
10751 }
else if (
Op.isDppFI()) {
10753 }
else if (
Op.isReg()) {
10754 Op.addRegOperands(Inst, 1);
10760 Op.addRegWithFPInputModsOperands(Inst, 2);
10761 }
else if (
Op.isReg()) {
10762 Op.addRegOperands(Inst, 1);
10763 }
else if (
Op.isDPPCtrl()) {
10764 Op.addImmOperands(Inst, 1);
10765 }
else if (
Op.isImm()) {
10767 OptionalIdx[
Op.getImmTy()] =
I;
10775 using namespace llvm::AMDGPU::DPP;
10779 AMDGPUOperand::ImmTyDppRowMask, 0xf);
10781 AMDGPUOperand::ImmTyDppBankMask, 0xf);
10783 AMDGPUOperand::ImmTyDppBoundCtrl);
10786 AMDGPUOperand::ImmTyDppFI);
10797 AMDGPUOperand::ImmTy
Type) {
10798 return parseStringOrIntWithPrefix(
10800 {
"BYTE_0",
"BYTE_1",
"BYTE_2",
"BYTE_3",
"WORD_0",
"WORD_1",
"DWORD"},
10805 return parseStringOrIntWithPrefix(
10806 Operands,
"dst_unused", {
"UNUSED_PAD",
"UNUSED_SEXT",
"UNUSED_PRESERVE"},
10807 AMDGPUOperand::ImmTySDWADstUnused);
10811 cvtSDWA(Inst,
Operands, SDWAInstType::VOP1);
10815 cvtSDWA(Inst,
Operands, SDWAInstType::VOP2);
10818void AMDGPUAsmParser::cvtSdwaVOP2b(MCInst &Inst,
10820 cvtSDWA(Inst,
Operands, SDWAInstType::VOP2,
true,
true);
10823void AMDGPUAsmParser::cvtSdwaVOP2e(MCInst &Inst,
10825 cvtSDWA(Inst,
Operands, SDWAInstType::VOP2,
false,
true);
10833 SDWAInstType BasicInstType,
bool SkipDstVcc,
10835 using namespace llvm::AMDGPU::SDWA;
10837 OptionalImmIndexMap OptionalIdx;
10838 bool SkipVcc = SkipDstVcc || SkipSrcVcc;
10839 bool SkippedVcc =
false;
10843 for (
unsigned J = 0; J <
Desc.getNumDefs(); ++J) {
10844 ((AMDGPUOperand &)*
Operands[
I++]).addRegOperands(Inst, 1);
10848 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[
I]);
10849 if (SkipVcc && !SkippedVcc &&
Op.isReg() &&
10850 (
Op.getReg() == AMDGPU::VCC ||
Op.getReg() == AMDGPU::VCC_LO)) {
10856 if (BasicInstType == SDWAInstType::VOP2 &&
10862 if (BasicInstType == SDWAInstType::VOPC && Inst.
getNumOperands() == 0) {
10868 Op.addRegOrImmWithInputModsOperands(Inst, 2);
10869 }
else if (
Op.isImm()) {
10871 OptionalIdx[
Op.getImmTy()] =
I;
10875 SkippedVcc =
false;
10879 if (
Opc != AMDGPU::V_NOP_sdwa_gfx10 &&
Opc != AMDGPU::V_NOP_sdwa_gfx9 &&
10880 Opc != AMDGPU::V_NOP_sdwa_vi) {
10882 switch (BasicInstType) {
10883 case SDWAInstType::VOP1:
10886 AMDGPUOperand::ImmTyClamp, 0);
10890 AMDGPUOperand::ImmTyOModSI, 0);
10894 AMDGPUOperand::ImmTySDWADstSel, SdwaSel::DWORD);
10898 AMDGPUOperand::ImmTySDWADstUnused,
10899 DstUnused::UNUSED_PRESERVE);
10902 AMDGPUOperand::ImmTySDWASrc0Sel, SdwaSel::DWORD);
10905 case SDWAInstType::VOP2:
10907 AMDGPUOperand::ImmTyClamp, 0);
10911 AMDGPUOperand::ImmTyOModSI, 0);
10914 AMDGPUOperand::ImmTySDWADstSel, SdwaSel::DWORD);
10916 AMDGPUOperand::ImmTySDWADstUnused,
10917 DstUnused::UNUSED_PRESERVE);
10919 AMDGPUOperand::ImmTySDWASrc0Sel, SdwaSel::DWORD);
10921 AMDGPUOperand::ImmTySDWASrc1Sel, SdwaSel::DWORD);
10924 case SDWAInstType::VOPC:
10927 AMDGPUOperand::ImmTyClamp, 0);
10929 AMDGPUOperand::ImmTySDWASrc0Sel, SdwaSel::DWORD);
10931 AMDGPUOperand::ImmTySDWASrc1Sel, SdwaSel::DWORD);
10938 if (Inst.
getOpcode() == AMDGPU::V_MAC_F32_sdwa_vi ||
10939 Inst.
getOpcode() == AMDGPU::V_MAC_F16_sdwa_vi) {
10940 auto *it = Inst.
begin();
10942 it, AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), AMDGPU::OpName::src2));
10955#define GET_MATCHER_IMPLEMENTATION
10956#define GET_MNEMONIC_SPELL_CHECKER
10957#define GET_MNEMONIC_CHECKER
10958#include "AMDGPUGenAsmMatcher.inc"
10964 return parseTokenOp(
"addr64",
Operands);
10966 return parseNamedBit(
"done",
Operands, AMDGPUOperand::ImmTyDone,
true);
10968 return parseTokenOp(
"idxen",
Operands);
10970 return parseNamedBit(
"lds",
Operands, AMDGPUOperand::ImmTyLDS,
10973 return parseTokenOp(
"offen",
Operands);
10975 return parseTokenOp(
"off",
Operands);
10976 case MCK_row_95_en:
10977 return parseNamedBit(
"row_en",
Operands, AMDGPUOperand::ImmTyRowEn,
true);
10979 return parseNamedBit(
"gds",
Operands, AMDGPUOperand::ImmTyGDS);
10981 return parseNamedBit(
"tfe",
Operands, AMDGPUOperand::ImmTyTFE);
10983 return tryCustomParseOperand(
Operands, MCK);
10988unsigned AMDGPUAsmParser::validateTargetOperandClass(MCParsedAsmOperand &
Op,
10994 AMDGPUOperand &Operand = (AMDGPUOperand &)
Op;
10997 return Operand.isAddr64() ? Match_Success : Match_InvalidOperand;
10999 return Operand.isGDS() ? Match_Success : Match_InvalidOperand;
11001 return Operand.isLDS() ? Match_Success : Match_InvalidOperand;
11003 return Operand.isIdxen() ? Match_Success : Match_InvalidOperand;
11005 return Operand.isOffen() ? Match_Success : Match_InvalidOperand;
11007 return Operand.isTFE() ? Match_Success : Match_InvalidOperand;
11009 return Operand.isDone() ? Match_Success : Match_InvalidOperand;
11010 case MCK_row_95_en:
11011 return Operand.isRowEn() ? Match_Success : Match_InvalidOperand;
11019 return Operand.isSSrc_b32() ? Match_Success : Match_InvalidOperand;
11021 return Operand.isSSrc_f32() ? Match_Success : Match_InvalidOperand;
11022 case MCK_SOPPBrTarget:
11023 return Operand.isSOPPBrTarget() ? Match_Success : Match_InvalidOperand;
11024 case MCK_VReg32OrOff:
11025 return Operand.isVReg32OrOff() ? Match_Success : Match_InvalidOperand;
11026 case MCK_InterpSlot:
11027 return Operand.isInterpSlot() ? Match_Success : Match_InvalidOperand;
11028 case MCK_InterpAttr:
11029 return Operand.isInterpAttr() ? Match_Success : Match_InvalidOperand;
11030 case MCK_InterpAttrChan:
11031 return Operand.isInterpAttrChan() ? Match_Success : Match_InvalidOperand;
11033 case MCK_SReg_64_XEXEC:
11043 return Operand.isNull() ? Match_Success : Match_InvalidOperand;
11045 return Match_InvalidOperand;
11054 SMLoc S = getLoc();
11063 return Error(S,
"expected a 16-bit value");
11066 AMDGPUOperand::CreateImm(
this,
Imm, S, AMDGPUOperand::ImmTyEndpgm));
11070bool AMDGPUOperand::isEndpgm()
const {
return isImmTy(ImmTyEndpgm); }
11076bool AMDGPUOperand::isSplitBarrier()
const {
static const TargetRegisterClass * getRegClass(const MachineInstr &MI, Register Reg)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
SmallVector< int16_t, MAX_SRC_OPERANDS_NUM > OperandIndices
static bool checkWriteLane(const MCInst &Inst)
static bool getRegNum(StringRef Str, unsigned &Num)
static void addSrcModifiersAndSrc(MCInst &Inst, const OperandVector &Operands, unsigned i, unsigned Opc, AMDGPU::OpName OpName)
static constexpr RegInfo RegularRegisters[]
static const RegInfo * getRegularRegInfo(StringRef Str)
static ArrayRef< unsigned > getAllVariants()
static OperandIndices getSrcOperandIndices(unsigned Opcode, bool AddMandatoryLiterals=false)
static int IsAGPROperand(const MCInst &Inst, AMDGPU::OpName Name, const MCRegisterInfo *MRI)
static bool IsMovrelsSDWAOpcode(const unsigned Opcode)
static const fltSemantics * getFltSemantics(unsigned Size)
static bool isRegularReg(RegisterKind Kind)
LLVM_ABI LLVM_EXTERNAL_VISIBILITY void LLVMInitializeAMDGPUAsmParser()
Force static initialization.
static bool ConvertOmodMul(int64_t &Mul)
#define PARSE_BITS_ENTRY(FIELD, ENTRY, VALUE, RANGE)
static bool isInlineableLiteralOp16(int64_t Val, MVT VT, bool HasInv2Pi)
static bool canLosslesslyConvertToFPType(APFloat &FPLiteral, MVT VT)
static bool AMDGPUCheckMnemonic(StringRef Mnemonic, const FeatureBitset &AvailableFeatures, unsigned VariantID)
static void applyMnemonicAliases(StringRef &Mnemonic, const FeatureBitset &Features, unsigned VariantID)
constexpr unsigned MAX_SRC_OPERANDS_NUM
#define EXPR_RESOLVE_OR_ERROR(RESOLVED)
static bool ConvertOmodDiv(int64_t &Div)
static bool IsRevOpcode(const unsigned Opcode)
static bool encodeCnt(const AMDGPU::IsaVersion ISA, int64_t &IntVal, int64_t CntVal, bool Saturate, unsigned(*encode)(const IsaVersion &Version, unsigned, unsigned), unsigned(*decode)(const IsaVersion &Version, unsigned))
static MCRegister getSpecialRegForName(StringRef RegName)
static void addOptionalImmOperand(MCInst &Inst, const OperandVector &Operands, AMDGPUAsmParser::OptionalImmIndexMap &OptionalIdx, AMDGPUOperand::ImmTy ImmT, int64_t Default=0, std::optional< unsigned > InsertAt=std::nullopt)
static void cvtVOP3DstOpSelOnly(MCInst &Inst, const MCRegisterInfo &MRI)
static bool isRegOrImmWithInputMods(const MCInstrDesc &Desc, unsigned OpNum)
static const fltSemantics * getOpFltSemantics(uint8_t OperandType)
static bool isInvalidVOPDY(const OperandVector &Operands, uint64_t InvalidOprIdx)
static std::string AMDGPUMnemonicSpellCheck(StringRef S, const FeatureBitset &FBS, unsigned VariantID=0)
static LLVM_READNONE unsigned encodeBitmaskPerm(const unsigned AndMask, const unsigned OrMask, const unsigned XorMask)
static bool isSafeTruncation(int64_t Val, unsigned Size)
AMDHSA kernel descriptor MCExpr struct for use in MC layer.
Provides AMDGPU specific target descriptions.
Enums shared between the AMDGPU backend (LLVM) and the ELF linker (LLD) for the .amdgpu....
AMDHSA kernel descriptor definitions.
static bool parseExpr(MCAsmParser &MCParser, const MCExpr *&Value, raw_ostream &Err)
MC layer struct for AMDGPUMCKernelCodeT, provides MCExpr functionality where required.
@ AMD_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32
This file declares a class to represent arbitrary precision floating point values and provide a varie...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
#define LLVM_EXTERNAL_VISIBILITY
static llvm::Expected< InlineInfo > decode(GsymDataExtractor &Data, uint64_t &Offset, uint64_t BaseAddr)
Decode an InlineInfo in Data at the specified offset.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
Loop::LoopBounds::Direction Direction
static bool hasFeature(StringRef Feature, const FeatureBitset &FeatureBits, ArrayRef< SubtargetFeatureKV > ProcFeatures)
Register const TargetRegisterInfo * TRI
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
static bool isReg(const MCInst &MI, unsigned OpNo)
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t IntrinsicInst * II
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
Interface definition for SIInstrInfo.
Func getContext().diagnose(DiagnosticInfoUnsupported(Func
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
This file implements the SmallBitVector class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static void initialize(TargetLibraryInfoImpl &TLI, const Triple &T, const llvm::StringTable &StandardNames, VectorLibrary VecLib)
Initialize the set of available library functions based on the specified target triple.
static const char * getRegisterName(MCRegister Reg)
static const AMDGPUMCExpr * createMax(ArrayRef< const MCExpr * > Args, MCContext &Ctx)
static unsigned getNumExpectedArgs(VariantKind Kind)
static const AMDGPUMCExpr * createLit(LitModifier Lit, int64_t Value, MCContext &Ctx)
static const AMDGPUMCExpr * create(VariantKind Kind, ArrayRef< const MCExpr * > Args, MCContext &Ctx)
static const AMDGPUMCExpr * createExtraSGPRs(const MCExpr *VCCUsed, const MCExpr *FlatScrUsed, bool XNACKUsed, MCContext &Ctx)
Allow delayed MCExpr resolve of ExtraSGPRs (in case VCCUsed or FlatScrUsed are unresolvable but neede...
static const AMDGPUMCExpr * createAlignTo(const MCExpr *Value, const MCExpr *Align, MCContext &Ctx)
static std::optional< TargetID > parseTargetIDString(StringRef TargetIDDirective)
Parse and validate a TargetID from a full "<triple>-<processor>:<features>" directive string.
TargetIDSetting getXnackSetting() const
GPUKind getGPUKind() const
StringRef getTargetTripleString() const
std::string toString() const
TargetIDSetting getSramEccSetting() const
static const fltSemantics & IEEEsingle()
static const fltSemantics & BFloat()
static const fltSemantics & IEEEdouble()
static constexpr roundingMode rmNearestTiesToEven
static const fltSemantics & IEEEhalf()
opStatus
IEEE-754R 7: Default exception handling.
LLVM_ABI opStatus convert(const fltSemantics &ToSemantics, roundingMode RM, bool *losesInfo)
Represent a constant reference to an array (0 or more elements consecutively in memory),...
ArrayRef< T > take_front(size_t N=1) const
Return a copy of *this with only the first N elements.
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
const T & front() const
Get the first element.
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
StringRef getString() const
Get the string for the current token, this includes all characters (for example, the quotes on string...
bool is(TokenKind K) const
Container class for subtarget features.
constexpr bool test(unsigned I) const
constexpr FeatureBitset & flip(unsigned I)
void printExpr(raw_ostream &, const MCExpr &) const
virtual void Initialize(MCAsmParser &Parser)
Initialize the extension for parsing using the given Parser.
static const MCBinaryExpr * createAdd(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx, SMLoc Loc=SMLoc())
static const MCBinaryExpr * createDiv(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
static const MCBinaryExpr * createSub(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
static LLVM_ABI const MCConstantExpr * create(int64_t Value, MCContext &Ctx, bool PrintInHex=false, unsigned SizeInBytes=0)
Context object for machine code objects.
LLVM_ABI MCSymbol * getOrCreateSymbol(const Twine &Name)
Lookup the symbol inside with the specified Name.
Instances of this class represent a single low-level machine instruction.
unsigned getNumOperands() const
unsigned getOpcode() const
iterator insert(iterator I, const MCOperand &Op)
void addOperand(const MCOperand Op)
const MCOperand & getOperand(unsigned i) const
Describe properties that are true of each instruction in the target description file.
const MCInstrDesc & get(unsigned Opcode) const
Return the machine instruction descriptor that corresponds to the specified instruction opcode.
const int16_t * getRegClassByHwModeTable(unsigned ModeId) const
int16_t getOpRegClassID(const MCOperandInfo &OpInfo, unsigned HwModeId) const
Return the ID of the register class to use for OpInfo, for the active HwMode HwModeId.
Instances of this class represent operands of the MCInst class.
static MCOperand createExpr(const MCExpr *Val)
static MCOperand createReg(MCRegister Reg)
static MCOperand createImm(int64_t Val)
void setReg(MCRegister Reg)
Set the register number.
MCRegister getReg() const
Returns the register number.
const MCExpr * getExpr() const
MCParsedAsmOperand - This abstract class represents a source-level assembly instruction operand.
MCRegisterClass - Base class of TargetRegisterClass.
MCRegister getRegister(unsigned i) const
getRegister - Return the specified register in the class.
unsigned getNumRegs() const
getNumRegs - Return the number of registers in this class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
MCRegisterInfo base class - We assume that the target defines a static array of MCRegisterDesc object...
bool regsOverlap(MCRegister RegA, MCRegister RegB) const
Returns true if the two registers are equal or alias each other.
const MCRegisterClass & getRegClass(unsigned i) const
Returns the register class associated with the enumeration value.
MCRegister getSubReg(MCRegister Reg, unsigned Idx) const
Returns the physical register number of sub-register "Index" for physical register RegNo.
Wrapper class representing physical registers. Should be passed by value.
constexpr bool isValid() const
Generic base class for all target subtargets.
MCSymbol - Instances of this class represent a symbol name in the MC file, and MCSymbols are created ...
StringRef getName() const
getName - Get the symbol name.
bool isVariable() const
isVariable - Check if this is a variable symbol.
LLVM_ABI void setVariableValue(const MCExpr *Value)
void setRedefinable(bool Value)
Mark this symbol as redefinable.
const MCExpr * getVariableValue() const
Get the expression of the variable symbol.
MCTargetAsmParser - Generic interface to target specific assembly parsers.
uint64_t getScalarSizeInBits() const
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
Ternary parse status returned by various parse* methods.
constexpr bool isFailure() const
static constexpr StatusTy Failure
constexpr bool isSuccess() const
static constexpr StatusTy Success
static constexpr StatusTy NoMatch
constexpr bool isNoMatch() const
constexpr unsigned id() const
Represents a location in source code.
static SMLoc getFromPointer(const char *Ptr)
constexpr const char * getPointer() const
constexpr bool isValid() const
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
A wrapper around a string literal that serves as a proxy for constructing global tables of StringRefs...
Represent a constant reference to a string, i.e.
bool consume_back(StringRef Suffix)
Returns true if this StringRef has the given suffix and removes that suffix.
constexpr StringRef substr(size_t Start, size_t N=npos) const
Return a reference to the substring from [Start, Start + N).
bool starts_with(StringRef Prefix) const
Check if this string starts with the given Prefix.
constexpr bool empty() const
Check if the string is empty.
StringRef drop_front(size_t N=1) const
Return a StringRef equal to 'this' but with the first N elements dropped.
constexpr size_t size() const
Get the string size.
constexpr const char * data() const
Get a pointer to the start of the string (which may not be null terminated).
bool ends_with(StringRef Suffix) const
Check if this string ends with the given Suffix.
bool consume_front(char Prefix)
Returns true if this StringRef has the given prefix and removes that prefix.
bool contains(StringRef key) const
Check if the set contains the given key.
std::pair< typename Base::iterator, bool > insert(StringRef key)
A switch()-like statement whose cases are string literals.
StringSwitch & Case(StringLiteral S, T Value)
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
LLVM_ABI std::string str() const
Return the twine contents as a std::string.
std::pair< iterator, bool > insert(const ValueT &V)
This class implements an extremely fast bulk output stream that can only output to a stream.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
int encodeDepCtr(const StringRef Name, int64_t Val, unsigned &UsedOprMask, const MCSubtargetInfo &STI)
int getDefaultDepCtrEncoding(const MCSubtargetInfo &STI)
bool isSupportedTgtId(unsigned Id, const MCSubtargetInfo &STI)
unsigned getTgtId(const StringRef Name)
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char NumSGPRs[]
Key for Kernel::CodeProps::Metadata::mNumSGPRs.
constexpr char SymbolName[]
Key for Kernel::Metadata::mSymbolName.
constexpr char AssemblerDirectiveBegin[]
HSA metadata beginning assembler directive.
constexpr char AssemblerDirectiveEnd[]
HSA metadata ending assembler directive.
constexpr char AssemblerDirectiveBegin[]
Old HSA metadata beginning assembler directive for V2.
int64_t getHwregId(StringRef Name, const MCSubtargetInfo &STI)
@ FIXED_NUM_SGPRS_FOR_INIT_BUG
unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI, std::optional< bool > EnableWavefrontSize32)
unsigned getSGPREncodingGranule(const MCSubtargetInfo &STI)
bool targetIDSettingsConflict(TargetIDSetting Lhs, TargetIDSetting Rhs)
Returns true if Lhs and Rhs are incompatible (both specific but different).
unsigned getLocalMemorySize(const MCSubtargetInfo &STI)
constexpr char AssemblerDirective[]
PAL metadata (old linear format) assembler directive.
constexpr char AssemblerDirectiveBegin[]
PAL metadata (new MsgPack format) beginning assembler directive.
constexpr char AssemblerDirectiveEnd[]
PAL metadata (new MsgPack format) ending assembler directive.
int64_t getMsgOpId(int64_t MsgId, StringRef Name, const MCSubtargetInfo &STI)
Map from a symbolic name for a sendmsg operation to the operation portion of the immediate encoding.
int64_t getMsgId(StringRef Name, const MCSubtargetInfo &STI)
Map from a symbolic name for a msg_id to the message portion of the immediate encoding.
uint64_t encodeMsg(uint64_t MsgId, uint64_t OpId, uint64_t StreamId)
bool msgSupportsStream(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI)
bool isValidMsgId(int64_t MsgId, const MCSubtargetInfo &STI)
bool isValidMsgStream(int64_t MsgId, int64_t OpId, int64_t StreamId, const MCSubtargetInfo &STI, bool Strict)
bool msgRequiresOp(int64_t MsgId, const MCSubtargetInfo &STI)
bool isValidMsgOp(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI, bool Strict)
ArrayRef< GFXVersion > getGFXVersions()
constexpr unsigned COMPONENTS[]
constexpr const char *const ModMatrixFmt[]
constexpr const char *const ModMatrixScaleFmt[]
constexpr const char *const ModMatrixScale[]
bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi)
bool isGFX10_BEncoding(const MCSubtargetInfo &STI)
bool isInlineValue(MCRegister Reg)
bool isPKFMACF16InlineConstant(uint32_t Literal, bool IsGFX11Plus)
LLVM_READONLY const MIMGInfo * getMIMGInfo(unsigned Opc)
bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi)
bool isSGPR(MCRegister Reg, const MCRegisterInfo *TRI)
Is Reg - scalar register.
MCRegister getMCReg(MCRegister Reg, const MCSubtargetInfo &STI)
If Reg is a pseudo reg, return the correct hardware register given STI otherwise return Reg.
FuncInfoFlags
Per-function flags packed into INFO_FLAGS entries.
uint8_t wmmaScaleF8F6F4FormatToNumRegs(unsigned Fmt)
const int OPR_ID_UNSUPPORTED
bool isInlinableLiteralV2I16(uint32_t Literal)
bool isHi16Reg(MCRegister Reg, const MCRegisterInfo &MRI)
unsigned getTemporalHintType(const MCInstrDesc TID)
bool isGFX10(const MCSubtargetInfo &STI)
LLVM_READONLY bool isLitExpr(const MCExpr *Expr)
bool isInlinableLiteralV2BF16(uint32_t Literal)
LLVM_ABI bool isCPUValidForSubArch(Triple::SubArchType SubArch, GPUKind AK)
Return true if the GPU AK is usable with the triple subarch SubArch.
unsigned getMaxNumUserSGPRs(const MCSubtargetInfo &STI)
unsigned getNumFlatOffsetBits(const MCSubtargetInfo &ST)
For pre-GFX12 FLAT instructions the offset must be positive; MSB is ignored and forced to zero.
bool hasA16(const MCSubtargetInfo &STI)
bool isLegalSMRDEncodedSignedOffset(const MCSubtargetInfo &ST, int64_t EncodedOffset, bool IsBuffer)
bool isGFX12Plus(const MCSubtargetInfo &STI)
unsigned getNSAMaxSize(const MCSubtargetInfo &STI, bool HasSampler)
bool hasPackedD16(const MCSubtargetInfo &STI)
bool isGFX940(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2F16(uint32_t Literal)
bool isHsaAbi(const MCSubtargetInfo &STI)
bool isGFX11(const MCSubtargetInfo &STI)
const int OPR_VAL_INVALID
bool getSMEMIsBuffer(unsigned Opc)
bool isPackedSingleSGPRFP32Inst(unsigned Opc)
The opcode is a packed fp32 instruction which only reads low 32 bits of a scalar operand and propagat...
bool isGFX13(const MCSubtargetInfo &STI)
LLVM_ABI unsigned getAddressableNumSGPRs(GPUKind AK)
uint8_t mfmaScaleF8F6F4FormatToNumRegs(unsigned EncodingVal)
LLVM_ABI IsaVersion getIsaVersion(StringRef GPU)
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
bool isValid32BitLiteral(uint64_t Val, bool IsFP64)
LLVM_ABI unsigned getTotalNumVGPRs(GPUKind AK, bool IsWave32)
CanBeVOPD getCanBeVOPD(unsigned Opc, unsigned EncodingFamily, bool VOPD3)
LLVM_READNONE bool isLegalDPALU_DPPControl(const MCSubtargetInfo &ST, unsigned DC)
bool isSI(const MCSubtargetInfo &STI)
bool hasPrivateApertureRegs(const MCSubtargetInfo &STI)
unsigned decodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt)
unsigned getWaitcntBitMask(const IsaVersion &Version)
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
bool isGFX9(const MCSubtargetInfo &STI)
unsigned getVOPDEncodingFamily(const MCSubtargetInfo &ST)
bool isKImmOperand(const MCInstrDesc &Desc, unsigned OpNo)
Is this a KImm operand?
GPUKind
GPU kinds supported by the AMDGPU target.
bool isGFX90A(const MCSubtargetInfo &STI)
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfoByEncoding(uint8_t DimEnc)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
bool isGFX12(const MCSubtargetInfo &STI)
unsigned encodeExpcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Expcnt)
bool hasMAIInsts(const MCSubtargetInfo &STI)
constexpr bool isSISrcOperand(const MCOperandInfo &OpInfo)
Is this an AMDGPU specific source operand?
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfoByAsmSuffix(StringRef AsmSuffix)
bool hasMIMG_R128(const MCSubtargetInfo &STI)
LLVM_ABI GPUKind parseArchAMDGCN(StringRef CPU)
bool hasG16(const MCSubtargetInfo &STI)
unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode, const MIMGDimInfo *Dim, bool IsA16, bool IsG16Supported)
bool isGFX13Plus(const MCSubtargetInfo &STI)
bool hasArchitectedFlatScratch(const MCSubtargetInfo &STI)
LLVM_READONLY int64_t getLitValue(const MCExpr *Expr)
bool isGFX11Plus(const MCSubtargetInfo &STI)
bool isSISrcFPOperand(const MCInstrDesc &Desc, unsigned OpNo)
Is this floating-point operand?
bool isGFX10Plus(const MCSubtargetInfo &STI)
AMDGPU::TargetID TargetID
int64_t encode32BitLiteral(int64_t Imm, OperandType Type, bool IsLit)
bool isValidWMMAScaleFmtCombination(unsigned AFmt, unsigned AScale, unsigned BFmt, unsigned BScale)
@ OPERAND_KIMM32
Operand with 32-bit immediate that uses the constant bus.
@ OPERAND_REG_INLINE_C_FP64
@ OPERAND_REG_IMM_NOINLINE_FP16
@ OPERAND_REG_INLINE_C_BF16
@ OPERAND_REG_INLINE_C_V2BF16
@ OPERAND_REG_IMM_V2INT64
@ OPERAND_REG_IMM_V2INT16
@ OPERAND_REG_IMM_INT32
Operands with register, 32-bit, or 64-bit immediate.
@ OPERAND_REG_IMM_V2FP16_SPLAT
@ OPERAND_REG_INLINE_C_INT64
@ OPERAND_REG_INLINE_C_INT16
Operands with register or inline constant.
@ OPERAND_REG_IMM_NOINLINE_V2FP16
@ OPERAND_REG_INLINE_C_V2FP16
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
@ OPERAND_REG_INLINE_AC_FP32
@ OPERAND_REG_IMM_V2INT32
@ OPERAND_REG_INLINE_C_FP32
@ OPERAND_REG_INLINE_C_INT32
@ OPERAND_REG_INLINE_C_V2INT16
@ OPERAND_REG_INLINE_AC_FP64
@ OPERAND_REG_INLINE_C_FP16
@ OPERAND_INLINE_SPLIT_BARRIER_INT32
constexpr bool isBF16SrcOperand(const MCOperandInfo &OpInfo)
Is this a scalar (i.e. not packed) bf16 source operand?
bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII, const MCSubtargetInfo &ST)
LLVM_ABI StringRef getArchNameAMDGCN(GPUKind AK)
bool hasGDS(const MCSubtargetInfo &STI)
bool isLegalSMRDEncodedUnsignedOffset(const MCSubtargetInfo &ST, int64_t EncodedOffset)
bool isGFX9Plus(const MCSubtargetInfo &STI)
bool hasDPPSrc1SGPR(const MCSubtargetInfo &STI)
const int OPR_ID_DUPLICATE
bool isVOPD(unsigned Opc)
VOPD::InstInfo getVOPDInstInfo(const MCInstrDesc &OpX, const MCInstrDesc &OpY)
unsigned encodeVmcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Vmcnt)
unsigned decodeExpcnt(const IsaVersion &Version, unsigned Waitcnt)
bool isGFX1250(const MCSubtargetInfo &STI)
const MIMGBaseOpcodeInfo * getMIMGBaseOpcode(unsigned Opc)
bool isVI(const MCSubtargetInfo &STI)
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
MCRegister mc2PseudoReg(MCRegister Reg)
Convert hardware register Reg to a pseudo register.
unsigned hasKernargPreload(const MCSubtargetInfo &STI)
bool supportsWGP(const MCSubtargetInfo &STI)
LLVM_READNONE unsigned getOperandSize(const MCOperandInfo &OpInfo)
bool isCI(const MCSubtargetInfo &STI)
unsigned encodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Lgkmcnt)
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
bool isGFX1250Plus(const MCSubtargetInfo &STI)
bool hasPopsExitingWaveID(const MCSubtargetInfo &STI)
unsigned decodeVmcnt(const IsaVersion &Version, unsigned Waitcnt)
bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi)
bool hasVOPD(const MCSubtargetInfo &STI)
bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi)
Is this literal inlinable.
bool isPermlane16(unsigned Opc)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ UNDEF
UNDEF - An undefined node.
Predicate getPredicate(unsigned Condition, unsigned Hint)
Return predicate consisting of specified condition and hint bits.
void validate(const Triple &TT, const FeatureBitset &FeatureBits)
constexpr bool isAtomicRet(const T &...O)
constexpr bool isVOPC(const T &...O)
constexpr bool isVOP3(const T &...O)
constexpr bool isVOP1(const T &...O)
constexpr bool usesTENSOR_CNT(const T &...O)
constexpr bool isMAI(const T &...O)
constexpr bool isVOP2(const T &...O)
constexpr bool isSWMMAC(const T &...O)
constexpr bool isSOP2(const T &...O)
constexpr bool isFLAT(const T &...O)
constexpr bool isVOP3P(const T &...O)
constexpr bool isBuffer(const T &...O)
constexpr bool hasIntClamp(const T &...O)
constexpr bool isAtomicNoRet(const T &...O)
constexpr bool isSMRD(const T &...O)
constexpr bool isVOP3Like(const T &...O)
constexpr bool isMIMG(const T &...O)
constexpr bool isVMEM(const T &...O)
constexpr bool isImage(const T &...O)
constexpr bool isWMMA(const T &...O)
constexpr bool isVOPD3(const T &...O)
constexpr bool isGWS(const T &...O)
constexpr bool isMUBUF(const T &...O)
constexpr bool isSDWA(const T &...O)
constexpr bool isSOPC(const T &...O)
constexpr bool isDOT(const T &...O)
constexpr bool isVSAMPLE(const T &...O)
constexpr bool isDS(const T &...O)
constexpr bool isAtomic(const T &...O)
constexpr bool isGather4(const T &...O)
constexpr bool isPacked(const T &...O)
constexpr bool isDPP(const T &...O)
constexpr bool isSegmentSpecificFLAT(const T &...O)
@ Valid
The data is already valid.
Scope
Defines the scope in which this symbol should be visible: Default – Visible in the public interface o...
EnumSet< Modifier > Modifiers
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
This is an optimization pass for GlobalISel generic memory operations.
bool errorToBool(Error Err)
Helper for converting an Error to a bool.
StringMapEntry< Value * > ValueName
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Printable print(const GCNRegPressure &RP, const GCNSubtarget *ST=nullptr, unsigned DynamicVGPRBlockSize=0)
unsigned encode(MaybeAlign A)
Returns a representation of the alignment that encodes undefined as 0.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
static bool isMem(const MachineInstr &MI, unsigned Op)
LLVM_ABI std::pair< StringRef, StringRef > getToken(StringRef Source, StringRef Delimiters=" \t\n\v\f\r")
getToken - This function extracts one token from source, ignoring any leading characters that appear ...
static StringRef getCPU(StringRef CPU)
Processes a CPU name.
testing::Matcher< const detail::ErrorHolder & > Failed()
LLVM_ABI void PrintError(const Twine &Msg)
constexpr bool isUIntN(unsigned N, uint64_t x)
Checks if an unsigned integer fits into the given (dynamic) bit width.
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
T bit_ceil(T Value)
Returns the smallest integral power of two no smaller than Value if Value is nonzero.
Target & getTheR600Target()
The target for R600 GPUs.
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
SmallVectorImpl< std::unique_ptr< MCParsedAsmOperand > > OperandVector
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
MutableArrayRef(T &OneElt) -> MutableArrayRef< T >
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
Target & getTheGCNTarget()
The target for GCN GPUs.
@ Sub
Subtraction of integers.
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
DWARFExpression::Operation Op
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
unsigned M0(unsigned Val)
ArrayRef(const T &OneElt) -> ArrayRef< T >
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
constexpr bool isIntN(unsigned N, int64_t x)
Checks if an signed integer fits into the given (dynamic) bit width.
Target & getTheGCNLegacyTarget()
The target for GCN GPUs, registered under the legacy "amdgcn" architecture name for use with -march.
@ Enabled
Convert any .debug_str_offsets tables to DWARF64 if needed.
@ Default
The result value is uniform if and only if all operands are uniform.
void initDefault(const MCSubtargetInfo &STI, MCContext &Ctx, bool InitMCExpr=true)
void validate(const MCSubtargetInfo *STI, MCContext &Ctx)
uint32_t PrivateSegmentSize
SmallVector< std::pair< MCSymbol *, std::string >, 4 > IndirectCalls
SmallVector< std::pair< MCSymbol *, MCSymbol * >, 8 > Calls
SmallVector< FuncInfo, 8 > Funcs
SmallVector< std::pair< MCSymbol *, std::string >, 4 > TypeIds
SmallVector< std::pair< MCSymbol *, MCSymbol * >, 4 > Uses
Instruction set architecture version.
const MCExpr * compute_pgm_rsrc2
const MCExpr * kernarg_size
const MCExpr * kernarg_preload
const MCExpr * compute_pgm_rsrc3
const MCExpr * private_segment_fixed_size
const MCExpr * compute_pgm_rsrc1
static void bits_set(const MCExpr *&Dst, const MCExpr *Value, uint32_t Shift, uint32_t Mask, MCContext &Ctx)
const MCExpr * group_segment_fixed_size
static MCKernelDescriptor getDefaultAmdhsaKernelDescriptor(const MCSubtargetInfo *STI, MCContext &Ctx)
const MCExpr * kernel_code_properties
RegisterMCAsmParser - Helper template for registering a target specific assembly parser,...
uint32_t group_segment_fixed_size
uint32_t private_segment_fixed_size