LLVM 24.0.0git
AMDGPUBaseInfo.cpp
Go to the documentation of this file.
1//===- AMDGPUBaseInfo.cpp - AMDGPU Base encoding information --------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
9#include "AMDGPUBaseInfo.h"
10#include "AMDGPU.h"
11#include "AMDGPUAsmUtils.h"
12#include "AMDKernelCodeT.h"
17#include "llvm/IR/Attributes.h"
18#include "llvm/IR/Constants.h"
19#include "llvm/IR/Function.h"
20#include "llvm/IR/GlobalValue.h"
21#include "llvm/IR/IntrinsicsAMDGPU.h"
22#include "llvm/IR/IntrinsicsR600.h"
23#include "llvm/IR/LLVMContext.h"
24#include "llvm/IR/Metadata.h"
25#include "llvm/MC/MCInstrInfo.h"
30#include <optional>
31
32#define GET_INSTRINFO_NAMED_OPS
33#define GET_INSTRMAP_INFO
34#include "AMDGPUGenInstrInfo.inc"
35
37 "amdhsa-code-object-version", llvm::cl::Hidden,
39 llvm::cl::desc("Set default AMDHSA Code Object Version (module flag "
40 "or asm directive still take priority if present)"));
41
42namespace {
43
44/// \returns Bit mask for given bit \p Shift and bit \p Width.
45unsigned getBitMask(unsigned Shift, unsigned Width) {
46 return ((1 << Width) - 1) << Shift;
47}
48
49/// Packs \p Src into \p Dst for given bit \p Shift and bit \p Width.
50///
51/// \returns Packed \p Dst.
52unsigned packBits(unsigned Src, unsigned Dst, unsigned Shift, unsigned Width) {
53 unsigned Mask = getBitMask(Shift, Width);
54 return ((Src << Shift) & Mask) | (Dst & ~Mask);
55}
56
57/// Unpacks bits from \p Src for given bit \p Shift and bit \p Width.
58///
59/// \returns Unpacked bits.
60unsigned unpackBits(unsigned Src, unsigned Shift, unsigned Width) {
61 return (Src & getBitMask(Shift, Width)) >> Shift;
62}
63
64/// \returns Vmcnt bit shift (lower bits).
65unsigned getVmcntBitShiftLo(unsigned VersionMajor) {
66 return VersionMajor >= 11 ? 10 : 0;
67}
68
69/// \returns Vmcnt bit width (lower bits).
70unsigned getVmcntBitWidthLo(unsigned VersionMajor) {
71 return VersionMajor >= 11 ? 6 : 4;
72}
73
74/// \returns Expcnt bit shift.
75unsigned getExpcntBitShift(unsigned VersionMajor) {
76 return VersionMajor >= 11 ? 0 : 4;
77}
78
79/// \returns Expcnt bit width.
80unsigned getExpcntBitWidth(unsigned VersionMajor) { return 3; }
81
82/// \returns Lgkmcnt bit shift.
83unsigned getLgkmcntBitShift(unsigned VersionMajor) {
84 return VersionMajor >= 11 ? 4 : 8;
85}
86
87/// \returns Lgkmcnt bit width.
88unsigned getLgkmcntBitWidth(unsigned VersionMajor) {
89 return VersionMajor >= 10 ? 6 : 4;
90}
91
92/// \returns Vmcnt bit shift (higher bits).
93unsigned getVmcntBitShiftHi(unsigned VersionMajor) { return 14; }
94
95/// \returns Vmcnt bit width (higher bits).
96unsigned getVmcntBitWidthHi(unsigned VersionMajor) {
97 return (VersionMajor == 9 || VersionMajor == 10) ? 2 : 0;
98}
99
100/// \returns Loadcnt bit width
101unsigned getLoadcntBitWidth(unsigned VersionMajor) {
102 return VersionMajor >= 12 ? 6 : 0;
103}
104
105/// \returns Samplecnt bit width.
106unsigned getSamplecntBitWidth(unsigned VersionMajor) {
107 return VersionMajor >= 12 ? 6 : 0;
108}
109
110/// \returns Bvhcnt bit width.
111unsigned getBvhcntBitWidth(unsigned VersionMajor) {
112 return VersionMajor >= 12 ? 3 : 0;
113}
114
115/// \returns Dscnt bit width.
116unsigned getDscntBitWidth(unsigned VersionMajor) {
117 return VersionMajor >= 12 ? 6 : 0;
118}
119
120/// \returns Dscnt bit shift in combined S_WAIT instructions.
121unsigned getDscntBitShift(unsigned VersionMajor) { return 0; }
122
123/// \returns Storecnt or Vscnt bit width, depending on VersionMajor.
124unsigned getStorecntBitWidth(unsigned VersionMajor) {
125 return VersionMajor >= 10 ? 6 : 0;
126}
127
128/// \returns Kmcnt bit width.
129unsigned getKmcntBitWidth(unsigned VersionMajor) {
130 return VersionMajor >= 12 ? 5 : 0;
131}
132
133/// \returns Xcnt bit width.
134unsigned getXcntBitWidth(unsigned VersionMajor, unsigned VersionMinor) {
135 return VersionMajor == 12 && VersionMinor == 5 ? 6 : 0;
136}
137
138/// \returns Asynccnt bit width.
139unsigned getAsynccntBitWidth(unsigned VersionMajor, unsigned VersionMinor) {
140 return VersionMajor == 12 && VersionMinor == 5 ? 6 : 0;
141}
142
143/// \returns shift for Loadcnt/Storecnt in combined S_WAIT instructions.
144unsigned getLoadcntStorecntBitShift(unsigned VersionMajor) {
145 return VersionMajor >= 12 ? 8 : 0;
146}
147
148/// \returns VaSdst bit width
149inline unsigned getVaSdstBitWidth() { return 3; }
150
151/// \returns VaSdst bit shift
152inline unsigned getVaSdstBitShift() { return 9; }
153
154/// \returns VmVsrc bit width
155inline unsigned getVmVsrcBitWidth() { return 3; }
156
157/// \returns VmVsrc bit shift
158inline unsigned getVmVsrcBitShift() { return 2; }
159
160/// \returns VaVdst bit width
161inline unsigned getVaVdstBitWidth() { return 4; }
162
163/// \returns VaVdst bit shift
164inline unsigned getVaVdstBitShift() { return 12; }
165
166/// \returns VaVcc bit width
167inline unsigned getVaVccBitWidth() { return 1; }
168
169/// \returns VaVcc bit shift
170inline unsigned getVaVccBitShift() { return 1; }
171
172/// \returns SaSdst bit width
173inline unsigned getSaSdstBitWidth() { return 1; }
174
175/// \returns SaSdst bit shift
176inline unsigned getSaSdstBitShift() { return 0; }
177
178/// \returns VaSsrc width
179inline unsigned getVaSsrcBitWidth() { return 1; }
180
181/// \returns VaSsrc bit shift
182inline unsigned getVaSsrcBitShift() { return 8; }
183
184/// \returns HoldCnt bit shift
185inline unsigned getHoldCntWidth(unsigned VersionMajor, unsigned VersionMinor) {
186 static constexpr const unsigned MinMajor = 10;
187 static constexpr const unsigned MinMinor = 3;
188 return std::tie(VersionMajor, VersionMinor) >= std::tie(MinMajor, MinMinor)
189 ? 1
190 : 0;
191}
192
193/// \returns HoldCnt bit shift
194inline unsigned getHoldCntBitShift() { return 7; }
195
196} // end anonymous namespace
197
198namespace llvm {
199
200namespace AMDGPU {
201
202/// \returns true if the target supports signed immediate offset for SMRD
203/// instructions.
205 return isGFX9Plus(ST);
206}
207
208/// \returns True if \p STI is AMDHSA.
209bool isHsaAbi(const MCSubtargetInfo &STI) {
210 return STI.getTargetTriple().getOS() == Triple::AMDHSA;
211}
212
215 M.getModuleFlag("amdhsa_code_object_version"))) {
216 return (unsigned)Ver->getZExtValue() / 100;
217 }
218
220}
221
225
226unsigned getAMDHSACodeObjectVersion(unsigned ABIVersion) {
227 switch (ABIVersion) {
229 return 4;
231 return 5;
233 return 6;
234 default:
236 }
237}
238
239uint8_t getELFABIVersion(const Triple &T, unsigned CodeObjectVersion) {
240 if (T.getOS() != Triple::AMDHSA)
241 return 0;
242
243 switch (CodeObjectVersion) {
244 case 4:
246 case 5:
248 case 6:
250 default:
251 report_fatal_error("Unsupported AMDHSA Code Object Version " +
252 Twine(CodeObjectVersion));
253 }
254}
255
256unsigned getMultigridSyncArgImplicitArgPosition(unsigned CodeObjectVersion) {
257 switch (CodeObjectVersion) {
258 case AMDHSA_COV4:
259 return 48;
260 case AMDHSA_COV5:
261 case AMDHSA_COV6:
262 default:
264 }
265}
266
267// FIXME: All such magic numbers about the ABI should be in a
268// central TD file.
269unsigned getHostcallImplicitArgPosition(unsigned CodeObjectVersion) {
270 switch (CodeObjectVersion) {
271 case AMDHSA_COV4:
272 return 24;
273 case AMDHSA_COV5:
274 case AMDHSA_COV6:
275 default:
277 }
278}
279
280unsigned getDefaultQueueImplicitArgPosition(unsigned CodeObjectVersion) {
281 switch (CodeObjectVersion) {
282 case AMDHSA_COV4:
283 return 32;
284 case AMDHSA_COV5:
285 case AMDHSA_COV6:
286 default:
288 }
289}
290
291unsigned getCompletionActionImplicitArgPosition(unsigned CodeObjectVersion) {
292 switch (CodeObjectVersion) {
293 case AMDHSA_COV4:
294 return 40;
295 case AMDHSA_COV5:
296 case AMDHSA_COV6:
297 default:
299 }
300}
301
302#define GET_MIMGBaseOpcodesTable_IMPL
303#define GET_MIMGDimInfoTable_IMPL
304#define GET_MIMGInfoTable_IMPL
305#define GET_MIMGLZMappingTable_IMPL
306#define GET_MIMGMIPMappingTable_IMPL
307#define GET_MIMGBiasMappingTable_IMPL
308#define GET_MIMGOffsetMappingTable_IMPL
309#define GET_MIMGG16MappingTable_IMPL
310#define GET_MAIInstInfoTable_IMPL
311#define GET_WMMAInstInfoTable_IMPL
312#include "AMDGPUGenSearchableTables.inc"
313
314int getMIMGOpcode(unsigned BaseOpcode, unsigned MIMGEncoding,
315 unsigned VDataDwords, unsigned VAddrDwords) {
316 const MIMGInfo *Info =
317 getMIMGOpcodeHelper(BaseOpcode, MIMGEncoding, VDataDwords, VAddrDwords);
318 return Info ? Info->Opcode : -1;
319}
320
322 const MIMGInfo *Info = getMIMGInfo(Opc);
323 return Info ? getMIMGBaseOpcodeInfo(Info->BaseOpcode) : nullptr;
324}
325
326int getMaskedMIMGOp(unsigned Opc, unsigned NewChannels) {
327 const MIMGInfo *OrigInfo = getMIMGInfo(Opc);
328 const MIMGInfo *NewInfo =
329 getMIMGOpcodeHelper(OrigInfo->BaseOpcode, OrigInfo->MIMGEncoding,
330 NewChannels, OrigInfo->VAddrDwords);
331 return NewInfo ? NewInfo->Opcode : -1;
332}
333
334unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode,
335 const MIMGDimInfo *Dim, bool IsA16,
336 bool IsG16Supported) {
337 unsigned AddrWords = BaseOpcode->NumExtraArgs;
338 unsigned AddrComponents = (BaseOpcode->Coordinates ? Dim->NumCoords : 0) +
339 (BaseOpcode->LodOrClampOrMip ? 1 : 0);
340 if (IsA16)
341 AddrWords += divideCeil(AddrComponents, 2);
342 else
343 AddrWords += AddrComponents;
344
345 // Note: For subtargets that support A16 but not G16, enabling A16 also
346 // enables 16 bit gradients.
347 // For subtargets that support A16 (operand) and G16 (done with a different
348 // instruction encoding), they are independent.
349
350 if (BaseOpcode->Gradients) {
351 if ((IsA16 && !IsG16Supported) || BaseOpcode->G16)
352 // There are two gradients per coordinate, we pack them separately.
353 // For the 3d case,
354 // we get (dy/du, dx/du) (-, dz/du) (dy/dv, dx/dv) (-, dz/dv)
355 AddrWords += alignTo<2>(Dim->NumGradients / 2);
356 else
357 AddrWords += Dim->NumGradients;
358 }
359 return AddrWords;
360}
361
372
381
386
391
395
399
403
408
416
421
424 bool IsX;
425 bool IsY;
426};
427
428#define GET_FP4FP8DstByteSelTable_DECL
429#define GET_FP4FP8DstByteSelTable_IMPL
430
435
441
442#define GET_DPMACCInstructionTable_DECL
443#define GET_DPMACCInstructionTable_IMPL
444#define GET_MTBUFInfoTable_DECL
445#define GET_MTBUFInfoTable_IMPL
446#define GET_MUBUFInfoTable_DECL
447#define GET_MUBUFInfoTable_IMPL
448#define GET_SMInfoTable_DECL
449#define GET_SMInfoTable_IMPL
450#define GET_VOP1InfoTable_DECL
451#define GET_VOP1InfoTable_IMPL
452#define GET_VOP2InfoTable_DECL
453#define GET_VOP2InfoTable_IMPL
454#define GET_VOP3InfoTable_DECL
455#define GET_VOP3InfoTable_IMPL
456#define GET_VOPC64DPPTable_DECL
457#define GET_VOPC64DPPTable_IMPL
458#define GET_VOPC64DPP8Table_DECL
459#define GET_VOPC64DPP8Table_IMPL
460#define GET_VOPCAsmOnlyInfoTable_DECL
461#define GET_VOPCAsmOnlyInfoTable_IMPL
462#define GET_VOP3CAsmOnlyInfoTable_DECL
463#define GET_VOP3CAsmOnlyInfoTable_IMPL
464#define GET_VOPDComponentTable_DECL
465#define GET_VOPDComponentTable_IMPL
466#define GET_VOPDPairs_DECL
467#define GET_VOPDPairs_IMPL
468#define GET_VOPDXYTable_DECL
469#define GET_VOPDXYTable_IMPL
470#define GET_VOPTrue16Table_DECL
471#define GET_VOPTrue16Table_IMPL
472#define GET_True16D16Table_IMPL
473#define GET_WMMAOpcode2AddrMappingTable_DECL
474#define GET_WMMAOpcode2AddrMappingTable_IMPL
475#define GET_WMMAOpcode3AddrMappingTable_DECL
476#define GET_WMMAOpcode3AddrMappingTable_IMPL
477#define GET_getMFMA_F8F6F4_WithSize_DECL
478#define GET_getMFMA_F8F6F4_WithSize_IMPL
479#define GET_isMFMA_F8F6F4Table_IMPL
480#define GET_isCvtScaleF32_F32F16ToF8F4Table_IMPL
481
482#include "AMDGPUGenSearchableTables.inc"
483
484int getMTBUFBaseOpcode(unsigned Opc) {
485 const MTBUFInfo *Info = getMTBUFInfoFromOpcode(Opc);
486 return Info ? Info->BaseOpcode : -1;
487}
488
489int getMTBUFOpcode(unsigned BaseOpc, unsigned Elements) {
490 const MTBUFInfo *Info =
491 getMTBUFInfoFromBaseOpcodeAndElements(BaseOpc, Elements);
492 return Info ? Info->Opcode : -1;
493}
494
495int getMTBUFElements(unsigned Opc) {
496 const MTBUFInfo *Info = getMTBUFOpcodeHelper(Opc);
497 return Info ? Info->elements : 0;
498}
499
500bool getMTBUFHasVAddr(unsigned Opc) {
501 const MTBUFInfo *Info = getMTBUFOpcodeHelper(Opc);
502 return Info && Info->has_vaddr;
503}
504
505bool getMTBUFHasSrsrc(unsigned Opc) {
506 const MTBUFInfo *Info = getMTBUFOpcodeHelper(Opc);
507 return Info && Info->has_srsrc;
508}
509
510bool getMTBUFHasSoffset(unsigned Opc) {
511 const MTBUFInfo *Info = getMTBUFOpcodeHelper(Opc);
512 return Info && Info->has_soffset;
513}
514
515int getMUBUFBaseOpcode(unsigned Opc) {
516 const MUBUFInfo *Info = getMUBUFInfoFromOpcode(Opc);
517 return Info ? Info->BaseOpcode : -1;
518}
519
520int getMUBUFOpcode(unsigned BaseOpc, unsigned Elements) {
521 const MUBUFInfo *Info =
522 getMUBUFInfoFromBaseOpcodeAndElements(BaseOpc, Elements);
523 return Info ? Info->Opcode : -1;
524}
525
526int getMUBUFElements(unsigned Opc) {
527 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
528 return Info ? Info->elements : 0;
529}
530
531bool getMUBUFHasVAddr(unsigned Opc) {
532 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
533 return Info && Info->has_vaddr;
534}
535
536bool getMUBUFHasSrsrc(unsigned Opc) {
537 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
538 return Info && Info->has_srsrc;
539}
540
541bool getMUBUFHasSoffset(unsigned Opc) {
542 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
543 return Info && Info->has_soffset;
544}
545
546bool getMUBUFIsBufferInv(unsigned Opc) {
547 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
548 return Info && Info->IsBufferInv;
549}
550
551bool getMUBUFTfe(unsigned Opc) {
552 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
553 return Info && Info->tfe;
554}
555
556bool getSMEMIsBuffer(unsigned Opc) {
557 const SMInfo *Info = getSMEMOpcodeHelper(Opc);
558 return Info && Info->IsBuffer;
559}
560
561bool getVOP1IsSingle(unsigned Opc) {
562 const VOPInfo *Info = getVOP1OpcodeHelper(Opc);
563 return !Info || Info->IsSingle;
564}
565
566bool getVOP2IsSingle(unsigned Opc) {
567 const VOPInfo *Info = getVOP2OpcodeHelper(Opc);
568 return !Info || Info->IsSingle;
569}
570
571bool getVOP3IsSingle(unsigned Opc) {
572 const VOPInfo *Info = getVOP3OpcodeHelper(Opc);
573 return !Info || Info->IsSingle;
574}
575
576bool isVOPC64DPP(unsigned Opc) {
577 return isVOPC64DPPOpcodeHelper(Opc) || isVOPC64DPP8OpcodeHelper(Opc);
578}
579
580bool isVOPCAsmOnly(unsigned Opc) { return isVOPCAsmOnlyOpcodeHelper(Opc); }
581
582bool getMAIIsDGEMM(unsigned Opc) {
583 const MAIInstInfo *Info = getMAIInstInfoHelper(Opc);
584 return Info && Info->is_dgemm;
585}
586
587bool getMAIIsGFX940XDL(unsigned Opc) {
588 const MAIInstInfo *Info = getMAIInstInfoHelper(Opc);
589 return Info && Info->is_gfx940_xdl;
590}
591
592bool getWMMAIsXDL(unsigned Opc) {
593 const WMMAInstInfo *Info = getWMMAInstInfoHelper(Opc);
594 return Info ? Info->is_wmma_xdl : false;
595}
596
597bool getHasMatrixScale(unsigned Opc) {
598 const WMMAInstInfo *Info = getWMMAInstInfoHelper(Opc);
599 return Info && Info->HasMatrixScale;
600}
601
603 switch (EncodingVal) {
606 return 6;
608 return 4;
611 default:
612 return 8;
613 }
614
615 llvm_unreachable("covered switch over mfma scale formats");
616}
617
619 unsigned BLGP,
620 unsigned F8F8Opcode) {
621 uint8_t SrcANumRegs = mfmaScaleF8F6F4FormatToNumRegs(CBSZ);
622 uint8_t SrcBNumRegs = mfmaScaleF8F6F4FormatToNumRegs(BLGP);
623 return getMFMA_F8F6F4_InstWithNumRegs(SrcANumRegs, SrcBNumRegs, F8F8Opcode);
624}
625
627 switch (Fmt) {
630 return 16;
633 return 12;
635 return 8;
636 }
637
638 llvm_unreachable("covered switch over wmma scale formats");
639}
640
642 unsigned FmtB,
643 unsigned F8F8Opcode) {
644 uint8_t SrcANumRegs = wmmaScaleF8F6F4FormatToNumRegs(FmtA);
645 uint8_t SrcBNumRegs = wmmaScaleF8F6F4FormatToNumRegs(FmtB);
646 return getMFMA_F8F6F4_InstWithNumRegs(SrcANumRegs, SrcBNumRegs, F8F8Opcode);
647}
648
649bool isValidWMMAScaleFmtCombination(unsigned AFmt, unsigned AScale,
650 unsigned BFmt, unsigned BScale) {
651 auto isValid = [](unsigned Fmt, unsigned Scale) -> bool {
652 switch (Fmt) {
657 if (Scale != WMMA::MATRIX_SCALE_FMT_E8)
658 return false;
659 break;
661 if (Scale != WMMA::MATRIX_SCALE_FMT_E8 &&
664 return false;
665 break;
666 }
667 return true;
668 };
669
670 if (!isValid(AFmt, AScale) || !isValid(BFmt, BScale))
671 return false;
672
673 if (AFmt == WMMA::MATRIX_FMT_FP4 && BFmt == WMMA::MATRIX_FMT_FP4 &&
674 AScale != BScale)
675 return false;
676
677 return true;
678}
679
681 if (ST.hasFeature(AMDGPU::FeatureGFX13Insts))
683 if (ST.hasFeature(AMDGPU::FeatureGFX1250Insts))
685 if (ST.hasFeature(AMDGPU::FeatureGFX12Insts))
687 if (ST.hasFeature(AMDGPU::FeatureGFX11_7Insts))
689 if (ST.hasFeature(AMDGPU::FeatureGFX11Insts))
691 llvm_unreachable("Subtarget generation does not support VOPD!");
692}
693
694CanBeVOPD getCanBeVOPD(unsigned Opc, unsigned EncodingFamily, bool VOPD3) {
695 bool IsConvertibleToBitOp = VOPD3 ? getBitOp2(Opc) : 0;
696 Opc = IsConvertibleToBitOp ? (unsigned)AMDGPU::V_BITOP3_B32_e64 : Opc;
697 // Normalize through VOPDComponentTable so that e32 and e64 variants
698 // of the same logical opcode all share a single entry.
699 const VOPDComponentInfo *Info = getVOPDComponentHelper(Opc);
700 if (!Info)
701 return {false, false};
702 unsigned Key =
703 (Info->VOPDOp << 5) | (EncodingFamily << 1) | (VOPD3 ? 1u : 0u);
704 const VOPDXYInfo *XYInfo = getVOPDXYInfo(Key);
705 if (!XYInfo)
706 return {false, false};
707 return {XYInfo->IsX, XYInfo->IsY};
708}
709
710unsigned getVOPDOpcode(unsigned Opc, bool VOPD3) {
711 bool IsConvertibleToBitOp = VOPD3 ? getBitOp2(Opc) : 0;
712 Opc = IsConvertibleToBitOp ? (unsigned)AMDGPU::V_BITOP3_B32_e64 : Opc;
713 const VOPDComponentInfo *Info = getVOPDComponentHelper(Opc);
714 return Info ? Info->VOPDOp : ~0u;
715}
716
717bool isVOPD(unsigned Opc) {
718 return AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::src0X);
719}
720
721bool isMAC(unsigned Opc) {
722 return Opc == AMDGPU::V_MAC_F32_e64_gfx6_gfx7 ||
723 Opc == AMDGPU::V_MAC_F32_e64_gfx10 ||
724 Opc == AMDGPU::V_MAC_F32_e64_vi ||
725 Opc == AMDGPU::V_MAC_LEGACY_F32_e64_gfx6_gfx7 ||
726 Opc == AMDGPU::V_MAC_LEGACY_F32_e64_gfx10 ||
727 Opc == AMDGPU::V_MAC_F16_e64_vi ||
728 Opc == AMDGPU::V_FMAC_F64_e64_gfx90a ||
729 Opc == AMDGPU::V_FMAC_F64_e64_gfx12 ||
730 Opc == AMDGPU::V_FMAC_F64_e64_gfx13 ||
731 Opc == AMDGPU::V_FMAC_F32_e64_gfx10 ||
732 Opc == AMDGPU::V_FMAC_F32_e64_gfx11 ||
733 Opc == AMDGPU::V_FMAC_F32_e64_gfx12 ||
734 Opc == AMDGPU::V_FMAC_F32_e64_gfx13 ||
735 Opc == AMDGPU::V_FMAC_F32_e64_vi ||
736 Opc == AMDGPU::V_FMAC_LEGACY_F32_e64_gfx10 ||
737 Opc == AMDGPU::V_FMAC_DX9_ZERO_F32_e64_gfx11 ||
738 Opc == AMDGPU::V_FMAC_F16_e64_gfx10 ||
739 Opc == AMDGPU::V_FMAC_F16_t16_e64_gfx11 ||
740 Opc == AMDGPU::V_FMAC_F16_fake16_e64_gfx11 ||
741 Opc == AMDGPU::V_FMAC_F16_t16_e64_gfx12 ||
742 Opc == AMDGPU::V_FMAC_F16_fake16_e64_gfx12 ||
743 Opc == AMDGPU::V_FMAC_F16_t16_e64_gfx13 ||
744 Opc == AMDGPU::V_FMAC_F16_fake16_e64_gfx13 ||
745 Opc == AMDGPU::V_DOT2C_F32_F16_e64_vi ||
746 Opc == AMDGPU::V_DOT2C_F32_BF16_e64_vi ||
747 Opc == AMDGPU::V_DOT2C_I32_I16_e64_vi ||
748 Opc == AMDGPU::V_DOT4C_I32_I8_e64_vi ||
749 Opc == AMDGPU::V_DOT8C_I32_I4_e64_vi;
750}
751
752bool isPermlane16(unsigned Opc) {
753 return Opc == AMDGPU::V_PERMLANE16_B32_gfx10 ||
754 Opc == AMDGPU::V_PERMLANEX16_B32_gfx10 ||
755 Opc == AMDGPU::V_PERMLANE16_B32_e64_gfx11 ||
756 Opc == AMDGPU::V_PERMLANEX16_B32_e64_gfx11 ||
757 Opc == AMDGPU::V_PERMLANE16_B32_e64_gfx12 ||
758 Opc == AMDGPU::V_PERMLANE16_B32_e64_gfx13 ||
759 Opc == AMDGPU::V_PERMLANEX16_B32_e64_gfx12 ||
760 Opc == AMDGPU::V_PERMLANEX16_B32_e64_gfx13 ||
761 Opc == AMDGPU::V_PERMLANE16_VAR_B32_e64_gfx12 ||
762 Opc == AMDGPU::V_PERMLANE16_VAR_B32_e64_gfx13 ||
763 Opc == AMDGPU::V_PERMLANEX16_VAR_B32_e64_gfx12 ||
764 Opc == AMDGPU::V_PERMLANEX16_VAR_B32_e64_gfx13;
765}
766
768 return Opc == AMDGPU::V_CVT_F32_BF8_e64_gfx12 ||
769 Opc == AMDGPU::V_CVT_F32_FP8_e64_gfx12 ||
770 Opc == AMDGPU::V_CVT_F32_BF8_e64_dpp_gfx12 ||
771 Opc == AMDGPU::V_CVT_F32_FP8_e64_dpp_gfx12 ||
772 Opc == AMDGPU::V_CVT_F32_BF8_e64_dpp8_gfx12 ||
773 Opc == AMDGPU::V_CVT_F32_FP8_e64_dpp8_gfx12 ||
774 Opc == AMDGPU::V_CVT_PK_F32_BF8_fake16_e64_gfx12 ||
775 Opc == AMDGPU::V_CVT_PK_F32_FP8_fake16_e64_gfx12 ||
776 Opc == AMDGPU::V_CVT_PK_F32_BF8_t16_e64_gfx12 ||
777 Opc == AMDGPU::V_CVT_PK_F32_FP8_t16_e64_gfx12;
778}
779
780bool isGenericAtomic(unsigned Opc) {
781 return Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SWAP ||
782 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_ADD ||
783 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SUB ||
784 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SMIN ||
785 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_UMIN ||
786 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SMAX ||
787 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_UMAX ||
788 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_AND ||
789 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_OR ||
790 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_XOR ||
791 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_INC ||
792 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_DEC ||
793 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_FADD ||
794 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_FMIN ||
795 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_FMAX ||
796 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_CMPSWAP ||
797 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SUB_CLAMP_U32 ||
798 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_COND_SUB_U32 ||
799 Opc == AMDGPU::G_AMDGPU_ATOMIC_CMPXCHG;
800}
801
802bool isAsyncStore(unsigned Opc) {
803 return Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B8_gfx1250 ||
804 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B32_gfx1250 ||
805 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B64_gfx1250 ||
806 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B128_gfx1250 ||
807 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B8_SADDR_gfx1250 ||
808 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B32_SADDR_gfx1250 ||
809 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B64_SADDR_gfx1250 ||
810 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B128_SADDR_gfx1250;
811}
812
813bool isTensorStore(unsigned Opc) {
814 return Opc == TENSOR_STORE_FROM_LDS_d2_gfx1250 ||
815 Opc == TENSOR_STORE_FROM_LDS_d4_gfx1250;
816}
817
818unsigned getTemporalHintType(const MCInstrDesc TID) {
819 if (SIInstrFlags::isAtomic(TID))
821 unsigned Opc = TID.getOpcode();
822 // Async and Tensor store should have the temporal hint type of TH_TYPE_STORE
823 if (TID.mayStore() &&
824 (isAsyncStore(Opc) || isTensorStore(Opc) || !TID.mayLoad()))
825 return CPol::TH_TYPE_STORE;
826
827 // This will default to returning TH_TYPE_LOAD when neither MayStore nor
828 // MayLoad flag is present which is the case with instructions like
829 // image_get_resinfo.
830 return CPol::TH_TYPE_LOAD;
831}
832
833bool isTrue16Inst(unsigned Opc) {
834 const VOPTrue16Info *Info = getTrue16OpcodeHelper(Opc);
835 return Info && Info->IsTrue16;
836}
837
839 const FP4FP8DstByteSelInfo *Info = getFP4FP8DstByteSelHelper(Opc);
840 if (!Info)
841 return FPType::None;
842 if (Info->HasFP8DstByteSel)
843 return FPType::FP8;
844 if (Info->HasFP4DstByteSel)
845 return FPType::FP4;
846
847 return FPType::None;
848}
849
850bool isDPMACCInstruction(unsigned Opc) {
851 const DPMACCInstructionInfo *Info = getDPMACCInstructionHelper(Opc);
852 return Info && Info->IsDPMACCInstruction;
853}
854
855unsigned mapWMMA2AddrTo3AddrOpcode(unsigned Opc) {
856 const WMMAOpcodeMappingInfo *Info = getWMMAMappingInfoFrom2AddrOpcode(Opc);
857 return Info ? Info->Opcode3Addr : ~0u;
858}
859
860unsigned mapWMMA3AddrTo2AddrOpcode(unsigned Opc) {
861 const WMMAOpcodeMappingInfo *Info = getWMMAMappingInfoFrom3AddrOpcode(Opc);
862 return Info ? Info->Opcode2Addr : ~0u;
863}
864
865// Wrapper for Tablegen'd function. enum Subtarget is not defined in any
866// header files, so we need to wrap it in a function that takes unsigned
867// instead.
868int32_t getMCOpcode(uint32_t Opcode, unsigned Gen) {
869 return getMCOpcodeGen(Opcode, static_cast<Subtarget>(Gen));
870}
871
872unsigned getBitOp2(unsigned Opc) {
873 switch (Opc) {
874 default:
875 return 0;
876 case AMDGPU::V_AND_B32_e32:
877 return 0x40;
878 case AMDGPU::V_OR_B32_e32:
879 return 0x54;
880 case AMDGPU::V_XOR_B32_e32:
881 return 0x14;
882 case AMDGPU::V_XNOR_B32_e32:
883 return 0x41;
884 }
885}
886
887int getVOPDFull(unsigned OpX, unsigned OpY, unsigned EncodingFamily,
888 bool VOPD3) {
889 bool IsConvertibleToBitOp = VOPD3 ? getBitOp2(OpY) : 0;
890 OpY = IsConvertibleToBitOp ? (unsigned)AMDGPU::V_BITOP3_B32_e64 : OpY;
891 const VOPDInfo *Info =
892 getVOPDInfoFromComponentOpcodes(OpX, OpY, EncodingFamily, VOPD3);
893 return Info ? Info->Opcode : -1;
894}
895
896std::pair<unsigned, unsigned> getVOPDComponents(unsigned VOPDOpcode) {
897 const VOPDInfo *Info = getVOPDOpcodeHelper(VOPDOpcode);
898 assert(Info);
899 const auto *OpX = getVOPDBaseFromComponent(Info->OpX);
900 const auto *OpY = getVOPDBaseFromComponent(Info->OpY);
901 assert(OpX && OpY);
902 return {OpX->BaseVOP, OpY->BaseVOP};
903}
904
905namespace VOPD {
906
907ComponentProps::ComponentProps(const MCInstrDesc &OpDesc, bool VOP3Layout) {
909
912 auto TiedIdx = OpDesc.getOperandConstraint(Component::SRC2, MCOI::TIED_TO);
913 assert(TiedIdx == -1 || TiedIdx == Component::DST);
914 HasSrc2Acc = TiedIdx != -1;
915 Opcode = OpDesc.getOpcode();
916
917 IsVOP3 = VOP3Layout || SIInstrFlags::isVOP3(OpDesc);
918 SrcOperandsNum = AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::src2) ? 3
919 : AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::imm) ? 3
920 : AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::src1) ? 2
921 : 1;
922 assert(SrcOperandsNum <= Component::MAX_SRC_NUM);
923
924 if (Opcode == AMDGPU::V_CNDMASK_B32_e32 ||
925 Opcode == AMDGPU::V_CNDMASK_B32_e64) {
926 // CNDMASK is an awkward exception, it has FP modifiers, but not FP
927 // operands.
928 NumVOPD3Mods = 2;
929 if (IsVOP3)
930 SrcOperandsNum = 3;
931 } else if (Opcode == AMDGPU::V_DOT2_F32_F16 ||
932 Opcode == AMDGPU::V_DOT2_F32_BF16) {
933 // VOP3P opcodes that have VOPD but don't have VOP2 version. Using VOPD3
934 // path in getIndexOfSrcInMCOperands to get correct src operand indexes,
935 // but generating VOPD, not VOPD3.
936 NumVOPD3Mods = SrcOperandsNum;
937 } else if (isSISrcFPOperand(OpDesc,
938 getNamedOperandIdx(Opcode, OpName::src0))) {
939 // All FP VOPD instructions have Neg modifiers for all operands except
940 // for tied src2.
941 NumVOPD3Mods = SrcOperandsNum;
942 if (HasSrc2Acc)
943 --NumVOPD3Mods;
944 }
945
946 if (SIInstrFlags::isVOP3(OpDesc))
947 return;
948
949 auto OperandsNum = OpDesc.getNumOperands();
950 unsigned CompOprIdx;
951 for (CompOprIdx = Component::SRC1; CompOprIdx < OperandsNum; ++CompOprIdx) {
952 if (OpDesc.operands()[CompOprIdx].OperandType == AMDGPU::OPERAND_KIMM32) {
953 MandatoryLiteralIdx = CompOprIdx;
954 break;
955 }
956 }
957}
958
960 return getNamedOperandIdx(Opcode, OpName::bitop3);
961}
962
963unsigned ComponentInfo::getIndexInParsedOperands(unsigned CompOprIdx) const {
964 assert(CompOprIdx < Component::MAX_OPR_NUM);
965
966 if (CompOprIdx == Component::DST)
968
969 auto CompSrcIdx = CompOprIdx - Component::DST_NUM;
970 if (CompSrcIdx < getCompParsedSrcOperandsNum())
971 return getIndexOfSrcInParsedOperands(CompSrcIdx);
972
973 // The specified operand does not exist.
974 return 0;
975}
976
978 std::function<MCRegister(unsigned, unsigned)> GetRegIdx,
979 const MCRegisterInfo &MRI, bool SkipSrc, bool AllowSameVGPR,
980 bool VOPD3) const {
981
982 auto OpXRegs = getRegIndices(ComponentIndex::X, GetRegIdx,
983 CompInfo[ComponentIndex::X].isVOP3());
984 auto OpYRegs = getRegIndices(ComponentIndex::Y, GetRegIdx,
985 CompInfo[ComponentIndex::Y].isVOP3());
986
987 const auto banksOverlap = [&MRI](MCRegister X, MCRegister Y,
988 unsigned BanksMask) -> bool {
989 MCRegister BaseX = MRI.getSubReg(X, AMDGPU::sub0);
990 MCRegister BaseY = MRI.getSubReg(Y, AMDGPU::sub0);
991 if (!BaseX)
992 BaseX = X;
993 if (!BaseY)
994 BaseY = Y;
995 if ((BaseX.id() & BanksMask) == (BaseY.id() & BanksMask))
996 return true;
997 if (BaseX != X /* This is 64-bit register */ &&
998 ((BaseX.id() + 1) & BanksMask) == (BaseY.id() & BanksMask))
999 return true;
1000 if (BaseY != Y &&
1001 (BaseX.id() & BanksMask) == ((BaseY.id() + 1) & BanksMask))
1002 return true;
1003
1004 // If both are 64-bit bank conflict will be detected yet while checking
1005 // the first subreg.
1006 return false;
1007 };
1008
1009 unsigned CompOprIdx;
1010 for (CompOprIdx = 0; CompOprIdx < Component::MAX_OPR_NUM; ++CompOprIdx) {
1011 unsigned BanksMasks = VOPD3 ? VOPD3_VGPR_BANK_MASKS[CompOprIdx]
1012 : VOPD_VGPR_BANK_MASKS[CompOprIdx];
1013 if (!OpXRegs[CompOprIdx] || !OpYRegs[CompOprIdx])
1014 continue;
1015
1016 if (getVGPREncodingMSBs(OpXRegs[CompOprIdx], MRI) !=
1017 getVGPREncodingMSBs(OpYRegs[CompOprIdx], MRI))
1018 return CompOprIdx;
1019
1020 if (SkipSrc && CompOprIdx >= Component::DST_NUM)
1021 continue;
1022
1023 if (CompOprIdx < Component::DST_NUM) {
1024 // Even if we do not check vdst parity, vdst operands still shall not
1025 // overlap.
1026 if (MRI.regsOverlap(OpXRegs[CompOprIdx], OpYRegs[CompOprIdx]))
1027 return CompOprIdx;
1028 if (VOPD3) // No need to check dst parity.
1029 continue;
1030 }
1031
1032 if (banksOverlap(OpXRegs[CompOprIdx], OpYRegs[CompOprIdx], BanksMasks) &&
1033 (!AllowSameVGPR || CompOprIdx < Component::DST_NUM ||
1034 OpXRegs[CompOprIdx] != OpYRegs[CompOprIdx]))
1035 return CompOprIdx;
1036 }
1037
1038 return {};
1039}
1040
1041// Return an array of VGPR registers [DST,SRC0,SRC1,SRC2] used
1042// by the specified component. If an operand is unused
1043// or is not a VGPR, the corresponding value is 0.
1044//
1045// GetRegIdx(Component, MCOperandIdx) must return a VGPR register index
1046// for the specified component and MC operand. The callback must return 0
1047// if the operand is not a register or not a VGPR.
1049InstInfo::getRegIndices(unsigned CompIdx,
1050 std::function<MCRegister(unsigned, unsigned)> GetRegIdx,
1051 bool VOPD3) const {
1052 assert(CompIdx < COMPONENTS_NUM);
1053
1054 const auto &Comp = CompInfo[CompIdx];
1056
1057 RegIndices[DST] = GetRegIdx(CompIdx, Comp.getIndexOfDstInMCOperands());
1058
1059 for (unsigned CompOprIdx : {SRC0, SRC1, SRC2}) {
1060 unsigned CompSrcIdx = CompOprIdx - DST_NUM;
1061 RegIndices[CompOprIdx] =
1062 Comp.hasRegSrcOperand(CompSrcIdx)
1063 ? GetRegIdx(CompIdx,
1064 Comp.getIndexOfSrcInMCOperands(CompSrcIdx, VOPD3))
1065 : MCRegister();
1066 }
1067 return RegIndices;
1068}
1069
1070} // namespace VOPD
1071
1073 return VOPD::InstInfo(OpX, OpY);
1074}
1075
1077 const MCInstrInfo *InstrInfo) {
1078 auto [OpX, OpY] = getVOPDComponents(VOPDOpcode);
1079 const auto &OpXDesc = InstrInfo->get(OpX);
1080 const auto &OpYDesc = InstrInfo->get(OpY);
1081 bool VOPD3 = SIInstrFlags::isVOPD3(*InstrInfo, VOPDOpcode);
1083 VOPD::ComponentInfo OpYInfo(OpYDesc, OpXInfo, VOPD3);
1084 return VOPD::InstInfo(OpXInfo, OpYInfo);
1085}
1086
1088 StringRef FeatureString) {
1089 // In codegen the mode comes from module flags and FeatureString is empty, so
1090 // the processor defaults apply. The assembler has no target directive, so it
1091 // pins the mode via the +xnack/-xnack/+sramecc/-sramecc feature string.
1093 STI.getCPU(), FeatureString);
1094}
1095
1096namespace IsaInfo {
1097
1099 if (STI.getFeatureBits().test(FeatureInstCacheLineSize128))
1100 return 128;
1101 if (STI.getFeatureBits().test(FeatureInstCacheLineSize64))
1102 return 64;
1103 return 64;
1104}
1105
1106unsigned getWavefrontSize(const MCSubtargetInfo &STI) {
1107 if (STI.getFeatureBits().test(FeatureWavefrontSize16))
1108 return 16;
1109 if (STI.getFeatureBits().test(FeatureWavefrontSize32))
1110 return 32;
1111
1112 return 64;
1113}
1114
1115// Maximum LDS a single work-group can address. This is a fixed HW cap. It does
1116// not depend on how many SIMDs a work-group runs on.
1118 if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize32768))
1119 return 32768;
1120 if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize65536))
1121 return 65536;
1122 if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize163840))
1123 return 163840;
1124 if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize196608))
1125 return 196608;
1126 if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize327680))
1127 return 327680;
1128 return 32768;
1129}
1130
1131// Total physical size of LDS on the block, in bytes. On targets with
1132// FeatureHalfAddressablePhysicalLocalMemory the physical block is twice the
1133// addressable size (gfx10/11/12, 128k physical and 64k addressable). On other
1134// targets it is equal to the addressable size.
1135static unsigned getPhysicalLocalMemorySize(const MCSubtargetInfo &STI) {
1136 unsigned Addressable = getMaxHWAddressableLocalMemorySize(STI);
1137 if (STI.getFeatureBits().test(FeatureHalfAddressablePhysicalLocalMemory))
1138 return 2 * Addressable;
1139 return Addressable;
1140}
1141
1142// Sizes in use, by generation (addressable / physical block):
1143// gfx6 : 32 KiB
1144// gfx7 / gfx8 / gfx9: 64 KiB
1145// gfx9.5 (gfx950) : 160 KiB
1146// gfx10 / 11 / 12 : 64 KiB addressable, 128 KiB physical block
1147// gfx12.5 (gfx1250) : 320 KiB (always runs on four SIMDs)
1148// gfx13 : 192 KiB on four SIMDs, 96 KiB on two
1149// Total available in the current mode. The physical size is halved when a
1150// work-group runs on two SIMDs.
1152 unsigned Size = getPhysicalLocalMemorySize(STI);
1153 if (!isFullSIMDMode(STI))
1154 Size /= 2;
1155 return Size;
1156}
1157
1158// What one work-group can allocate in the current mode. This is the HW
1159// addressable cap, but never more than the total available in the current mode.
1161 return std::min(getMaxHWAddressableLocalMemorySize(STI),
1162 getLocalMemorySize(STI));
1163}
1164
1166 unsigned FlatWorkGroupSize) {
1167 assert(FlatWorkGroupSize != 0);
1168 if (!STI.getTargetTriple().isAMDGCN())
1169 return 8;
1170 GPUKind Kind = parseArchAMDGCN(STI.getCPU());
1171 unsigned MaxWaves =
1173 unsigned N = getWavesPerWorkGroup(STI, FlatWorkGroupSize);
1174 if (N == 1) {
1175 // Single-wave workgroups don't consume barrier resources.
1176 return MaxWaves;
1177 }
1178
1179 unsigned MaxBarriers = 16;
1180 if (isGFX10Plus(STI) && !STI.getFeatureBits().test(FeatureCuMode))
1181 MaxBarriers = 32;
1182
1183 return std::min(MaxWaves / N, MaxBarriers);
1184}
1185
1187 unsigned FlatWorkGroupSize) {
1188 return divideCeil(getWavesPerWorkGroup(STI, FlatWorkGroupSize),
1190}
1191
1193 unsigned FlatWorkGroupSize) {
1194 return divideCeil(FlatWorkGroupSize, getWavefrontSize(STI));
1195}
1196
1197unsigned getSGPREncodingGranule(const MCSubtargetInfo &STI) { return 8; }
1198
1199// Per-wave SGPRs reserved for the trap handler when enabled.
1200static unsigned getSGPRTrapHandlerReserve(const MCSubtargetInfo &STI) {
1201 return STI.getFeatureBits().test(FeatureTrapHandler) ? TRAP_NUM_SGPRS : 0;
1202}
1203
1204// Per-wave SGPR budget (before the addressable clamp): take off the trap
1205// reserve, round down to \p Granule. Shared by getMinNumSGPRs() and
1206// getMaxNumSGPRs(); getOccupancyWithNumSGPRs() is the closed-form algebraic
1207// inverse of this same budget (it does not call this helper), so the two encode
1208// one model.
1209static unsigned getSGPRBudgetPerWave(unsigned TotalNumSGPRs,
1210 unsigned WavesPerEU, unsigned TrapReserve,
1211 unsigned Granule) {
1212 assert(WavesPerEU != 0 && Granule != 0);
1213 unsigned Budget = TotalNumSGPRs / WavesPerEU;
1214 Budget -= std::min(Budget, TrapReserve);
1215 return alignDown(Budget, Granule);
1216}
1217
1218unsigned getMinNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU) {
1219 assert(WavesPerEU != 0);
1220
1222 if (Version.Major >= 10)
1223 return 0;
1224
1225 GPUKind Kind = parseArchAMDGCN(STI.getCPU());
1226 if (WavesPerEU >= getMaxWavesPerEU(Kind))
1227 return 0;
1228
1229 unsigned MinNumSGPRs =
1230 getSGPRBudgetPerWave(getTotalNumSGPRs(Kind), WavesPerEU + 1,
1232 getSGPRAllocGranule(Kind)) +
1233 1;
1234 return std::min(MinNumSGPRs, getAddressableNumSGPRs(Kind));
1235}
1236
1237unsigned getMaxNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
1238 bool Addressable) {
1239 assert(WavesPerEU != 0);
1240
1241 GPUKind Kind = parseArchAMDGCN(STI.getCPU());
1242 unsigned AddressableNumSGPRs = getAddressableNumSGPRs(Kind);
1244 if (Version.Major >= 10)
1245 return Addressable ? AddressableNumSGPRs : 108;
1246 if (Version.Major >= 8 && !Addressable)
1247 AddressableNumSGPRs = 112;
1248 unsigned MaxNumSGPRs = getSGPRBudgetPerWave(
1249 getTotalNumSGPRs(Kind), WavesPerEU, getSGPRTrapHandlerReserve(STI),
1250 getSGPRAllocGranule(Kind));
1251 return std::min(MaxNumSGPRs, AddressableNumSGPRs);
1252}
1253
1255 // From GFX10 on the SGPR file is large enough that SGPRs never limit
1256 // occupancy. Kept as one capability so callers don't each test the version.
1257 return getIsaVersion(STI.getCPU()).Major < 10;
1258}
1259
1260unsigned getNumExtraSGPRs(const MCSubtargetInfo &STI, bool VCCUsed,
1261 bool FlatScrUsed, bool XNACKUsed) {
1262 unsigned ExtraSGPRs = 0;
1263 if (VCCUsed)
1264 ExtraSGPRs = 2;
1265
1267 if (Version.Major >= 10)
1268 return ExtraSGPRs;
1269
1270 if (Version.Major < 8) {
1271 if (FlatScrUsed)
1272 ExtraSGPRs = 4;
1273 } else {
1274 if (XNACKUsed)
1275 ExtraSGPRs = 4;
1276
1277 if (FlatScrUsed ||
1278 STI.getFeatureBits().test(AMDGPU::FeatureArchitectedFlatScratch))
1279 ExtraSGPRs = 6;
1280 }
1281
1282 return ExtraSGPRs;
1283}
1284
1285static unsigned getGranulatedNumRegisterBlocks(unsigned NumRegs,
1286 unsigned Granule) {
1287 return divideCeil(std::max(1u, NumRegs), Granule);
1288}
1289
1290unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs) {
1291 // SGPRBlocks is actual number of SGPR blocks minus 1.
1293 1;
1294}
1295
1297 unsigned DynamicVGPRBlockSize,
1298 std::optional<bool> EnableWavefrontSize32) {
1299 if (STI.getFeatureBits().test(FeatureGFX90AInsts))
1300 return 8;
1301
1302 if (DynamicVGPRBlockSize != 0)
1303 return DynamicVGPRBlockSize;
1304
1305 bool IsWave32 = EnableWavefrontSize32
1306 ? *EnableWavefrontSize32
1307 : STI.getFeatureBits().test(FeatureWavefrontSize32);
1308
1309 if (STI.getFeatureBits().test(Feature1536VGPRs))
1310 return IsWave32 ? 24 : 12;
1311
1312 if (hasGFX10_3Insts(STI))
1313 return IsWave32 ? 16 : 8;
1314
1315 return IsWave32 ? 8 : 4;
1316}
1317
1319 std::optional<bool> EnableWavefrontSize32) {
1320 if (STI.getFeatureBits().test(FeatureGFX90AInsts))
1321 return 8;
1322
1323 bool IsWave32 = EnableWavefrontSize32
1324 ? *EnableWavefrontSize32
1325 : STI.getFeatureBits().test(FeatureWavefrontSize32);
1326
1327 if (STI.getFeatureBits().test(Feature1024AddressableVGPRs))
1328 return IsWave32 ? 16 : 8;
1329
1330 return IsWave32 ? 8 : 4;
1331}
1332
1333unsigned getArchVGPRAllocGranule() { return 4; }
1334
1336 const auto &Features = STI.getFeatureBits();
1337 if (Features.test(Feature1024AddressableVGPRs))
1338 return Features.test(FeatureWavefrontSize32) ? 1024 : 512;
1339 return 256;
1340}
1341
1343 unsigned DynamicVGPRBlockSize) {
1344 const auto &Features = STI.getFeatureBits();
1345 if (Features.test(FeatureGFX90AInsts))
1346 return 512;
1347
1348 if (DynamicVGPRBlockSize != 0) {
1349 // On GFX12 we can allocate at most MaxDynamicVGPRBlocks blocks of VGPRs.
1350 return MaxDynamicVGPRBlocks *
1351 getVGPRAllocGranule(STI, DynamicVGPRBlockSize);
1352 }
1353 return getAddressableNumArchVGPRs(STI);
1354}
1355
1357 unsigned NumVGPRs,
1358 unsigned DynamicVGPRBlockSize) {
1359 GPUKind Kind = parseArchAMDGCN(STI.getCPU());
1360 bool IsWave32 = STI.getFeatureBits().test(FeatureWavefrontSize32);
1362 NumVGPRs, getVGPRAllocGranule(STI, DynamicVGPRBlockSize),
1363 getMaxWavesPerEU(Kind), AMDGPU::getTotalNumVGPRs(Kind, IsWave32));
1364}
1365
1366unsigned getNumWavesPerEUWithNumVGPRs(unsigned NumVGPRs, unsigned Granule,
1367 unsigned MaxWaves,
1368 unsigned TotalNumVGPRs) {
1369 if (NumVGPRs < Granule)
1370 return MaxWaves;
1371 unsigned RoundedRegs = alignTo(NumVGPRs, Granule);
1372 return std::min(std::max(TotalNumVGPRs / RoundedRegs, 1u), MaxWaves);
1373}
1374
1375unsigned getOccupancyWithNumSGPRs(unsigned SGPRs, unsigned MaxWaves,
1376 unsigned TotalNumSGPRs, unsigned Granule,
1377 unsigned TrapReserve) {
1378 // Closed-form inverse of getMaxNumSGPRs(): the budget condition
1379 // SGPRs <= alignDown(TotalNumSGPRs / W - TrapReserve, Granule)
1380 // solves to W <= TotalNumSGPRs / (alignTo(SGPRs, Granule) + TrapReserve).
1381 unsigned PerWave = alignTo(SGPRs, Granule) + TrapReserve;
1382 return PerWave ? std::clamp(TotalNumSGPRs / PerWave, 1u, MaxWaves) : MaxWaves;
1383}
1384
1385unsigned getOccupancyWithNumSGPRs(const MCSubtargetInfo &STI, unsigned SGPRs) {
1386 GPUKind Kind = parseArchAMDGCN(STI.getCPU());
1387 unsigned MaxWaves = getMaxWavesPerEU(Kind);
1388
1389 if (!isSGPROccupancyLimited(STI))
1390 return MaxWaves;
1391
1392 return getOccupancyWithNumSGPRs(SGPRs, MaxWaves, getTotalNumSGPRs(Kind),
1393 getSGPRAllocGranule(Kind),
1395}
1396
1397unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
1398 unsigned DynamicVGPRBlockSize) {
1399 assert(WavesPerEU != 0);
1400
1401 // In dynamic VGPR mode, (static) occupancy does not depend on VGPR usage,
1402 // so getMaxNumVGPRs does not depend on WavesPerEU, and thus we need to return
1403 // zero because there is no nonzero VGPR usage N where going below N
1404 // achieves higher (static) occupancy.
1405 bool DynamicVGPREnabled = (DynamicVGPRBlockSize != 0);
1406 if (DynamicVGPREnabled)
1407 return 0;
1408
1409 GPUKind Kind = parseArchAMDGCN(STI.getCPU());
1410 unsigned MaxWavesPerEU = getMaxWavesPerEU(Kind);
1411 if (WavesPerEU >= MaxWavesPerEU)
1412 return 0;
1413
1414 unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(
1415 Kind, STI.getFeatureBits().test(FeatureWavefrontSize32));
1416 unsigned AddrsableNumVGPRs =
1417 getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
1418 unsigned Granule = getVGPRAllocGranule(STI, DynamicVGPRBlockSize);
1419 unsigned MaxNumVGPRs = alignDown(TotNumVGPRs / WavesPerEU, Granule);
1420
1421 if (MaxNumVGPRs == alignDown(TotNumVGPRs / MaxWavesPerEU, Granule))
1422 return 0;
1423
1424 unsigned MinWavesPerEU = getNumWavesPerEUWithNumVGPRs(STI, AddrsableNumVGPRs,
1425 DynamicVGPRBlockSize);
1426 if (WavesPerEU < MinWavesPerEU)
1427 return getMinNumVGPRs(STI, MinWavesPerEU, DynamicVGPRBlockSize);
1428
1429 unsigned MaxNumVGPRsNext = alignDown(TotNumVGPRs / (WavesPerEU + 1), Granule);
1430 unsigned MinNumVGPRs = 1 + std::min(MaxNumVGPRs - Granule, MaxNumVGPRsNext);
1431 return std::min(MinNumVGPRs, AddrsableNumVGPRs);
1432}
1433
1434unsigned getMaxNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
1435 unsigned DynamicVGPRBlockSize) {
1436 assert(WavesPerEU != 0);
1437
1438 unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(
1439 parseArchAMDGCN(STI.getCPU()),
1440 STI.getFeatureBits().test(FeatureWavefrontSize32));
1441
1442 // In dynamic VGPR mode, WavesPerEU does not imply a VGPR limit.
1443 bool DynamicVGPREnabled = (DynamicVGPRBlockSize != 0);
1444 unsigned MaxNumVGPRs =
1445 DynamicVGPREnabled
1446 ? TotNumVGPRs
1447 : alignDown(TotNumVGPRs / WavesPerEU,
1448 getVGPRAllocGranule(STI, DynamicVGPRBlockSize));
1449 unsigned AddressableNumVGPRs =
1450 getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
1451 return std::min(MaxNumVGPRs, AddressableNumVGPRs);
1452}
1453
1454unsigned getEncodedNumVGPRBlocks(const MCSubtargetInfo &STI, unsigned NumVGPRs,
1455 std::optional<bool> EnableWavefrontSize32) {
1457 NumVGPRs, getVGPREncodingGranule(STI, EnableWavefrontSize32)) -
1458 1;
1459}
1460
1462 unsigned NumVGPRs,
1463 unsigned DynamicVGPRBlockSize,
1464 std::optional<bool> EnableWavefrontSize32) {
1466 NumVGPRs,
1467 getVGPRAllocGranule(STI, DynamicVGPRBlockSize, EnableWavefrontSize32));
1468}
1469} // end namespace IsaInfo
1470
1472 const MCSubtargetInfo &STI) {
1474 KernelCode.amd_kernel_code_version_major = 1;
1475 KernelCode.amd_kernel_code_version_minor = 2;
1476 KernelCode.amd_machine_kind = 1; // AMD_MACHINE_KIND_AMDGPU
1477 KernelCode.amd_machine_version_major = Version.Major;
1478 KernelCode.amd_machine_version_minor = Version.Minor;
1479 KernelCode.amd_machine_version_stepping = Version.Stepping;
1481 if (STI.getFeatureBits().test(FeatureWavefrontSize32)) {
1482 KernelCode.wavefront_size = 5;
1484 } else {
1485 KernelCode.wavefront_size = 6;
1486 }
1487
1488 // If the code object does not support indirect functions, then the value must
1489 // be 0xffffffff.
1490 KernelCode.call_convention = -1;
1491
1492 // These alignment values are specified in powers of two, so alignment =
1493 // 2^n. The minimum alignment is 2^4 = 16.
1494 KernelCode.kernarg_segment_alignment = 4;
1495 KernelCode.group_segment_alignment = 4;
1496 KernelCode.private_segment_alignment = 4;
1497
1498 if (Version.Major >= 10) {
1499 KernelCode.compute_pgm_resource_registers |=
1500 S_00B848_WGP_MODE(STI.getFeatureBits().test(FeatureCuMode) ? 0 : 1) |
1502 }
1503}
1504
1507}
1508
1511}
1512
1514 unsigned AS = GV->getAddressSpace();
1515 return AS == AMDGPUAS::CONSTANT_ADDRESS ||
1517}
1518
1520 return TT.getArch() == Triple::r600;
1521}
1522
1523static bool isValidRegPrefix(char C) {
1524 return C == 'v' || C == 's' || C == 'a';
1525}
1526
1527std::tuple<char, unsigned, unsigned> parseAsmPhysRegName(StringRef RegName) {
1528 char Kind = RegName.front();
1529 if (!isValidRegPrefix(Kind))
1530 return {};
1531
1532 RegName = RegName.drop_front();
1533 if (RegName.consume_front("[")) {
1534 unsigned Idx, End;
1535 bool Failed = RegName.consumeInteger(10, Idx);
1536 Failed |= !RegName.consume_front(":");
1537 Failed |= RegName.consumeInteger(10, End);
1538 Failed |= !RegName.consume_back("]");
1539 if (!Failed) {
1540 unsigned NumRegs = End - Idx + 1;
1541 if (NumRegs > 1)
1542 return {Kind, Idx, NumRegs};
1543 }
1544 } else {
1545 unsigned Idx;
1546 bool Failed = RegName.getAsInteger(10, Idx);
1547 if (!Failed)
1548 return {Kind, Idx, 1};
1549 }
1550
1551 return {};
1552}
1553
1554std::tuple<char, unsigned, unsigned>
1556 StringRef RegName = Constraint;
1557 if (!RegName.consume_front("{") || !RegName.consume_back("}"))
1558 return {};
1560}
1561
1562std::pair<unsigned, unsigned>
1564 std::pair<unsigned, unsigned> Default,
1565 bool OnlyFirstRequired) {
1566 if (auto Attr = getIntegerPairAttribute(F, Name, OnlyFirstRequired))
1567 return {Attr->first, Attr->second.value_or(Default.second)};
1568 return Default;
1569}
1570
1571std::optional<std::pair<unsigned, std::optional<unsigned>>>
1573 bool OnlyFirstRequired) {
1574 Attribute A = F.getFnAttribute(Name);
1575 if (!A.isStringAttribute())
1576 return std::nullopt;
1577
1578 LLVMContext &Ctx = F.getContext();
1579 std::pair<unsigned, std::optional<unsigned>> Ints;
1580 std::pair<StringRef, StringRef> Strs = A.getValueAsString().split(',');
1581 if (Strs.first.trim().getAsInteger(0, Ints.first)) {
1582 Ctx.emitError("can't parse first integer attribute " + Name);
1583 return std::nullopt;
1584 }
1585 unsigned Second = 0;
1586 if (Strs.second.trim().getAsInteger(0, Second)) {
1587 if (!OnlyFirstRequired || !Strs.second.trim().empty()) {
1588 Ctx.emitError("can't parse second integer attribute " + Name);
1589 return std::nullopt;
1590 }
1591 } else {
1592 Ints.second = Second;
1593 }
1594
1595 return Ints;
1596}
1597
1599 unsigned Size,
1600 unsigned DefaultVal) {
1601 std::optional<SmallVector<unsigned>> R =
1603 return R.has_value() ? *R : SmallVector<unsigned>(Size, DefaultVal);
1604}
1605
1606std::optional<SmallVector<unsigned>>
1608 assert(Size > 2);
1609 LLVMContext &Ctx = F.getContext();
1610
1611 Attribute A = F.getFnAttribute(Name);
1612 if (!A.isValid())
1613 return std::nullopt;
1614 if (!A.isStringAttribute()) {
1615 Ctx.emitError(Name + " is not a string attribute");
1616 return std::nullopt;
1617 }
1618
1620
1621 StringRef S = A.getValueAsString();
1622 unsigned i = 0;
1623 for (; !S.empty() && i < Size; i++) {
1624 std::pair<StringRef, StringRef> Strs = S.split(',');
1625 unsigned IntVal;
1626 if (Strs.first.trim().getAsInteger(0, IntVal)) {
1627 Ctx.emitError("can't parse integer attribute " + Strs.first + " in " +
1628 Name);
1629 return std::nullopt;
1630 }
1631 Vals[i] = IntVal;
1632 S = Strs.second;
1633 }
1634
1635 if (!S.empty() || i < Size) {
1636 Ctx.emitError("attribute " + Name +
1637 " has incorrect number of integers; expected " +
1639 return std::nullopt;
1640 }
1641 return Vals;
1642}
1643
1645 return getIntegerVecAttribute(F, "amdgpu-max-num-workgroups", 3,
1646 std::numeric_limits<uint32_t>::max());
1647}
1648
1649bool hasValueInRangeLikeMetadata(const MDNode &MD, int64_t Val) {
1650 assert((MD.getNumOperands() % 2 == 0) && "invalid number of operands!");
1651 for (unsigned I = 0, E = MD.getNumOperands() / 2; I != E; ++I) {
1652 auto Low =
1653 mdconst::extract<ConstantInt>(MD.getOperand(2 * I + 0))->getValue();
1654 auto High =
1655 mdconst::extract<ConstantInt>(MD.getOperand(2 * I + 1))->getValue();
1656 // There are two types of [A; B) ranges:
1657 // A < B, e.g. [4; 5) which is a range that only includes 4.
1658 // A > B, e.g. [5; 4) which is a range that wraps around and includes
1659 // everything except 4.
1660 if (Low.ult(High)) {
1661 if (Low.ule(Val) && High.ugt(Val))
1662 return true;
1663 } else {
1664 if (Low.uge(Val) && High.ult(Val))
1665 return true;
1666 }
1667 }
1668
1669 return false;
1670}
1671
1673 return (1 << (getVmcntBitWidthLo(Version.Major) +
1674 getVmcntBitWidthHi(Version.Major))) -
1675 1;
1676}
1677
1679 return (1 << getLoadcntBitWidth(Version.Major)) - 1;
1680}
1681
1683 return (1 << getSamplecntBitWidth(Version.Major)) - 1;
1684}
1685
1687 return (1 << getBvhcntBitWidth(Version.Major)) - 1;
1688}
1689
1691 return (1 << getExpcntBitWidth(Version.Major)) - 1;
1692}
1693
1695 return (1 << getLgkmcntBitWidth(Version.Major)) - 1;
1696}
1697
1699 return (1 << getDscntBitWidth(Version.Major)) - 1;
1700}
1701
1703 return (1 << getKmcntBitWidth(Version.Major)) - 1;
1704}
1705
1707 return (1 << getXcntBitWidth(Version.Major, Version.Minor)) - 1;
1708}
1709
1711 return (1 << getAsynccntBitWidth(Version.Major, Version.Minor)) - 1;
1712}
1713
1715 return (1 << getStorecntBitWidth(Version.Major)) - 1;
1716}
1717
1719 unsigned VmcntLo = getBitMask(getVmcntBitShiftLo(Version.Major),
1720 getVmcntBitWidthLo(Version.Major));
1721 unsigned Expcnt = getBitMask(getExpcntBitShift(Version.Major),
1722 getExpcntBitWidth(Version.Major));
1723 unsigned Lgkmcnt = getBitMask(getLgkmcntBitShift(Version.Major),
1724 getLgkmcntBitWidth(Version.Major));
1725 unsigned VmcntHi = getBitMask(getVmcntBitShiftHi(Version.Major),
1726 getVmcntBitWidthHi(Version.Major));
1727 return VmcntLo | Expcnt | Lgkmcnt | VmcntHi;
1728}
1729
1730unsigned decodeVmcnt(const IsaVersion &Version, unsigned Waitcnt) {
1731 unsigned VmcntLo = unpackBits(Waitcnt, getVmcntBitShiftLo(Version.Major),
1732 getVmcntBitWidthLo(Version.Major));
1733 unsigned VmcntHi = unpackBits(Waitcnt, getVmcntBitShiftHi(Version.Major),
1734 getVmcntBitWidthHi(Version.Major));
1735 return VmcntLo | VmcntHi << getVmcntBitWidthLo(Version.Major);
1736}
1737
1738unsigned decodeExpcnt(const IsaVersion &Version, unsigned Waitcnt) {
1739 return unpackBits(Waitcnt, getExpcntBitShift(Version.Major),
1740 getExpcntBitWidth(Version.Major));
1741}
1742
1743unsigned decodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt) {
1744 return unpackBits(Waitcnt, getLgkmcntBitShift(Version.Major),
1745 getLgkmcntBitWidth(Version.Major));
1746}
1747
1748unsigned decodeLoadcnt(const IsaVersion &Version, unsigned Waitcnt) {
1749 return unpackBits(Waitcnt, getLoadcntStorecntBitShift(Version.Major),
1750 getLoadcntBitWidth(Version.Major));
1751}
1752
1753unsigned decodeStorecnt(const IsaVersion &Version, unsigned Waitcnt) {
1754 return unpackBits(Waitcnt, getLoadcntStorecntBitShift(Version.Major),
1755 getStorecntBitWidth(Version.Major));
1756}
1757
1758unsigned decodeDscnt(const IsaVersion &Version, unsigned Waitcnt) {
1759 return unpackBits(Waitcnt, getDscntBitShift(Version.Major),
1760 getDscntBitWidth(Version.Major));
1761}
1762
1763void decodeWaitcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned &Vmcnt,
1764 unsigned &Expcnt, unsigned &Lgkmcnt) {
1765 Vmcnt = decodeVmcnt(Version, Waitcnt);
1766 Expcnt = decodeExpcnt(Version, Waitcnt);
1767 Lgkmcnt = decodeLgkmcnt(Version, Waitcnt);
1768}
1769
1770unsigned encodeVmcnt(const IsaVersion &Version, unsigned Waitcnt,
1771 unsigned Vmcnt) {
1772 Waitcnt = packBits(Vmcnt, Waitcnt, getVmcntBitShiftLo(Version.Major),
1773 getVmcntBitWidthLo(Version.Major));
1774 return packBits(Vmcnt >> getVmcntBitWidthLo(Version.Major), Waitcnt,
1775 getVmcntBitShiftHi(Version.Major),
1776 getVmcntBitWidthHi(Version.Major));
1777}
1778
1779unsigned encodeExpcnt(const IsaVersion &Version, unsigned Waitcnt,
1780 unsigned Expcnt) {
1781 return packBits(Expcnt, Waitcnt, getExpcntBitShift(Version.Major),
1782 getExpcntBitWidth(Version.Major));
1783}
1784
1785unsigned encodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt,
1786 unsigned Lgkmcnt) {
1787 return packBits(Lgkmcnt, Waitcnt, getLgkmcntBitShift(Version.Major),
1788 getLgkmcntBitWidth(Version.Major));
1789}
1790
1791unsigned encodeWaitcnt(const IsaVersion &Version, unsigned Vmcnt,
1792 unsigned Expcnt, unsigned Lgkmcnt) {
1793 unsigned Waitcnt = getWaitcntBitMask(Version);
1795 Waitcnt = encodeExpcnt(Version, Waitcnt, Expcnt);
1796 Waitcnt = encodeLgkmcnt(Version, Waitcnt, Lgkmcnt);
1797 return Waitcnt;
1798}
1799
1801 bool IsStore) {
1802 unsigned Dscnt = getBitMask(getDscntBitShift(Version.Major),
1803 getDscntBitWidth(Version.Major));
1804 if (IsStore) {
1805 unsigned Storecnt = getBitMask(getLoadcntStorecntBitShift(Version.Major),
1806 getStorecntBitWidth(Version.Major));
1807 return Dscnt | Storecnt;
1808 }
1809 unsigned Loadcnt = getBitMask(getLoadcntStorecntBitShift(Version.Major),
1810 getLoadcntBitWidth(Version.Major));
1811 return Dscnt | Loadcnt;
1812}
1813
1814static unsigned encodeLoadcnt(const IsaVersion &Version, unsigned Waitcnt,
1815 unsigned Loadcnt) {
1816 return packBits(Loadcnt, Waitcnt, getLoadcntStorecntBitShift(Version.Major),
1817 getLoadcntBitWidth(Version.Major));
1818}
1819
1820static unsigned encodeStorecnt(const IsaVersion &Version, unsigned Waitcnt,
1821 unsigned Storecnt) {
1822 return packBits(Storecnt, Waitcnt, getLoadcntStorecntBitShift(Version.Major),
1823 getStorecntBitWidth(Version.Major));
1824}
1825
1826static unsigned encodeDscnt(const IsaVersion &Version, unsigned Waitcnt,
1827 unsigned Dscnt) {
1828 return packBits(Dscnt, Waitcnt, getDscntBitShift(Version.Major),
1829 getDscntBitWidth(Version.Major));
1830}
1831
1832unsigned encodeLoadcntDscnt(const IsaVersion &Version, unsigned Loadcnt,
1833 unsigned Dscnt) {
1834 unsigned Waitcnt = getCombinedCountBitMask(Version, false);
1835 Waitcnt = encodeLoadcnt(Version, Waitcnt, Loadcnt);
1837 return Waitcnt;
1838}
1839
1840unsigned encodeStorecntDscnt(const IsaVersion &Version, unsigned Storecnt,
1841 unsigned Dscnt) {
1842 unsigned Waitcnt = getCombinedCountBitMask(Version, true);
1843 Waitcnt = encodeStorecnt(Version, Waitcnt, Storecnt);
1845 return Waitcnt;
1846}
1847
1848//===----------------------------------------------------------------------===//
1849// Custom Operand Values
1850//===----------------------------------------------------------------------===//
1851
1853 int Size,
1854 const MCSubtargetInfo &STI) {
1855 unsigned Enc = 0;
1856 for (int Idx = 0; Idx < Size; ++Idx) {
1857 const auto &Op = Opr[Idx];
1858 if (Op.isSupported(STI))
1859 Enc |= Op.encode(Op.Default);
1860 }
1861 return Enc;
1862}
1863
1865 int Size, unsigned Code,
1866 bool &HasNonDefaultVal,
1867 const MCSubtargetInfo &STI) {
1868 unsigned UsedOprMask = 0;
1869 HasNonDefaultVal = false;
1870 for (int Idx = 0; Idx < Size; ++Idx) {
1871 const auto &Op = Opr[Idx];
1872 if (!Op.isSupported(STI))
1873 continue;
1874 UsedOprMask |= Op.getMask();
1875 unsigned Val = Op.decode(Code);
1876 if (!Op.isValid(Val))
1877 return false;
1878 HasNonDefaultVal |= (Val != Op.Default);
1879 }
1880 return (Code & ~UsedOprMask) == 0;
1881}
1882
1883static bool decodeCustomOperand(const CustomOperandVal *Opr, int Size,
1884 unsigned Code, int &Idx, StringRef &Name,
1885 unsigned &Val, bool &IsDefault,
1886 const MCSubtargetInfo &STI) {
1887 while (Idx < Size) {
1888 const auto &Op = Opr[Idx++];
1889 if (Op.isSupported(STI)) {
1890 Name = Op.Name;
1891 Val = Op.decode(Code);
1892 IsDefault = (Val == Op.Default);
1893 return true;
1894 }
1895 }
1896
1897 return false;
1898}
1899
1901 int64_t InputVal) {
1902 if (InputVal < 0 || InputVal > Op.Max)
1903 return OPR_VAL_INVALID;
1904 return Op.encode(InputVal);
1905}
1906
1907static int encodeCustomOperand(const CustomOperandVal *Opr, int Size,
1908 const StringRef Name, int64_t InputVal,
1909 unsigned &UsedOprMask,
1910 const MCSubtargetInfo &STI) {
1911 int InvalidId = OPR_ID_UNKNOWN;
1912 for (int Idx = 0; Idx < Size; ++Idx) {
1913 const auto &Op = Opr[Idx];
1914 if (Op.Name == Name) {
1915 if (!Op.isSupported(STI)) {
1916 InvalidId = OPR_ID_UNSUPPORTED;
1917 continue;
1918 }
1919 auto OprMask = Op.getMask();
1920 if (OprMask & UsedOprMask)
1921 return OPR_ID_DUPLICATE;
1922 UsedOprMask |= OprMask;
1923 return encodeCustomOperandVal(Op, InputVal);
1924 }
1925 }
1926 return InvalidId;
1927}
1928
1929//===----------------------------------------------------------------------===//
1930// DepCtr
1931//===----------------------------------------------------------------------===//
1932
1933namespace DepCtr {
1934
1936 static int Default = -1;
1937 if (Default == -1)
1939 return Default;
1940}
1941
1942bool isSymbolicDepCtrEncoding(unsigned Code, bool &HasNonDefaultVal,
1943 const MCSubtargetInfo &STI) {
1945 HasNonDefaultVal, STI);
1946}
1947
1948bool decodeDepCtr(unsigned Code, int &Id, StringRef &Name, unsigned &Val,
1949 bool &IsDefault, const MCSubtargetInfo &STI) {
1950 return decodeCustomOperand(DepCtrInfo, DEP_CTR_SIZE, Code, Id, Name, Val,
1951 IsDefault, STI);
1952}
1953
1954int encodeDepCtr(const StringRef Name, int64_t Val, unsigned &UsedOprMask,
1955 const MCSubtargetInfo &STI) {
1956 return encodeCustomOperand(DepCtrInfo, DEP_CTR_SIZE, Name, Val, UsedOprMask,
1957 STI);
1958}
1959
1960unsigned getVaVdstBitMask() { return (1 << getVaVdstBitWidth()) - 1; }
1961
1962unsigned getVaSdstBitMask() { return (1 << getVaSdstBitWidth()) - 1; }
1963
1964unsigned getVaSsrcBitMask() { return (1 << getVaSsrcBitWidth()) - 1; }
1965
1967 return (1 << getHoldCntWidth(Version.Major, Version.Minor)) - 1;
1968}
1969
1970unsigned getVmVsrcBitMask() { return (1 << getVmVsrcBitWidth()) - 1; }
1971
1972unsigned getVaVccBitMask() { return (1 << getVaVccBitWidth()) - 1; }
1973
1974unsigned getSaSdstBitMask() { return (1 << getSaSdstBitWidth()) - 1; }
1975
1976unsigned decodeFieldVmVsrc(unsigned Encoded) {
1977 return unpackBits(Encoded, getVmVsrcBitShift(), getVmVsrcBitWidth());
1978}
1979
1980unsigned decodeFieldVaVdst(unsigned Encoded) {
1981 return unpackBits(Encoded, getVaVdstBitShift(), getVaVdstBitWidth());
1982}
1983
1984unsigned decodeFieldSaSdst(unsigned Encoded) {
1985 return unpackBits(Encoded, getSaSdstBitShift(), getSaSdstBitWidth());
1986}
1987
1988unsigned decodeFieldVaSdst(unsigned Encoded) {
1989 return unpackBits(Encoded, getVaSdstBitShift(), getVaSdstBitWidth());
1990}
1991
1992unsigned decodeFieldVaVcc(unsigned Encoded) {
1993 return unpackBits(Encoded, getVaVccBitShift(), getVaVccBitWidth());
1994}
1995
1996unsigned decodeFieldVaSsrc(unsigned Encoded) {
1997 return unpackBits(Encoded, getVaSsrcBitShift(), getVaSsrcBitWidth());
1998}
1999
2000unsigned decodeFieldHoldCnt(unsigned Encoded, const IsaVersion &Version) {
2001 return unpackBits(Encoded, getHoldCntBitShift(),
2002 getHoldCntWidth(Version.Major, Version.Minor));
2003}
2004
2005unsigned encodeFieldVmVsrc(unsigned Encoded, unsigned VmVsrc) {
2006 return packBits(VmVsrc, Encoded, getVmVsrcBitShift(), getVmVsrcBitWidth());
2007}
2008
2009unsigned encodeFieldVmVsrc(unsigned VmVsrc, const MCSubtargetInfo &STI) {
2010 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2011 return encodeFieldVmVsrc(Encoded, VmVsrc);
2012}
2013
2014unsigned encodeFieldVaVdst(unsigned Encoded, unsigned VaVdst) {
2015 return packBits(VaVdst, Encoded, getVaVdstBitShift(), getVaVdstBitWidth());
2016}
2017
2018unsigned encodeFieldVaVdst(unsigned VaVdst, const MCSubtargetInfo &STI) {
2019 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2020 return encodeFieldVaVdst(Encoded, VaVdst);
2021}
2022
2023unsigned encodeFieldSaSdst(unsigned Encoded, unsigned SaSdst) {
2024 return packBits(SaSdst, Encoded, getSaSdstBitShift(), getSaSdstBitWidth());
2025}
2026
2027unsigned encodeFieldSaSdst(unsigned SaSdst, const MCSubtargetInfo &STI) {
2028 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2029 return encodeFieldSaSdst(Encoded, SaSdst);
2030}
2031
2032unsigned encodeFieldVaSdst(unsigned Encoded, unsigned VaSdst) {
2033 return packBits(VaSdst, Encoded, getVaSdstBitShift(), getVaSdstBitWidth());
2034}
2035
2036unsigned encodeFieldVaSdst(unsigned VaSdst, const MCSubtargetInfo &STI) {
2037 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2038 return encodeFieldVaSdst(Encoded, VaSdst);
2039}
2040
2041unsigned encodeFieldVaVcc(unsigned Encoded, unsigned VaVcc) {
2042 return packBits(VaVcc, Encoded, getVaVccBitShift(), getVaVccBitWidth());
2043}
2044
2045unsigned encodeFieldVaVcc(unsigned VaVcc, const MCSubtargetInfo &STI) {
2046 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2047 return encodeFieldVaVcc(Encoded, VaVcc);
2048}
2049
2050unsigned encodeFieldVaSsrc(unsigned Encoded, unsigned VaSsrc) {
2051 return packBits(VaSsrc, Encoded, getVaSsrcBitShift(), getVaSsrcBitWidth());
2052}
2053
2054unsigned encodeFieldVaSsrc(unsigned VaSsrc, const MCSubtargetInfo &STI) {
2055 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2056 return encodeFieldVaSsrc(Encoded, VaSsrc);
2057}
2058
2059unsigned encodeFieldHoldCnt(unsigned Encoded, unsigned HoldCnt,
2060 const IsaVersion &Version) {
2061 return packBits(HoldCnt, Encoded, getHoldCntBitShift(),
2062 getHoldCntWidth(Version.Major, Version.Minor));
2063}
2064
2065unsigned encodeFieldHoldCnt(unsigned HoldCnt, const MCSubtargetInfo &STI) {
2066 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2067 return encodeFieldHoldCnt(Encoded, HoldCnt, getIsaVersion(STI.getCPU()));
2068}
2069
2070} // namespace DepCtr
2071
2072//===----------------------------------------------------------------------===//
2073// exp tgt
2074//===----------------------------------------------------------------------===//
2075
2076namespace Exp {
2077
2078struct ExpTgt {
2080 unsigned Tgt;
2081 unsigned MaxIndex;
2082};
2083
2084// clang-format off
2085static constexpr ExpTgt ExpTgtInfo[] = {
2086 {{"null"}, ET_NULL, ET_NULL_MAX_IDX},
2087 {{"mrtz"}, ET_MRTZ, ET_MRTZ_MAX_IDX},
2088 {{"prim"}, ET_PRIM, ET_PRIM_MAX_IDX},
2089 {{"mrt"}, ET_MRT0, ET_MRT_MAX_IDX},
2090 {{"pos"}, ET_POS0, ET_POS_MAX_IDX},
2091 {{"dual_src_blend"},ET_DUAL_SRC_BLEND0, ET_DUAL_SRC_BLEND_MAX_IDX},
2092 {{"param"}, ET_PARAM0, ET_PARAM_MAX_IDX},
2093};
2094// clang-format on
2095
2096bool getTgtName(unsigned Id, StringRef &Name, int &Index) {
2097 for (const ExpTgt &Val : ExpTgtInfo) {
2098 if (Val.Tgt <= Id && Id <= Val.Tgt + Val.MaxIndex) {
2099 Index = (Val.MaxIndex == 0) ? -1 : (Id - Val.Tgt);
2100 Name = Val.Name;
2101 return true;
2102 }
2103 }
2104 return false;
2105}
2106
2107unsigned getTgtId(const StringRef Name) {
2108
2109 for (const ExpTgt &Val : ExpTgtInfo) {
2110 if (Val.MaxIndex == 0 && Name == Val.Name)
2111 return Val.Tgt;
2112
2113 if (Val.MaxIndex > 0 && Name.starts_with(Val.Name)) {
2114 StringRef Suffix = Name.drop_front(Val.Name.size());
2115
2116 unsigned Id;
2117 if (Suffix.getAsInteger(10, Id) || Id > Val.MaxIndex)
2118 return ET_INVALID;
2119
2120 // Disable leading zeroes
2121 if (Suffix.size() > 1 && Suffix[0] == '0')
2122 return ET_INVALID;
2123
2124 return Val.Tgt + Id;
2125 }
2126 }
2127 return ET_INVALID;
2128}
2129
2130bool isSupportedTgtId(unsigned Id, const MCSubtargetInfo &STI) {
2131 switch (Id) {
2132 case ET_NULL:
2133 return !isGFX11Plus(STI);
2134 case ET_POS4:
2135 case ET_PRIM:
2136 return isGFX10Plus(STI);
2137 case ET_DUAL_SRC_BLEND0:
2138 case ET_DUAL_SRC_BLEND1:
2139 return isGFX11Plus(STI);
2140 default:
2141 if (Id >= ET_PARAM0 && Id <= ET_PARAM31)
2142 return !isGFX11Plus(STI) || isGFX13Plus(STI);
2143 return true;
2144 }
2145}
2146
2147} // namespace Exp
2148
2149//===----------------------------------------------------------------------===//
2150// MTBUF Format
2151//===----------------------------------------------------------------------===//
2152
2153namespace MTBUFFormat {
2154
2155int64_t getDfmt(const StringRef Name) {
2156 for (int Id = DFMT_MIN; Id <= DFMT_MAX; ++Id) {
2157 if (Name == DfmtSymbolic[Id])
2158 return Id;
2159 }
2160 return DFMT_UNDEF;
2161}
2162
2164 assert(Id <= DFMT_MAX);
2165 return DfmtSymbolic[Id];
2166}
2167
2169 if (isSI(STI) || isCI(STI))
2170 return NfmtSymbolicSICI;
2171 if (isVI(STI) || isGFX9(STI))
2172 return NfmtSymbolicVI;
2173 return NfmtSymbolicGFX10;
2174}
2175
2176int64_t getNfmt(const StringRef Name, const MCSubtargetInfo &STI) {
2177 const auto *lookupTable = getNfmtLookupTable(STI);
2178 for (int Id = NFMT_MIN; Id <= NFMT_MAX; ++Id) {
2179 if (Name == lookupTable[Id])
2180 return Id;
2181 }
2182 return NFMT_UNDEF;
2183}
2184
2185StringRef getNfmtName(unsigned Id, const MCSubtargetInfo &STI) {
2186 assert(Id <= NFMT_MAX);
2187 return getNfmtLookupTable(STI)[Id];
2188}
2189
2190bool isValidDfmtNfmt(unsigned Id, const MCSubtargetInfo &STI) {
2191 unsigned Dfmt;
2192 unsigned Nfmt;
2193 decodeDfmtNfmt(Id, Dfmt, Nfmt);
2194 return isValidNfmt(Nfmt, STI);
2195}
2196
2197bool isValidNfmt(unsigned Id, const MCSubtargetInfo &STI) {
2198 return !getNfmtName(Id, STI).empty();
2199}
2200
2201int64_t encodeDfmtNfmt(unsigned Dfmt, unsigned Nfmt) {
2202 return (Dfmt << DFMT_SHIFT) | (Nfmt << NFMT_SHIFT);
2203}
2204
2205void decodeDfmtNfmt(unsigned Format, unsigned &Dfmt, unsigned &Nfmt) {
2206 Dfmt = (Format >> DFMT_SHIFT) & DFMT_MASK;
2207 Nfmt = (Format >> NFMT_SHIFT) & NFMT_MASK;
2208}
2209
2210int64_t getUnifiedFormat(const StringRef Name, const MCSubtargetInfo &STI) {
2211 if (isGFX11Plus(STI)) {
2212 for (int Id = UfmtGFX11::UFMT_FIRST; Id <= UfmtGFX11::UFMT_LAST; ++Id) {
2213 if (Name == UfmtSymbolicGFX11[Id])
2214 return Id;
2215 }
2216 } else {
2217 for (int Id = UfmtGFX10::UFMT_FIRST; Id <= UfmtGFX10::UFMT_LAST; ++Id) {
2218 if (Name == UfmtSymbolicGFX10[Id])
2219 return Id;
2220 }
2221 }
2222 return UFMT_UNDEF;
2223}
2224
2226 if (isValidUnifiedFormat(Id, STI))
2227 return isGFX10(STI) ? UfmtSymbolicGFX10[Id] : UfmtSymbolicGFX11[Id];
2228 return "";
2229}
2230
2231bool isValidUnifiedFormat(unsigned Id, const MCSubtargetInfo &STI) {
2232 return isGFX10(STI) ? Id <= UfmtGFX10::UFMT_LAST : Id <= UfmtGFX11::UFMT_LAST;
2233}
2234
2235int64_t convertDfmtNfmt2Ufmt(unsigned Dfmt, unsigned Nfmt,
2236 const MCSubtargetInfo &STI) {
2237 int64_t Fmt = encodeDfmtNfmt(Dfmt, Nfmt);
2238 if (isGFX11Plus(STI)) {
2239 for (int Id = UfmtGFX11::UFMT_FIRST; Id <= UfmtGFX11::UFMT_LAST; ++Id) {
2240 if (Fmt == DfmtNfmt2UFmtGFX11[Id])
2241 return Id;
2242 }
2243 } else {
2244 for (int Id = UfmtGFX10::UFMT_FIRST; Id <= UfmtGFX10::UFMT_LAST; ++Id) {
2245 if (Fmt == DfmtNfmt2UFmtGFX10[Id])
2246 return Id;
2247 }
2248 }
2249 return UFMT_UNDEF;
2250}
2251
2252bool isValidFormatEncoding(unsigned Val, const MCSubtargetInfo &STI) {
2253 return isGFX10Plus(STI) ? (Val <= UFMT_MAX) : (Val <= DFMT_NFMT_MAX);
2254}
2255
2257 if (isGFX10Plus(STI))
2258 return UFMT_DEFAULT;
2259 return DFMT_NFMT_DEFAULT;
2260}
2261
2262} // namespace MTBUFFormat
2263
2264//===----------------------------------------------------------------------===//
2265// SendMsg
2266//===----------------------------------------------------------------------===//
2267
2268namespace SendMsg {
2269
2273
2274bool isValidMsgId(int64_t MsgId, const MCSubtargetInfo &STI) {
2275 return (MsgId & ~(getMsgIdMask(STI))) == 0;
2276}
2277
2278bool isValidMsgOp(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI,
2279 bool Strict) {
2280 assert(isValidMsgId(MsgId, STI));
2281
2282 if (!Strict)
2283 return 0 <= OpId && isUInt<OP_WIDTH_>(OpId);
2284
2285 if (msgRequiresOp(MsgId, STI)) {
2286 if (MsgId == ID_GS_PreGFX11 && OpId == OP_GS_NOP)
2287 return false;
2288
2289 return !getMsgOpName(MsgId, OpId, STI).empty();
2290 }
2291
2292 return OpId == OP_NONE_;
2293}
2294
2295bool isValidMsgStream(int64_t MsgId, int64_t OpId, int64_t StreamId,
2296 const MCSubtargetInfo &STI, bool Strict) {
2297 assert(isValidMsgOp(MsgId, OpId, STI, Strict));
2298
2299 if (!Strict)
2301
2302 if (!isGFX11Plus(STI)) {
2303 switch (MsgId) {
2304 case ID_GS_PreGFX11:
2307 return (OpId == OP_GS_NOP)
2310 }
2311 }
2312 return StreamId == STREAM_ID_NONE_;
2313}
2314
2315bool msgRequiresOp(int64_t MsgId, const MCSubtargetInfo &STI) {
2316 return MsgId == ID_SYSMSG ||
2317 (!isGFX11Plus(STI) &&
2318 (MsgId == ID_GS_PreGFX11 || MsgId == ID_GS_DONE_PreGFX11));
2319}
2320
2321bool msgSupportsStream(int64_t MsgId, int64_t OpId,
2322 const MCSubtargetInfo &STI) {
2323 return !isGFX11Plus(STI) &&
2324 (MsgId == ID_GS_PreGFX11 || MsgId == ID_GS_DONE_PreGFX11) &&
2325 OpId != OP_GS_NOP;
2326}
2327
2328void decodeMsg(unsigned Val, uint16_t &MsgId, uint16_t &OpId,
2329 uint16_t &StreamId, const MCSubtargetInfo &STI) {
2330 MsgId = Val & getMsgIdMask(STI);
2331 if (isGFX11Plus(STI)) {
2332 OpId = 0;
2333 StreamId = 0;
2334 } else {
2335 OpId = (Val & OP_MASK_) >> OP_SHIFT_;
2337 }
2338}
2339
2341 return MsgId | (OpId << OP_SHIFT_) | (StreamId << STREAM_ID_SHIFT_);
2342}
2343
2344bool msgDoesNotUseM0(int64_t MsgId, const MCSubtargetInfo &STI) {
2345 // Explicitly list message types that are known to not use m0.
2346 // This is safer than excluding only GS_ALLOC_REQ, in case new message
2347 // types are added in the future that do use m0.
2348 if (isGFX11Plus(STI)) {
2349 switch (MsgId) {
2351 return true;
2352 default:
2353 break;
2354 }
2355 }
2356 switch (MsgId) {
2357 case ID_SAVEWAVE:
2358 case ID_STALL_WAVE_GEN:
2359 case ID_HALT_WAVES:
2360 case ID_ORDERED_PS_DONE:
2362 case ID_GET_DOORBELL:
2363 case ID_GET_DDID:
2364 case ID_SYSMSG:
2365 return true;
2366 default:
2367 return false;
2368 }
2369}
2370
2371} // namespace SendMsg
2372
2373//===----------------------------------------------------------------------===//
2374//
2375//===----------------------------------------------------------------------===//
2376
2378 return F.getFnAttributeAsParsedInteger("InitialPSInputAddr", 0);
2379}
2380
2382 // As a safe default always respond as if PS has color exports.
2383 return F.getFnAttributeAsParsedInteger(
2384 "amdgpu-color-export",
2385 F.getCallingConv() == CallingConv::AMDGPU_PS ? 1 : 0) != 0;
2386}
2387
2389 return F.getFnAttributeAsParsedInteger("amdgpu-depth-export", 0) != 0;
2390}
2391
2393 unsigned BlockSize =
2394 F.getFnAttributeAsParsedInteger("amdgpu-dynamic-vgpr-block-size", 0);
2395
2396 if (BlockSize == 16 || BlockSize == 32)
2397 return BlockSize;
2398
2399 return 0;
2400}
2401
2402bool hasXNACK(const MCSubtargetInfo &STI) {
2403 // Only hardwired-on xnack (gfx1250) is knowable from the subtarget alone;
2404 // toggleable targets take their mode from the TargetID.
2405 return STI.hasFeature(AMDGPU::FeatureSupportsXNACK) &&
2406 !STI.hasFeature(AMDGPU::FeatureXNACKOnOffModes);
2407}
2408
2410 return STI.hasFeature(AMDGPU::FeatureMIMG_R128) &&
2411 !STI.hasFeature(AMDGPU::FeatureR128A16);
2412}
2413
2414bool hasA16(const MCSubtargetInfo &STI) {
2415 return STI.hasFeature(AMDGPU::FeatureA16);
2416}
2417
2418bool hasG16(const MCSubtargetInfo &STI) {
2419 return STI.hasFeature(AMDGPU::FeatureG16);
2420}
2421
2423 return !STI.hasFeature(AMDGPU::FeatureUnpackedD16VMem) && !isCI(STI) &&
2424 !isSI(STI);
2425}
2426
2427bool hasGDS(const MCSubtargetInfo &STI) {
2428 return STI.hasFeature(AMDGPU::FeatureGDS);
2429}
2430
2431unsigned getNSAMaxSize(const MCSubtargetInfo &STI, bool HasSampler) {
2432 auto Version = getIsaVersion(STI.getCPU());
2433 if (Version.Major == 10)
2434 return Version.Minor >= 3 ? 13 : 5;
2435 if (Version.Major == 11)
2436 return 5;
2437 if (Version.Major >= 12)
2438 return HasSampler ? 4 : 5;
2439 return 0;
2440}
2441
2443 if (isGFX1250Plus(STI))
2444 return 32;
2445 return 16;
2446}
2447
2448bool isSI(const MCSubtargetInfo &STI) {
2449 return STI.hasFeature(AMDGPU::FeatureSouthernIslands);
2450}
2451
2452bool isCI(const MCSubtargetInfo &STI) {
2453 return STI.hasFeature(AMDGPU::FeatureSeaIslands);
2454}
2455
2456bool isVI(const MCSubtargetInfo &STI) {
2457 return STI.hasFeature(AMDGPU::FeatureVolcanicIslands);
2458}
2459
2460bool isGFX9(const MCSubtargetInfo &STI) {
2461 return STI.hasFeature(AMDGPU::FeatureGFX9);
2462}
2463
2465 return isGFX9(STI) || isGFX10(STI);
2466}
2467
2469 return isGFX9(STI) || isGFX10(STI) || isGFX11(STI);
2470}
2471
2473 return isVI(STI) || isGFX9(STI) || isGFX10(STI);
2474}
2475
2476bool isGFX8Plus(const MCSubtargetInfo &STI) {
2477 return isVI(STI) || isGFX9Plus(STI);
2478}
2479
2480bool isGFX9Plus(const MCSubtargetInfo &STI) {
2481 return isGFX9(STI) || isGFX10Plus(STI);
2482}
2483
2484bool isNotGFX9Plus(const MCSubtargetInfo &STI) { return !isGFX9Plus(STI); }
2485
2487 return STI.hasFeature(AMDGPU::FeaturePopsExitingWaveID);
2488}
2489
2491 return STI.hasFeature(AMDGPU::FeatureApertureRegs) &&
2492 !STI.hasFeature(AMDGPU::FeatureGloballyAddressableScratch);
2493}
2494
2495bool isGFX10(const MCSubtargetInfo &STI) {
2496 return STI.hasFeature(AMDGPU::FeatureGFX10);
2497}
2498
2500 return isGFX10(STI) || isGFX11(STI);
2501}
2502
2504 return isGFX10(STI) || isGFX11Plus(STI);
2505}
2506
2507bool isGFX11(const MCSubtargetInfo &STI) {
2508 return STI.hasFeature(AMDGPU::FeatureGFX11);
2509}
2510
2512 return isGFX11(STI) || isGFX12Plus(STI);
2513}
2514
2515bool isGFX12(const MCSubtargetInfo &STI) {
2516 return STI.getFeatureBits()[AMDGPU::FeatureGFX12];
2517}
2518
2520 return isGFX12(STI) || isGFX13Plus(STI);
2521}
2522
2523bool isNotGFX12Plus(const MCSubtargetInfo &STI) { return !isGFX12Plus(STI); }
2524
2525bool isGFX1250(const MCSubtargetInfo &STI) {
2526 return STI.getFeatureBits()[AMDGPU::FeatureGFX1250Insts] && !isGFX13(STI);
2527}
2528
2530 return isGFX1250(STI) || !STI.getFeatureBits().test(FeatureCuMode);
2531}
2532
2534 return STI.getFeatureBits()[AMDGPU::FeatureGFX1250Insts];
2535}
2536
2537bool isGFX13(const MCSubtargetInfo &STI) {
2538 return STI.getFeatureBits()[AMDGPU::FeatureGFX13];
2539}
2540
2541bool isGFX13Plus(const MCSubtargetInfo &STI) { return isGFX13(STI); }
2542
2544 if (isGFX1250(STI))
2545 return false;
2546 return isGFX10Plus(STI);
2547}
2548
2549bool isNotGFX11Plus(const MCSubtargetInfo &STI) { return !isGFX11Plus(STI); }
2550
2552 return isSI(STI) || isCI(STI) || isVI(STI) || isGFX9(STI);
2553}
2554
2556 return isGFX10(STI) && !AMDGPU::isGFX10_BEncoding(STI);
2557}
2558
2560 return STI.hasFeature(AMDGPU::FeatureGCN3Encoding);
2561}
2562
2564 return STI.hasFeature(AMDGPU::FeatureGFX10_BEncoding);
2565}
2566
2568 return STI.hasFeature(AMDGPU::FeatureGFX10_3Insts);
2569}
2570
2572 return isGFX10_BEncoding(STI) && !isGFX12Plus(STI);
2573}
2574
2575bool isGFX90A(const MCSubtargetInfo &STI) {
2576 return STI.hasFeature(AMDGPU::FeatureGFX90AInsts);
2577}
2578
2579bool isGFX940(const MCSubtargetInfo &STI) {
2580 return STI.hasFeature(AMDGPU::FeatureGFX940Insts);
2581}
2582
2584 return STI.hasFeature(AMDGPU::FeatureArchitectedFlatScratch);
2585}
2586
2588 return STI.hasFeature(AMDGPU::FeatureMAIInsts);
2589}
2590
2591bool hasVOPD(const MCSubtargetInfo &STI) {
2592 return STI.hasFeature(AMDGPU::FeatureVOPDInsts);
2593}
2594
2596 return STI.hasFeature(AMDGPU::FeatureDPPSrc1SGPR);
2597}
2598
2600 return STI.hasFeature(AMDGPU::FeatureKernargPreload);
2601}
2602
2603int32_t getTotalNumVGPRs(bool has90AInsts, int32_t ArgNumAGPR,
2604 int32_t ArgNumVGPR) {
2605 if (has90AInsts && ArgNumAGPR)
2606 return alignTo(ArgNumVGPR, 4) + ArgNumAGPR;
2607 return std::max(ArgNumVGPR, ArgNumAGPR);
2608}
2609
2611 const MCRegisterClass &SGPRClass =
2612 TRI->getRegClass(AMDGPU::SReg_32RegClassID);
2613 const MCRegister FirstSubReg = TRI->getSubReg(Reg, AMDGPU::sub0);
2614 return SGPRClass.contains(FirstSubReg != 0 ? FirstSubReg : Reg) ||
2615 Reg == AMDGPU::SCC;
2616}
2617
2619 return MRI.getRegClass(AMDGPU::RsrcReg32RegClassID).contains(Reg);
2620}
2621
2625
2626#define MAP_REG2REG \
2627 using namespace AMDGPU; \
2628 switch (Reg.id()) { \
2629 default: \
2630 return Reg; \
2631 CASE_CI_VI(FLAT_SCR) \
2632 CASE_CI_VI(FLAT_SCR_LO) \
2633 CASE_CI_VI(FLAT_SCR_HI) \
2634 CASE_VI_GFX9PLUS(TTMP0) \
2635 CASE_VI_GFX9PLUS(TTMP1) \
2636 CASE_VI_GFX9PLUS(TTMP2) \
2637 CASE_VI_GFX9PLUS(TTMP3) \
2638 CASE_VI_GFX9PLUS(TTMP4) \
2639 CASE_VI_GFX9PLUS(TTMP5) \
2640 CASE_VI_GFX9PLUS(TTMP6) \
2641 CASE_VI_GFX9PLUS(TTMP7) \
2642 CASE_VI_GFX9PLUS(TTMP8) \
2643 CASE_VI_GFX9PLUS(TTMP9) \
2644 CASE_VI_GFX9PLUS(TTMP10) \
2645 CASE_VI_GFX9PLUS(TTMP11) \
2646 CASE_VI_GFX9PLUS(TTMP12) \
2647 CASE_VI_GFX9PLUS(TTMP13) \
2648 CASE_VI_GFX9PLUS(TTMP14) \
2649 CASE_VI_GFX9PLUS(TTMP15) \
2650 CASE_VI_GFX9PLUS(TTMP0_TTMP1) \
2651 CASE_VI_GFX9PLUS(TTMP2_TTMP3) \
2652 CASE_VI_GFX9PLUS(TTMP4_TTMP5) \
2653 CASE_VI_GFX9PLUS(TTMP6_TTMP7) \
2654 CASE_VI_GFX9PLUS(TTMP8_TTMP9) \
2655 CASE_VI_GFX9PLUS(TTMP10_TTMP11) \
2656 CASE_VI_GFX9PLUS(TTMP12_TTMP13) \
2657 CASE_VI_GFX9PLUS(TTMP14_TTMP15) \
2658 CASE_VI_GFX9PLUS(TTMP0_TTMP1_TTMP2_TTMP3) \
2659 CASE_VI_GFX9PLUS(TTMP4_TTMP5_TTMP6_TTMP7) \
2660 CASE_VI_GFX9PLUS(TTMP8_TTMP9_TTMP10_TTMP11) \
2661 CASE_VI_GFX9PLUS(TTMP12_TTMP13_TTMP14_TTMP15) \
2662 CASE_VI_GFX9PLUS(TTMP0_TTMP1_TTMP2_TTMP3_TTMP4_TTMP5_TTMP6_TTMP7) \
2663 CASE_VI_GFX9PLUS(TTMP4_TTMP5_TTMP6_TTMP7_TTMP8_TTMP9_TTMP10_TTMP11) \
2664 CASE_VI_GFX9PLUS(TTMP8_TTMP9_TTMP10_TTMP11_TTMP12_TTMP13_TTMP14_TTMP15) \
2665 CASE_VI_GFX9PLUS( \
2666 TTMP0_TTMP1_TTMP2_TTMP3_TTMP4_TTMP5_TTMP6_TTMP7_TTMP8_TTMP9_TTMP10_TTMP11_TTMP12_TTMP13_TTMP14_TTMP15) \
2667 CASE_GFXPRE11_GFX11PLUS(M0) \
2668 CASE_GFXPRE11_GFX11PLUS(SGPR_NULL) \
2669 CASE_GFXPRE11_GFX11PLUS_TO(SGPR_NULL64, SGPR_NULL) \
2670 }
2671
2672#define CASE_CI_VI(node) \
2673 assert(!isSI(STI)); \
2674 case node: \
2675 return isCI(STI) ? node##_ci : node##_vi;
2676
2677#define CASE_VI_GFX9PLUS(node) \
2678 case node: \
2679 return isGFX9Plus(STI) ? node##_gfx9plus : node##_vi;
2680
2681#define CASE_GFXPRE11_GFX11PLUS(node) \
2682 case node: \
2683 return isGFX11Plus(STI) ? node##_gfx11plus : node##_gfxpre11;
2684
2685#define CASE_GFXPRE11_GFX11PLUS_TO(node, result) \
2686 case node: \
2687 return isGFX11Plus(STI) ? result##_gfx11plus : result##_gfxpre11;
2688
2690 if (STI.getTargetTriple().getArch() == Triple::r600)
2691 return Reg;
2693}
2694
2695#undef CASE_CI_VI
2696#undef CASE_VI_GFX9PLUS
2697#undef CASE_GFXPRE11_GFX11PLUS
2698#undef CASE_GFXPRE11_GFX11PLUS_TO
2699
2700#define CASE_CI_VI(node) \
2701 case node##_ci: \
2702 case node##_vi: \
2703 return node;
2704#define CASE_VI_GFX9PLUS(node) \
2705 case node##_vi: \
2706 case node##_gfx9plus: \
2707 return node;
2708#define CASE_GFXPRE11_GFX11PLUS(node) \
2709 case node##_gfx11plus: \
2710 case node##_gfxpre11: \
2711 return node;
2712#define CASE_GFXPRE11_GFX11PLUS_TO(node, result)
2713
2715
2717 switch (Reg.id()) {
2718 case AMDGPU::SRC_SHARED_BASE_LO:
2719 case AMDGPU::SRC_SHARED_BASE:
2720 case AMDGPU::SRC_SHARED_LIMIT_LO:
2721 case AMDGPU::SRC_SHARED_LIMIT:
2722 case AMDGPU::SRC_PRIVATE_BASE_LO:
2723 case AMDGPU::SRC_PRIVATE_BASE:
2724 case AMDGPU::SRC_PRIVATE_LIMIT_LO:
2725 case AMDGPU::SRC_PRIVATE_LIMIT:
2726 case AMDGPU::SRC_FLAT_SCRATCH_BASE_LO:
2727 case AMDGPU::SRC_FLAT_SCRATCH_BASE_HI:
2728 case AMDGPU::SRC_POPS_EXITING_WAVE_ID:
2729 return true;
2730 case AMDGPU::SRC_VCCZ:
2731 case AMDGPU::SRC_EXECZ:
2732 case AMDGPU::SRC_SCC:
2733 return true;
2734 case AMDGPU::SGPR_NULL:
2735 return true;
2736 default:
2737 return false;
2738 }
2739}
2740
2741#undef CASE_CI_VI
2742#undef CASE_VI_GFX9PLUS
2743#undef CASE_GFXPRE11_GFX11PLUS
2744#undef CASE_GFXPRE11_GFX11PLUS_TO
2745#undef MAP_REG2REG
2746
2747bool isKImmOperand(const MCInstrDesc &Desc, unsigned OpNo) {
2748 assert(OpNo < Desc.NumOperands);
2749 unsigned OpType = Desc.operands()[OpNo].OperandType;
2750 return OpType >= AMDGPU::OPERAND_KIMM_FIRST &&
2751 OpType <= AMDGPU::OPERAND_KIMM_LAST;
2752}
2753
2754bool isSISrcFPOperand(const MCInstrDesc &Desc, unsigned OpNo) {
2755 assert(OpNo < Desc.NumOperands);
2756 unsigned OpType = Desc.operands()[OpNo].OperandType;
2757 switch (OpType) {
2773 return true;
2774 default:
2775 return false;
2776 }
2777}
2778
2779bool isSISrcInlinableOperand(const MCInstrDesc &Desc, unsigned OpNo) {
2780 assert(OpNo < Desc.NumOperands);
2781 unsigned OpType = Desc.operands()[OpNo].OperandType;
2782 return (OpType >= AMDGPU::OPERAND_REG_INLINE_C_FIRST &&
2786}
2787
2788// Avoid using MCRegisterClass::getSize, since that function will go away
2789// (move from MC* level to Target* level). Return size in bits.
2790unsigned getRegBitWidth(unsigned RCID) {
2791 switch (RCID) {
2792 case AMDGPU::VGPR_16RegClassID:
2793 case AMDGPU::VGPR_16_Lo128RegClassID:
2794 case AMDGPU::SGPR_LO16RegClassID:
2795 case AMDGPU::AGPR_LO16RegClassID:
2796 return 16;
2797 case AMDGPU::SGPR_32RegClassID:
2798 case AMDGPU::VGPR_32RegClassID:
2799 case AMDGPU::VGPR_32_Lo256RegClassID:
2800 case AMDGPU::VRegOrLds_32RegClassID:
2801 case AMDGPU::AGPR_32RegClassID:
2802 case AMDGPU::VS_32RegClassID:
2803 case AMDGPU::AV_32RegClassID:
2804 case AMDGPU::SReg_32RegClassID:
2805 case AMDGPU::SReg_32_XM0RegClassID:
2806 case AMDGPU::SRegOrLds_32RegClassID:
2807 return 32;
2808 case AMDGPU::SGPR_64RegClassID:
2809 case AMDGPU::VS_64RegClassID:
2810 case AMDGPU::SReg_64RegClassID:
2811 case AMDGPU::VReg_64RegClassID:
2812 case AMDGPU::AReg_64RegClassID:
2813 case AMDGPU::SReg_64_XEXECRegClassID:
2814 case AMDGPU::VReg_64_Align2RegClassID:
2815 case AMDGPU::AReg_64_Align2RegClassID:
2816 case AMDGPU::AV_64RegClassID:
2817 case AMDGPU::AV_64_Align2RegClassID:
2818 case AMDGPU::VReg_64_Lo256_Align2RegClassID:
2819 case AMDGPU::VS_64_Lo256RegClassID:
2820 return 64;
2821 case AMDGPU::SGPR_96RegClassID:
2822 case AMDGPU::SReg_96RegClassID:
2823 case AMDGPU::VReg_96RegClassID:
2824 case AMDGPU::AReg_96RegClassID:
2825 case AMDGPU::VReg_96_Align2RegClassID:
2826 case AMDGPU::AReg_96_Align2RegClassID:
2827 case AMDGPU::AV_96RegClassID:
2828 case AMDGPU::AV_96_Align2RegClassID:
2829 case AMDGPU::VReg_96_Lo256_Align2RegClassID:
2830 return 96;
2831 case AMDGPU::SGPR_128RegClassID:
2832 case AMDGPU::SReg_128RegClassID:
2833 case AMDGPU::VReg_128RegClassID:
2834 case AMDGPU::AReg_128RegClassID:
2835 case AMDGPU::VReg_128_Align2RegClassID:
2836 case AMDGPU::AReg_128_Align2RegClassID:
2837 case AMDGPU::AV_128RegClassID:
2838 case AMDGPU::AV_128_Align2RegClassID:
2839 case AMDGPU::SReg_128_XNULLRegClassID:
2840 case AMDGPU::VReg_128_Lo256_Align2RegClassID:
2841 return 128;
2842 case AMDGPU::SGPR_160RegClassID:
2843 case AMDGPU::SReg_160RegClassID:
2844 case AMDGPU::VReg_160RegClassID:
2845 case AMDGPU::AReg_160RegClassID:
2846 case AMDGPU::VReg_160_Align2RegClassID:
2847 case AMDGPU::AReg_160_Align2RegClassID:
2848 case AMDGPU::AV_160RegClassID:
2849 case AMDGPU::AV_160_Align2RegClassID:
2850 case AMDGPU::VReg_160_Lo256_Align2RegClassID:
2851 return 160;
2852 case AMDGPU::SGPR_192RegClassID:
2853 case AMDGPU::SReg_192RegClassID:
2854 case AMDGPU::VReg_192RegClassID:
2855 case AMDGPU::AReg_192RegClassID:
2856 case AMDGPU::VReg_192_Align2RegClassID:
2857 case AMDGPU::AReg_192_Align2RegClassID:
2858 case AMDGPU::AV_192RegClassID:
2859 case AMDGPU::AV_192_Align2RegClassID:
2860 case AMDGPU::VReg_192_Lo256_Align2RegClassID:
2861 return 192;
2862 case AMDGPU::SGPR_224RegClassID:
2863 case AMDGPU::SReg_224RegClassID:
2864 case AMDGPU::VReg_224RegClassID:
2865 case AMDGPU::AReg_224RegClassID:
2866 case AMDGPU::VReg_224_Align2RegClassID:
2867 case AMDGPU::AReg_224_Align2RegClassID:
2868 case AMDGPU::AV_224RegClassID:
2869 case AMDGPU::AV_224_Align2RegClassID:
2870 case AMDGPU::VReg_224_Lo256_Align2RegClassID:
2871 return 224;
2872 case AMDGPU::SGPR_256RegClassID:
2873 case AMDGPU::SReg_256RegClassID:
2874 case AMDGPU::VReg_256RegClassID:
2875 case AMDGPU::AReg_256RegClassID:
2876 case AMDGPU::VReg_256_Align2RegClassID:
2877 case AMDGPU::AReg_256_Align2RegClassID:
2878 case AMDGPU::AV_256RegClassID:
2879 case AMDGPU::AV_256_Align2RegClassID:
2880 case AMDGPU::SReg_256_XNULLRegClassID:
2881 case AMDGPU::VReg_256_Lo256_Align2RegClassID:
2882 return 256;
2883 case AMDGPU::SGPR_288RegClassID:
2884 case AMDGPU::SReg_288RegClassID:
2885 case AMDGPU::VReg_288RegClassID:
2886 case AMDGPU::AReg_288RegClassID:
2887 case AMDGPU::VReg_288_Align2RegClassID:
2888 case AMDGPU::AReg_288_Align2RegClassID:
2889 case AMDGPU::AV_288RegClassID:
2890 case AMDGPU::AV_288_Align2RegClassID:
2891 case AMDGPU::VReg_288_Lo256_Align2RegClassID:
2892 return 288;
2893 case AMDGPU::SGPR_320RegClassID:
2894 case AMDGPU::SReg_320RegClassID:
2895 case AMDGPU::VReg_320RegClassID:
2896 case AMDGPU::AReg_320RegClassID:
2897 case AMDGPU::VReg_320_Align2RegClassID:
2898 case AMDGPU::AReg_320_Align2RegClassID:
2899 case AMDGPU::AV_320RegClassID:
2900 case AMDGPU::AV_320_Align2RegClassID:
2901 case AMDGPU::VReg_320_Lo256_Align2RegClassID:
2902 return 320;
2903 case AMDGPU::SGPR_352RegClassID:
2904 case AMDGPU::SReg_352RegClassID:
2905 case AMDGPU::VReg_352RegClassID:
2906 case AMDGPU::AReg_352RegClassID:
2907 case AMDGPU::VReg_352_Align2RegClassID:
2908 case AMDGPU::AReg_352_Align2RegClassID:
2909 case AMDGPU::AV_352RegClassID:
2910 case AMDGPU::AV_352_Align2RegClassID:
2911 case AMDGPU::VReg_352_Lo256_Align2RegClassID:
2912 return 352;
2913 case AMDGPU::SGPR_384RegClassID:
2914 case AMDGPU::SReg_384RegClassID:
2915 case AMDGPU::VReg_384RegClassID:
2916 case AMDGPU::AReg_384RegClassID:
2917 case AMDGPU::VReg_384_Align2RegClassID:
2918 case AMDGPU::AReg_384_Align2RegClassID:
2919 case AMDGPU::AV_384RegClassID:
2920 case AMDGPU::AV_384_Align2RegClassID:
2921 case AMDGPU::VReg_384_Lo256_Align2RegClassID:
2922 return 384;
2923 case AMDGPU::SGPR_512RegClassID:
2924 case AMDGPU::SReg_512RegClassID:
2925 case AMDGPU::VReg_512RegClassID:
2926 case AMDGPU::AReg_512RegClassID:
2927 case AMDGPU::VReg_512_Align2RegClassID:
2928 case AMDGPU::AReg_512_Align2RegClassID:
2929 case AMDGPU::AV_512RegClassID:
2930 case AMDGPU::AV_512_Align2RegClassID:
2931 case AMDGPU::VReg_512_Lo256_Align2RegClassID:
2932 return 512;
2933 case AMDGPU::SGPR_1024RegClassID:
2934 case AMDGPU::SReg_1024RegClassID:
2935 case AMDGPU::VReg_1024RegClassID:
2936 case AMDGPU::AReg_1024RegClassID:
2937 case AMDGPU::VReg_1024_Align2RegClassID:
2938 case AMDGPU::AReg_1024_Align2RegClassID:
2939 case AMDGPU::AV_1024RegClassID:
2940 case AMDGPU::AV_1024_Align2RegClassID:
2941 case AMDGPU::VReg_1024_Lo256_Align2RegClassID:
2942 return 1024;
2943 default:
2944 llvm_unreachable("Unexpected register class");
2945 }
2946}
2947
2948unsigned getRegBitWidth(const MCRegisterClass &RC) {
2949 return getRegBitWidth(RC.getID());
2950}
2951
2952bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi) {
2954 return true;
2955
2956 uint64_t Val = static_cast<uint64_t>(Literal);
2957 return (Val == llvm::bit_cast<uint64_t>(0.0)) ||
2958 (Val == llvm::bit_cast<uint64_t>(1.0)) ||
2959 (Val == llvm::bit_cast<uint64_t>(-1.0)) ||
2960 (Val == llvm::bit_cast<uint64_t>(0.5)) ||
2961 (Val == llvm::bit_cast<uint64_t>(-0.5)) ||
2962 (Val == llvm::bit_cast<uint64_t>(2.0)) ||
2963 (Val == llvm::bit_cast<uint64_t>(-2.0)) ||
2964 (Val == llvm::bit_cast<uint64_t>(4.0)) ||
2965 (Val == llvm::bit_cast<uint64_t>(-4.0)) ||
2966 (Val == 0x3fc45f306dc9c882 && HasInv2Pi);
2967}
2968
2969bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi) {
2971 return true;
2972
2973 // The actual type of the operand does not seem to matter as long
2974 // as the bits match one of the inline immediate values. For example:
2975 //
2976 // -nan has the hexadecimal encoding of 0xfffffffe which is -2 in decimal,
2977 // so it is a legal inline immediate.
2978 //
2979 // 1065353216 has the hexadecimal encoding 0x3f800000 which is 1.0f in
2980 // floating-point, so it is a legal inline immediate.
2981
2982 uint32_t Val = static_cast<uint32_t>(Literal);
2983 return (Val == llvm::bit_cast<uint32_t>(0.0f)) ||
2984 (Val == llvm::bit_cast<uint32_t>(1.0f)) ||
2985 (Val == llvm::bit_cast<uint32_t>(-1.0f)) ||
2986 (Val == llvm::bit_cast<uint32_t>(0.5f)) ||
2987 (Val == llvm::bit_cast<uint32_t>(-0.5f)) ||
2988 (Val == llvm::bit_cast<uint32_t>(2.0f)) ||
2989 (Val == llvm::bit_cast<uint32_t>(-2.0f)) ||
2990 (Val == llvm::bit_cast<uint32_t>(4.0f)) ||
2991 (Val == llvm::bit_cast<uint32_t>(-4.0f)) ||
2992 (Val == 0x3e22f983 && HasInv2Pi);
2993}
2994
2995bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi) {
2996 if (!HasInv2Pi)
2997 return false;
2999 return true;
3000 uint16_t Val = static_cast<uint16_t>(Literal);
3001 return Val == 0x3F00 || // 0.5
3002 Val == 0xBF00 || // -0.5
3003 Val == 0x3F80 || // 1.0
3004 Val == 0xBF80 || // -1.0
3005 Val == 0x4000 || // 2.0
3006 Val == 0xC000 || // -2.0
3007 Val == 0x4080 || // 4.0
3008 Val == 0xC080 || // -4.0
3009 Val == 0x3E22; // 1.0 / (2.0 * pi)
3010}
3011
3012bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi) {
3013 return isInlinableLiteral32(Literal, HasInv2Pi);
3014}
3015
3016bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi) {
3017 if (!HasInv2Pi)
3018 return false;
3020 return true;
3021 uint16_t Val = static_cast<uint16_t>(Literal);
3022 return Val == 0x3C00 || // 1.0
3023 Val == 0xBC00 || // -1.0
3024 Val == 0x3800 || // 0.5
3025 Val == 0xB800 || // -0.5
3026 Val == 0x4000 || // 2.0
3027 Val == 0xC000 || // -2.0
3028 Val == 0x4400 || // 4.0
3029 Val == 0xC400 || // -4.0
3030 Val == 0x3118; // 1/2pi
3031}
3032
3033std::optional<unsigned> getInlineEncodingV216(bool IsFloat, uint32_t Literal) {
3034 // Unfortunately, the Instruction Set Architecture Reference Guide is
3035 // misleading about how the inline operands work for (packed) 16-bit
3036 // instructions. In a nutshell, the actual HW behavior is:
3037 //
3038 // - integer encodings (-16 .. 64) are always produced as sign-extended
3039 // 32-bit values
3040 // - float encodings are produced as:
3041 // - for F16 instructions: corresponding half-precision float values in
3042 // the LSBs, 0 in the MSBs
3043 // - for UI16 instructions: corresponding single-precision float value
3044 int32_t Signed = static_cast<int32_t>(Literal);
3045 if (Signed >= 0 && Signed <= 64)
3046 return 128 + Signed;
3047
3048 if (Signed >= -16 && Signed <= -1)
3049 return 192 + std::abs(Signed);
3050
3051 if (IsFloat) {
3052 // clang-format off
3053 switch (Literal) {
3054 case 0x3800: return 240; // 0.5
3055 case 0xB800: return 241; // -0.5
3056 case 0x3C00: return 242; // 1.0
3057 case 0xBC00: return 243; // -1.0
3058 case 0x4000: return 244; // 2.0
3059 case 0xC000: return 245; // -2.0
3060 case 0x4400: return 246; // 4.0
3061 case 0xC400: return 247; // -4.0
3062 case 0x3118: return 248; // 1.0 / (2.0 * pi)
3063 default: break;
3064 }
3065 // clang-format on
3066 } else {
3067 // clang-format off
3068 switch (Literal) {
3069 case 0x3F000000: return 240; // 0.5
3070 case 0xBF000000: return 241; // -0.5
3071 case 0x3F800000: return 242; // 1.0
3072 case 0xBF800000: return 243; // -1.0
3073 case 0x40000000: return 244; // 2.0
3074 case 0xC0000000: return 245; // -2.0
3075 case 0x40800000: return 246; // 4.0
3076 case 0xC0800000: return 247; // -4.0
3077 case 0x3E22F983: return 248; // 1.0 / (2.0 * pi)
3078 default: break;
3079 }
3080 // clang-format on
3081 }
3082
3083 return {};
3084}
3085
3086// Encoding of the literal as an inline constant for a V_PK_*_IU16 instruction
3087// or nullopt.
3088std::optional<unsigned> getInlineEncodingV2I16(uint32_t Literal) {
3089 return getInlineEncodingV216(false, Literal);
3090}
3091
3092// Encoding of the literal as an inline constant for a V_PK_*_BF16 instruction
3093// or nullopt.
3094std::optional<unsigned> getInlineEncodingV2BF16(uint32_t Literal) {
3095 int32_t Signed = static_cast<int32_t>(Literal);
3096 if (Signed >= 0 && Signed <= 64)
3097 return 128 + Signed;
3098
3099 if (Signed >= -16 && Signed <= -1)
3100 return 192 + std::abs(Signed);
3101
3102 // clang-format off
3103 switch (Literal) {
3104 case 0x3F00: return 240; // 0.5
3105 case 0xBF00: return 241; // -0.5
3106 case 0x3F80: return 242; // 1.0
3107 case 0xBF80: return 243; // -1.0
3108 case 0x4000: return 244; // 2.0
3109 case 0xC000: return 245; // -2.0
3110 case 0x4080: return 246; // 4.0
3111 case 0xC080: return 247; // -4.0
3112 case 0x3E22: return 248; // 1.0 / (2.0 * pi)
3113 default: break;
3114 }
3115 // clang-format on
3116
3117 return std::nullopt;
3118}
3119
3120// Encoding of the literal as an inline constant for a V_PK_*_F16 instruction
3121// or nullopt.
3122std::optional<unsigned> getInlineEncodingV2F16(uint32_t Literal) {
3123 return getInlineEncodingV216(true, Literal);
3124}
3125
3126// Encoding of the literal as an inline constant for V_PK_FMAC_F16 instruction
3127// or nullopt. This accounts for different inline constant behavior:
3128// - Pre-GFX11: fp16 inline constants have the value in low 16 bits, 0 in high
3129// - GFX11+: fp16 inline constants are duplicated into both halves
3131 bool IsGFX11Plus) {
3132 // Pre-GFX11 behavior: f16 in low bits, 0 in high bits
3133 if (!IsGFX11Plus)
3134 return getInlineEncodingV216(/*IsFloat=*/true, Literal);
3135
3136 // GFX11+ behavior: f16 duplicated in both halves
3137 // First, check for sign-extended integer inline constants (-16 to 64)
3138 // These work the same across all generations
3139 int32_t Signed = static_cast<int32_t>(Literal);
3140 if (Signed >= 0 && Signed <= 64)
3141 return 128 + Signed;
3142
3143 if (Signed >= -16 && Signed <= -1)
3144 return 192 + std::abs(Signed);
3145
3146 // For float inline constants on GFX11+, both halves must be equal
3147 uint16_t Lo = static_cast<uint16_t>(Literal);
3148 uint16_t Hi = static_cast<uint16_t>(Literal >> 16);
3149 if (Lo != Hi)
3150 return std::nullopt;
3151 return getInlineEncodingV216(/*IsFloat=*/true, Lo);
3152}
3153
3154// Whether the given literal can be inlined for a V_PK_* instruction.
3156 switch (OpType) {
3159 return getInlineEncodingV216(false, Literal).has_value();
3162 return getInlineEncodingV216(true, Literal).has_value();
3164 llvm_unreachable("OPERAND_REG_IMM_V2FP16_SPLAT is not supported");
3169 return false;
3170 default:
3171 llvm_unreachable("bad packed operand type");
3172 }
3173}
3174
3175// Whether the given literal can be inlined for a V_PK_*_IU16 instruction.
3179
3180// Whether the given literal can be inlined for a V_PK_*_BF16 instruction.
3184
3185// Whether the given literal can be inlined for a V_PK_*_F16 instruction.
3189
3190// Whether the given literal can be inlined for V_PK_FMAC_F16 instruction.
3192 return getPKFMACF16InlineEncoding(Literal, IsGFX11Plus).has_value();
3193}
3194
3195bool isValid32BitLiteral(uint64_t Val, bool IsFP64) {
3196 if (IsFP64)
3197 return !Lo_32(Val);
3198
3199 return isUInt<32>(Val) || isInt<32>(Val);
3200}
3201
3202int64_t encode32BitLiteral(int64_t Imm, OperandType Type, bool IsLit) {
3203 switch (Type) {
3204 default:
3205 break;
3211 return Imm & 0xffff;
3225 return Lo_32(Imm);
3228 return IsLit ? Imm : Hi_32(Imm);
3229 }
3230 return Imm;
3231}
3232
3234 const Function *F = A->getParent();
3235
3236 // Arguments to compute shaders are never a source of divergence.
3237 CallingConv::ID CC = F->getCallingConv();
3238 switch (CC) {
3241 return true;
3252 // For non-compute shaders, SGPR inputs are marked with either inreg or
3253 // byval. Everything else is in VGPRs.
3254 return A->hasAttribute(Attribute::InReg) ||
3255 A->hasAttribute(Attribute::ByVal);
3256 default:
3257 // TODO: treat i1 as divergent?
3258 return A->hasAttribute(Attribute::InReg);
3259 }
3260}
3261
3262bool isArgPassedInSGPR(const CallBase *CB, unsigned ArgNo) {
3263 // Arguments to compute shaders are never a source of divergence.
3265 switch (CC) {
3268 return true;
3279 // For non-compute shaders, SGPR inputs are marked with either inreg or
3280 // byval. Everything else is in VGPRs.
3281 return CB->paramHasAttr(ArgNo, Attribute::InReg) ||
3282 CB->isByValArgument(ArgNo);
3283 default:
3284 return CB->paramHasAttr(ArgNo, Attribute::InReg);
3285 }
3286}
3287
3288static bool hasSMEMByteOffset(const MCSubtargetInfo &ST) {
3289 return isGCN3Encoding(ST) || isGFX10Plus(ST);
3290}
3291
3293 int64_t EncodedOffset) {
3294 if (isGFX12Plus(ST))
3295 return isUInt<23>(EncodedOffset);
3296
3297 return hasSMEMByteOffset(ST) ? isUInt<20>(EncodedOffset)
3298 : isUInt<8>(EncodedOffset);
3299}
3300
3302 int64_t EncodedOffset, bool IsBuffer) {
3303 if (isGFX12Plus(ST)) {
3304 if (IsBuffer && EncodedOffset < 0)
3305 return false;
3306 return isInt<24>(EncodedOffset);
3307 }
3308
3309 return !IsBuffer && hasSMRDSignedImmOffset(ST) && isInt<21>(EncodedOffset);
3310}
3311
3312static bool isDwordAligned(uint64_t ByteOffset) {
3313 return (ByteOffset & 3) == 0;
3314}
3315
3317 uint64_t ByteOffset) {
3318 if (hasSMEMByteOffset(ST))
3319 return ByteOffset;
3320
3321 assert(isDwordAligned(ByteOffset));
3322 return ByteOffset >> 2;
3323}
3324
3325std::optional<int64_t> getSMRDEncodedOffset(const MCSubtargetInfo &ST,
3326 int64_t ByteOffset, bool IsBuffer,
3327 bool HasSOffset) {
3328 // For unbuffered smem loads, it is illegal for the Immediate Offset to be
3329 // negative if the resulting (Offset + (M0 or SOffset or zero) is negative.
3330 // Handle case where SOffset is not present.
3331 if (!IsBuffer && !HasSOffset && ByteOffset < 0 && hasSMRDSignedImmOffset(ST))
3332 return std::nullopt;
3333
3334 if (isGFX12Plus(ST)) // 24 bit signed offsets
3335 return isInt<24>(ByteOffset) ? std::optional<int64_t>(ByteOffset)
3336 : std::nullopt;
3337
3338 // The signed version is always a byte offset.
3339 if (!IsBuffer && hasSMRDSignedImmOffset(ST)) {
3341 return isInt<20>(ByteOffset) ? std::optional<int64_t>(ByteOffset)
3342 : std::nullopt;
3343 }
3344
3345 if (!isDwordAligned(ByteOffset) && !hasSMEMByteOffset(ST))
3346 return std::nullopt;
3347
3348 int64_t EncodedOffset = convertSMRDOffsetUnits(ST, ByteOffset);
3349 return isLegalSMRDEncodedUnsignedOffset(ST, EncodedOffset)
3350 ? std::optional<int64_t>(EncodedOffset)
3351 : std::nullopt;
3352}
3353
3354std::optional<int64_t> getSMRDEncodedLiteralOffset32(const MCSubtargetInfo &ST,
3355 int64_t ByteOffset) {
3356 if (!isCI(ST) || !isDwordAligned(ByteOffset))
3357 return std::nullopt;
3358
3359 int64_t EncodedOffset = convertSMRDOffsetUnits(ST, ByteOffset);
3360 return isUInt<32>(EncodedOffset) ? std::optional<int64_t>(EncodedOffset)
3361 : std::nullopt;
3362}
3363
3365 if (ST.getFeatureBits().test(FeatureFlatOffsetBits12))
3366 return 12;
3367 if (ST.getFeatureBits().test(FeatureFlatOffsetBits24))
3368 return 24;
3369 return 13;
3370}
3371
3372namespace {
3373
3374struct SourceOfDivergence {
3375 unsigned Intr;
3376};
3377const SourceOfDivergence *lookupSourceOfDivergence(unsigned Intr);
3378
3379struct AlwaysUniform {
3380 unsigned Intr;
3381};
3382const AlwaysUniform *lookupAlwaysUniform(unsigned Intr);
3383
3384#define GET_SourcesOfDivergence_IMPL
3385#define GET_UniformIntrinsics_IMPL
3386#define GET_Gfx9BufferFormat_IMPL
3387#define GET_Gfx10BufferFormat_IMPL
3388#define GET_Gfx11PlusBufferFormat_IMPL
3389
3390#include "AMDGPUGenSearchableTables.inc"
3391
3392} // end anonymous namespace
3393
3394bool isIntrinsicSourceOfDivergence(unsigned IntrID) {
3395 return lookupSourceOfDivergence(IntrID);
3396}
3397
3398bool isIntrinsicAlwaysUniform(unsigned IntrID) {
3399 return lookupAlwaysUniform(IntrID);
3400}
3401
3403 uint8_t NumComponents,
3404 uint8_t NumFormat,
3405 const MCSubtargetInfo &STI) {
3406 return isGFX11Plus(STI) ? getGfx11PlusBufferFormatInfo(
3407 BitsPerComp, NumComponents, NumFormat)
3408 : isGFX10(STI)
3409 ? getGfx10BufferFormatInfo(BitsPerComp, NumComponents, NumFormat)
3410 : getGfx9BufferFormatInfo(BitsPerComp, NumComponents, NumFormat);
3411}
3412
3414 const MCSubtargetInfo &STI) {
3415 return isGFX11Plus(STI) ? getGfx11PlusBufferFormatInfo(Format)
3416 : isGFX10(STI) ? getGfx10BufferFormatInfo(Format)
3417 : getGfx9BufferFormatInfo(Format);
3418}
3419
3421 const MCRegisterInfo &MRI) {
3422 const unsigned VGPRClasses[] = {
3423 AMDGPU::VGPR_16RegClassID, AMDGPU::VGPR_32RegClassID,
3424 AMDGPU::VReg_64RegClassID, AMDGPU::VReg_96RegClassID,
3425 AMDGPU::VReg_128RegClassID, AMDGPU::VReg_160RegClassID,
3426 AMDGPU::VReg_192RegClassID, AMDGPU::VReg_224RegClassID,
3427 AMDGPU::VReg_256RegClassID, AMDGPU::VReg_288RegClassID,
3428 AMDGPU::VReg_320RegClassID, AMDGPU::VReg_352RegClassID,
3429 AMDGPU::VReg_384RegClassID, AMDGPU::VReg_512RegClassID,
3430 AMDGPU::VReg_1024RegClassID};
3431
3432 for (unsigned RCID : VGPRClasses) {
3433 const MCRegisterClass &RC = MRI.getRegClass(RCID);
3434 if (RC.contains(Reg))
3435 return &RC;
3436 }
3437
3438 return nullptr;
3439}
3440
3442 unsigned Enc = MRI.getEncodingValue(Reg);
3443 unsigned Idx = Enc & AMDGPU::HWEncoding::REG_IDX_MASK;
3444 return Idx >> 8;
3445}
3446
3448 const MCRegisterInfo &MRI) {
3449 unsigned Enc = MRI.getEncodingValue(Reg);
3450 unsigned Idx = Enc & AMDGPU::HWEncoding::REG_IDX_MASK;
3451 if (Idx >= 0x100)
3452 return MCRegister();
3453
3454 const MCRegisterClass *RC = getVGPRPhysRegClass(Reg, MRI);
3455 if (!RC)
3456 return MCRegister();
3457
3458 Idx |= MSBs << 8;
3459 if (RC->getID() == AMDGPU::VGPR_16RegClassID) {
3460 // This class has 2048 registers with interleaved lo16 and hi16.
3461 Idx *= 2;
3463 ++Idx;
3464 }
3465
3466 return RC->getRegister(Idx);
3467}
3468
3469static std::optional<unsigned>
3470convertSetRegImmToVgprMSBs(unsigned Imm, unsigned Simm16,
3471 bool HasSetregVGPRMSBFixup) {
3472 constexpr unsigned VGPRMSBShift =
3474
3475 auto [HwRegId, Offset, Size] = Hwreg::HwregEncoding::decode(Simm16);
3476 if (HwRegId != Hwreg::ID_MODE ||
3477 (!HasSetregVGPRMSBFixup && (Offset + Size) < VGPRMSBShift))
3478 return {};
3479 // If there is SetregVGPRMSBFixup then Offset is ignored.
3480 if (!HasSetregVGPRMSBFixup)
3481 Imm <<= Offset;
3482 Imm = (Imm & Hwreg::VGPR_MSB_MASK) >> VGPRMSBShift;
3483 if (!HasSetregVGPRMSBFixup)
3485 return llvm::rotr<uint8_t>(static_cast<uint8_t>(Imm), /*R=*/2);
3486}
3487
3488std::optional<unsigned> convertSetRegImmToVgprMSBs(const MachineInstr &MI,
3489 bool HasSetregVGPRMSBFixup) {
3490 assert(MI.getOpcode() == AMDGPU::S_SETREG_IMM32_B32);
3491 return convertSetRegImmToVgprMSBs(MI.getOperand(0).getImm(),
3492 MI.getOperand(1).getImm(),
3493 HasSetregVGPRMSBFixup);
3494}
3495
3496std::optional<unsigned> convertSetRegImmToVgprMSBs(const MCInst &MI,
3497 bool HasSetregVGPRMSBFixup) {
3498 assert(MI.getOpcode() == AMDGPU::S_SETREG_IMM32_B32_gfx12);
3499 return convertSetRegImmToVgprMSBs(MI.getOperand(0).getImm(),
3500 MI.getOperand(1).getImm(),
3501 HasSetregVGPRMSBFixup);
3502}
3503
3504std::pair<const AMDGPU::OpName *, const AMDGPU::OpName *>
3506 static const AMDGPU::OpName VOPOps[4] = {
3507 AMDGPU::OpName::src0, AMDGPU::OpName::src1, AMDGPU::OpName::src2,
3508 AMDGPU::OpName::vdst};
3509 static const AMDGPU::OpName VDSOps[4] = {
3510 AMDGPU::OpName::addr, AMDGPU::OpName::data0, AMDGPU::OpName::data1,
3511 AMDGPU::OpName::vdst};
3512 static const AMDGPU::OpName FLATOps[4] = {
3513 AMDGPU::OpName::vaddr, AMDGPU::OpName::vdata,
3514 AMDGPU::OpName::NUM_OPERAND_NAMES, AMDGPU::OpName::vdst};
3515 static const AMDGPU::OpName BUFOps[4] = {
3516 AMDGPU::OpName::vaddr, AMDGPU::OpName::NUM_OPERAND_NAMES,
3517 AMDGPU::OpName::NUM_OPERAND_NAMES, AMDGPU::OpName::vdata};
3518 static const AMDGPU::OpName VIMGOps[4] = {
3519 AMDGPU::OpName::vaddr0, AMDGPU::OpName::vaddr1, AMDGPU::OpName::vaddr2,
3520 AMDGPU::OpName::vdata};
3521
3522 // For VOPD instructions MSB of a corresponding Y component operand VGPR
3523 // address is supposed to match X operand, otherwise VOPD shall not be
3524 // combined.
3525 static const AMDGPU::OpName VOPDOpsX[4] = {
3526 AMDGPU::OpName::src0X, AMDGPU::OpName::vsrc1X, AMDGPU::OpName::vsrc2X,
3527 AMDGPU::OpName::vdstX};
3528 static const AMDGPU::OpName VOPDOpsY[4] = {
3529 AMDGPU::OpName::src0Y, AMDGPU::OpName::vsrc1Y, AMDGPU::OpName::vsrc2Y,
3530 AMDGPU::OpName::vdstY};
3531
3532 // VOP2 MADMK instructions use src0, imm, src1 scheme.
3533 static const AMDGPU::OpName VOP2MADMKOps[4] = {
3534 AMDGPU::OpName::src0, AMDGPU::OpName::NUM_OPERAND_NAMES,
3535 AMDGPU::OpName::src1, AMDGPU::OpName::vdst};
3536 static const AMDGPU::OpName VOPDFMAMKOpsX[4] = {
3537 AMDGPU::OpName::src0X, AMDGPU::OpName::NUM_OPERAND_NAMES,
3538 AMDGPU::OpName::vsrc1X, AMDGPU::OpName::vdstX};
3539 static const AMDGPU::OpName VOPDFMAMKOpsY[4] = {
3540 AMDGPU::OpName::src0Y, AMDGPU::OpName::NUM_OPERAND_NAMES,
3541 AMDGPU::OpName::vsrc1Y, AMDGPU::OpName::vdstY};
3542
3546 switch (Desc.getOpcode()) {
3547 // LD_SCALE operands ignore MSB.
3548 case AMDGPU::V_WMMA_LD_SCALE_PAIRED_B32:
3549 case AMDGPU::V_WMMA_LD_SCALE_PAIRED_B32_gfx1250:
3550 case AMDGPU::V_WMMA_LD_SCALE16_PAIRED_B64:
3551 case AMDGPU::V_WMMA_LD_SCALE16_PAIRED_B64_gfx1250:
3552 return {};
3553 case AMDGPU::V_FMAMK_F16:
3554 case AMDGPU::V_FMAMK_F16_t16:
3555 case AMDGPU::V_FMAMK_F16_t16_gfx12:
3556 case AMDGPU::V_FMAMK_F16_fake16:
3557 case AMDGPU::V_FMAMK_F16_fake16_gfx12:
3558 case AMDGPU::V_FMAMK_F32:
3559 case AMDGPU::V_FMAMK_F32_gfx12:
3560 case AMDGPU::V_FMAMK_F64:
3561 case AMDGPU::V_FMAMK_F64_gfx1250:
3562 return {VOP2MADMKOps, nullptr};
3563 default:
3564 break;
3565 }
3566 return {VOPOps, nullptr};
3567 }
3568
3570 return {VDSOps, nullptr};
3571
3573 return {FLATOps, nullptr};
3574
3576 return {BUFOps, nullptr};
3577
3579 return {VIMGOps, nullptr};
3580
3581 if (AMDGPU::isVOPD(Desc.getOpcode())) {
3582 auto [OpX, OpY] = getVOPDComponents(Desc.getOpcode());
3583 return {(OpX == AMDGPU::V_FMAMK_F32) ? VOPDFMAMKOpsX : VOPDOpsX,
3584 (OpY == AMDGPU::V_FMAMK_F32) ? VOPDFMAMKOpsY : VOPDOpsY};
3585 }
3586
3588
3590 llvm_unreachable("Sample and export VGPR lowering is not implemented and"
3591 " these instructions are not expected on gfx1250");
3592
3593 return {};
3594}
3595
3596bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode) {
3597 const MCInstrDesc &Desc = MII.get(Opcode);
3599 return Desc.mayLoad() && !Desc.mayStore() && !getSMEMIsBuffer(Opcode);
3601 return false;
3602
3603 // Only SV and SVS modes are supported.
3604 if (SIInstrFlags::isFlatScratch(MII, Opcode))
3605 return hasNamedOperand(Opcode, OpName::vaddr);
3606
3607 // Only GVS mode is supported.
3608 return hasNamedOperand(Opcode, OpName::vaddr) &&
3609 hasNamedOperand(Opcode, OpName::saddr);
3610
3611 return false;
3612}
3613
3614bool hasAny64BitVGPROperands(const MCInstrDesc &OpDesc, const MCInstrInfo &MII,
3615 const MCSubtargetInfo &ST) {
3616 for (auto OpName : {OpName::vdst, OpName::src0, OpName::src1, OpName::src2}) {
3617 int Idx = getNamedOperandIdx(OpDesc.getOpcode(), OpName);
3618 if (Idx == -1)
3619 continue;
3620
3621 const MCOperandInfo &OpInfo = OpDesc.operands()[Idx];
3622 int16_t RegClass = MII.getOpRegClassID(
3623 OpInfo, ST.getHwMode(MCSubtargetInfo::HwMode_RegInfo));
3624 if (RegClass == AMDGPU::VReg_64RegClassID ||
3625 RegClass == AMDGPU::VReg_64_Align2RegClassID)
3626 return true;
3627 }
3628
3629 return false;
3630}
3631
3632bool isDPALU_DPP32BitOpc(unsigned Opc) {
3633 switch (Opc) {
3634 case AMDGPU::V_MUL_LO_U32_e64:
3635 case AMDGPU::V_MUL_LO_U32_e64_dpp:
3636 case AMDGPU::V_MUL_LO_U32_e64_dpp_gfx1250:
3637 case AMDGPU::V_MUL_HI_U32_e64:
3638 case AMDGPU::V_MUL_HI_U32_e64_dpp:
3639 case AMDGPU::V_MUL_HI_U32_e64_dpp_gfx1250:
3640 case AMDGPU::V_MUL_HI_I32_e64:
3641 case AMDGPU::V_MUL_HI_I32_e64_dpp:
3642 case AMDGPU::V_MUL_HI_I32_e64_dpp_gfx1250:
3643 case AMDGPU::V_MAD_U32_e64:
3644 case AMDGPU::V_MAD_U32_e64_dpp:
3645 case AMDGPU::V_MAD_U32_e64_dpp_gfx1250:
3646 return true;
3647 default:
3648 return false;
3649 }
3650}
3651
3652bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII,
3653 const MCSubtargetInfo &ST) {
3654 if (!ST.hasFeature(AMDGPU::FeatureDPALU_DPP))
3655 return false;
3656
3657 if (isDPALU_DPP32BitOpc(OpDesc.getOpcode()))
3658 return ST.hasFeature(AMDGPU::FeatureGFX1250Insts);
3659
3660 return hasAny64BitVGPROperands(OpDesc, MII, ST);
3661}
3662
3664 if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize32768))
3665 return 64;
3666 if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize65536))
3667 return 128;
3668 if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize196608))
3669 return 256;
3670 if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize163840))
3671 return 320;
3672 if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize327680))
3673 return 512;
3674 return 64; // In sync with getAddressableLocalMemorySize
3675}
3676
3678 switch (Opc) {
3679 case AMDGPU::V_PK_ADD_F32_gfx1250:
3680 case AMDGPU::V_PK_ADD_F32_gfx1250_gfx12:
3681 case AMDGPU::V_PK_MUL_F32_gfx1250:
3682 case AMDGPU::V_PK_MUL_F32_gfx1250_gfx12:
3683 case AMDGPU::V_PK_FMA_F32_gfx1250:
3684 case AMDGPU::V_PK_FMA_F32_gfx1250_gfx12:
3685 return true;
3686 default:
3687 return false;
3688 }
3689}
3690
3691// NOTE: This function is currently only used before pseudo-expansion.
3693 switch (Opc) {
3694 case AMDGPU::V_PK_ADD_F64:
3695 case AMDGPU::V_PK_MUL_F64:
3696 case AMDGPU::V_PK_FMA_F64:
3697 case AMDGPU::V_PK_MAX_NUM_F64:
3698 case AMDGPU::V_PK_MIN_NUM_F64:
3699 case AMDGPU::V_PK_ADD_NC_U64:
3700 case AMDGPU::V_PK_SUB_NC_U64:
3701 case AMDGPU::V_PK_LSHL_ADD_U64:
3702 return true;
3703 default:
3704 return false;
3705 }
3706}
3707
3711
3712const std::array<unsigned, 3> &ClusterDimsAttr::getDims() const {
3713 assert(isFixedDims() && "expect kind to be FixedDims");
3714 return Dims;
3715}
3716
3717std::string ClusterDimsAttr::to_string() const {
3718 SmallString<10> Buffer;
3719 raw_svector_ostream OS(Buffer);
3720
3721 switch (getKind()) {
3722 case Kind::Unknown:
3723 return "";
3724 case Kind::NoCluster: {
3725 OS << EncoNoCluster << ',' << EncoNoCluster << ',' << EncoNoCluster;
3726 return Buffer.c_str();
3727 }
3728 case Kind::VariableDims: {
3729 OS << EncoVariableDims << ',' << EncoVariableDims << ','
3730 << EncoVariableDims;
3731 return Buffer.c_str();
3732 }
3733 case Kind::FixedDims: {
3734 OS << Dims[0] << ',' << Dims[1] << ',' << Dims[2];
3735 return Buffer.c_str();
3736 }
3737 }
3738 llvm_unreachable("Unknown ClusterDimsAttr kind");
3739}
3740
3742 std::optional<SmallVector<unsigned>> Attr =
3743 getIntegerVecAttribute(F, "amdgpu-cluster-dims", /*Size=*/3);
3745
3746 if (!Attr.has_value())
3747 AttrKind = Kind::Unknown;
3748 else if (all_of(*Attr, equal_to(EncoNoCluster)))
3749 AttrKind = Kind::NoCluster;
3750 else if (all_of(*Attr, equal_to(EncoVariableDims)))
3751 AttrKind = Kind::VariableDims;
3752
3753 ClusterDimsAttr A(AttrKind);
3754 if (AttrKind == Kind::FixedDims)
3755 A.Dims = {(*Attr)[0], (*Attr)[1], (*Attr)[2]};
3756
3757 return A;
3758}
3759
3760std::optional<APFloat> evaluateRcp(const APFloat &Val) {
3761 const fltSemantics &Sem = Val.getSemantics();
3762
3763 // v_rcp_f16/bf16 are correctly rounded.
3764 if (&Sem == &APFloat::IEEEhalf() || &Sem == &APFloat::BFloat())
3765 return APFloat::getOne(Sem) / Val;
3766
3767 // v_rcp_f32/f64 always flush a denormal input to zero (preserving sign)
3768 // before reciprocating.
3769 APFloat Arg = Val;
3770 if (Arg.isDenormal())
3771 Arg = APFloat::getZero(Sem, Arg.isNegative());
3772
3773 APFloat Result = APFloat::getOne(Sem) / Arg;
3774
3775 // v_rcp_f32/f64 always flush a denormal result to zero (preserving sign).
3776 if (Result.isDenormal())
3777 Result = APFloat::getZero(Sem, Result.isNegative());
3778
3779 // v_rcp_f32/f64 only approximate the reciprocal, except for these special
3780 // cases where the result is exact.
3781 if (!Result.isZero() && !Result.isInfinity() && !Result.isNaN() &&
3782 !Result.isOne() && !Result.isMinusOne())
3783 return std::nullopt;
3784
3785 return Result;
3786}
3787
3788} // namespace AMDGPU
3789
3791 switch (S) {
3792 case (AMDGPU::TargetIDSetting::Unsupported):
3793 OS << "Unsupported";
3794 break;
3795 case (AMDGPU::TargetIDSetting::Any):
3796 OS << "Any";
3797 break;
3798 case (AMDGPU::TargetIDSetting::Off):
3799 OS << "Off";
3800 break;
3801 case (AMDGPU::TargetIDSetting::On):
3802 OS << "On";
3803 break;
3804 }
3805 return OS;
3806}
3807
3808} // namespace llvm
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static llvm::cl::opt< unsigned > DefaultAMDHSACodeObjectVersion("amdhsa-code-object-version", llvm::cl::Hidden, llvm::cl::init(llvm::AMDGPU::AMDHSA_COV6), llvm::cl::desc("Set default AMDHSA Code Object Version (module flag " "or asm directive still take priority if present)"))
#define MAP_REG2REG
unsigned uint64_t
Provides AMDGPU specific target descriptions.
MC layer struct for AMDGPUMCKernelCodeT, provides MCExpr functionality where required.
@ AMD_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32
This file contains the simple types necessary to represent the attributes associated with functions a...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
IRTranslator LLVM IR MI
#define RegName(no)
#define F(x, y, z)
Definition MD5.cpp:54
Register Reg
Register const TargetRegisterInfo * TRI
This file contains the declarations for metadata subclasses.
#define T
uint64_t High
static bool isValid(const char C)
Returns true if C is a valid mangled character: <0-9a-zA-Z_>.
#define S_00B848_MEM_ORDERED(x)
Definition SIDefines.h:1492
#define S_00B848_WGP_MODE(x)
Definition SIDefines.h:1489
#define S_00B848_FWD_PROGRESS(x)
Definition SIDefines.h:1495
This file contains some functions that are useful when dealing with strings.
static const int BlockSize
Definition TarWriter.cpp:33
static ClusterDimsAttr get(const Function &F)
const std::array< unsigned, 3 > & getDims() const
static TargetID createFromSubtargetFeatures(const Triple &TT, StringRef CPU, StringRef FeatureString)
Construct a TargetID for triple TT and processor CPU, taking the xnack/sramecc modes from the subtarg...
unsigned getIndexInParsedOperands(unsigned CompOprIdx) const
unsigned getIndexOfSrcInParsedOperands(unsigned CompSrcIdx) const
std::optional< unsigned > getInvalidCompOperandIndex(std::function< MCRegister(unsigned, unsigned)> GetRegIdx, const MCRegisterInfo &MRI, bool SkipSrc=false, bool AllowSameVGPR=false, bool VOPD3=false) const
std::array< MCRegister, Component::MAX_OPR_NUM > RegIndices
Represents the counter values to wait for in an s_waitcnt instruction.
static const fltSemantics & BFloat()
Definition APFloat.h:303
static const fltSemantics & IEEEhalf()
Definition APFloat.h:302
bool isNegative() const
Definition APFloat.h:1583
bool isDenormal() const
Definition APFloat.h:1584
const fltSemantics & getSemantics() const
Definition APFloat.h:1591
static APFloat getOne(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative One.
Definition APFloat.h:1192
static APFloat getZero(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Zero.
Definition APFloat.h:1183
This class represents an incoming formal argument to a Function.
Definition Argument.h:32
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:106
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
CallingConv::ID getCallingConv() const
LLVM_ABI bool paramHasAttr(unsigned ArgNo, Attribute::AttrKind Kind) const
Determine whether the argument or parameter has the given attribute.
bool isByValArgument(unsigned ArgNo) const
Determine whether this argument is passed by value.
constexpr bool test(unsigned I) const
unsigned getAddressSpace() const
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Instances of this class represent a single low-level machine instruction.
Definition MCInst.h:188
Describe properties that are true of each instruction in the target description file.
unsigned getNumOperands() const
Return the number of declared MachineOperands for this MachineInstruction.
ArrayRef< MCOperandInfo > operands() const
bool mayStore() const
Return true if this instruction could possibly modify memory.
bool mayLoad() const
Return true if this instruction could possibly read memory.
unsigned getNumDefs() const
Return the number of MachineOperands that are register definitions.
int getOperandConstraint(unsigned OpNum, MCOI::OperandConstraint Constraint) const
Returns the value of the specified operand constraint if it is present.
unsigned getOpcode() const
Return the opcode number for this descriptor.
Interface to description of machine instruction set.
Definition MCInstrInfo.h:27
const MCInstrDesc & get(unsigned Opcode) const
Return the machine instruction descriptor that corresponds to the specified instruction opcode.
Definition MCInstrInfo.h:89
int16_t getOpRegClassID(const MCOperandInfo &OpInfo, unsigned HwModeId) const
Return the ID of the register class to use for OpInfo, for the active HwMode HwModeId.
Definition MCInstrInfo.h:79
This holds information about one operand of a machine instruction, indicating the register class for ...
Definition MCInstrDesc.h:88
MCRegisterClass - Base class of TargetRegisterClass.
unsigned getID() const
getID() - Return the register class ID number.
MCRegister getRegister(unsigned i) const
getRegister - Return the specified register in the class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
MCRegisterInfo base class - We assume that the target defines a static array of MCRegisterDesc object...
bool regsOverlap(MCRegister RegA, MCRegister RegB) const
Returns true if the two registers are equal or alias each other.
uint16_t getEncodingValue(MCRegister Reg) const
Returns the encoding for Reg.
const MCRegisterClass & getRegClass(unsigned i) const
Returns the register class associated with the enumeration value.
MCRegister getSubReg(MCRegister Reg, unsigned Idx) const
Returns the physical register number of sub-register "Index" for physical register RegNo.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
constexpr unsigned id() const
Definition MCRegister.h:82
Generic base class for all target subtargets.
bool hasFeature(unsigned Feature) const
const Triple & getTargetTriple() const
const FeatureBitset & getFeatureBits() const
StringRef getCPU() const
Metadata node.
Definition Metadata.h:1069
const MDOperand & getOperand(unsigned I) const
Definition Metadata.h:1426
unsigned getNumOperands() const
Return number of MDNode operands.
Definition Metadata.h:1432
Representation of each machine instruction.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
const char * c_str()
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
A wrapper around a string literal that serves as a proxy for constructing global tables of StringRefs...
Definition StringRef.h:888
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
std::pair< StringRef, StringRef > split(char Separator) const
Split into two substrings around the first occurrence of a separator character.
Definition StringRef.h:736
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Definition StringRef.h:490
constexpr bool empty() const
Check if the string is empty.
Definition StringRef.h:141
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
OSType getOS() const
Get the parsed operating system type of this triple.
Definition Triple.h:523
ArchType getArch() const
Get the parsed architecture type of this triple.
Definition Triple.h:514
bool isAMDGCN() const
Tests whether the target is AMDGCN.
Definition Triple.h:992
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
A raw_ostream that writes to an SmallVector or SmallString.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ CONSTANT_ADDRESS_32BIT
Address space for 32-bit constant memory.
@ LOCAL_ADDRESS
Address space for local memory.
@ CONSTANT_ADDRESS
Address space for constant memory (VTX2).
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
unsigned decodeFieldVaVcc(unsigned Encoded)
unsigned encodeFieldVaVcc(unsigned Encoded, unsigned VaVcc)
unsigned decodeFieldHoldCnt(unsigned Encoded, const IsaVersion &Version)
bool decodeDepCtr(unsigned Code, int &Id, StringRef &Name, unsigned &Val, bool &IsDefault, const MCSubtargetInfo &STI)
unsigned encodeFieldHoldCnt(unsigned Encoded, unsigned HoldCnt, const IsaVersion &Version)
unsigned encodeFieldVaSsrc(unsigned Encoded, unsigned VaSsrc)
unsigned encodeFieldVaVdst(unsigned Encoded, unsigned VaVdst)
unsigned decodeFieldSaSdst(unsigned Encoded)
unsigned getHoldCntBitMask(const IsaVersion &Version)
unsigned decodeFieldVaSdst(unsigned Encoded)
unsigned encodeFieldVmVsrc(unsigned Encoded, unsigned VmVsrc)
unsigned decodeFieldVaSsrc(unsigned Encoded)
int encodeDepCtr(const StringRef Name, int64_t Val, unsigned &UsedOprMask, const MCSubtargetInfo &STI)
unsigned encodeFieldSaSdst(unsigned Encoded, unsigned SaSdst)
const CustomOperandVal DepCtrInfo[]
bool isSymbolicDepCtrEncoding(unsigned Code, bool &HasNonDefaultVal, const MCSubtargetInfo &STI)
unsigned decodeFieldVaVdst(unsigned Encoded)
int getDefaultDepCtrEncoding(const MCSubtargetInfo &STI)
unsigned decodeFieldVmVsrc(unsigned Encoded)
unsigned encodeFieldVaSdst(unsigned Encoded, unsigned VaSdst)
bool isSupportedTgtId(unsigned Id, const MCSubtargetInfo &STI)
static constexpr ExpTgt ExpTgtInfo[]
bool getTgtName(unsigned Id, StringRef &Name, int &Index)
unsigned getTgtId(const StringRef Name)
constexpr uint32_t VersionMinor
HSA metadata minor version.
constexpr uint32_t VersionMajor
HSA metadata major version.
unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI, unsigned NumVGPRs, unsigned DynamicVGPRBlockSize)
static unsigned getMaxHWAddressableLocalMemorySize(const MCSubtargetInfo &STI)
unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI)
bool isSGPROccupancyLimited(const MCSubtargetInfo &STI)
unsigned getArchVGPRAllocGranule()
For subtargets with a unified VGPR file and mixed ArchVGPR/AGPR usage, returns the allocation granule...
static unsigned getPhysicalLocalMemorySize(const MCSubtargetInfo &STI)
static unsigned getSGPRTrapHandlerReserve(const MCSubtargetInfo &STI)
unsigned getAddressableLocalMemorySize(const MCSubtargetInfo &STI)
unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI, std::optional< bool > EnableWavefrontSize32)
unsigned getEncodedNumVGPRBlocks(const MCSubtargetInfo &STI, unsigned NumVGPRs, std::optional< bool > EnableWavefrontSize32)
unsigned getMaxWorkGroupsPerCU(const MCSubtargetInfo &STI, unsigned FlatWorkGroupSize)
unsigned getMinNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU)
unsigned getMaxNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU, bool Addressable)
unsigned getWavefrontSize(const MCSubtargetInfo &STI)
unsigned getWavesPerEUForWorkGroup(const MCSubtargetInfo &STI, unsigned FlatWorkGroupSize)
unsigned getInstCacheLineSize(const MCSubtargetInfo &STI)
static constexpr unsigned MaxDynamicVGPRBlocks
Maximum number of VGPR blocks that can be allocated in dynamic VGPR mode.
unsigned getSGPREncodingGranule(const MCSubtargetInfo &STI)
static unsigned getSGPRBudgetPerWave(unsigned TotalNumSGPRs, unsigned WavesPerEU, unsigned TrapReserve, unsigned Granule)
unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU, unsigned DynamicVGPRBlockSize)
unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize)
unsigned getWavesPerWorkGroup(const MCSubtargetInfo &STI, unsigned FlatWorkGroupSize)
unsigned getAllocatedNumVGPRBlocks(const MCSubtargetInfo &STI, unsigned NumVGPRs, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
unsigned getOccupancyWithNumSGPRs(unsigned SGPRs, unsigned MaxWaves, unsigned TotalNumSGPRs, unsigned Granule, unsigned TrapReserve)
unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs)
unsigned getNumExtraSGPRs(const MCSubtargetInfo &STI, bool VCCUsed, bool FlatScrUsed, bool XNACKUsed)
unsigned getLocalMemorySize(const MCSubtargetInfo &STI)
unsigned getMaxNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU, unsigned DynamicVGPRBlockSize)
static unsigned getGranulatedNumRegisterBlocks(unsigned NumRegs, unsigned Granule)
unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
StringLiteral const UfmtSymbolicGFX11[]
bool isValidUnifiedFormat(unsigned Id, const MCSubtargetInfo &STI)
unsigned getDefaultFormatEncoding(const MCSubtargetInfo &STI)
StringRef getUnifiedFormatName(unsigned Id, const MCSubtargetInfo &STI)
unsigned const DfmtNfmt2UFmtGFX10[]
StringLiteral const DfmtSymbolic[]
static StringLiteral const * getNfmtLookupTable(const MCSubtargetInfo &STI)
bool isValidNfmt(unsigned Id, const MCSubtargetInfo &STI)
StringLiteral const NfmtSymbolicGFX10[]
bool isValidDfmtNfmt(unsigned Id, const MCSubtargetInfo &STI)
int64_t convertDfmtNfmt2Ufmt(unsigned Dfmt, unsigned Nfmt, const MCSubtargetInfo &STI)
StringRef getDfmtName(unsigned Id)
int64_t encodeDfmtNfmt(unsigned Dfmt, unsigned Nfmt)
int64_t getUnifiedFormat(const StringRef Name, const MCSubtargetInfo &STI)
bool isValidFormatEncoding(unsigned Val, const MCSubtargetInfo &STI)
StringRef getNfmtName(unsigned Id, const MCSubtargetInfo &STI)
unsigned const DfmtNfmt2UFmtGFX11[]
StringLiteral const NfmtSymbolicVI[]
StringLiteral const NfmtSymbolicSICI[]
int64_t getNfmt(const StringRef Name, const MCSubtargetInfo &STI)
int64_t getDfmt(const StringRef Name)
StringLiteral const UfmtSymbolicGFX10[]
void decodeDfmtNfmt(unsigned Format, unsigned &Dfmt, unsigned &Nfmt)
uint64_t encodeMsg(uint64_t MsgId, uint64_t OpId, uint64_t StreamId)
bool msgSupportsStream(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI)
void decodeMsg(unsigned Val, uint16_t &MsgId, uint16_t &OpId, uint16_t &StreamId, const MCSubtargetInfo &STI)
bool isValidMsgId(int64_t MsgId, const MCSubtargetInfo &STI)
bool isValidMsgStream(int64_t MsgId, int64_t OpId, int64_t StreamId, const MCSubtargetInfo &STI, bool Strict)
bool msgDoesNotUseM0(int64_t MsgId, const MCSubtargetInfo &STI)
Returns true if the message does not use the m0 operand.
StringRef getMsgOpName(int64_t MsgId, uint64_t Encoding, const MCSubtargetInfo &STI)
Map from an encoding to the symbolic name for a sendmsg operation.
static uint64_t getMsgIdMask(const MCSubtargetInfo &STI)
bool msgRequiresOp(int64_t MsgId, const MCSubtargetInfo &STI)
bool isValidMsgOp(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI, bool Strict)
constexpr unsigned VOPD_VGPR_BANK_MASKS[]
constexpr unsigned COMPONENTS_NUM
constexpr unsigned VOPD3_VGPR_BANK_MASKS[]
bool isGCN3Encoding(const MCSubtargetInfo &STI)
bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi)
bool isGFX10_BEncoding(const MCSubtargetInfo &STI)
bool isInlineValue(MCRegister Reg)
bool isGFX10_GFX11(const MCSubtargetInfo &STI)
bool isInlinableLiteralV216(uint32_t Literal, uint8_t OpType)
bool isPKFMACF16InlineConstant(uint32_t Literal, bool IsGFX11Plus)
LLVM_READONLY const MIMGInfo * getMIMGInfo(unsigned Opc)
bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi)
bool isSGPR(MCRegister Reg, const MCRegisterInfo *TRI)
Is Reg - scalar register.
uint64_t convertSMRDOffsetUnits(const MCSubtargetInfo &ST, uint64_t ByteOffset)
Convert ByteOffset to dwords if the subtarget uses dword SMRD immediate offsets.
static unsigned encodeStorecnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Storecnt)
MCRegister getMCReg(MCRegister Reg, const MCSubtargetInfo &STI)
If Reg is a pseudo reg, return the correct hardware register given STI otherwise return Reg.
static bool hasSMEMByteOffset(const MCSubtargetInfo &ST)
LLVM_ABI unsigned getMaxWavesPerEU(GPUKind AK)
bool isVOPCAsmOnly(unsigned Opc)
int getMIMGOpcode(unsigned BaseOpcode, unsigned MIMGEncoding, unsigned VDataDwords, unsigned VAddrDwords)
bool getMTBUFHasSrsrc(unsigned Opc)
std::optional< int64_t > getSMRDEncodedLiteralOffset32(const MCSubtargetInfo &ST, int64_t ByteOffset)
bool getWMMAIsXDL(unsigned Opc)
static std::optional< unsigned > convertSetRegImmToVgprMSBs(unsigned Imm, unsigned Simm16, bool HasSetregVGPRMSBFixup)
uint8_t wmmaScaleF8F6F4FormatToNumRegs(unsigned Fmt)
static bool isSymbolicCustomOperandEncoding(const CustomOperandVal *Opr, int Size, unsigned Code, bool &HasNonDefaultVal, const MCSubtargetInfo &STI)
bool isGFX10Before1030(const MCSubtargetInfo &STI)
bool isSISrcInlinableOperand(const MCInstrDesc &Desc, unsigned OpNo)
Does this operand support only inlinable literals?
unsigned mapWMMA2AddrTo3AddrOpcode(unsigned Opc)
const int OPR_ID_UNSUPPORTED
void initDefaultAMDKernelCodeT(AMDGPUMCKernelCodeT &KernelCode, const MCSubtargetInfo &STI)
bool shouldEmitConstantsToTextSection(const Triple &TT)
bool isInlinableLiteralV2I16(uint32_t Literal)
bool isDPMACCInstruction(unsigned Opc)
int getMTBUFElements(unsigned Opc)
constexpr unsigned getNumWorkGroupSIMDs(bool FullSIMDMode)
bool isHi16Reg(MCRegister Reg, const MCRegisterInfo &MRI)
static int encodeCustomOperandVal(const CustomOperandVal &Op, int64_t InputVal)
unsigned getTemporalHintType(const MCInstrDesc TID)
bool isGFX10(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2BF16(uint32_t Literal)
unsigned getMaxNumUserSGPRs(const MCSubtargetInfo &STI)
std::optional< unsigned > getInlineEncodingV216(bool IsFloat, uint32_t Literal)
FPType getFPDstSelType(unsigned Opc)
unsigned getNumFlatOffsetBits(const MCSubtargetInfo &ST)
For pre-GFX12 FLAT instructions the offset must be positive; MSB is ignored and forced to zero.
bool hasA16(const MCSubtargetInfo &STI)
bool isLegalSMRDEncodedSignedOffset(const MCSubtargetInfo &ST, int64_t EncodedOffset, bool IsBuffer)
bool isGFX12Plus(const MCSubtargetInfo &STI)
unsigned getNSAMaxSize(const MCSubtargetInfo &STI, bool HasSampler)
const MCRegisterClass * getVGPRPhysRegClass(MCRegister Reg, const MCRegisterInfo &MRI)
unsigned encodeLoadcntDscnt(const IsaVersion &Version, const Waitcnt &Decoded)
bool getHasMatrixScale(unsigned Opc)
bool hasPackedD16(const MCSubtargetInfo &STI)
unsigned getStorecntBitMask(const IsaVersion &Version)
bool isFullSIMDMode(const MCSubtargetInfo &STI)
unsigned getLdsDwGranularity(const MCSubtargetInfo &ST)
bool isGFX940(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2F16(uint32_t Literal)
bool isHsaAbi(const MCSubtargetInfo &STI)
bool isGFX11(const MCSubtargetInfo &STI)
const int OPR_VAL_INVALID
bool getSMEMIsBuffer(unsigned Opc)
bool isPackedSingleSGPRFP32Inst(unsigned Opc)
The opcode is a packed fp32 instruction which only reads low 32 bits of a scalar operand and propagat...
bool isGFX10_3_GFX11(const MCSubtargetInfo &STI)
bool isGFX13(const MCSubtargetInfo &STI)
unsigned getAsynccntBitMask(const IsaVersion &Version)
bool hasValueInRangeLikeMetadata(const MDNode &MD, int64_t Val)
Checks if Val is inside MD, a !range-like metadata.
LLVM_ABI unsigned getAddressableNumSGPRs(GPUKind AK)
TargetID createAMDGPUTargetID(const MCSubtargetInfo &STI, StringRef FeatureString)
Construct TargetID from MCSubtargetInfo.
uint8_t mfmaScaleF8F6F4FormatToNumRegs(unsigned EncodingVal)
unsigned getVOPDOpcode(unsigned Opc, bool VOPD3)
bool isGroupSegment(const GlobalValue *GV)
LLVM_ABI IsaVersion getIsaVersion(StringRef GPU)
bool getMTBUFHasSoffset(unsigned Opc)
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
bool hasXNACK(const MCSubtargetInfo &STI)
bool isValid32BitLiteral(uint64_t Val, bool IsFP64)
static unsigned getCombinedCountBitMask(const IsaVersion &Version, bool IsStore)
LLVM_ABI unsigned getTotalNumVGPRs(GPUKind AK, bool IsWave32)
CanBeVOPD getCanBeVOPD(unsigned Opc, unsigned EncodingFamily, bool VOPD3)
bool isVOPC64DPP(unsigned Opc)
int getMUBUFOpcode(unsigned BaseOpc, unsigned Elements)
bool getMAIIsGFX940XDL(unsigned Opc)
bool isSI(const MCSubtargetInfo &STI)
unsigned getDefaultAMDHSACodeObjectVersion()
LLVM_ABI unsigned getTotalNumSGPRs(GPUKind AK)
bool hasPrivateApertureRegs(const MCSubtargetInfo &STI)
bool isReadOnlySegment(const GlobalValue *GV)
Waitcnt decodeWaitcnt(const IsaVersion &Version, unsigned Encoded)
bool isArgPassedInSGPR(const Argument *A)
bool isIntrinsicAlwaysUniform(unsigned IntrID)
int getMUBUFBaseOpcode(unsigned Opc)
unsigned encodeWaitcnt(const IsaVersion &Version, const Waitcnt &Decoded)
unsigned getAMDHSACodeObjectVersion(const Module &M)
unsigned decodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt)
unsigned getWaitcntBitMask(const IsaVersion &Version)
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
bool getVOP3IsSingle(unsigned Opc)
bool isPackedSingleSGPR64BitInst(unsigned Opc)
The opcode is a packed 64-bit instruction which only reads low 64 bits of a scalar operand and propag...
bool isGFX9(const MCSubtargetInfo &STI)
bool isDPALU_DPP32BitOpc(unsigned Opc)
bool getVOP1IsSingle(unsigned Opc)
static bool isDwordAligned(uint64_t ByteOffset)
unsigned getVOPDEncodingFamily(const MCSubtargetInfo &ST)
bool isKImmOperand(const MCInstrDesc &Desc, unsigned OpNo)
Is this a KImm operand?
bool getHasColorExport(const Function &F)
GPUKind
GPU kinds supported by the AMDGPU target.
int getMTBUFBaseOpcode(unsigned Opc)
bool isGFX90A(const MCSubtargetInfo &STI)
unsigned getSamplecntBitMask(const IsaVersion &Version)
unsigned getDefaultQueueImplicitArgPosition(unsigned CodeObjectVersion)
std::tuple< char, unsigned, unsigned > parseAsmPhysRegName(StringRef RegName)
Returns a valid charcode or 0 in the first entry if this is a valid physical register name.
bool getHasDepthExport(const Function &F)
bool isGFX8_GFX9_GFX10(const MCSubtargetInfo &STI)
bool getMUBUFHasVAddr(unsigned Opc)
bool isTrue16Inst(unsigned Opc)
LLVM_ABI unsigned getSGPRAllocGranule(GPUKind AK)
unsigned getVGPREncodingMSBs(MCRegister Reg, const MCRegisterInfo &MRI)
std::pair< unsigned, unsigned > getVOPDComponents(unsigned VOPDOpcode)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
bool isGFX12(const MCSubtargetInfo &STI)
unsigned getInitialPSInputAddr(const Function &F)
unsigned encodeExpcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Expcnt)
bool isAsyncStore(unsigned Opc)
unsigned getDynamicVGPRBlockSize(const Function &F)
unsigned getKmcntBitMask(const IsaVersion &Version)
MCRegister getVGPRWithMSBs(MCRegister Reg, unsigned MSBs, const MCRegisterInfo &MRI)
If Reg is a low VGPR return a corresponding high VGPR with MSBs set.
unsigned getVmcntBitMask(const IsaVersion &Version)
bool isNotGFX10Plus(const MCSubtargetInfo &STI)
bool hasMAIInsts(const MCSubtargetInfo &STI)
unsigned getBitOp2(unsigned Opc)
bool isIntrinsicSourceOfDivergence(unsigned IntrID)
unsigned getXcntBitMask(const IsaVersion &Version)
bool isGenericAtomic(unsigned Opc)
const MFMA_F8F6F4_Info * getWMMA_F8F6F4_WithFormatArgs(unsigned FmtA, unsigned FmtB, unsigned F8F8Opcode)
bool isGFX8Plus(const MCSubtargetInfo &STI)
LLVM_READNONE bool isInlinableIntLiteral(int64_t Literal)
Is this literal inlinable, and not one of the values intended for floating point values.
unsigned getLgkmcntBitMask(const IsaVersion &Version)
bool getMUBUFTfe(unsigned Opc)
unsigned getBvhcntBitMask(const IsaVersion &Version)
bool hasSMRDSignedImmOffset(const MCSubtargetInfo &ST)
bool hasMIMG_R128(const MCSubtargetInfo &STI)
LLVM_ABI GPUKind parseArchAMDGCN(StringRef CPU)
bool hasGFX10_3Insts(const MCSubtargetInfo &STI)
unsigned decodeDscnt(const IsaVersion &Version, unsigned Waitcnt)
std::pair< const AMDGPU::OpName *, const AMDGPU::OpName * > getVGPRLoweringOperandTables(const MCInstrDesc &Desc)
bool hasG16(const MCSubtargetInfo &STI)
unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode, const MIMGDimInfo *Dim, bool IsA16, bool IsG16Supported)
int getMTBUFOpcode(unsigned BaseOpc, unsigned Elements)
bool isGFX13Plus(const MCSubtargetInfo &STI)
unsigned getExpcntBitMask(const IsaVersion &Version)
bool hasArchitectedFlatScratch(const MCSubtargetInfo &STI)
int32_t getMCOpcode(uint32_t Opcode, unsigned Gen)
bool getMUBUFHasSoffset(unsigned Opc)
bool isNotGFX11Plus(const MCSubtargetInfo &STI)
bool isGFX11Plus(const MCSubtargetInfo &STI)
std::optional< unsigned > getInlineEncodingV2F16(uint32_t Literal)
bool isSISrcFPOperand(const MCInstrDesc &Desc, unsigned OpNo)
Is this floating-point operand?
std::optional< APFloat > evaluateRcp(const APFloat &Val)
Evaluate the constant-folded result of v_rcp for Val, accounting for the hardware's denormal flushing...
std::tuple< char, unsigned, unsigned > parseAsmConstraintPhysReg(StringRef Constraint)
Returns a valid charcode or 0 in the first entry if this is a valid physical register constraint.
unsigned getHostcallImplicitArgPosition(unsigned CodeObjectVersion)
static unsigned getDefaultCustomOperandEncoding(const CustomOperandVal *Opr, int Size, const MCSubtargetInfo &STI)
static unsigned encodeLoadcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Loadcnt)
bool isGFX10Plus(const MCSubtargetInfo &STI)
static bool decodeCustomOperand(const CustomOperandVal *Opr, int Size, unsigned Code, int &Idx, StringRef &Name, unsigned &Val, bool &IsDefault, const MCSubtargetInfo &STI)
static bool isValidRegPrefix(char C)
std::optional< int64_t > getSMRDEncodedOffset(const MCSubtargetInfo &ST, int64_t ByteOffset, bool IsBuffer, bool HasSOffset)
bool isGlobalSegment(const GlobalValue *GV)
SmallVector< unsigned > getMaxNumWorkGroups(const Function &F)
int64_t encode32BitLiteral(int64_t Imm, OperandType Type, bool IsLit)
bool isValidWMMAScaleFmtCombination(unsigned AFmt, unsigned AScale, unsigned BFmt, unsigned BScale)
@ OPERAND_REG_IMM_V2FP64
Definition SIDefines.h:447
@ OPERAND_KIMM32
Operand with 32-bit immediate that uses the constant bus.
Definition SIDefines.h:465
@ OPERAND_REG_INLINE_C_LAST
Definition SIDefines.h:488
@ OPERAND_REG_IMM_V2FP16
Definition SIDefines.h:440
@ OPERAND_REG_INLINE_C_FP64
Definition SIDefines.h:456
@ OPERAND_REG_IMM_NOINLINE_FP16
Definition SIDefines.h:438
@ OPERAND_REG_INLINE_C_BF16
Definition SIDefines.h:453
@ OPERAND_REG_INLINE_C_V2BF16
Definition SIDefines.h:458
@ OPERAND_REG_IMM_V2INT16
Definition SIDefines.h:442
@ OPERAND_REG_IMM_BF16
Definition SIDefines.h:436
@ OPERAND_REG_IMM_INT32
Operands with register, 32-bit, or 64-bit immediate.
Definition SIDefines.h:431
@ OPERAND_REG_IMM_V2BF16
Definition SIDefines.h:439
@ OPERAND_REG_INLINE_AC_FIRST
Definition SIDefines.h:490
@ OPERAND_REG_IMM_FP16
Definition SIDefines.h:437
@ OPERAND_REG_IMM_V2FP16_SPLAT
Definition SIDefines.h:441
@ OPERAND_REG_IMM_NOINLINE_V2FP16
Definition SIDefines.h:444
@ OPERAND_REG_IMM_FP64
Definition SIDefines.h:435
@ OPERAND_REG_INLINE_C_V2FP16
Definition SIDefines.h:459
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
Definition SIDefines.h:470
@ OPERAND_REG_INLINE_AC_FP32
Definition SIDefines.h:471
@ OPERAND_REG_IMM_V2INT32
Definition SIDefines.h:445
@ OPERAND_REG_IMM_FP32
Definition SIDefines.h:434
@ OPERAND_REG_INLINE_C_FIRST
Definition SIDefines.h:487
@ OPERAND_REG_INLINE_C_FP32
Definition SIDefines.h:455
@ OPERAND_REG_INLINE_AC_LAST
Definition SIDefines.h:491
@ OPERAND_REG_INLINE_C_INT32
Definition SIDefines.h:451
@ OPERAND_REG_INLINE_C_V2INT16
Definition SIDefines.h:457
@ OPERAND_REG_IMM_V2FP32
Definition SIDefines.h:446
@ OPERAND_REG_INLINE_AC_FP64
Definition SIDefines.h:472
@ OPERAND_REG_INLINE_C_FP16
Definition SIDefines.h:454
@ OPERAND_INLINE_SPLIT_BARRIER_INT32
Definition SIDefines.h:462
std::optional< unsigned > getPKFMACF16InlineEncoding(uint32_t Literal, bool IsGFX11Plus)
bool isNotGFX9Plus(const MCSubtargetInfo &STI)
bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII, const MCSubtargetInfo &ST)
bool hasGDS(const MCSubtargetInfo &STI)
bool isLegalSMRDEncodedUnsignedOffset(const MCSubtargetInfo &ST, int64_t EncodedOffset)
bool isGFX9Plus(const MCSubtargetInfo &STI)
bool hasDPPSrc1SGPR(const MCSubtargetInfo &STI)
const int OPR_ID_DUPLICATE
bool isVOPD(unsigned Opc)
VOPD::InstInfo getVOPDInstInfo(const MCInstrDesc &OpX, const MCInstrDesc &OpY)
unsigned encodeVmcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Vmcnt)
unsigned decodeExpcnt(const IsaVersion &Version, unsigned Waitcnt)
bool isRsrcIndexReg(MCRegister Reg, const MCRegisterInfo &MRI)
bool isCvt_F32_Fp8_Bf8_e64(unsigned Opc)
std::optional< unsigned > getInlineEncodingV2I16(uint32_t Literal)
unsigned encodeStorecntDscnt(const IsaVersion &Version, const Waitcnt &Decoded)
bool isGFX1250(const MCSubtargetInfo &STI)
const MIMGBaseOpcodeInfo * getMIMGBaseOpcode(unsigned Opc)
bool isVI(const MCSubtargetInfo &STI)
bool isSingleSGPRReadInst(unsigned Opc)
Packed instructions that read a single SGPR for SGPR operands, except for 64-bit elements which read ...
bool isTensorStore(unsigned Opc)
bool getMUBUFIsBufferInv(unsigned Opc)
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
MCRegister mc2PseudoReg(MCRegister Reg)
Convert hardware register Reg to a pseudo register.
std::optional< unsigned > getInlineEncodingV2BF16(uint32_t Literal)
static int encodeCustomOperand(const CustomOperandVal *Opr, int Size, const StringRef Name, int64_t InputVal, unsigned &UsedOprMask, const MCSubtargetInfo &STI)
unsigned hasKernargPreload(const MCSubtargetInfo &STI)
bool supportsWGP(const MCSubtargetInfo &STI)
bool isMAC(unsigned Opc)
bool isCI(const MCSubtargetInfo &STI)
unsigned encodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Lgkmcnt)
bool getVOP2IsSingle(unsigned Opc)
bool getMAIIsDGEMM(unsigned Opc)
Returns true if MAI operation is a double precision GEMM.
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
const int OPR_ID_UNKNOWN
unsigned getCompletionActionImplicitArgPosition(unsigned CodeObjectVersion)
SmallVector< unsigned > getIntegerVecAttribute(const Function &F, StringRef Name, unsigned Size, unsigned DefaultVal)
unsigned decodeStorecnt(const IsaVersion &Version, unsigned Waitcnt)
bool isGFX1250Plus(const MCSubtargetInfo &STI)
int getMaskedMIMGOp(unsigned Opc, unsigned NewChannels)
bool isNotGFX12Plus(const MCSubtargetInfo &STI)
bool getMTBUFHasVAddr(unsigned Opc)
bool hasPopsExitingWaveID(const MCSubtargetInfo &STI)
unsigned decodeVmcnt(const IsaVersion &Version, unsigned Waitcnt)
uint8_t getELFABIVersion(const Triple &T, unsigned CodeObjectVersion)
std::pair< unsigned, unsigned > getIntegerPairAttribute(const Function &F, StringRef Name, std::pair< unsigned, unsigned > Default, bool OnlyFirstRequired)
unsigned getLoadcntBitMask(const IsaVersion &Version)
bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi)
bool hasVOPD(const MCSubtargetInfo &STI)
int getVOPDFull(unsigned OpX, unsigned OpY, unsigned EncodingFamily, bool VOPD3)
static unsigned encodeDscnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Dscnt)
bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi)
Is this literal inlinable.
const MFMA_F8F6F4_Info * getMFMA_F8F6F4_WithFormatArgs(unsigned CBSZ, unsigned BLGP, unsigned F8F8Opcode)
unsigned decodeLoadcnt(const IsaVersion &Version, unsigned Waitcnt)
unsigned getMultigridSyncArgImplicitArgPosition(unsigned CodeObjectVersion)
bool isGFX9_GFX10_GFX11(const MCSubtargetInfo &STI)
bool isGFX9_GFX10(const MCSubtargetInfo &STI)
int getMUBUFElements(unsigned Opc)
const GcnBufferFormatInfo * getGcnBufferFormatInfo(uint8_t BitsPerComp, uint8_t NumComponents, uint8_t NumFormat, const MCSubtargetInfo &STI)
unsigned mapWMMA3AddrTo2AddrOpcode(unsigned Opc)
bool isPermlane16(unsigned Opc)
bool getMUBUFHasSrsrc(unsigned Opc)
unsigned getDscntBitMask(const IsaVersion &Version)
bool hasAny64BitVGPROperands(const MCInstrDesc &OpDesc, const MCInstrInfo &MII, const MCSubtargetInfo &ST)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
@ AMDGPU_VS
Used for Mesa vertex shaders, or AMDPAL last shader stage before rasterization (vertex shader if tess...
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ AMDGPU_Gfx
Used for AMD graphics targets.
@ AMDGPU_CS_ChainPreserve
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_HS
Used for Mesa/AMDPAL hull shaders (= tessellation control shaders).
@ AMDGPU_GS
Used for Mesa/AMDPAL geometry shaders.
@ AMDGPU_CS_Chain
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_PS
Used for Mesa/AMDPAL pixel shaders.
@ SPIR_KERNEL
Used for SPIR kernel functions.
@ AMDGPU_ES
Used for AMDPAL shader stage before geometry shader if geometry is in use.
@ AMDGPU_LS
Used for AMDPAL vertex shader if tessellation is in use.
@ ELFABIVERSION_AMDGPU_HSA_V4
Definition ELF.h:384
@ ELFABIVERSION_AMDGPU_HSA_V5
Definition ELF.h:385
@ ELFABIVERSION_AMDGPU_HSA_V6
Definition ELF.h:386
constexpr bool isVOPC(const T &...O)
Definition SIDefines.h:236
constexpr bool isVOP3(const T &...O)
Definition SIDefines.h:239
constexpr bool isVOP1(const T &...O)
Definition SIDefines.h:230
constexpr bool isVOP2(const T &...O)
Definition SIDefines.h:233
constexpr bool isFLAT(const T &...O)
Definition SIDefines.h:286
constexpr bool isBuffer(const T &...O)
Definition SIDefines.h:267
constexpr bool isVIMAGE(const T &...O)
Definition SIDefines.h:277
constexpr bool isSMRD(const T &...O)
Definition SIDefines.h:271
constexpr bool isVOP3Like(const T &...O)
Definition SIDefines.h:245
constexpr bool isFlatScratch(const T &...O)
Definition SIDefines.h:361
constexpr bool isMIMG(const T &...O)
Definition SIDefines.h:274
constexpr bool isVOPD3(const T &...O)
Definition SIDefines.h:385
constexpr bool isEXP(const T &...O)
Definition SIDefines.h:283
constexpr bool isVSAMPLE(const T &...O)
Definition SIDefines.h:280
constexpr bool isDS(const T &...O)
Definition SIDefines.h:289
constexpr bool isAtomic(const T &...O)
Definition SIDefines.h:396
constexpr bool isDPP(const T &...O)
Definition SIDefines.h:255
initializer< Ty > init(const Ty &Val)
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract_or_null(Y &&MD)
Extract a Value from Metadata, allowing null.
Definition Metadata.h:683
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract(Y &&MD)
Extract a Value from Metadata.
Definition Metadata.h:668
This is an optimization pass for GlobalISel generic memory operations.
@ Low
Lower the current thread's priority such that it does not affect foreground tasks significantly.
Definition Threading.h:280
@ Offset
Definition DWP.cpp:577
constexpr T rotr(T V, int R)
Definition bit.h:399
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1739
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
testing::Matcher< const detail::ErrorHolder & > Failed()
Definition Error.h:198
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
std::string utostr(uint64_t X, bool isNeg=false)
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
Definition STLExtras.h:2173
Op::Description Desc
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
To bit_cast(const From &from) noexcept
Definition bit.h:90
DWARFExpression::Operation Op
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
constexpr int countr_zero_constexpr(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:190
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:78
@ AlwaysUniform
The result value is always uniform.
Definition Uniformity.h:23
@ Default
The result value is uniform if and only if all operands are uniform.
Definition Uniformity.h:20
#define N
AMD Kernel Code Object (amd_kernel_code_t).
static std::tuple< typename Fields::ValueType... > decode(uint64_t Encoded)
Instruction set architecture version.