LLVM 24.0.0git
GCNSubtarget.h
Go to the documentation of this file.
1//=====-- GCNSubtarget.h - Define GCN Subtarget for AMDGPU ------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//==-----------------------------------------------------------------------===//
8//
9/// \file
10/// AMD GCN specific subclass of TargetSubtarget.
11//
12//===----------------------------------------------------------------------===//
13
14#ifndef LLVM_LIB_TARGET_AMDGPU_GCNSUBTARGET_H
15#define LLVM_LIB_TARGET_AMDGPU_GCNSUBTARGET_H
16
17#include "AMDGPUCallLowering.h"
19#include "AMDGPUSubtarget.h"
20#include "SIFrameLowering.h"
21#include "SIISelLowering.h"
22#include "SIInstrInfo.h"
26
27#define GET_SUBTARGETINFO_HEADER
28#include "AMDGPUGenSubtargetInfo.inc"
29
30namespace llvm {
31
32class GCNTargetMachine;
33
34/// Module flag names controlling out-of-bounds buffer access semantics.
35/// Each flag is an i32 with Module::Max merge behaviour and tri-state values:
36/// 0 = any (absent/default - backend currently treats as strict)
37/// 1 = relaxed
38/// 2 = strict
39namespace AMDGPUOOBMode {
40inline constexpr StringLiteral BufferFlag("amdgpu.buffer.oob.mode");
41inline constexpr StringLiteral TBufferFlag("amdgpu.tbuffer.oob.mode");
42} // namespace AMDGPUOOBMode
43
45 public AMDGPUSubtarget {
46public:
48
49 // Following 2 enums are documented at:
50 // - https://llvm.org/docs/AMDGPUUsage.html#trap-handler-abi
51 enum class TrapHandlerAbi {
52 NONE = 0x00,
53 AMDHSA = 0x01,
54 };
55
56 enum class TrapID {
59 };
60
61private:
62 /// SelectionDAGISel related APIs.
63 std::unique_ptr<const SelectionDAGTargetInfo> TSInfo;
64
65 /// GlobalISel related APIs.
66 std::unique_ptr<AMDGPUCallLowering> CallLoweringInfo;
67 std::unique_ptr<InlineAsmLowering> InlineAsmLoweringInfo;
68 std::unique_ptr<InstructionSelector> InstSelector;
69 std::unique_ptr<LegalizerInfo> Legalizer;
70 std::unique_ptr<AMDGPURegisterBankInfo> RegBankInfo;
71
72protected:
73 // Basic subtarget description.
75 unsigned Gen = INVALID;
77 int LDSBankCount = 0;
79
80 // Instruction cache line size in bytes; set from TableGen subtarget features.
81 unsigned InstCacheLineSize = 0;
82
83 // Data (VMEM) cache line size in bytes; set from TableGen subtarget features.
84 unsigned DataCacheLineSize = 0;
85
86 // Dynamically set bits that enable features.
87 bool ScalarizeGlobal = false;
88 const bool BufferOOBRelaxed;
90
91 /// The maximum number of instructions that may be placed within an S_CLAUSE,
92 /// which is one greater than the maximum argument to S_CLAUSE. A value of 0
93 /// indicates a lack of S_CLAUSE support.
94 unsigned MaxHardClauseLength = 0;
95
96#define GET_SUBTARGETINFO_MACRO(ATTRIBUTE, DEFAULT, GETTER) \
97 bool ATTRIBUTE = DEFAULT;
98#include "AMDGPUGenSubtargetInfo.inc"
99
100private:
101 SIInstrInfo InstrInfo;
102 SITargetLowering TLInfo;
103 SIFrameLowering FrameLowering;
104
105 /// Get the register that represents the actual dependency between the
106 /// definition and the use. The definition might only affect a subregister
107 /// that is not actually used. Works for both virtual and physical registers.
108 /// Note: Currently supports VOP3P instructions (without WMMA an SWMMAC).
109 /// Returns the definition register if there is a real dependency and no
110 /// better match is found.
111 Register getRealSchedDependency(const MachineInstr &DefI, int DefOpIdx,
112 const MachineInstr &UseI, int UseOpIdx) const;
113
114public:
116 const Triple &TT, StringRef GPU, StringRef FS, const GCNTargetMachine &TM,
117 bool BufferOOBRelaxed = false, bool TBufferOOBRelaxed = false,
118 AMDGPU::TargetIDSetting XnackSetting = AMDGPU::TargetIDSetting::Any,
119 AMDGPU::TargetIDSetting SramEccSetting = AMDGPU::TargetIDSetting::Any);
120 ~GCNSubtarget() override;
121
123 StringRef FS);
124
125 /// Diagnose inconsistent subtarget features before attempting to codegen
126 /// function \p F.
127 void checkSubtargetFeatures(const Function &F) const;
128
129 const SIInstrInfo *getInstrInfo() const override { return &InstrInfo; }
130
131 const SIFrameLowering *getFrameLowering() const override {
132 return &FrameLowering;
133 }
134
135 const SITargetLowering *getTargetLowering() const override { return &TLInfo; }
136
137 const SIRegisterInfo *getRegisterInfo() const override {
138 return &InstrInfo.getRegisterInfo();
139 }
140
141 const SelectionDAGTargetInfo *getSelectionDAGInfo() const override;
142
143 const CallLowering *getCallLowering() const override {
144 return CallLoweringInfo.get();
145 }
146
147 const InlineAsmLowering *getInlineAsmLowering() const override {
148 return InlineAsmLoweringInfo.get();
149 }
150
152 return InstSelector.get();
153 }
154
155 const LegalizerInfo *getLegalizerInfo() const override {
156 return Legalizer.get();
157 }
158
159 const AMDGPURegisterBankInfo *getRegBankInfo() const override {
160 return RegBankInfo.get();
161 }
162
163 const AMDGPU::TargetID &getTargetID() const { return TargetID; }
164
166 return &InstrItins;
167 }
168
170
172
173 bool isGFX11Plus() const { return getGeneration() >= GFX11; }
174
175#define GET_SUBTARGETINFO_MACRO(ATTRIBUTE, DEFAULT, GETTER) \
176 bool GETTER() const override { return ATTRIBUTE; }
177#include "AMDGPUGenSubtargetInfo.inc"
178
179 unsigned getMaxWaveScratchSize() const {
180 // See COMPUTE_TMPRING_SIZE.WAVESIZE.
181 if (getGeneration() >= GFX12) {
182 // 18-bit field in units of 64-dword.
183 return (64 * 4) * ((1 << 18) - 1);
184 }
185 if (getGeneration() == GFX11) {
186 // 15-bit field in units of 64-dword.
187 return (64 * 4) * ((1 << 15) - 1);
188 }
189 // 13-bit field in units of 256-dword.
190 return (256 * 4) * ((1 << 13) - 1);
191 }
192
193 /// Return the number of high bits known to be zero for a frame index.
197
198 int getLDSBankCount() const { return LDSBankCount; }
199
200 /// Instruction cache line size in bytes (64 for pre-GFX11, 128 for GFX11+).
201 unsigned getInstCacheLineSize() const { return InstCacheLineSize; }
202
203 /// Data (VMEM) cache line size in bytes (128 for gfx12), has no use before
204 /// GFX12.
205 unsigned getDataCacheLineSize() const { return DataCacheLineSize; }
206
207 unsigned getMaxPrivateElementSize(bool ForBufferRSrc = false) const {
208 return (ForBufferRSrc || !hasFlatScratchEnabled()) ? MaxPrivateElementSize
209 : 16;
210 }
211
212 unsigned getConstantBusLimit(unsigned Opcode) const;
213
214 /// Returns if the result of this instruction with a 16-bit result returned in
215 /// a 32-bit register implicitly zeroes the high 16-bits, rather than preserve
216 /// the original value.
217 bool zeroesHigh16BitsOfDest(unsigned Opcode) const;
218
219 bool hasHWFP64() const { return HasFP64; }
220
221 bool hasAddr64() const {
223 }
224
225 bool hasFlat() const {
227 }
228
229 // Return true if the target only has the reverse operand versions of VALU
230 // shift instructions (e.g. v_lshrrev_b32, and no v_lshr_b32).
231 bool hasOnlyRevVALUShifts() const {
233 }
234
235 bool hasFractBug() const { return getGeneration() == SOUTHERN_ISLANDS; }
236
237 bool hasMed3_16() const { return getGeneration() >= AMDGPUSubtarget::GFX9; }
238
239 bool hasMin3Max3_16() const {
241 }
242
243 bool hasSwap() const { return HasGFX9Insts; }
244
245 bool hasScalarPackInsts() const { return HasGFX9Insts; }
246
247 bool hasScalarMulHiInsts() const { return HasGFX9Insts; }
248
249 bool hasScalarSubwordLoads() const { return getGeneration() >= GFX12; }
250
251 bool hasAsyncMark() const { return hasVMemToLDSLoad() || HasAsynccnt; }
252
256
258 // The S_GETREG DOORBELL_ID is supported by all GFX9 onward targets.
259 return getGeneration() >= GFX9;
260 }
261
262 /// True if the offset field of DS instructions works as expected. On SI, the
263 /// offset uses a 16-bit adder and does not always wrap properly.
264 bool hasUsableDSOffset() const { return getGeneration() >= SEA_ISLANDS; }
265
267 return EnableUnsafeDSOffsetFolding;
268 }
269
270 /// Condition output from div_scale is usable.
274
275 /// Extra wait hazard is needed in some cases before
276 /// s_cbranch_vccnz/s_cbranch_vccz.
277 bool hasReadVCCZBug() const { return getGeneration() <= SEA_ISLANDS; }
278
279 /// Writes to VCC_LO/VCC_HI update the VCCZ flag.
280 bool partialVCCWritesUpdateVCCZ() const { return getGeneration() >= GFX10; }
281
282 /// A read of an SGPR by SMRD instruction requires 4 wait states when the SGPR
283 /// was written by a VALU instruction.
286 }
287
288 /// A read of an SGPR by a VMEM instruction requires 5 wait states when the
289 /// SGPR was written by a VALU Instruction.
292 }
293
294 bool hasRFEHazards() const { return getGeneration() >= VOLCANIC_ISLANDS; }
295
296 /// Number of hazard wait states for s_setreg_b32/s_setreg_imm32_b32.
297 unsigned getSetRegWaitStates() const {
298 return getGeneration() <= SEA_ISLANDS ? 1 : 2;
299 }
300
301 /// Return the amount of LDS that can be used that will not restrict the
302 /// occupancy lower than WaveCount.
303 unsigned getMaxLocalMemSizeWithWaveCount(unsigned WaveCount,
304 const Function &) const;
305
308 }
309
310 /// \returns If target supports S_DENORM_MODE.
311 bool hasDenormModeInst() const {
313 }
314
315 /// \returns If target supports ds_read/write_b128 and user enables generation
316 /// of ds_read/write_b128.
317 bool useDS128() const { return HasCIInsts && EnableDS128; }
318
319 /// \return If target supports ds_read/write_b96/128.
320 bool hasDS96AndDS128() const { return HasCIInsts; }
321
322 /// Have v_trunc_f64, v_ceil_f64, v_rndne_f64
323 bool haveRoundOpsF64() const { return HasCIInsts; }
324
325 /// \returns If MUBUF instructions always perform range checking, even for
326 /// buffer resources used for private memory access.
330
331 /// \returns If target requires PRT Struct NULL support (zero result registers
332 /// for sparse texture support).
333 bool usePRTStrictNull() const { return EnablePRTStrictNull; }
334
336 return HasUnalignedBufferAccess && HasUnalignedAccessMode;
337 }
338
340 return HasUnalignedDSAccess && HasUnalignedAccessMode;
341 }
342
344 return HasUnalignedScratchAccess && HasUnalignedAccessMode;
345 }
346
347 bool isXNACKEnabled() const {
348 return enableXNACK() || TargetID.isXnackOnOrAny();
349 }
350
353
354 bool isCuModeEnabled() const { return EnableCuMode; }
355
356 bool isPreciseMemoryEnabled() const { return EnablePreciseMemory; }
357
358 bool hasFlatScrRegister() const { return hasFlatAddressSpace(); }
359
360 // Check if target supports ST addressing mode with FLAT scratch instructions.
361 // The ST addressing mode means no registers are used, either VGPR or SGPR,
362 // but only immediate offset is swizzled and added to the FLAT scratch base.
363 bool hasFlatScratchSTMode() const {
364 return hasFlatScratchInsts() && (hasGFX10_3Insts() || hasGFX940Insts());
365 }
366
367 bool hasFlatScratchSVSMode() const { return HasGFX940Insts || HasGFX11Insts; }
368
370 return hasArchitectedFlatScratch() ||
371 (EnableFlatScratch && hasFlatScratchInsts());
372 }
373
374 bool hasGlobalAddTidInsts() const { return HasGFX10_BEncoding; }
375
376 bool hasAtomicCSub() const { return HasGFX10_BEncoding; }
377
378 bool hasExportInsts() const {
379 return !hasGFX940Insts() && !hasGFX1250Insts();
380 }
381
382 bool hasVINTERPEncoding() const {
383 return HasGFX11Insts && !hasGFX1250Insts();
384 }
385
387 return getGeneration() >= GFX9;
388 }
389
390 bool hasFlatLgkmVMemCountInOrder() const { return getGeneration() > GFX9; }
391
392 bool hasD16LoadStore() const { return getGeneration() >= GFX9; }
393
395 return hasD16LoadStore() && !TargetID.isSramEccOnOrAny();
396 }
397
398 bool hasD16Images() const { return getGeneration() >= VOLCANIC_ISLANDS; }
399
400 /// Return if most LDS instructions have an m0 use that require m0 to be
401 /// initialized.
402 bool ldsRequiresM0Init() const { return getGeneration() < GFX9; }
403
404 // True if the hardware rewinds and replays GWS operations if a wave is
405 // preempted.
406 //
407 // If this is false, a GWS operation requires testing if a nack set the
408 // MEM_VIOL bit, and repeating if so.
409 bool hasGWSAutoReplay() const { return getGeneration() >= GFX9; }
410
411 /// \returns if target has ds_gws_sema_release_all instruction.
412 bool hasGWSSemaReleaseAll() const { return HasCIInsts; }
413
414 bool hasScalarAddSub64() const { return getGeneration() >= GFX12; }
415
416 bool hasScalarSMulU64() const { return getGeneration() >= GFX12; }
417
418 // Covers VS/PS/CS graphics shaders
419 bool isMesaGfxShader(const Function &F) const {
420 return isMesa3DOS() && AMDGPU::isShader(F.getCallingConv());
421 }
422
423 bool hasMad64_32() const { return getGeneration() >= SEA_ISLANDS; }
424
425 bool hasAtomicFaddInsts() const {
426 return HasAtomicFaddRtnInsts || HasAtomicFaddNoRtnInsts;
427 }
428
430 return getGeneration() < SEA_ISLANDS;
431 }
432
433 bool hasInstPrefetch() const {
434 return getGeneration() == GFX10 || getGeneration() == GFX11;
435 }
436
437 bool hasPrefetch() const { return HasGFX12Insts; }
438
439 bool hasInstPrefSize() const { return isGFX11Plus(); }
440
441 void getInstPrefSizeArgs(uint32_t &Mask, uint32_t &Shift, uint32_t &Width,
442 uint32_t &CacheLineSize) const {
445 if (getGeneration() == GFX11) {
446 Mask = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE;
447 Shift = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_SHIFT;
448 Width = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_WIDTH;
449 } else {
450 Mask = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE;
451 Shift = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_SHIFT;
452 Width = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_WIDTH;
453 }
454 }
455
456 // Has s_cmpk_* instructions.
457 bool hasSCmpK() const { return getGeneration() < GFX12; }
458
459 // Scratch is allocated in 256 dword per wave blocks for the entire
460 // wavefront. When viewed from the perspective of an arbitrary workitem, this
461 // is 4-byte aligned.
462 //
463 // Only 4-byte alignment is really needed to access anything. Transformations
464 // on the pointer value itself may rely on the alignment / known low bits of
465 // the pointer. Set this to something above the minimum to avoid needing
466 // dynamic realignment in common cases.
467 Align getStackAlignment() const { return Align(16); }
468
469 bool enableMachineScheduler() const override { return true; }
470
471 bool useAA() const override;
472
473 bool enableSubRegLiveness() const override { return true; }
474
477
478 // static wrappers
479 static bool hasHalfRate64Ops(const TargetSubtargetInfo &STI);
480
481 // XXX - Why is this here if it isn't in the default pass set?
482 bool enableEarlyIfConversion() const override { return true; }
483
485 const SchedRegion &Region) const override;
486
488 const SchedRegion &Region) const override;
489
490 void mirFileLoaded(MachineFunction &MF) const override;
491
492 unsigned getMaxNumUserSGPRs() const {
493 return AMDGPU::getMaxNumUserSGPRs(*this);
494 }
495
496 bool useVGPRIndexMode() const;
497
498 bool hasScalarCompareEq64() const {
500 }
501
502 bool hasLDSFPAtomicAddF32() const { return HasGFX8Insts; }
503 bool hasLDSFPAtomicAddF64() const {
504 return HasGFX90AInsts || HasGFX1250Insts;
505 }
506
507 /// \returns true if the subtarget has the v_permlane64_b32 instruction.
508 bool hasPermLane64() const { return getGeneration() >= GFX11; }
509
510 /// \returns true if the subtarget supports the ds_swizzle rotate and FFT
511 /// swizzle modes (GFX9+).
512 bool hasDsSwizzleRotateMode() const { return getGeneration() >= GFX9; }
513
514 bool hasDPPRowShare() const {
515 return HasDPP && (HasGFX90AInsts || getGeneration() >= GFX10);
516 }
517
518 // Has V_PK_MOV_B32 opcode
519 bool hasPkMovB32() const { return HasGFX90AInsts; }
520
522 return getGeneration() >= GFX10 || hasGFX940Insts();
523 }
524
525 bool hasFmaakFmamkF64Insts() const { return hasGFX1250Insts(); }
526
527 bool hasNonNSAEncoding() const { return getGeneration() < GFX12; }
528
529 unsigned getNSAMaxSize(bool HasSampler = false) const {
530 return AMDGPU::getNSAMaxSize(*this, HasSampler);
531 }
532
533 bool hasMadF16() const;
534
535 // Scalar and global loads support scale_offset bit.
536 bool hasScaleOffset() const { return HasGFX1250Insts; }
537
538 // FLAT GLOBAL VOffset is signed
539 bool hasSignedGVSOffset() const { return HasGFX1250Insts; }
540
542
544 return HasUserSGPRInit16Bug && isWave32();
545 }
546
550
551 // \returns true if the subtarget supports DWORDX3 load/store instructions.
552 bool hasDwordx3LoadStores() const { return HasCIInsts; }
553
557
562
565 }
566
569 }
570
572 return HasLDSMisalignedBug && !EnableCuMode;
573 }
574
575 // Shift amount of a 64 bit shift cannot be a highest allocated register
576 // if also at the end of the allocation block.
577 bool hasShift64HighRegBug() const { return HasGFX90AInsts; }
578
579 // v_dot2c_f32_f16 unconditionally flushes f16 subnormal inputs to zero
580 // regardless of the MODE register, unlike v_fma_mix_f32 which respects it.
582 return HasGFX90AInsts && !HasGFX940Insts;
583 }
584
585 // Has one cycle hazard on transcendental instruction feeding a
586 // non transcendental VALU.
587 bool hasTransForwardingHazard() const { return HasGFX940Insts; }
588
589 // Has one cycle hazard on a VALU instruction partially writing dst with
590 // a shift of result bits feeding another VALU instruction.
591 bool hasDstSelForwardingHazard() const { return HasGFX940Insts; }
592
593 // Cannot use op_sel with v_dot instructions.
594 bool hasDOTOpSelHazard() const { return HasGFX940Insts || HasGFX11Insts; }
595
596 // Does not have HW interlocs for VALU writing and then reading SGPRs.
597 bool hasVDecCoExecHazard() const { return HasGFX940Insts; }
598
599 bool hasHardClauses() const { return MaxHardClauseLength > 0; }
600
602 return getGeneration() == GFX10;
603 }
604
605 bool hasVOP3DPP() const { return getGeneration() >= GFX11; }
606
607 bool hasLdsDirect() const { return getGeneration() >= GFX11; }
608
609 bool hasLdsWaitVMSRC() const { return getGeneration() >= GFX12; }
610
612 return getGeneration() == GFX11;
613 }
614
615 bool hasCvtScaleForwardingHazard() const { return HasGFX950Insts; }
616
617 // All GFX9 targets experience a fetch delay when an instruction at the start
618 // of a loop header is split by a 32-byte fetch window boundary, but GFX950
619 // is uniquely sensitive to this: the delay triggers further performance
620 // degradation beyond the fetch latency itself.
621 bool hasLoopHeadInstSplitSensitivity() const { return HasGFX950Insts; }
622
623 bool requiresCodeObjectV6() const { return RequiresCOV6; }
624
625 bool useVGPRBlockOpsForCSR() const { return UseBlockVGPROpsForCSR; }
626
627 bool hasVALUMaskWriteHazard() const { return getGeneration() == GFX11; }
628
630 return HasGFX12Insts && !HasGFX1250Insts;
631 }
632
633 bool setRegModeNeedsVNOPs() const {
634 return HasGFX1250Insts && getGeneration() == GFX12;
635 }
636
637 /// Return if operations acting on VGPR tuples require even alignment.
638 bool needsAlignedVGPRs() const { return RequiresAlignVGPR; }
639
640 /// Return true if the target has the S_PACK_HL_B32_B16 instruction.
641 bool hasSPackHL() const { return HasGFX11Insts; }
642
643 /// Return true if the target has the V_CVT_PK_I16_F32/V_CVT_PK_U16_F32
644 /// instructions.
645 bool hasVCvtPkIU16F32() const { return HasGFX11Insts; }
646
647 /// Return true if the target's EXP instruction supports the NULL export
648 /// target.
649 bool hasNullExportTarget() const { return !HasGFX11Insts; }
650
651 bool hasFlatScratchSVSSwizzleBug() const { return getGeneration() == GFX11; }
652
653 /// Return true if the target has the S_DELAY_ALU instruction.
654 bool hasDelayAlu() const { return HasGFX11Insts; }
655
656 /// Returns true if the target supports
657 /// global_load_lds_dwordx3/global_load_lds_dwordx4 or
658 /// buffer_load_dwordx3/buffer_load_dwordx4 with the lds bit.
659 bool hasLDSLoadB96_B128() const { return hasGFX950Insts(); }
660
661 /// \returns true if the target uses LOADcnt/SAMPLEcnt/BVHcnt, DScnt/KMcnt
662 /// and STOREcnt rather than VMcnt, LGKMcnt and VScnt respectively.
663 bool hasExtendedWaitCounts() const { return getGeneration() >= GFX12; }
664
665 /// \returns true if the target has packed f32 instructions that only read 32
666 /// bits from a scalar operand (SGPR or literal) and replicates the bits to
667 /// both channels.
669 return getGeneration() == GFX12 && HasGFX1250Insts;
670 }
671
672 bool hasAddPC64Inst() const { return HasGFX1250Insts; }
673
674 /// \returns true if the target supports expert scheduling mode 2 which relies
675 /// on the compiler to insert waits to avoid hazards between VMEM and VALU
676 /// instructions in some instances.
677 bool hasExpertSchedulingMode() const { return getGeneration() >= GFX12; }
678
679 /// \returns The maximum number of instructions that can be enclosed in an
680 /// S_CLAUSE on the given subtarget, or 0 for targets that do not support that
681 /// instruction.
682 unsigned maxHardClauseLength() const { return MaxHardClauseLength; }
683
684 /// Return the maximum number of waves per SIMD for kernels using \p SGPRs
685 /// SGPRs
686 unsigned getOccupancyWithNumSGPRs(unsigned SGPRs) const;
687
688 /// Return the maximum number of waves per SIMD for kernels using \p VGPRs
689 /// VGPRs
690 unsigned getOccupancyWithNumVGPRs(unsigned VGPRs,
691 unsigned DynamicVGPRBlockSize) const;
692
693 /// Subtarget's minimum/maximum occupancy, in number of waves per EU, that can
694 /// be achieved when the only function running on a CU is \p F, each workgroup
695 /// uses \p LDSSize bytes of LDS, and each wave uses \p NumSGPRs SGPRs and \p
696 /// NumVGPRs VGPRs. The flat workgroup sizes associated to the function are a
697 /// range, so this returns a range as well.
698 ///
699 /// Note that occupancy can be affected by the scratch allocation as well, but
700 /// we do not have enough information to compute it.
701 std::pair<unsigned, unsigned> computeOccupancy(const Function &F,
702 unsigned LDSSize = 0,
703 unsigned NumSGPRs = 0,
704 unsigned NumVGPRs = 0) const;
705
706 /// \returns true if the flat_scratch register should be initialized with the
707 /// pointer to the wave's scratch memory rather than a size and offset.
708 bool flatScratchIsPointer() const {
710 }
711
712 /// \returns true if the machine has merged shaders in which s0-s7 are
713 /// reserved by the hardware and user SGPRs start at s8
714 bool hasMergedShaders() const { return getGeneration() >= GFX9; }
715
716 // \returns true if the target supports the pre-NGG legacy geometry path.
717 bool hasLegacyGeometry() const { return getGeneration() < GFX11; }
718
719 // \returns true if the target has split barriers feature
720 bool hasSplitBarriers() const { return getGeneration() >= GFX12; }
721
722 // \returns true if the target has WG_RR_MODE kernel descriptor mode bit
723 bool hasRrWGMode() const { return getGeneration() >= GFX12; }
724
725 /// \returns true if VADDR and SADDR fields in VSCRATCH can use negative
726 /// values.
727 bool hasSignedScratchOffsets() const { return getGeneration() >= GFX12; }
728
729 bool hasINVWBL2WaitCntRequirement() const { return HasGFX1250Insts; }
730
731 bool hasVOPD3() const { return HasGFX1250Insts; }
732
733 // \returns true if the target has V_PK_{MIN|MAX}3_{I|U}16 instructions.
734 bool hasPkMinMax3Insts() const { return HasGFX1250Insts; }
735
736 // \returns ture if target has S_GET_SHADER_CYCLES_U64 instruction.
737 bool hasSGetShaderCyclesInst() const { return HasGFX1250Insts; }
738
739 // \returns true if S_GETPC_B64 zero-extends the result from 48 bits instead
740 // of sign-extending. Note that GFX1250 has not only fixed the bug but also
741 // extended VA to 57 bits.
743 return HasGFX12Insts && !HasGFX1250Insts;
744 }
745
746 // \returns true if the target needs to create a prolog for backward
747 // compatibility when preloading kernel arguments.
749 return hasKernargPreload() && !HasGFX1250Insts;
750 }
751
752 bool hasCondSubInsts() const { return HasGFX12Insts; }
753
754 bool hasSubClampInsts() const { return hasGFX10_3Insts(); }
755
756 bool hasAnyPackedFP32Ops() const {
757 return hasPackedFP32Ops() || hasPackedFP32SingleSGPROps();
758 };
759
760 bool hasAnyPackedFP64Ops() const { return hasPackedFP64SingleSGPROps(); };
761
762 bool hasAnyPackedU64Ops() const { return hasPackedU64SingleSGPROps(); };
763
764 /// \returns SGPR allocation granularity supported by the subtarget.
765 unsigned getSGPRAllocGranule() const {
766 return AMDGPU::getSGPRAllocGranule(getTargetID().getGPUKind());
767 }
768
769 /// \returns SGPR encoding granularity supported by the subtarget.
770 unsigned getSGPREncodingGranule() const {
772 }
773
774 /// \returns Total number of SGPRs supported by the subtarget.
775 unsigned getTotalNumSGPRs() const {
776 return AMDGPU::getTotalNumSGPRs(getTargetID().getGPUKind());
777 }
778
779 /// \returns Addressable number of SGPRs supported by the subtarget.
780 unsigned getAddressableNumSGPRs() const {
781 return AMDGPU::getAddressableNumSGPRs(getTargetID().getGPUKind());
782 }
783
784 /// \returns Minimum number of SGPRs that meets the given number of waves per
785 /// execution unit requirement supported by the subtarget.
786 unsigned getMinNumSGPRs(unsigned WavesPerEU) const {
787 return AMDGPU::IsaInfo::getMinNumSGPRs(*this, WavesPerEU);
788 }
789
790 /// \returns Maximum number of SGPRs that meets the given number of waves per
791 /// execution unit requirement supported by the subtarget.
792 unsigned getMaxNumSGPRs(unsigned WavesPerEU, bool Addressable) const {
793 return AMDGPU::IsaInfo::getMaxNumSGPRs(*this, WavesPerEU, Addressable);
794 }
795
796 /// \returns Reserved number of SGPRs. This is common
797 /// utility function called by MachineFunction and
798 /// Function variants of getReservedNumSGPRs.
799 unsigned getBaseReservedNumSGPRs(const bool HasFlatScratch) const;
800 /// \returns Reserved number of SGPRs for given machine function \p MF.
801 unsigned getReservedNumSGPRs(const MachineFunction &MF) const;
802
803 /// \returns Reserved number of SGPRs for given function \p F.
804 unsigned getReservedNumSGPRs(const Function &F) const;
805
806 /// \returns Maximum number of preloaded SGPRs for the subtarget.
807 unsigned getMaxNumPreloadedSGPRs() const;
808
809 /// \returns max num SGPRs. This is the common utility
810 /// function called by MachineFunction and Function
811 /// variants of getMaxNumSGPRs.
812 unsigned getBaseMaxNumSGPRs(const Function &F,
813 std::pair<unsigned, unsigned> WavesPerEU,
814 unsigned PreloadedSGPRs,
815 unsigned ReservedNumSGPRs) const;
816
817 /// \returns Maximum number of SGPRs that meets number of waves per execution
818 /// unit requirement for function \p MF, or number of SGPRs explicitly
819 /// requested using "amdgpu-num-sgpr" attribute attached to function \p MF.
820 ///
821 /// \returns Value that meets number of waves per execution unit requirement
822 /// if explicitly requested value cannot be converted to integer, violates
823 /// subtarget's specifications, or does not meet number of waves per execution
824 /// unit requirement.
825 unsigned getMaxNumSGPRs(const MachineFunction &MF) const;
826
827 /// \returns Maximum number of SGPRs that meets number of waves per execution
828 /// unit requirement for function \p F, or number of SGPRs explicitly
829 /// requested using "amdgpu-num-sgpr" attribute attached to function \p F.
830 ///
831 /// \returns Value that meets number of waves per execution unit requirement
832 /// if explicitly requested value cannot be converted to integer, violates
833 /// subtarget's specifications, or does not meet number of waves per execution
834 /// unit requirement.
835 unsigned getMaxNumSGPRs(const Function &F) const;
836
837 /// \returns VGPR allocation granularity supported by the subtarget.
838 unsigned getVGPRAllocGranule(unsigned DynamicVGPRBlockSize) const {
839 return AMDGPU::IsaInfo::getVGPRAllocGranule(*this, DynamicVGPRBlockSize);
840 }
841
842 /// \returns VGPR encoding granularity supported by the subtarget.
843 unsigned getVGPREncodingGranule() const {
845 }
846
847 /// \returns Total number of VGPRs supported by the subtarget.
848 unsigned getTotalNumVGPRs() const {
850 }
851
852 /// \returns Addressable number of architectural VGPRs supported by the
853 /// subtarget.
857
858 /// \returns Addressable number of VGPRs supported by the subtarget.
859 unsigned getAddressableNumVGPRs(unsigned DynamicVGPRBlockSize) const {
860 return AMDGPU::IsaInfo::getAddressableNumVGPRs(*this, DynamicVGPRBlockSize);
861 }
862
863 /// \returns the minimum number of VGPRs that will prevent achieving more than
864 /// the specified number of waves \p WavesPerEU.
865 unsigned getMinNumVGPRs(unsigned WavesPerEU,
866 unsigned DynamicVGPRBlockSize) const {
867 return AMDGPU::IsaInfo::getMinNumVGPRs(*this, WavesPerEU,
868 DynamicVGPRBlockSize);
869 }
870
871 /// \returns the maximum number of VGPRs that can be used and still achieved
872 /// at least the specified number of waves \p WavesPerEU.
873 unsigned getMaxNumVGPRs(unsigned WavesPerEU,
874 unsigned DynamicVGPRBlockSize) const {
875 return AMDGPU::IsaInfo::getMaxNumVGPRs(*this, WavesPerEU,
876 DynamicVGPRBlockSize);
877 }
878
879 /// \returns max num VGPRs. This is the common utility function
880 /// called by MachineFunction and Function variants of getMaxNumVGPRs.
881 unsigned
883 std::pair<unsigned, unsigned> NumVGPRBounds) const;
884
885 /// \returns Maximum number of VGPRs that meets number of waves per execution
886 /// unit requirement for function \p F, or number of VGPRs explicitly
887 /// requested using "amdgpu-num-vgpr" attribute attached to function \p F.
888 ///
889 /// \returns Value that meets number of waves per execution unit requirement
890 /// if explicitly requested value cannot be converted to integer, violates
891 /// subtarget's specifications, or does not meet number of waves per execution
892 /// unit requirement.
893 unsigned getMaxNumVGPRs(const Function &F) const;
894
895 unsigned getMaxNumAGPRs(const Function &F) const { return getMaxNumVGPRs(F); }
896
897 /// Return a pair of maximum numbers of VGPRs and AGPRs that meet the number
898 /// of waves per execution unit required for the function \p MF.
899 std::pair<unsigned, unsigned> getMaxNumVectorRegs(const Function &F) const;
900
901 /// \returns Maximum number of VGPRs that meets number of waves per execution
902 /// unit requirement for function \p MF, or number of VGPRs explicitly
903 /// requested using "amdgpu-num-vgpr" attribute attached to function \p MF.
904 ///
905 /// \returns Value that meets number of waves per execution unit requirement
906 /// if explicitly requested value cannot be converted to integer, violates
907 /// subtarget's specifications, or does not meet number of waves per execution
908 /// unit requirement.
909 unsigned getMaxNumVGPRs(const MachineFunction &MF) const;
910
911 bool isWave32() const { return getWavefrontSize() == 32; }
912
913 bool isWave64() const { return getWavefrontSize() == 64; }
914
915 /// Returns if the wavesize of this subtarget is known reliable. This is false
916 /// only for the a default target-cpu that does not have an explicit
917 /// +wavefrontsize target feature.
918 bool isWaveSizeKnown() const {
919 return hasFeature(AMDGPU::FeatureWavefrontSize32) ||
920 hasFeature(AMDGPU::FeatureWavefrontSize64);
921 }
922
924 return getRegisterInfo()->getBoolRC();
925 }
926
927 /// \returns Maximum number of work groups per compute unit supported by the
928 /// subtarget and limited by given \p FlatWorkGroupSize.
929 unsigned getMaxWorkGroupsPerCU(unsigned FlatWorkGroupSize) const override {
930 return AMDGPU::IsaInfo::getMaxWorkGroupsPerCU(*this, FlatWorkGroupSize);
931 }
932
933 /// \returns Minimum flat work group size supported by the subtarget.
934 unsigned getMinFlatWorkGroupSize() const override {
936 }
937
938 /// \returns Maximum flat work group size supported by the subtarget.
939 unsigned getMaxFlatWorkGroupSize() const override {
941 }
942
943 /// \returns Number of waves per execution unit required to support the given
944 /// \p FlatWorkGroupSize.
945 unsigned
946 getWavesPerEUForWorkGroup(unsigned FlatWorkGroupSize) const override {
947 return AMDGPU::IsaInfo::getWavesPerEUForWorkGroup(*this, FlatWorkGroupSize);
948 }
949
950 /// \returns Minimum number of waves per execution unit supported by the
951 /// subtarget.
952 unsigned getMinWavesPerEU() const override {
954 }
955
956 void adjustSchedDependency(SUnit *Def, int DefOpIdx, SUnit *Use, int UseOpIdx,
957 SDep &Dep,
958 const TargetSchedModel *SchedModel) const override;
959
960 // \returns true if it's beneficial on this subtarget for the scheduler to
961 // cluster stores as well as loads.
962 bool shouldClusterStores() const { return getGeneration() >= GFX11; }
963
964 // \returns the number of address arguments from which to enable MIMG NSA
965 // on supported architectures.
966 unsigned getNSAThreshold(const MachineFunction &MF) const;
967
968 // \returns true if the subtarget has a hazard requiring an "s_nop 0"
969 // instruction before "s_sendmsg sendmsg(MSG_DEALLOC_VGPRS)".
970 bool requiresNopBeforeDeallocVGPRs() const { return !HasGFX1250Insts; }
971
972 // \returns true if the subtarget needs S_WAIT_ALU 0 before S_GETREG_B32 on
973 // STATUS, STATE_PRIV, EXCP_FLAG_PRIV, or EXCP_FLAG_USER.
974 bool requiresWaitIdleBeforeGetReg() const { return HasGFX1250Insts; }
975
977 // AMDGPU doesn't care if early-clobber and undef operands are allocated
978 // to the same register.
979 return false;
980 }
981
982 // DS_ATOMIC_ASYNC_BARRIER_ARRIVE_B64 shall not be claused with anything
983 // and surronded by S_WAIT_ALU(0xFFE3).
985 return getGeneration() == GFX12;
986 }
987
988 // Requires s_wait_alu(0) after s102/s103 write and src_flat_scratch_base
989 // read.
991 return HasGFX1250Insts && getGeneration() == GFX12;
992 }
993
994 // src_flat_scratch_hi cannot be used as a source in SALU producing a 64-bit
995 // result.
997 return HasGFX1250Insts && getGeneration() == GFX12;
998 }
999
1000 /// \returns true if the subtarget requires a wait for xcnt before VMEM
1001 /// accesses that must never be repeated in the event of a page fault/re-try.
1002 /// Atomic stores/rmw and all volatile accesses fall under this criteria.
1004 return HasGFX1250Insts;
1005 }
1006
1007 /// \returns the number of significant bits in the immediate field of the
1008 /// S_NOP instruction.
1009 unsigned getSNopBits() const {
1011 return 7;
1013 return 4;
1014 return 3;
1015 }
1016
1020
1022 return (getGeneration() <= AMDGPUSubtarget::GFX9 ||
1024 isWave32();
1025 }
1026
1027 /// Return true if real (non-fake) variants of True16 instructions using
1028 /// 16-bit registers should be code-generated. Fake True16 instructions are
1029 /// identical to non-fake ones except that they take 32-bit registers as
1030 /// operands and always use their low halves.
1031 // TODO: Remove and use hasTrue16BitInsts() instead once True16 is fully
1032 // supported and the support for fake True16 instructions is removed.
1033 bool useRealTrue16Insts() const {
1034 return hasTrue16BitInsts() && EnableRealTrue16Insts;
1035 }
1036
1037 bool requiresWaitOnWorkgroupReleaseFence(bool TgSplit) const {
1038 return getGeneration() >= GFX10 || TgSplit;
1039 }
1040};
1041
1043public:
1044 bool hasImplicitBufferPtr() const { return ImplicitBufferPtr; }
1045
1046 bool hasPrivateSegmentBuffer() const { return PrivateSegmentBuffer; }
1047
1048 bool hasDispatchPtr() const { return DispatchPtr; }
1049
1050 bool hasQueuePtr() const { return QueuePtr; }
1051
1052 bool hasKernargSegmentPtr() const { return KernargSegmentPtr; }
1053
1054 bool hasDispatchID() const { return DispatchID; }
1055
1056 bool hasFlatScratchInit() const { return FlatScratchInit; }
1057
1058 bool hasPrivateSegmentSize() const { return PrivateSegmentSize; }
1059
1060 unsigned getNumKernargPreloadSGPRs() const { return NumKernargPreloadSGPRs; }
1061
1062 unsigned getNumUsedUserSGPRs() const { return NumUsedUserSGPRs; }
1063
1064 unsigned getNumFreeUserSGPRs();
1065
1066 void allocKernargPreloadSGPRs(unsigned NumSGPRs);
1067
1078
1079 // Returns the size in number of SGPRs for preload user SGPR field.
1081 switch (ID) {
1083 return 2;
1085 return 4;
1086 case DispatchPtrID:
1087 return 2;
1088 case QueuePtrID:
1089 return 2;
1091 return 2;
1092 case DispatchIdID:
1093 return 2;
1094 case FlatScratchInitID:
1095 return 2;
1097 return 1;
1098 }
1099 llvm_unreachable("Unknown UserSGPRID.");
1100 }
1101
1102 GCNUserSGPRUsageInfo(const Function &F, const GCNSubtarget &ST);
1103
1104private:
1105 const GCNSubtarget &ST;
1106
1107 // Private memory buffer
1108 // Compute directly in sgpr[0:1]
1109 // Other shaders indirect 64-bits at sgpr[0:1]
1110 bool ImplicitBufferPtr = false;
1111
1112 bool PrivateSegmentBuffer = false;
1113
1114 bool DispatchPtr = false;
1115
1116 bool QueuePtr = false;
1117
1118 bool KernargSegmentPtr = false;
1119
1120 bool DispatchID = false;
1121
1122 bool FlatScratchInit = false;
1123
1124 bool PrivateSegmentSize = false;
1125
1126 unsigned NumKernargPreloadSGPRs = 0;
1127
1128 unsigned NumUsedUserSGPRs = 0;
1129};
1130
1131} // end namespace llvm
1132
1133#endif // LLVM_LIB_TARGET_AMDGPU_GCNSUBTARGET_H
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static cl::opt< bool > EnableLoadStoreOpt("aarch64-enable-ldst-opt", cl::desc("Enable the load/store pair" " optimization pass"), cl::init(true), cl::Hidden)
This file describes how to lower LLVM calls to machine code calls.
This file declares the targeting of the RegisterBankInfo class for AMDGPU.
Base class for AMDGPU specific classes of TargetSubtarget.
static cl::opt< bool > SramEccSetting("amdgpu-sramecc", cl::desc("Force amdgpu.sramecc for testing"), cl::ReallyHidden)
static cl::opt< bool > XnackSetting("amdgpu-xnack", cl::desc("Force amdgpu.xnack value for testing"), cl::ReallyHidden)
AMDHSA kernel descriptor definitions.
DXIL Legalizer
static bool hasFeature(StringRef Feature, const FeatureBitset &FeatureBits, ArrayRef< SubtargetFeatureKV > ProcFeatures)
#define F(x, y, z)
Definition MD5.cpp:54
Promote Memory to Register
Definition Mem2Reg.cpp:110
SI DAG Lowering interface definition.
Interface definition for SIInstrInfo.
static cl::opt< unsigned > CacheLineSize("cache-line-size", cl::init(0), cl::Hidden, cl::desc("Use this to override the target cache line size when " "specified by the user."))
unsigned getWavefrontSizeLog2() const
AMDGPUSubtarget(const Triple &TT)
unsigned getMaxWavesPerEU() const
unsigned getWavefrontSize() const
bool hasPrefetch() const
bool hasFlat() const
bool hasD16Images() const
InstrItineraryData InstrItins
bool useVGPRIndexMode() const
bool dot2UnconditionalFlush() const
bool partialVCCWritesUpdateVCCZ() const
Writes to VCC_LO/VCC_HI update the VCCZ flag.
bool hasSwap() const
bool hasPkMinMax3Insts() const
bool hasD16LoadStore() const
bool hasMergedShaders() const
bool hasRrWGMode() const
bool hasScalarCompareEq64() const
int getLDSBankCount() const
bool hasOnlyRevVALUShifts() const
bool hasNonNSAEncoding() const
bool hasUsableDivScaleConditionOutput() const
Condition output from div_scale is usable.
bool hasExpertSchedulingMode() const
void mirFileLoaded(MachineFunction &MF) const override
bool hasUsableDSOffset() const
True if the offset field of DS instructions works as expected.
bool loadStoreOptEnabled() const
bool enableSubRegLiveness() const override
unsigned getSGPRAllocGranule() const
bool hasFlatLgkmVMemCountInOrder() const
bool flatScratchIsPointer() const
bool hasShift64HighRegBug() const
unsigned MaxPrivateElementSize
bool unsafeDSOffsetFoldingEnabled() const
bool hasFPAtomicToDenormModeHazard() const
unsigned getAddressableNumArchVGPRs() const
bool vmemWriteNeedsExpWaitcnt() const
bool shouldClusterStores() const
unsigned getMinNumSGPRs(unsigned WavesPerEU) const
bool hasUserSGPRInit16BugInWave32() const
unsigned getSGPREncodingGranule() const
void ParseSubtargetFeatures(StringRef CPU, StringRef TuneCPU, StringRef FS)
bool hasFlatScratchHiInB64InstHazard() const
bool hasDstSelForwardingHazard() const
void setScalarizeGlobalBehavior(bool b)
bool hasFlatScratchEnabled() const
bool hasRelaxedBufferOOBMode() const
unsigned DataCacheLineSize
unsigned getSNopBits() const
bool hasLDSLoadB96_B128() const
Returns true if the target supports global_load_lds_dwordx3/global_load_lds_dwordx4 or buffer_load_dw...
bool hasMultiDwordFlatScratchAddressing() const
bool hasFmaakFmamkF64Insts() const
bool hasDsSwizzleRotateMode() const
bool hasHWFP64() const
bool hasScaleOffset() const
bool hasAnyPackedFP64Ops() const
bool hasDenormModeInst() const
bool hasCvtScaleForwardingHazard() const
unsigned getTotalNumVGPRs() const
unsigned getMinWavesPerEU() const override
bool hasUnalignedDSAccessEnabled() const
const SIInstrInfo * getInstrInfo() const override
unsigned getMaxWorkGroupsPerCU(unsigned FlatWorkGroupSize) const override
unsigned getConstantBusLimit(unsigned Opcode) const
bool hasVALUMaskWriteHazard() const
bool hasCondSubInsts() const
const InlineAsmLowering * getInlineAsmLowering() const override
unsigned getTotalNumSGPRs() const
const InstrItineraryData * getInstrItineraryData() const override
void adjustSchedDependency(SUnit *Def, int DefOpIdx, SUnit *Use, int UseOpIdx, SDep &Dep, const TargetSchedModel *SchedModel) const override
void overridePostRASchedPolicy(MachineSchedPolicy &Policy, const SchedRegion &Region) const override
unsigned getMaxLocalMemSizeWithWaveCount(unsigned WaveCount, const Function &) const
Return the amount of LDS that can be used that will not restrict the occupancy lower than WaveCount.
bool hasPkMovB32() const
bool needsAlignedVGPRs() const
Return if operations acting on VGPR tuples require even alignment.
Align getStackAlignment() const
bool privateMemoryResourceIsRangeChecked() const
bool hasScalarSubwordLoads() const
const bool BufferOOBRelaxed
bool hasMadF16() const
bool hasDsAtomicAsyncBarrierArriveB64PipeBug() const
unsigned getInstCacheLineSize() const
Instruction cache line size in bytes (64 for pre-GFX11, 128 for GFX11+).
bool hasLoopHeadInstSplitSensitivity() const
bool hasDwordx3LoadStores() const
bool hasSignedScratchOffsets() const
bool hasGlobalAddTidInsts() const
bool hasFlatScrRegister() const
bool hasGetPCZeroExtension() const
bool hasPermLane64() const
bool requiresNopBeforeDeallocVGPRs() const
unsigned getMinNumVGPRs(unsigned WavesPerEU, unsigned DynamicVGPRBlockSize) const
bool supportsGetDoorbellID() const
unsigned getMaxNumAGPRs(const Function &F) const
bool hasReadM0MovRelInterpHazard() const
bool hasInstPrefSize() const
const SIRegisterInfo * getRegisterInfo() const override
bool hasDOTOpSelHazard() const
bool hasLdsWaitVMSRC() const
const TargetRegisterClass * getBoolRC() const
unsigned getBaseMaxNumVGPRs(const Function &F, std::pair< unsigned, unsigned > NumVGPRBounds) const
bool hasFmaakFmamkF32Insts() const
bool hasMad64_32() const
InstructionSelector * getInstructionSelector() const override
unsigned getVGPREncodingGranule() const
bool hasHardClauses() const
bool useDS128() const
bool hasExtendedWaitCounts() const
bool d16PreservesUnusedBits() const
bool hasInstPrefetch() const
bool hasAddPC64Inst() const
unsigned maxHardClauseLength() const
bool hasAnyPackedFP32Ops() const
bool isMesaGfxShader(const Function &F) const
bool hasExportInsts() const
bool hasVINTERPEncoding() const
const AMDGPURegisterBankInfo * getRegBankInfo() const override
bool hasLegacyGeometry() const
TrapHandlerAbi getTrapHandlerAbi() const
bool isCuModeEnabled() const
const SIFrameLowering * getFrameLowering() const override
bool hasDPPRowShare() const
bool zeroesHigh16BitsOfDest(unsigned Opcode) const
Returns if the result of this instruction with a 16-bit result returned in a 32-bit register implicit...
unsigned getBaseMaxNumSGPRs(const Function &F, std::pair< unsigned, unsigned > WavesPerEU, unsigned PreloadedSGPRs, unsigned ReservedNumSGPRs) const
unsigned getMaxNumPreloadedSGPRs() const
GCNSubtarget & initializeSubtargetDependencies(const Triple &TT, StringRef GPU, StringRef FS)
bool has12DWordStoreHazard() const
bool hasVALUPartialForwardingHazard() const
void overrideSchedPolicy(MachineSchedPolicy &Policy, const SchedRegion &Region) const override
bool useVGPRBlockOpsForCSR() const
std::pair< unsigned, unsigned > computeOccupancy(const Function &F, unsigned LDSSize=0, unsigned NumSGPRs=0, unsigned NumVGPRs=0) const
Subtarget's minimum/maximum occupancy, in number of waves per EU, that can be achieved when the only ...
bool requiresWaitOnWorkgroupReleaseFence(bool TgSplit) const
bool needsKernArgPreloadProlog() const
bool hasMin3Max3_16() const
unsigned getMaxNumVGPRs(unsigned WavesPerEU, unsigned DynamicVGPRBlockSize) const
unsigned getVGPRAllocGranule(unsigned DynamicVGPRBlockSize) const
AMDGPU::TargetID TargetID
unsigned getSetRegWaitStates() const
Number of hazard wait states for s_setreg_b32/s_setreg_imm32_b32.
const SITargetLowering * getTargetLowering() const override
bool hasTransForwardingHazard() const
bool enableMachineScheduler() const override
bool hasLDSFPAtomicAddF64() const
unsigned getNSAThreshold(const MachineFunction &MF) const
bool getScalarizeGlobalBehavior() const
bool hasPKF32InstsReplicatingLower32BitsOfScalarInput() const
bool hasReadM0LdsDmaHazard() const
bool hasScalarSMulU64() const
const AMDGPU::TargetID & getTargetID() const
unsigned getKnownHighZeroBitsForFrameIndex() const
Return the number of high bits known to be zero for a frame index.
bool hasScratchBaseForwardingHazard() const
GCNSubtarget(const Triple &TT, StringRef GPU, StringRef FS, const GCNTargetMachine &TM, bool BufferOOBRelaxed=false, bool TBufferOOBRelaxed=false, AMDGPU::TargetIDSetting XnackSetting=AMDGPU::TargetIDSetting::Any, AMDGPU::TargetIDSetting SramEccSetting=AMDGPU::TargetIDSetting::Any)
bool hasRelaxedTBufferOOBMode() const
bool hasScalarPackInsts() const
bool requiresDisjointEarlyClobberAndUndef() const override
bool hasVALUReadSGPRHazard() const
bool usePRTStrictNull() const
unsigned getAddressableNumVGPRs(unsigned DynamicVGPRBlockSize) const
bool supportsWaveWideBPermute() const
bool hasMed3_16() const
unsigned getReservedNumSGPRs(const MachineFunction &MF) const
bool hasUnalignedScratchAccessEnabled() const
bool hasNullExportTarget() const
Return true if the target's EXP instruction supports the NULL export target.
bool ldsRequiresM0Init() const
Return if most LDS instructions have an m0 use that require m0 to be initialized.
bool useRealTrue16Insts() const
Return true if real (non-fake) variants of True16 instructions using 16-bit registers should be code-...
const bool TBufferOOBRelaxed
bool useAA() const override
bool isWave32() const
bool isGFX11Plus() const
unsigned getOccupancyWithNumVGPRs(unsigned VGPRs, unsigned DynamicVGPRBlockSize) const
Return the maximum number of waves per SIMD for kernels using VGPRs VGPRs.
bool hasUnalignedBufferAccessEnabled() const
bool isWaveSizeKnown() const
Returns if the wavesize of this subtarget is known reliable.
unsigned getMaxPrivateElementSize(bool ForBufferRSrc=false) const
unsigned getMinFlatWorkGroupSize() const override
bool hasAsyncMark() const
bool hasSPackHL() const
Return true if the target has the S_PACK_HL_B32_B16 instruction.
bool supportsMinMaxDenormModes() const
bool supportsBPermute() const
bool hasFlatScratchSVSMode() const
unsigned InstCacheLineSize
bool hasAtomicFaddInsts() const
bool hasSubClampInsts() const
bool requiresWaitXCntForSingleAccessInstructions() const
unsigned getNSAMaxSize(bool HasSampler=false) const
unsigned getOccupancyWithNumSGPRs(unsigned SGPRs) const
Return the maximum number of waves per SIMD for kernels using SGPRs SGPRs.
bool hasVOP3DPP() const
void getInstPrefSizeArgs(uint32_t &Mask, uint32_t &Shift, uint32_t &Width, uint32_t &CacheLineSize) const
unsigned getMaxFlatWorkGroupSize() const override
unsigned getMaxNumUserSGPRs() const
unsigned MaxHardClauseLength
The maximum number of instructions that may be placed within an S_CLAUSE, which is one greater than t...
bool hasFlatScratchSVSSwizzleBug() const
bool hasVDecCoExecHazard() const
bool hasSignedGVSOffset() const
bool hasLDSFPAtomicAddF32() const
unsigned getWavesPerEUForWorkGroup(unsigned FlatWorkGroupSize) const override
bool haveRoundOpsF64() const
Have v_trunc_f64, v_ceil_f64, v_rndne_f64.
bool hasDelayAlu() const
Return true if the target has the S_DELAY_ALU instruction.
unsigned getDataCacheLineSize() const
Data (VMEM) cache line size in bytes (128 for gfx12), has no use before GFX12.
bool hasReadM0SendMsgHazard() const
bool hasScalarMulHiInsts() const
bool hasSCmpK() const
bool hasVCvtPkIU16F32() const
Return true if the target has the V_CVT_PK_I16_F32/V_CVT_PK_U16_F32 instructions.
const LegalizerInfo * getLegalizerInfo() const override
bool requiresWaitIdleBeforeGetReg() const
bool hasDS96AndDS128() const
bool hasReadM0LdsDirectHazard() const
static bool hasHalfRate64Ops(const TargetSubtargetInfo &STI)
Generation getGeneration() const
unsigned getMaxNumSGPRs(unsigned WavesPerEU, bool Addressable) const
std::pair< unsigned, unsigned > getMaxNumVectorRegs(const Function &F) const
Return a pair of maximum numbers of VGPRs and AGPRs that meet the number of waves per execution unit ...
bool isXNACKEnabled() const
bool hasScalarAddSub64() const
bool hasSplitBarriers() const
bool enableEarlyIfConversion() const override
bool hasSMRDReadVALUDefHazard() const
A read of an SGPR by SMRD instruction requires 4 wait states when the SGPR was written by a VALU inst...
bool hasSGetShaderCyclesInst() const
bool hasINVWBL2WaitCntRequirement() const
bool hasRFEHazards() const
bool hasVMEMReadSGPRVALUDefHazard() const
A read of an SGPR by a VMEM instruction requires 5 wait states when the SGPR was written by a VALU In...
bool hasFlatScratchSTMode() const
unsigned getBaseReservedNumSGPRs(const bool HasFlatScratch) const
bool hasGWSSemaReleaseAll() const
bool hasAddr64() const
unsigned getAddressableNumSGPRs() const
bool hasReadVCCZBug() const
Extra wait hazard is needed in some cases before s_cbranch_vccnz/s_cbranch_vccz.
bool isWave64() const
bool setRegModeNeedsVNOPs() const
bool hasFractBug() const
bool isPreciseMemoryEnabled() const
unsigned getMaxWaveScratchSize() const
bool hasLDSMisalignedBugInWGPMode() const
bool hasAnyPackedU64Ops() const
void checkSubtargetFeatures(const Function &F) const
Diagnose inconsistent subtarget features before attempting to codegen function F.
~GCNSubtarget() override
const SelectionDAGTargetInfo * getSelectionDAGInfo() const override
bool hasVOPD3() const
bool hasAtomicCSub() const
bool requiresCodeObjectV6() const
const CallLowering * getCallLowering() const override
bool hasLdsDirect() const
bool hasGWSAutoReplay() const
static unsigned getNumUserSGPRForField(UserSGPRID ID)
void allocKernargPreloadSGPRs(unsigned NumSGPRs)
bool hasPrivateSegmentBuffer() const
unsigned getNumKernargPreloadSGPRs() const
unsigned getNumUsedUserSGPRs() const
GCNUserSGPRUsageInfo(const Function &F, const GCNSubtarget &ST)
Itinerary data supplied by a subtarget to be used by a target.
Scheduling dependency.
Definition ScheduleDAG.h:52
const TargetRegisterClass * getBoolRC() const
Scheduling unit. This is a node in the scheduling DAG.
Targets can subclass this to parameterize the SelectionDAG lowering and instruction selection process...
A wrapper around a string literal that serves as a proxy for constructing global tables of StringRefs...
Definition StringRef.h:888
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Provide an instruction scheduling machine model to CodeGen passes.
TargetSubtargetInfo - Generic base class for all target subtargets.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
Module flag names controlling out-of-bounds buffer access semantics.
constexpr StringLiteral BufferFlag("amdgpu.buffer.oob.mode")
constexpr StringLiteral TBufferFlag("amdgpu.tbuffer.oob.mode")
unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI)
unsigned getMinFlatWorkGroupSize(const MCSubtargetInfo &STI)
unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI, std::optional< bool > EnableWavefrontSize32)
unsigned getMaxWorkGroupsPerCU(const MCSubtargetInfo &STI, unsigned FlatWorkGroupSize)
unsigned getMinNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU)
unsigned getMaxNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU, bool Addressable)
unsigned getWavesPerEUForWorkGroup(const MCSubtargetInfo &STI, unsigned FlatWorkGroupSize)
constexpr unsigned getMaxFlatWorkGroupSize()
unsigned getSGPREncodingGranule(const MCSubtargetInfo &STI)
unsigned getTotalNumVGPRs(const MCSubtargetInfo &STI)
unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU, unsigned DynamicVGPRBlockSize)
unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize)
unsigned getMaxNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU, unsigned DynamicVGPRBlockSize)
unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
unsigned getMinWavesPerEU(const MCSubtargetInfo &STI)
LLVM_READNONE constexpr bool isShader(CallingConv::ID CC)
unsigned getMaxNumUserSGPRs(const MCSubtargetInfo &STI)
unsigned getNSAMaxSize(const MCSubtargetInfo &STI, bool HasSampler)
LLVM_ABI unsigned getAddressableNumSGPRs(GPUKind AK)
LLVM_ABI unsigned getTotalNumSGPRs(GPUKind AK)
LLVM_ABI unsigned getSGPRAllocGranule(GPUKind AK)
This is an optimization pass for GlobalISel generic memory operations.
int countl_zero(T Val)
Count number of 0's from the most significant bit to the least stopping at the first 1.
Definition bit.h:263
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Define a generic scheduling policy for targets that don't provide their own MachineSchedStrategy.
A region of an MBB for scheduling.