LLVM 24.0.0git
GCNSubtarget.cpp
Go to the documentation of this file.
1//===-- GCNSubtarget.cpp - GCN Subtarget Information ----------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// Implements the GCN specific subclass of TargetSubtarget.
11//
12//===----------------------------------------------------------------------===//
13
14#include "GCNSubtarget.h"
15#include "AMDGPUCallLowering.h"
17#include "AMDGPULegalizerInfo.h"
20#include "AMDGPUTargetMachine.h"
29#include "llvm/IR/MDBuilder.h"
31#include <algorithm>
32
33using namespace llvm;
34
35#define DEBUG_TYPE "gcn-subtarget"
36
37#define GET_SUBTARGETINFO_TARGET_DESC
38#define GET_SUBTARGETINFO_CTOR
39#define AMDGPUSubtarget GCNSubtarget
40#include "AMDGPUGenSubtargetInfo.inc"
41#undef AMDGPUSubtarget
42
44 "amdgpu-vgpr-index-mode",
45 cl::desc("Use GPR indexing mode instead of movrel for vector indexing"),
46 cl::init(false));
47
48static cl::opt<bool> UseAA("amdgpu-use-aa-in-codegen",
49 cl::desc("Enable the use of AA during codegen."),
50 cl::init(true));
51
53 NSAThreshold("amdgpu-nsa-threshold",
54 cl::desc("Number of addresses from which to enable MIMG NSA."),
56
58
60 // Legacy triples without a subarch default to the first target that supports
61 // flat addressing for HSA, otherwise the first amdgcn target.
62 if (TT.getSubArch() == Triple::NoSubArch)
63 return TT.getOS() == Triple::AMDHSA ? AMDGPUSubtarget::SEA_ISLANDS
65
66 switch (AMDGPU::getMajorSubArch(TT.getSubArch())) {
91 default:
92 reportFatalUsageError("invalid subarch for amdgpu");
93 }
94}
95
97 StringRef GPU,
98 StringRef FS) {
99 // Determine default and user-specified characteristics
100 //
101 // We want to be able to turn these off, but making this a subtarget feature
102 // for SI has the unhelpful behavior that it unsets everything else if you
103 // disable it.
104 //
105 // Similarly we want enable-prt-strict-null to be on by default and not to
106 // unset everything else if it is disabled
107
108 SmallString<256> FullFS("+load-store-opt,+enable-ds128,");
109
110 // Turn on features that HSA ABI requires. Also turn on FlatForGlobal by
111 // default
112 if (isAmdHsaOS())
113 FullFS += "+flat-for-global,+unaligned-access-mode,+trap-handler,";
114
115 FullFS += "+enable-prt-strict-null,"; // This is overridden by a disable in FS
116
117 // Disable mutually exclusive bits.
118 if (FS.contains_insensitive("+wavefrontsize")) {
119 if (!FS.contains_insensitive("wavefrontsize16"))
120 FullFS += "-wavefrontsize16,";
121 if (!FS.contains_insensitive("wavefrontsize32"))
122 FullFS += "-wavefrontsize32,";
123 if (!FS.contains_insensitive("wavefrontsize64"))
124 FullFS += "-wavefrontsize64,";
125 }
126
127 FullFS += FS;
128
129 ParseSubtargetFeatures(GPU, /*TuneCPU*/ GPU, FullFS);
130
131 // Implement the "generic" processors, which acts as the default when no
132 // generation features are enabled (e.g for -mcpu=''). HSA OS defaults to
133 // the first amdgcn target that supports flat addressing. Other OSes defaults
134 // to the first amdgcn target.
137 // Assume wave64 for the unknown target, if not explicitly set.
138 if (getWavefrontSizeLog2() == 0)
140 } else if (!hasFeature(AMDGPU::FeatureWavefrontSize32) &&
141 !hasFeature(AMDGPU::FeatureWavefrontSize64)) {
142 // If there is no default wave size it must be a generation before gfx10,
143 // these have FeatureWavefrontSize64 in their definition already. For gfx10+
144 // set wave32 as a default.
145 ToggleFeature(AMDGPU::FeatureWavefrontSize32);
147 }
148
149 // We don't support FP64 for EG/NI atm.
151
152 // Targets must either support 64-bit offsets for MUBUF instructions, and/or
153 // support flat operations, otherwise they cannot access a 64-bit global
154 // address space
155 assert(hasAddr64() || hasFlat());
156 // Unless +-flat-for-global is specified, turn on FlatForGlobal for targets
157 // that do not support ADDR64 variants of MUBUF instructions. Such targets
158 // cannot use a 64 bit offset with a MUBUF instruction to access the global
159 // address space
160 if (!hasAddr64() && !FS.contains("flat-for-global") && !UseFlatForGlobal) {
161 ToggleFeature(AMDGPU::FeatureUseFlatForGlobal);
162 UseFlatForGlobal = true;
163 }
164 // Unless +-flat-for-global is specified, use MUBUF instructions for global
165 // address space access if flat operations are not available.
166 if (!hasFlat() && !FS.contains("flat-for-global") && UseFlatForGlobal) {
167 ToggleFeature(AMDGPU::FeatureUseFlatForGlobal);
168 UseFlatForGlobal = false;
169 }
170
171 // Set defaults if needed.
172 if (MaxPrivateElementSize == 0)
174
175 if (LDSBankCount == 0)
176 LDSBankCount = 32;
177
178 if (MaxWavesPerEU == 0)
179 MaxWavesPerEU = 10;
180
181 if (FlatOffsetBitWidth == 0)
183
187 getTargetID().getGPUKind(), isFullSIMDMode());
188 // LDS allocation granularity is in bytes.
191
194
195 // InstCacheLineSize is set from TableGen subtarget features
196 // (FeatureInstCacheLineSize64 / FeatureInstCacheLineSize128).
197 // Fall back to 64 if no feature was specified (e.g. generic targets).
198 if (InstCacheLineSize == 0)
200
202 "InstCacheLineSize must be a power of 2");
203
204 return *this;
205}
206
208 LLVMContext &Ctx = F.getContext();
209 if (hasFeature(AMDGPU::FeatureWavefrontSize32) &&
210 hasFeature(AMDGPU::FeatureWavefrontSize64)) {
211 Ctx.diagnose(DiagnosticInfoUnsupported(
212 F, "must specify exactly one of wavefrontsize32 and wavefrontsize64"));
213 }
214}
215
216// TODO: Validate subarch for subtarget
217
219 const GCNTargetMachine &TM, bool BufferOOBRelaxed,
223 : // clang-format off
224 AMDGPUGenSubtargetInfo(TT, GPU, /*TuneCPU*/ GPU, FS),
225 AMDGPUSubtarget(TT),
226 TargetID(AMDGPU::createAMDGPUTargetID(*this, "")),
227 InstrItins(getInstrItineraryForCPU(GPU)),
230 InstrInfo(initializeSubtargetDependencies(TT, GPU, FS)),
231 TLInfo(TM, *this),
232 // Frame index expansion sometimes assumes the low bit of SP is 0
233 FrameLowering(TargetFrameLowering::StackGrowsUp, getStackAlignment(), 0,
234 /*TransAl=*/Align(4)) {
235
236 // clang-format on
237
238 // Apply the module flag's xnack setting if the target supports on/off modes.
239 // Targets without on/off mode support have xnack always on and ignore module
240 // flags.
241 if (hasXNACKOnOffModes())
242 TargetID.setXnackSetting(XnackSetting);
243
244 // Apply the module flag's sramecc setting if the target supports on/off
245 // modes. Targets with sramecc hardwired on ignore module flags.
246 if (hasSRAMECCOnOffModes())
247 TargetID.setSramEccSetting(SramEccSetting);
248
249 LLVM_DEBUG(dbgs() << "xnack setting for subtarget: "
250 << TargetID.getXnackSetting() << '\n');
251 LLVM_DEBUG(dbgs() << "sramecc setting for subtarget: "
252 << TargetID.getSramEccSetting() << '\n');
253
255
256 TSInfo = std::make_unique<AMDGPUSelectionDAGInfo>();
257
258 CallLoweringInfo = std::make_unique<AMDGPUCallLowering>(*getTargetLowering());
259 InlineAsmLoweringInfo =
260 std::make_unique<InlineAsmLowering>(getTargetLowering());
261 Legalizer = std::make_unique<AMDGPULegalizerInfo>(*this, TM);
262 RegBankInfo = std::make_unique<AMDGPURegisterBankInfo>(*this);
263 InstSelector =
264 std::make_unique<AMDGPUInstructionSelector>(*this, *RegBankInfo);
265}
266
268 return TSInfo.get();
269}
270
271unsigned GCNSubtarget::getConstantBusLimit(unsigned Opcode) const {
272 if (getGeneration() < GFX10)
273 return 1;
274
275 switch (Opcode) {
276 case AMDGPU::V_LSHLREV_B64_e64:
277 case AMDGPU::V_LSHLREV_B64_gfx10:
278 case AMDGPU::V_LSHLREV_B64_e64_gfx11:
279 case AMDGPU::V_LSHLREV_B64_e32_gfx12:
280 case AMDGPU::V_LSHLREV_B64_e64_gfx12:
281 case AMDGPU::V_LSHL_B64_e64:
282 case AMDGPU::V_LSHRREV_B64_e64:
283 case AMDGPU::V_LSHRREV_B64_gfx10:
284 case AMDGPU::V_LSHRREV_B64_e64_gfx11:
285 case AMDGPU::V_LSHRREV_B64_e64_gfx12:
286 case AMDGPU::V_LSHR_B64_e64:
287 case AMDGPU::V_ASHRREV_I64_e64:
288 case AMDGPU::V_ASHRREV_I64_gfx10:
289 case AMDGPU::V_ASHRREV_I64_e64_gfx11:
290 case AMDGPU::V_ASHRREV_I64_e64_gfx12:
291 case AMDGPU::V_ASHR_I64_e64:
292 return 1;
293 }
294
295 return 2;
296}
297
298/// This list was mostly derived from experimentation.
299bool GCNSubtarget::zeroesHigh16BitsOfDest(unsigned Opcode) const {
300 switch (Opcode) {
301 case AMDGPU::V_CVT_F16_F32_e32:
302 case AMDGPU::V_CVT_F16_F32_e64:
303 case AMDGPU::V_CVT_F16_U16_e32:
304 case AMDGPU::V_CVT_F16_U16_e64:
305 case AMDGPU::V_CVT_F16_I16_e32:
306 case AMDGPU::V_CVT_F16_I16_e64:
307 case AMDGPU::V_RCP_F16_e64:
308 case AMDGPU::V_RCP_F16_e32:
309 case AMDGPU::V_RSQ_F16_e64:
310 case AMDGPU::V_RSQ_F16_e32:
311 case AMDGPU::V_SQRT_F16_e64:
312 case AMDGPU::V_SQRT_F16_e32:
313 case AMDGPU::V_LOG_F16_e64:
314 case AMDGPU::V_LOG_F16_e32:
315 case AMDGPU::V_EXP_F16_e64:
316 case AMDGPU::V_EXP_F16_e32:
317 case AMDGPU::V_SIN_F16_e64:
318 case AMDGPU::V_SIN_F16_e32:
319 case AMDGPU::V_COS_F16_e64:
320 case AMDGPU::V_COS_F16_e32:
321 case AMDGPU::V_FLOOR_F16_e64:
322 case AMDGPU::V_FLOOR_F16_e32:
323 case AMDGPU::V_CEIL_F16_e64:
324 case AMDGPU::V_CEIL_F16_e32:
325 case AMDGPU::V_TRUNC_F16_e64:
326 case AMDGPU::V_TRUNC_F16_e32:
327 case AMDGPU::V_RNDNE_F16_e64:
328 case AMDGPU::V_RNDNE_F16_e32:
329 case AMDGPU::V_FRACT_F16_e64:
330 case AMDGPU::V_FRACT_F16_e32:
331 case AMDGPU::V_FREXP_MANT_F16_e64:
332 case AMDGPU::V_FREXP_MANT_F16_e32:
333 case AMDGPU::V_FREXP_EXP_I16_F16_e64:
334 case AMDGPU::V_FREXP_EXP_I16_F16_e32:
335 case AMDGPU::V_LDEXP_F16_e64:
336 case AMDGPU::V_LDEXP_F16_e32:
337 case AMDGPU::V_LSHLREV_B16_e64:
338 case AMDGPU::V_LSHLREV_B16_e32:
339 case AMDGPU::V_LSHRREV_B16_e64:
340 case AMDGPU::V_LSHRREV_B16_e32:
341 case AMDGPU::V_ASHRREV_I16_e64:
342 case AMDGPU::V_ASHRREV_I16_e32:
343 case AMDGPU::V_ADD_U16_e64:
344 case AMDGPU::V_ADD_U16_e32:
345 case AMDGPU::V_SUB_U16_e64:
346 case AMDGPU::V_SUB_U16_e32:
347 case AMDGPU::V_SUBREV_U16_e64:
348 case AMDGPU::V_SUBREV_U16_e32:
349 case AMDGPU::V_MUL_LO_U16_e64:
350 case AMDGPU::V_MUL_LO_U16_e32:
351 case AMDGPU::V_ADD_F16_e64:
352 case AMDGPU::V_ADD_F16_e32:
353 case AMDGPU::V_SUB_F16_e64:
354 case AMDGPU::V_SUB_F16_e32:
355 case AMDGPU::V_SUBREV_F16_e64:
356 case AMDGPU::V_SUBREV_F16_e32:
357 case AMDGPU::V_MUL_F16_e64:
358 case AMDGPU::V_MUL_F16_e32:
359 case AMDGPU::V_MAX_F16_e64:
360 case AMDGPU::V_MAX_F16_e32:
361 case AMDGPU::V_MIN_F16_e64:
362 case AMDGPU::V_MIN_F16_e32:
363 case AMDGPU::V_MAX_U16_e64:
364 case AMDGPU::V_MAX_U16_e32:
365 case AMDGPU::V_MIN_U16_e64:
366 case AMDGPU::V_MIN_U16_e32:
367 case AMDGPU::V_MAX_I16_e64:
368 case AMDGPU::V_MAX_I16_e32:
369 case AMDGPU::V_MIN_I16_e64:
370 case AMDGPU::V_MIN_I16_e32:
371 case AMDGPU::V_MAD_F16_e64:
372 case AMDGPU::V_MAD_U16_e64:
373 case AMDGPU::V_MAD_I16_e64:
374 case AMDGPU::V_FMA_F16_e64:
375 case AMDGPU::V_DIV_FIXUP_F16_e64:
376 // On gfx10, all 16-bit instructions preserve the high bits.
378 case AMDGPU::V_MADAK_F16:
379 case AMDGPU::V_MADMK_F16:
380 case AMDGPU::V_MAC_F16_e64:
381 case AMDGPU::V_MAC_F16_e32:
382 case AMDGPU::V_FMAMK_F16:
383 case AMDGPU::V_FMAAK_F16:
384 case AMDGPU::V_FMAC_F16_e64:
385 case AMDGPU::V_FMAC_F16_e32:
386 // In gfx9, the preferred handling of the unused high 16-bits changed. Most
387 // instructions maintain the legacy behavior of 0ing. Some instructions
388 // changed to preserving the high bits.
390 case AMDGPU::V_MAD_MIXLO_F16:
391 case AMDGPU::V_MAD_MIXHI_F16:
392 default:
393 return false;
394 }
395}
396
398 const SchedRegion &Region) const {
399 // Track register pressure so the scheduler can try to decrease
400 // pressure once register usage is above the threshold defined by
401 // SIRegisterInfo::getRegPressureSetLimit()
402 Policy.ShouldTrackPressure = true;
403
404 const Function &F = Region.RegionBegin->getMF()->getFunction();
405 if (AMDGPU::getSchedStrategy(F) == "coexec") {
406 Policy.OnlyTopDown = true;
407 Policy.OnlyBottomUp = false;
408 return;
409 }
410
411 // Enabling both top down and bottom up scheduling seems to give us less
412 // register spills than just using one of these approaches on its own.
413 Policy.OnlyTopDown = false;
414 Policy.OnlyBottomUp = false;
415
416 // Enabling ShouldTrackLaneMasks crashes the SI Machine Scheduler.
417 if (!enableSIScheduler())
418 Policy.ShouldTrackLaneMasks = true;
419}
420
422 const SchedRegion &Region) const {
423 const Function &F = Region.RegionBegin->getMF()->getFunction();
424 Attribute PostRADirectionAttr = F.getFnAttribute("amdgpu-post-ra-direction");
425 if (!PostRADirectionAttr.isValid())
426 return;
427
428 StringRef PostRADirectionStr = PostRADirectionAttr.getValueAsString();
429 if (PostRADirectionStr == "topdown") {
430 Policy.OnlyTopDown = true;
431 Policy.OnlyBottomUp = false;
432 } else if (PostRADirectionStr == "bottomup") {
433 Policy.OnlyTopDown = false;
434 Policy.OnlyBottomUp = true;
435 } else if (PostRADirectionStr == "bidirectional") {
436 Policy.OnlyTopDown = false;
437 Policy.OnlyBottomUp = false;
438 } else {
440 F, F.getSubprogram(), "invalid value for postRA direction attribute");
441 F.getContext().diagnose(Diag);
442 }
443
444 LLVM_DEBUG({
445 const char *DirStr = "default";
446 if (Policy.OnlyTopDown && !Policy.OnlyBottomUp)
447 DirStr = "topdown";
448 else if (!Policy.OnlyTopDown && Policy.OnlyBottomUp)
449 DirStr = "bottomup";
450 else if (!Policy.OnlyTopDown && !Policy.OnlyBottomUp)
451 DirStr = "bidirectional";
452
453 dbgs() << "Post-MI-sched direction (" << F.getName() << "): " << DirStr
454 << '\n';
455 });
456}
457
462
464 if (isWave32()) {
465 // Fix implicit $vcc operands after MIParser has verified that they match
466 // the instruction definitions.
467 for (auto &MBB : MF) {
468 for (auto &MI : MBB)
469 InstrInfo.fixImplicitOperands(MI);
470 }
471 }
472}
473
475 return InstrInfo.pseudoToMCOpcode(AMDGPU::V_MAD_F16_e64) != -1;
476}
477
479 return hasVGPRIndexMode() && (!hasMovrel() || EnableVGPRIndexMode);
480}
481
482bool GCNSubtarget::useAA() const { return UseAA; }
483
484unsigned GCNSubtarget::getOccupancyWithNumSGPRs(unsigned SGPRs) const {
486}
487
488unsigned
490 unsigned DynamicVGPRBlockSize) const {
492 DynamicVGPRBlockSize);
493}
494
495unsigned
496GCNSubtarget::getBaseReservedNumSGPRs(const bool HasFlatScratch) const {
498 return 2; // VCC. FLAT_SCRATCH and XNACK are no longer in SGPRs.
499
500 if (HasFlatScratch || HasArchitectedFlatScratch) {
502 return 6; // FLAT_SCRATCH, XNACK, VCC (in that order).
504 return 4; // FLAT_SCRATCH, VCC (in that order).
505 }
506
507 if (isXNACKEnabled())
508 return 4; // XNACK, VCC (in that order).
509 return 2; // VCC.
510}
511
516
518 // In principle we do not need to reserve SGPR pair used for flat_scratch if
519 // we know flat instructions do not access the stack anywhere in the
520 // program. For now assume it's needed if we have flat instructions.
521 const bool KernelUsesFlatScratch = hasFlatAddressSpace();
522 return getBaseReservedNumSGPRs(KernelUsesFlatScratch);
523}
524
525std::pair<unsigned, unsigned>
527 unsigned NumSGPRs, unsigned NumVGPRs) const {
528 unsigned DynamicVGPRBlockSize = AMDGPU::getDynamicVGPRBlockSize(F);
529 auto [MinOcc, MaxOcc] = getOccupancyWithWorkGroupSizes(LDSSize, F);
530 unsigned SGPROcc = getOccupancyWithNumSGPRs(NumSGPRs);
531 unsigned VGPROcc = getOccupancyWithNumVGPRs(NumVGPRs, DynamicVGPRBlockSize);
532
533 // Maximum occupancy may be further limited by high SGPR/VGPR usage.
534 MaxOcc = std::min({MaxOcc, SGPROcc, VGPROcc});
535 return {std::min(MinOcc, MaxOcc), MaxOcc};
536}
537
539 const Function &F, std::pair<unsigned, unsigned> WavesPerEU,
540 unsigned PreloadedSGPRs, unsigned ReservedNumSGPRs) const {
541 // Compute maximum number of SGPRs function can use using default/requested
542 // minimum number of waves per execution unit.
543 unsigned MaxNumSGPRs = getMaxNumSGPRs(WavesPerEU.first, false);
544 unsigned MaxAddressableNumSGPRs = getMaxNumSGPRs(WavesPerEU.first, true);
545
546 // Check if maximum number of SGPRs was explicitly requested using
547 // "amdgpu-num-sgpr" attribute.
548 unsigned Requested =
549 F.getFnAttributeAsParsedInteger("amdgpu-num-sgpr", MaxNumSGPRs);
550
551 if (Requested != MaxNumSGPRs) {
552 // Make sure requested value does not violate subtarget's specifications.
553 if (Requested && (Requested <= ReservedNumSGPRs))
554 Requested = 0;
555
556 // If more SGPRs are required to support the input user/system SGPRs,
557 // increase to accommodate them.
558 //
559 // FIXME: This really ends up using the requested number of SGPRs + number
560 // of reserved special registers in total. Theoretically you could re-use
561 // the last input registers for these special registers, but this would
562 // require a lot of complexity to deal with the weird aliasing.
563 unsigned InputNumSGPRs = PreloadedSGPRs;
564 if (Requested && Requested < InputNumSGPRs)
565 Requested = InputNumSGPRs;
566
567 // Make sure requested value is compatible with values implied by
568 // default/requested minimum/maximum number of waves per execution unit.
569 if (Requested && Requested > getMaxNumSGPRs(WavesPerEU.first, false))
570 Requested = 0;
571 if (WavesPerEU.second && Requested &&
572 Requested < getMinNumSGPRs(WavesPerEU.second))
573 Requested = 0;
574
575 if (Requested)
576 MaxNumSGPRs = Requested;
577 }
578
579 if (hasSGPRInitBug())
581
582 return std::min(MaxNumSGPRs - ReservedNumSGPRs, MaxAddressableNumSGPRs);
583}
584
586 const Function &F = MF.getFunction();
590}
591
593 using USI = GCNUserSGPRUsageInfo;
594 // Max number of user SGPRs
595 const unsigned MaxUserSGPRs =
596 USI::getNumUserSGPRForField(USI::PrivateSegmentBufferID) +
597 USI::getNumUserSGPRForField(USI::DispatchPtrID) +
598 USI::getNumUserSGPRForField(USI::QueuePtrID) +
599 USI::getNumUserSGPRForField(USI::KernargSegmentPtrID) +
600 USI::getNumUserSGPRForField(USI::DispatchIdID) +
601 USI::getNumUserSGPRForField(USI::FlatScratchInitID) +
602 USI::getNumUserSGPRForField(USI::ImplicitBufferPtrID);
603
604 // Max number of system SGPRs
605 const unsigned MaxSystemSGPRs = 1 + // WorkGroupIDX
606 1 + // WorkGroupIDY
607 1 + // WorkGroupIDZ
608 1 + // WorkGroupInfo
609 1; // private segment wave byte offset
610
611 // Max number of synthetic SGPRs
612 const unsigned SyntheticSGPRs = 1; // LDSKernelId
613
614 return MaxUserSGPRs + MaxSystemSGPRs + SyntheticSGPRs;
615}
616
621
623 const Function &F, std::pair<unsigned, unsigned> NumVGPRBounds) const {
624 const auto [Min, Max] = NumVGPRBounds;
625
626 // Check if maximum number of VGPRs was explicitly requested using
627 // "amdgpu-num-vgpr" attribute.
628
629 unsigned Requested = F.getFnAttributeAsParsedInteger("amdgpu-num-vgpr", Max);
630 if (Requested != Max && hasGFX90AInsts())
631 Requested *= 2;
632
633 // Make sure requested value is inside the range of possible VGPR usage.
634 return std::clamp(Requested, Min, Max);
635}
636
638 unsigned DynamicVGPRBlockSize = AMDGPU::getDynamicVGPRBlockSize(F);
639 std::pair<unsigned, unsigned> Waves = getWavesPerEU(F);
640
641 unsigned MaxNumVGPRs = getBaseMaxNumVGPRs(
642 F, {getMinNumVGPRs(Waves.second, DynamicVGPRBlockSize),
643 getMaxNumVGPRs(Waves.first, DynamicVGPRBlockSize)});
644
645 // In DVGPR mode, a wave launches with a single VGPR block allocated. Applied
646 // after getBaseMaxNumVGPRs so "amdgpu-num-vgpr" cannot raise it back up.
647 if (DynamicVGPRBlockSize != 0 &&
648 AMDGPU::isEntryFunctionCC(F.getCallingConv()))
649 MaxNumVGPRs = std::min(MaxNumVGPRs, DynamicVGPRBlockSize);
650
651 return MaxNumVGPRs;
652}
653
655 return getMaxNumVGPRs(MF.getFunction());
656}
657
658std::pair<unsigned, unsigned>
660 const unsigned MaxVectorRegs = getMaxNumVGPRs(F);
661
662 unsigned MaxNumVGPRs = MaxVectorRegs;
663 unsigned MaxNumAGPRs = 0;
664 unsigned NumArchVGPRs = getAddressableNumArchVGPRs();
665
666 // On GFX90A, the number of VGPRs and AGPRs need not be equal. Theoretically,
667 // a wave may have up to 512 total vector registers combining together both
668 // VGPRs and AGPRs. Hence, in an entry function without calls and without
669 // AGPRs used within it, it is possible to use the whole vector register
670 // budget for VGPRs.
671 //
672 // TODO: it shall be possible to estimate maximum AGPR/VGPR pressure and split
673 // register file accordingly.
674 if (hasGFX90AInsts()) {
675 unsigned MinNumAGPRs = 0;
676 const unsigned TotalNumAGPRs = AMDGPU::AGPR_32RegClass.getNumRegs();
677
678 const std::pair<unsigned, unsigned> DefaultNumAGPR = {~0u, ~0u};
679
680 // TODO: The lower bound should probably force the number of required
681 // registers up, overriding amdgpu-waves-per-eu.
682 std::tie(MinNumAGPRs, MaxNumAGPRs) =
683 AMDGPU::getIntegerPairAttribute(F, "amdgpu-agpr-alloc", DefaultNumAGPR,
684 /*OnlyFirstRequired=*/true);
685
686 if (MinNumAGPRs == DefaultNumAGPR.first) {
687 // Default to splitting half the registers if AGPRs are required.
688 MinNumAGPRs = MaxNumAGPRs = MaxVectorRegs / 2;
689 } else {
690 // Align to accum_offset's allocation granularity.
691 MinNumAGPRs = alignTo(MinNumAGPRs, 4);
692
693 MinNumAGPRs = std::min(MinNumAGPRs, TotalNumAGPRs);
694 }
695
696 // Clamp values to be inbounds of our limits, and ensure min <= max.
697
698 MaxNumAGPRs = std::min(std::max(MinNumAGPRs, MaxNumAGPRs), MaxVectorRegs);
699 MinNumAGPRs = std::min({MinNumAGPRs, TotalNumAGPRs, MaxNumAGPRs});
700
701 MaxNumVGPRs = std::min(MaxVectorRegs - MinNumAGPRs, NumArchVGPRs);
702 MaxNumAGPRs = std::min(MaxVectorRegs - MaxNumVGPRs, MaxNumAGPRs);
703
704 assert(MaxNumVGPRs + MaxNumAGPRs <= MaxVectorRegs &&
705 MaxNumAGPRs <= TotalNumAGPRs && MaxNumVGPRs <= NumArchVGPRs &&
706 "invalid register counts");
707 } else if (hasMAIInsts()) {
708 // On gfx908 the number of AGPRs always equals the number of VGPRs.
709 MaxNumAGPRs = MaxNumVGPRs = MaxVectorRegs;
710 }
711
712 return std::pair(MaxNumVGPRs, MaxNumAGPRs);
713}
714
715// Check to which source operand UseOpIdx points to and return a pointer to the
716// operand of the corresponding source modifier.
717// Return nullptr if UseOpIdx either doesn't point to src0/1/2 or if there is no
718// operand for the corresponding source modifier.
719static const MachineOperand *
721 const SIInstrInfo &InstrInfo) {
722 AMDGPU::OpName UseName =
723 AMDGPU::getOperandIdxName(UseI.getOpcode(), UseOpIdx);
724 switch (UseName) {
725 case AMDGPU::OpName::src0:
726 return InstrInfo.getNamedOperand(UseI, AMDGPU::OpName::src0_modifiers);
727 case AMDGPU::OpName::src1:
728 return InstrInfo.getNamedOperand(UseI, AMDGPU::OpName::src1_modifiers);
729 case AMDGPU::OpName::src2:
730 return InstrInfo.getNamedOperand(UseI, AMDGPU::OpName::src2_modifiers);
731 default:
732 return nullptr;
733 }
734}
735
736// Get the subreg idx of the subreg that is used by the given instruction
737// operand, considering the given op_sel modifier.
738// Return 0 if the whole register is used or as a conservative fallback.
740 const SIInstrInfo &InstrInfo,
741 const MachineInstr &I,
742 const MachineOperand &Op) {
743 if (!InstrInfo.isVOP3P(I) || InstrInfo.isWMMA(I) || InstrInfo.isSWMMAC(I))
744 return AMDGPU::NoSubRegister;
745
746 const MachineOperand *OpMod =
747 getVOP3PSourceModifierFromOpIdx(I, Op.getOperandNo(), InstrInfo);
748 if (!OpMod)
749 return AMDGPU::NoSubRegister;
750
751 // Note: the FMA_MIX* and MAD_MIX* instructions have different semantics for
752 // the op_sel and op_sel_hi source modifiers:
753 // - op_sel: selects low/high operand bits as input to the operation;
754 // has only meaning for 16-bit source operands
755 // - op_sel_hi: specifies the size of the source operands (16 or 32 bits);
756 // a value of 0 indicates 32 bit, 1 indicates 16 bit
757 // For the other VOP3P instructions, the semantics are:
758 // - op_sel: selects low/high operand bits as input to the operation which
759 // results in the lower-half of the destination
760 // - op_sel_hi: selects the low/high operand bits as input to the operation
761 // which results in the higher-half of the destination
762 int64_t OpSel = OpMod->getImm() & SISrcMods::OP_SEL_0;
763 int64_t OpSelHi = OpMod->getImm() & SISrcMods::OP_SEL_1;
764
765 // Check if all parts of the register are being used (= op_sel and op_sel_hi
766 // differ for VOP3P or op_sel_hi=0 for VOP3PMix). In that case we can return
767 // early.
768 if ((!InstrInfo.isVOP3PMix(I) && (!OpSel || !OpSelHi) &&
769 (OpSel || OpSelHi)) ||
770 (InstrInfo.isVOP3PMix(I) && !OpSelHi))
771 return AMDGPU::NoSubRegister;
772
773 const MachineRegisterInfo &MRI = I.getParent()->getParent()->getRegInfo();
774 const TargetRegisterClass *RC = TRI.getRegClassForOperandReg(MRI, Op);
775
776 if (unsigned SubRegIdx = OpSel ? AMDGPU::sub1 : AMDGPU::sub0;
777 TRI.getSubClassWithSubReg(RC, SubRegIdx) == RC)
778 return SubRegIdx;
779 if (unsigned SubRegIdx = OpSel ? AMDGPU::hi16 : AMDGPU::lo16;
780 TRI.getSubClassWithSubReg(RC, SubRegIdx) == RC)
781 return SubRegIdx;
782
783 return AMDGPU::NoSubRegister;
784}
785
786Register GCNSubtarget::getRealSchedDependency(const MachineInstr &DefI,
787 int DefOpIdx,
788 const MachineInstr &UseI,
789 int UseOpIdx) const {
790 const SIRegisterInfo *TRI = getRegisterInfo();
791 const MachineOperand &DefOp = DefI.getOperand(DefOpIdx);
792 const MachineOperand &UseOp = UseI.getOperand(UseOpIdx);
793 Register DefReg = DefOp.getReg();
794 Register UseReg = UseOp.getReg();
795
796 // If the registers aren't restricted to a sub-register, there is no point in
797 // further analysis. This check makes only sense for virtual registers because
798 // physical registers may form a tuple and thus be part of a superregister
799 // although they are not a subregister themselves (vgpr0 is a "subreg" of
800 // vgpr0_vgpr1 without being a subreg in itself).
801 unsigned DefSubRegIdx = DefOp.getSubReg();
802 if (DefReg.isVirtual() && DefSubRegIdx == AMDGPU::NoSubRegister)
803 return DefReg;
804 unsigned UseSubRegIdx = getEffectiveSubRegIdx(*TRI, InstrInfo, UseI, UseOp);
805 if (UseReg.isVirtual() && UseSubRegIdx == AMDGPU::NoSubRegister)
806 return DefReg;
807
808 if (!TRI->checkSubRegInterference(DefReg, DefSubRegIdx, UseReg, UseSubRegIdx))
809 return Register(); // No real dependency
810
811 // UseReg might be smaller or larger than DefReg, depending on the subreg and
812 // on whether DefReg is a subreg, too. -> Find the smaller one. This does not
813 // apply to virtual registers because we cannot construct a subreg for them.
814 if (DefReg.isVirtual())
815 return DefReg;
816 MCRegister DefMCReg =
817 DefSubRegIdx ? TRI->getSubReg(DefReg, DefSubRegIdx) : DefReg.asMCReg();
818 MCRegister UseMCReg =
819 UseSubRegIdx ? TRI->getSubReg(UseReg, UseSubRegIdx) : UseReg.asMCReg();
820 return TRI->isSubRegisterEq(DefMCReg, UseMCReg) ? UseMCReg : DefMCReg;
821}
822
824 SUnit *Def, int DefOpIdx, SUnit *Use, int UseOpIdx, SDep &Dep,
825 const TargetSchedModel *SchedModel) const {
826 if (Dep.getKind() != SDep::Kind::Data || !Dep.getReg() || !Def->isInstr() ||
827 !Use->isInstr())
828 return;
829
830 MachineInstr *DefI = Def->getInstr();
831 MachineInstr *UseI = Use->getInstr();
832
833 // Check for false latency on $tensorcnt / $asynccnt dependencies
834 if (Dep.getReg() == AMDGPU::TENSORcnt || Dep.getReg() == AMDGPU::ASYNCcnt) {
835 unsigned UseOp = UseI->getOpcode();
836 // Do not adjust latency for load->s_wait
837 bool IsBarrierCase =
838 InstrInfo.isLDSDMA(*DefI) &&
839 (UseOp == AMDGPU::S_WAIT_TENSORCNT || UseOp == AMDGPU::S_WAIT_ASYNCCNT);
840 if (!IsBarrierCase) {
841 Dep.setLatency(1);
842 return;
843 }
844 }
845
846 if (Register Reg = getRealSchedDependency(*DefI, DefOpIdx, *UseI, UseOpIdx)) {
847 Dep.setReg(Reg);
848 } else {
849 Dep = SDep(Def, SDep::Artificial);
850 return; // This is not a data dependency anymore.
851 }
852
853 if (DefI->isBundle()) {
855 auto Reg = Dep.getReg();
858 unsigned Lat = 0;
859 for (++I; I != E && I->isBundledWithPred(); ++I) {
860 if (I->isMetaInstruction())
861 continue;
862 if (I->modifiesRegister(Reg, TRI))
863 Lat = InstrInfo.getInstrLatency(getInstrItineraryData(), *I);
864 else if (Lat)
865 --Lat;
866 }
867 Dep.setLatency(Lat);
868 } else if (UseI->isBundle()) {
870 auto Reg = Dep.getReg();
873 unsigned Lat = InstrInfo.getInstrLatency(getInstrItineraryData(), *DefI);
874 for (++I; I != E && I->isBundledWithPred() && Lat; ++I) {
875 if (I->isMetaInstruction())
876 continue;
877 if (I->readsRegister(Reg, TRI))
878 break;
879 --Lat;
880 }
881 Dep.setLatency(Lat);
882 } else if (Dep.getLatency() == 0 && Dep.getReg() == AMDGPU::VCC_LO) {
883 // Work around the fact that SIInstrInfo::fixImplicitOperands modifies
884 // implicit operands which come from the MCInstrDesc, which can fool
885 // ScheduleDAGInstrs::addPhysRegDataDeps into treating them as implicit
886 // pseudo operands.
887 Dep.setLatency(InstrInfo.getSchedModel().computeOperandLatency(
888 DefI, DefOpIdx, UseI, UseOpIdx));
889 }
890}
891
894 return 0; // Not MIMG encoding.
895
896 if (NSAThreshold.getNumOccurrences() > 0)
897 return std::max(NSAThreshold.getValue(), 2u);
898
900 "amdgpu-nsa-threshold", -1);
901 if (Value > 0)
902 return std::max(Value, 2);
903
904 return NSAThreshold;
905}
906
908 const GCNSubtarget &ST)
909 : ST(ST) {
910 const CallingConv::ID CC = F.getCallingConv();
911 const bool IsKernel =
913
914 if (IsKernel && (!F.arg_empty() || ST.getImplicitArgNumBytes(F) != 0))
915 KernargSegmentPtr = true;
916
917 bool IsAmdHsaOrMesa = ST.isAmdHsaOrMesa(F);
918 if (IsAmdHsaOrMesa && !ST.hasFlatScratchEnabled())
919 PrivateSegmentBuffer = true;
920 else if (ST.isMesaGfxShader(F))
921 ImplicitBufferPtr = true;
922
923 if (!AMDGPU::isGraphics(CC)) {
924 if (!F.hasFnAttribute("amdgpu-no-dispatch-ptr"))
925 DispatchPtr = true;
926
927 // FIXME: Can this always be disabled with < COv5?
928 if (!F.hasFnAttribute("amdgpu-no-queue-ptr"))
929 QueuePtr = true;
930
931 if (!F.hasFnAttribute("amdgpu-no-dispatch-id"))
932 DispatchID = true;
933 }
934
935 if (ST.hasFlatAddressSpace() && AMDGPU::isEntryFunctionCC(CC) &&
936 (IsAmdHsaOrMesa || ST.hasFlatScratchEnabled()) &&
937 // FlatScratchInit cannot be true for graphics CC if
938 // hasFlatScratchEnabled() is false.
939 (ST.hasFlatScratchEnabled() ||
940 (!AMDGPU::isGraphics(CC) &&
941 !F.hasFnAttribute("amdgpu-no-flat-scratch-init"))) &&
942 !ST.hasArchitectedFlatScratch()) {
943 FlatScratchInit = true;
944 }
945
947 NumUsedUserSGPRs += getNumUserSGPRForField(ImplicitBufferPtrID);
948
951
952 if (hasDispatchPtr())
953 NumUsedUserSGPRs += getNumUserSGPRForField(DispatchPtrID);
954
955 if (hasQueuePtr())
956 NumUsedUserSGPRs += getNumUserSGPRForField(QueuePtrID);
957
959 NumUsedUserSGPRs += getNumUserSGPRForField(KernargSegmentPtrID);
960
961 if (hasDispatchID())
962 NumUsedUserSGPRs += getNumUserSGPRForField(DispatchIdID);
963
964 if (hasFlatScratchInit())
965 NumUsedUserSGPRs += getNumUserSGPRForField(FlatScratchInitID);
966
968 NumUsedUserSGPRs += getNumUserSGPRForField(PrivateSegmentSizeID);
969}
970
972 assert(NumKernargPreloadSGPRs + NumSGPRs <= AMDGPU::getMaxNumUserSGPRs(ST));
973 NumKernargPreloadSGPRs += NumSGPRs;
974 NumUsedUserSGPRs += NumSGPRs;
975}
976
978 return AMDGPU::getMaxNumUserSGPRs(ST) - NumUsedUserSGPRs;
979}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static cl::opt< bool > UseAA("aarch64-use-aa", cl::init(true), cl::desc("Enable the use of AA during codegen."))
This file describes how to lower LLVM calls to machine code calls.
This file declares the targeting of the InstructionSelector class for AMDGPU.
This file declares the targeting of the Machinelegalizer class for AMDGPU.
This file declares the targeting of the RegisterBankInfo class for AMDGPU.
static cl::opt< bool > SramEccSetting("amdgpu-sramecc", cl::desc("Force amdgpu.sramecc for testing"), cl::ReallyHidden)
static cl::opt< bool > XnackSetting("amdgpu-xnack", cl::desc("Force amdgpu.xnack value for testing"), cl::ReallyHidden)
The AMDGPU TargetMachine interface definition for hw codegen targets.
MachineBasicBlock & MBB
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static AMDGPUSubtarget::Generation computeDefaultGeneration(const Triple &TT)
static cl::opt< unsigned > NSAThreshold("amdgpu-nsa-threshold", cl::desc("Number of addresses from which to enable MIMG NSA."), cl::init(2), cl::Hidden)
static cl::opt< bool > EnableVGPRIndexMode("amdgpu-vgpr-index-mode", cl::desc("Use GPR indexing mode instead of movrel for vector indexing"), cl::init(false))
static cl::opt< bool > UseAA("amdgpu-use-aa-in-codegen", cl::desc("Enable the use of AA during codegen."), cl::init(true))
static const MachineOperand * getVOP3PSourceModifierFromOpIdx(const MachineInstr &UseI, int UseOpIdx, const SIInstrInfo &InstrInfo)
static unsigned getEffectiveSubRegIdx(const SIRegisterInfo &TRI, const SIInstrInfo &InstrInfo, const MachineInstr &I, const MachineOperand &Op)
AMD GCN specific subclass of TargetSubtarget.
static Register UseReg(const MachineOperand &MO)
IRTranslator LLVM IR MI
This file describes how to lower LLVM inline asm to machine code INLINEASM.
static bool hasFeature(StringRef Feature, const FeatureBitset &FeatureBits, ArrayRef< SubtargetFeatureKV > ProcFeatures)
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
if(PassOpts->AAPipeline)
This file defines the SmallString class.
#define LLVM_DEBUG(...)
Definition Debug.h:119
std::pair< unsigned, unsigned > getWavesPerEU(const Function &F) const
std::pair< unsigned, unsigned > getOccupancyWithWorkGroupSizes(uint32_t LDSBytes, const Function &F) const
Subtarget's minimum/maximum occupancy, in number of waves per EU, that can be achieved when the only ...
unsigned getWavefrontSizeLog2() const
AMDGPUSubtarget(const Triple &TT)
unsigned AddressableLocalMemorySize
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:106
LLVM_ABI StringRef getValueAsString() const
Return the attribute's value as a string.
bool isValid() const
Return true if the attribute is any kind of attribute.
Definition Attributes.h:266
Diagnostic information for optimization failures.
Diagnostic information for unsupported feature in backend.
uint64_t getFnAttributeAsParsedInteger(StringRef Kind, uint64_t Default=0) const
For a string attribute Kind, parse attribute as an integer.
Definition Function.cpp:777
bool hasFlat() const
InstrItineraryData InstrItins
bool useVGPRIndexMode() const
void mirFileLoaded(MachineFunction &MF) const override
unsigned MaxPrivateElementSize
unsigned getAddressableNumArchVGPRs() const
unsigned getMinNumSGPRs(unsigned WavesPerEU) const
void ParseSubtargetFeatures(StringRef CPU, StringRef TuneCPU, StringRef FS)
void overridePipelinerPolicy(MachinePipelinerPolicy &Policy) const override
unsigned getConstantBusLimit(unsigned Opcode) const
const InstrItineraryData * getInstrItineraryData() const override
void adjustSchedDependency(SUnit *Def, int DefOpIdx, SUnit *Use, int UseOpIdx, SDep &Dep, const TargetSchedModel *SchedModel) const override
void overridePostRASchedPolicy(MachineSchedPolicy &Policy, const SchedRegion &Region) const override
bool isFullSIMDMode() const
Align getStackAlignment() const
const bool BufferOOBRelaxed
bool hasMadF16() const
unsigned getMinNumVGPRs(unsigned WavesPerEU, unsigned DynamicVGPRBlockSize) const
const SIRegisterInfo * getRegisterInfo() const override
unsigned getBaseMaxNumVGPRs(const Function &F, std::pair< unsigned, unsigned > NumVGPRBounds) const
bool zeroesHigh16BitsOfDest(unsigned Opcode) const
Returns if the result of this instruction with a 16-bit result returned in a 32-bit register implicit...
unsigned getBaseMaxNumSGPRs(const Function &F, std::pair< unsigned, unsigned > WavesPerEU, unsigned PreloadedSGPRs, unsigned ReservedNumSGPRs) const
unsigned getMaxNumPreloadedSGPRs() const
GCNSubtarget & initializeSubtargetDependencies(const Triple &TT, StringRef GPU, StringRef FS)
void overrideSchedPolicy(MachineSchedPolicy &Policy, const SchedRegion &Region) const override
std::pair< unsigned, unsigned > computeOccupancy(const Function &F, unsigned LDSSize=0, unsigned NumSGPRs=0, unsigned NumVGPRs=0) const
Subtarget's minimum/maximum occupancy, in number of waves per EU, that can be achieved when the only ...
unsigned getMaxNumVGPRs(unsigned WavesPerEU, unsigned DynamicVGPRBlockSize) const
AMDGPU::TargetID TargetID
const SITargetLowering * getTargetLowering() const override
unsigned getNSAThreshold(const MachineFunction &MF) const
const AMDGPU::TargetID & getTargetID() const
GCNSubtarget(const Triple &TT, StringRef GPU, StringRef FS, const GCNTargetMachine &TM, bool BufferOOBRelaxed=false, bool TBufferOOBRelaxed=false, AMDGPU::TargetIDSetting XnackSetting=AMDGPU::TargetIDSetting::Any, AMDGPU::TargetIDSetting SramEccSetting=AMDGPU::TargetIDSetting::Any)
unsigned getReservedNumSGPRs(const MachineFunction &MF) const
const bool TBufferOOBRelaxed
bool useAA() const override
bool isWave32() const
unsigned getOccupancyWithNumVGPRs(unsigned VGPRs, unsigned DynamicVGPRBlockSize) const
Return the maximum number of waves per SIMD for kernels using VGPRs VGPRs.
unsigned InstCacheLineSize
unsigned getOccupancyWithNumSGPRs(unsigned SGPRs) const
Return the maximum number of waves per SIMD for kernels using SGPRs SGPRs.
Generation getGeneration() const
unsigned getMaxNumSGPRs(unsigned WavesPerEU, bool Addressable) const
std::pair< unsigned, unsigned > getMaxNumVectorRegs(const Function &F) const
Return a pair of maximum numbers of VGPRs and AGPRs that meet the number of waves per execution unit ...
bool isXNACKEnabled() const
unsigned getBaseReservedNumSGPRs(const bool HasFlatScratch) const
bool hasAddr64() const
void checkSubtargetFeatures(const Function &F) const
Diagnose inconsistent subtarget features before attempting to codegen function F.
~GCNSubtarget() override
const SelectionDAGTargetInfo * getSelectionDAGInfo() const override
static unsigned getNumUserSGPRForField(UserSGPRID ID)
void allocKernargPreloadSGPRs(unsigned NumSGPRs)
bool hasPrivateSegmentBuffer() const
GCNUserSGPRUsageInfo(const Function &F, const GCNSubtarget &ST)
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Instructions::const_iterator const_instr_iterator
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
bool isBundle() const
const MachineOperand & getOperand(unsigned i) const
MachineOperand class - Representation of each machine instruction operand.
unsigned getSubReg() const
int64_t getImm() const
Register getReg() const
getReg - Returns the register number.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
Wrapper class representing virtual and physical registers.
Definition Register.h:20
MCRegister asMCReg() const
Utility to check-convert this value to a MCRegister.
Definition Register.h:107
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
Definition Register.h:79
Scheduling dependency.
Definition ScheduleDAG.h:53
Kind getKind() const
Returns an enum value representing the kind of the dependence.
@ Data
Regular data dependence (aka true-dependence).
Definition ScheduleDAG.h:57
void setLatency(unsigned Lat)
Sets the latency for this edge.
@ Artificial
Arbitrary strong DAG edge (no real dependence).
Definition ScheduleDAG.h:76
unsigned getLatency() const
Returns the latency value for this edge, which roughly means the minimum number of cycles that must e...
Register getReg() const
Returns the register associated with this edge.
void setReg(Register Reg)
Assigns the associated register for this edge.
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
std::pair< unsigned, unsigned > getWavesPerEU() const
GCNUserSGPRUsageInfo & getUserSGPRInfo()
Scheduling unit. This is a node in the scheduling DAG.
Targets can subclass this to parameterize the SelectionDAG lowering and instruction selection process...
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Information about stack frame layout on the target.
Provide an instruction scheduling machine model to CodeGen passes.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
@ AMDGPUSubArch1250S
Definition Triple.h:272
@ AMDGPUSubArch9
Definition Triple.h:219
@ AMDGPUSubArch9_4
Definition Triple.h:232
@ AMDGPUSubArch6
Definition Triple.h:197
@ AMDGPUSubArch10_3
Definition Triple.h:242
@ AMDGPUSubArch90A
Definition Triple.h:230
@ AMDGPUSubArch810
Definition Triple.h:217
@ AMDGPUSubArch11
Definition Triple.h:251
@ AMDGPUSubArch7
Definition Triple.h:202
@ AMDGPUSubArch12_5
Definition Triple.h:271
@ AMDGPUSubArch10_1
Definition Triple.h:236
@ AMDGPUSubArch11_7
Definition Triple.h:262
@ AMDGPUSubArch8
Definition Triple.h:210
@ AMDGPUSubArch13
Definition Triple.h:276
@ AMDGPUSubArch12
Definition Triple.h:267
@ AMDGPUSubArch908
Definition Triple.h:229
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM Value Representation.
Definition Value.h:75
self_iterator getIterator()
Definition ilist_node.h:123
unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI, unsigned NumVGPRs, unsigned DynamicVGPRBlockSize)
unsigned getOccupancyWithNumSGPRs(unsigned SGPRs, unsigned MaxWaves, unsigned TotalNumSGPRs, unsigned Granule, unsigned TrapReserve)
StringRef getSchedStrategy(const Function &F)
LLVM_ABI unsigned getLDSAllocGranule(GPUKind AK)
LLVM_ABI unsigned getAddressableLocalMemorySize(GPUKind AK, bool FullSIMDMode)
constexpr unsigned getNumWorkGroupSIMDs(bool FullSIMDMode)
unsigned getMaxNumUserSGPRs(const MCSubtargetInfo &STI)
LLVM_ABI unsigned getLocalMemorySize(GPUKind AK, bool FullSIMDMode)
LLVM_READNONE constexpr bool isEntryFunctionCC(CallingConv::ID CC)
unsigned getDynamicVGPRBlockSize(const Function &F)
LLVM_ABI Triple::SubArchType getMajorSubArch(Triple::SubArchType SubArch)
std::pair< unsigned, unsigned > getIntegerPairAttribute(const Function &F, StringRef Name, std::pair< unsigned, unsigned > Default, bool OnlyFirstRequired)
LLVM_READNONE constexpr bool isGraphics(CallingConv::ID CC)
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ SPIR_KERNEL
Used for SPIR kernel functions.
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
DWARFExpression::Operation Op
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Software pipelining policy for a loop, which a target can customize by implementing TargetSubtargetIn...
bool ShouldLimitRegPressure
Limit the register pressure of the scheduled loop, retrying at a higher II when a schedule needs too ...
Define a generic scheduling policy for targets that don't provide their own MachineSchedStrategy.
bool ShouldTrackLaneMasks
Track LaneMasks to allow reordering of independent subregister writes of the same vreg.
A region of an MBB for scheduling.