LLVM 24.0.0git
AMDGPUTargetMachine.cpp
Go to the documentation of this file.
1//===-- AMDGPUTargetMachine.cpp - TargetMachine for hw codegen targets-----===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This file contains both AMDGPU target machine and the CodeGen pass builder.
11/// The AMDGPU target machine contains all of the hardware specific information
12/// needed to emit code for SI+ GPUs in the legacy pass manager pipeline. The
13/// CodeGen pass builder handles the pass pipeline for new pass manager.
14//
15//===----------------------------------------------------------------------===//
16
17#include "AMDGPUTargetMachine.h"
18#include "AMDGPU.h"
19#include "AMDGPUAliasAnalysis.h"
20#include "AMDGPUAsmPrinter.h"
26#include "AMDGPUHazardLatency.h"
27#include "AMDGPUIGroupLP.h"
28#include "AMDGPUISelDAGToDAG.h"
30#include "AMDGPUMacroFusion.h"
38#include "AMDGPUSplitModule.h"
43#include "GCNDPPCombine.h"
45#include "GCNNSAReassign.h"
49#include "GCNSchedStrategy.h"
50#include "GCNVOPDUtils.h"
51#include "R600.h"
52#include "R600TargetMachine.h"
53#include "SIFixSGPRCopies.h"
54#include "SIFixVGPRCopies.h"
55#include "SIFoldOperands.h"
56#include "SIFormMemoryClauses.h"
58#include "SILowerControlFlow.h"
59#include "SILowerSGPRSpills.h"
60#include "SILowerWWMCopies.h"
62#include "SIMachineScheduler.h"
66#include "SIPeepholeSDWA.h"
67#include "SIPostRABundler.h"
70#include "SIWholeQuadMode.h"
95#include "llvm/CodeGen/Passes.h"
107#include "llvm/IR/IntrinsicsAMDGPU.h"
108#include "llvm/IR/Module.h"
109#include "llvm/IR/PassManager.h"
110#include "llvm/IR/PatternMatch.h"
119#include "llvm/Transforms/IPO.h"
144#include <optional>
145
146using namespace llvm;
147using namespace llvm::PatternMatch;
148
149namespace {
150//===----------------------------------------------------------------------===//
151// AMDGPU CodeGen Pass Builder interface.
152//===----------------------------------------------------------------------===//
153
154class AMDGPUCodeGenPassBuilder : public CodeGenPassBuilder {
155 using Base = CodeGenPassBuilder;
156
157 GCNTargetMachine &getTM() const {
158 return static_cast<GCNTargetMachine &>(TM);
159 }
160
161public:
162 AMDGPUCodeGenPassBuilder(GCNTargetMachine &TM,
163 const CGPassBuilderOption &Opts,
164 PassInstrumentationCallbacks *PIC);
165
166 void addIRPasses(PassManagerWrapper &PMW) override;
167 void addCodeGenPrepare(PassManagerWrapper &PMW) override;
168 void addPreISel(PassManagerWrapper &PMW) override;
169 void addILPOpts(PassManagerWrapper &PMW) override;
170 void addAsmPrinterBegin(PassManagerWrapper &PMW) override;
171 void addAsmPrinter(PassManagerWrapper &PMW) override;
172 void addAsmPrinterEnd(PassManagerWrapper &PMW) override;
173 Error addInstSelector(PassManagerWrapper &PMW) override;
174 void addPreRewrite(PassManagerWrapper &PMW) override;
175 void addMachineSSAOptimization(PassManagerWrapper &PMW) override;
176 void addPostRegAlloc(PassManagerWrapper &PMW) override;
177 void addPreEmitPass(PassManagerWrapper &PMW) override;
178 Error addRegAssignAndRewriteFast(PassManagerWrapper &PMW) override;
179 Expected<bool>
180 addRegAssignAndRewriteOptimized(PassManagerWrapper &PMW) override;
181 void addPreRegAlloc(PassManagerWrapper &PMW) override;
182 Error addFastRegAlloc(PassManagerWrapper &PMW) override;
183 Error addOptimizedRegAlloc(PassManagerWrapper &PMW) override;
184 void addPreSched2(PassManagerWrapper &PMW) override;
185 void addPostBBSections(PassManagerWrapper &PMW) override;
186
187private:
188 Error validateRegAllocOptions() const;
189
190public:
191 /// Check if a pass is enabled given \p Opt option. The option always
192 /// overrides defaults if explicitly used. Otherwise its default will be used
193 /// given that a pass shall work at an optimization \p Level minimum.
194 bool isPassEnabled(const cl::opt<bool> &Opt,
195 CodeGenOptLevel Level = CodeGenOptLevel::Default) const;
196 void addEarlyCSEOrGVNPass(PassManagerWrapper &PMW);
197 void addStraightLineScalarOptimizationPasses(PassManagerWrapper &PMW);
198};
199
200class SGPRRegisterRegAlloc : public RegisterRegAllocBase<SGPRRegisterRegAlloc> {
201public:
202 SGPRRegisterRegAlloc(const char *N, const char *D, FunctionPassCtor C)
203 : RegisterRegAllocBase(N, D, C) {}
204};
205
206class VGPRRegisterRegAlloc : public RegisterRegAllocBase<VGPRRegisterRegAlloc> {
207public:
208 VGPRRegisterRegAlloc(const char *N, const char *D, FunctionPassCtor C)
209 : RegisterRegAllocBase(N, D, C) {}
210};
211
212class WWMRegisterRegAlloc : public RegisterRegAllocBase<WWMRegisterRegAlloc> {
213public:
214 WWMRegisterRegAlloc(const char *N, const char *D, FunctionPassCtor C)
215 : RegisterRegAllocBase(N, D, C) {}
216};
217
218static bool onlyAllocateSGPRs(const TargetRegisterInfo &TRI,
219 const MachineRegisterInfo &MRI,
220 const Register Reg) {
221 const TargetRegisterClass *RC = MRI.getRegClass(Reg);
222 return static_cast<const SIRegisterInfo &>(TRI).isSGPRClass(RC);
223}
224
225static bool onlyAllocateVGPRs(const TargetRegisterInfo &TRI,
226 const MachineRegisterInfo &MRI,
227 const Register Reg) {
228 const TargetRegisterClass *RC = MRI.getRegClass(Reg);
229 return !static_cast<const SIRegisterInfo &>(TRI).isSGPRClass(RC);
230}
231
232static bool onlyAllocateWWMRegs(const TargetRegisterInfo &TRI,
233 const MachineRegisterInfo &MRI,
234 const Register Reg) {
235 const SIMachineFunctionInfo *MFI =
237 const TargetRegisterClass *RC = MRI.getRegClass(Reg);
238 return !static_cast<const SIRegisterInfo &>(TRI).isSGPRClass(RC) &&
240}
241
242/// -{sgpr|wwm|vgpr}-regalloc=... command line option.
243static FunctionPass *useDefaultRegisterAllocator() { return nullptr; }
244
245/// A dummy default pass factory indicates whether the register allocator is
246/// overridden on the command line.
247static llvm::once_flag InitializeDefaultSGPRRegisterAllocatorFlag;
248static llvm::once_flag InitializeDefaultVGPRRegisterAllocatorFlag;
249static llvm::once_flag InitializeDefaultWWMRegisterAllocatorFlag;
250
251static SGPRRegisterRegAlloc
252defaultSGPRRegAlloc("default",
253 "pick SGPR register allocator based on -O option",
255
256static cl::opt<SGPRRegisterRegAlloc::FunctionPassCtor, false,
258SGPRRegAlloc("sgpr-regalloc", cl::Hidden, cl::init(&useDefaultRegisterAllocator),
259 cl::desc("Register allocator to use for SGPRs"));
260
261static cl::opt<VGPRRegisterRegAlloc::FunctionPassCtor, false,
263VGPRRegAlloc("vgpr-regalloc", cl::Hidden, cl::init(&useDefaultRegisterAllocator),
264 cl::desc("Register allocator to use for VGPRs"));
265
266static cl::opt<WWMRegisterRegAlloc::FunctionPassCtor, false,
268 WWMRegAlloc("wwm-regalloc", cl::Hidden,
270 cl::desc("Register allocator to use for WWM registers"));
271
272// New pass manager register allocator options for AMDGPU
274 "sgpr-regalloc-npm", cl::Hidden, cl::init(RegAllocType::Default),
275 cl::desc("Register allocator for SGPRs (new pass manager)"));
276
278 "vgpr-regalloc-npm", cl::Hidden, cl::init(RegAllocType::Default),
279 cl::desc("Register allocator for VGPRs (new pass manager)"));
280
282 "wwm-regalloc-npm", cl::Hidden, cl::init(RegAllocType::Default),
283 cl::desc("Register allocator for WWM registers (new pass manager)"));
284
285/// Check if the given RegAllocType is supported for AMDGPU NPM register
286/// allocation. Only Fast and Greedy are supported; Basic and PBQP are not.
287static Error checkRegAllocSupported(RegAllocType RAType, StringRef RegName) {
288 if (RAType == RegAllocType::Basic || RAType == RegAllocType::PBQP) {
290 Twine("unsupported register allocator '") +
291 (RAType == RegAllocType::Basic ? "basic" : "pbqp") + "' for " +
292 RegName + " registers",
294 }
295 return Error::success();
296}
297
298Error AMDGPUCodeGenPassBuilder::validateRegAllocOptions() const {
299 // 1. Generic --regalloc-npm is not supported for AMDGPU.
300 if (Opt.RegAlloc != RegAllocType::Unset) {
302 "-regalloc-npm not supported for amdgcn. Use -sgpr-regalloc-npm, "
303 "-vgpr-regalloc-npm, and -wwm-regalloc-npm",
305 }
306
307 // 2. Legacy PM regalloc options are not compatible with NPM.
308 if (SGPRRegAlloc.getNumOccurrences() > 0 ||
309 VGPRRegAlloc.getNumOccurrences() > 0 ||
310 WWMRegAlloc.getNumOccurrences() > 0) {
312 "-sgpr-regalloc, -vgpr-regalloc, and -wwm-regalloc are legacy PM "
313 "options. Use -sgpr-regalloc-npm, -vgpr-regalloc-npm, and "
314 "-wwm-regalloc-npm with the new pass manager",
316 }
317
318 // 3. Only Fast and Greedy allocators are supported for AMDGPU.
319 if (auto Err = checkRegAllocSupported(SGPRRegAllocNPM, "SGPR"))
320 return Err;
321 if (auto Err = checkRegAllocSupported(WWMRegAllocNPM, "WWM"))
322 return Err;
323 if (auto Err = checkRegAllocSupported(VGPRRegAllocNPM, "VGPR"))
324 return Err;
325
326 return Error::success();
327}
328
329static void initializeDefaultSGPRRegisterAllocatorOnce() {
330 RegisterRegAlloc::FunctionPassCtor Ctor = SGPRRegisterRegAlloc::getDefault();
331
332 if (!Ctor) {
333 Ctor = SGPRRegAlloc;
334 SGPRRegisterRegAlloc::setDefault(SGPRRegAlloc);
335 }
336}
337
338static void initializeDefaultVGPRRegisterAllocatorOnce() {
339 RegisterRegAlloc::FunctionPassCtor Ctor = VGPRRegisterRegAlloc::getDefault();
340
341 if (!Ctor) {
342 Ctor = VGPRRegAlloc;
343 VGPRRegisterRegAlloc::setDefault(VGPRRegAlloc);
344 }
345}
346
347static void initializeDefaultWWMRegisterAllocatorOnce() {
348 RegisterRegAlloc::FunctionPassCtor Ctor = WWMRegisterRegAlloc::getDefault();
349
350 if (!Ctor) {
351 Ctor = WWMRegAlloc;
352 WWMRegisterRegAlloc::setDefault(WWMRegAlloc);
353 }
354}
355
356static FunctionPass *createBasicSGPRRegisterAllocator() {
357 return createBasicRegisterAllocator(onlyAllocateSGPRs);
358}
359
360static FunctionPass *createGreedySGPRRegisterAllocator() {
361 return createGreedyRegisterAllocator(onlyAllocateSGPRs);
362}
363
364static FunctionPass *createFastSGPRRegisterAllocator() {
365 return createFastRegisterAllocator(onlyAllocateSGPRs, false);
366}
367
368static FunctionPass *createBasicVGPRRegisterAllocator() {
369 return createBasicRegisterAllocator(onlyAllocateVGPRs);
370}
371
372static FunctionPass *createGreedyVGPRRegisterAllocator() {
373 return createGreedyRegisterAllocator(onlyAllocateVGPRs);
374}
375
376static FunctionPass *createFastVGPRRegisterAllocator() {
377 return createFastRegisterAllocator(onlyAllocateVGPRs, true);
378}
379
380static FunctionPass *createBasicWWMRegisterAllocator() {
381 return createBasicRegisterAllocator(onlyAllocateWWMRegs);
382}
383
384static FunctionPass *createGreedyWWMRegisterAllocator() {
385 return createGreedyRegisterAllocator(onlyAllocateWWMRegs);
386}
387
388static FunctionPass *createFastWWMRegisterAllocator() {
389 return createFastRegisterAllocator(onlyAllocateWWMRegs, false);
390}
391
392static SGPRRegisterRegAlloc basicRegAllocSGPR(
393 "basic", "basic register allocator", createBasicSGPRRegisterAllocator);
394static SGPRRegisterRegAlloc greedyRegAllocSGPR(
395 "greedy", "greedy register allocator", createGreedySGPRRegisterAllocator);
396
397static SGPRRegisterRegAlloc fastRegAllocSGPR(
398 "fast", "fast register allocator", createFastSGPRRegisterAllocator);
399
400
401static VGPRRegisterRegAlloc basicRegAllocVGPR(
402 "basic", "basic register allocator", createBasicVGPRRegisterAllocator);
403static VGPRRegisterRegAlloc greedyRegAllocVGPR(
404 "greedy", "greedy register allocator", createGreedyVGPRRegisterAllocator);
405
406static VGPRRegisterRegAlloc fastRegAllocVGPR(
407 "fast", "fast register allocator", createFastVGPRRegisterAllocator);
408static WWMRegisterRegAlloc basicRegAllocWWMReg("basic",
409 "basic register allocator",
410 createBasicWWMRegisterAllocator);
411static WWMRegisterRegAlloc
412 greedyRegAllocWWMReg("greedy", "greedy register allocator",
413 createGreedyWWMRegisterAllocator);
414static WWMRegisterRegAlloc fastRegAllocWWMReg("fast", "fast register allocator",
415 createFastWWMRegisterAllocator);
416
418 return Phase == ThinOrFullLTOPhase::FullLTOPreLink ||
419 Phase == ThinOrFullLTOPhase::ThinLTOPreLink;
420}
421} // anonymous namespace
422
423static cl::opt<bool>
425 cl::desc("Run early if-conversion"),
426 cl::init(false));
427
428static cl::opt<bool>
429OptExecMaskPreRA("amdgpu-opt-exec-mask-pre-ra", cl::Hidden,
430 cl::desc("Run pre-RA exec mask optimizations"),
431 cl::init(true));
432
433static cl::opt<bool>
434 LowerCtorDtor("amdgpu-lower-global-ctor-dtor",
435 cl::desc("Lower GPU ctor / dtors to globals on the device."),
436 cl::init(true), cl::Hidden);
437
438// Option to disable vectorizer for tests.
440 "amdgpu-load-store-vectorizer",
441 cl::desc("Enable load store vectorizer"),
442 cl::init(true),
443 cl::Hidden);
444
445// Option to control global loads scalarization
447 "amdgpu-scalarize-global-loads",
448 cl::desc("Enable global load scalarization"),
449 cl::init(true),
450 cl::Hidden);
451
452// Option to run internalize pass.
454 "amdgpu-internalize-symbols",
455 cl::desc("Enable elimination of non-kernel functions and unused globals"),
456 cl::init(false),
457 cl::Hidden);
458
459// Option to inline all early.
461 "amdgpu-early-inline-all",
462 cl::desc("Inline all functions early"),
463 cl::init(false),
464 cl::Hidden);
465
467 "amdgpu-enable-remove-incompatible-functions", cl::Hidden,
468 cl::desc("Enable removal of functions when they"
469 "use features not supported by the target GPU"),
470 cl::init(true));
471
473 "amdgpu-sdwa-peephole",
474 cl::desc("Enable SDWA peepholer"),
475 cl::init(true));
476
478 "amdgpu-dpp-combine",
479 cl::desc("Enable DPP combiner"),
480 cl::init(true));
481
482// Enable address space based alias analysis
484 cl::desc("Enable AMDGPU Alias Analysis"),
485 cl::init(true));
486
487static cl::opt<bool>
488 XnackSetting("amdgpu-xnack",
489 cl::desc("Force amdgpu.xnack value for testing"),
491
492static cl::opt<bool>
493 SramEccSetting("amdgpu-sramecc",
494 cl::desc("Force amdgpu.sramecc for testing"),
496
497// Enable lib calls simplifications
499 "amdgpu-simplify-libcall",
500 cl::desc("Enable amdgpu library simplifications"),
501 cl::init(true),
502 cl::Hidden);
503
505 "amdgpu-ir-lower-kernel-arguments",
506 cl::desc("Lower kernel argument loads in IR pass"),
507 cl::init(true),
508 cl::Hidden);
509
511 "amdgpu-reassign-regs",
512 cl::desc("Enable register reassign optimizations on gfx10+"),
513 cl::init(true),
514 cl::Hidden);
515
517 "amdgpu-opt-vgpr-liverange",
518 cl::desc("Enable VGPR liverange optimizations for if-else structure"),
519 cl::init(true), cl::Hidden);
520
522 "amdgpu-atomic-optimizer-strategy",
523 cl::desc("Select DPP or Iterative strategy for scan"),
526 clEnumValN(ScanOptions::DPP, "DPP", "Use DPP operations for scan"),
528 "Use Iterative approach for scan"),
529 clEnumValN(ScanOptions::None, "None", "Disable atomic optimizer")));
530
531// Enable Mode register optimization
533 "amdgpu-mode-register",
534 cl::desc("Enable mode register pass"),
535 cl::init(true),
536 cl::Hidden);
537
538// Enable GFX11+ s_delay_alu insertion
539static cl::opt<bool>
540 EnableInsertDelayAlu("amdgpu-enable-delay-alu",
541 cl::desc("Enable s_delay_alu insertion"),
542 cl::init(true), cl::Hidden);
543
544// Enable GFX11+ VOPD
545static cl::opt<bool>
546 EnableVOPD("amdgpu-enable-vopd",
547 cl::desc("Enable VOPD, dual issue of VALU in wave32"),
548 cl::init(true), cl::Hidden);
549
550// Option is used in lit tests to prevent deadcoding of patterns inspected.
551static cl::opt<bool>
552EnableDCEInRA("amdgpu-dce-in-ra",
553 cl::init(true), cl::Hidden,
554 cl::desc("Enable machine DCE inside regalloc"));
555
556static cl::opt<bool> EnableSetWavePriority("amdgpu-set-wave-priority",
557 cl::desc("Adjust wave priority"),
558 cl::init(false), cl::Hidden);
559
561 "amdgpu-scalar-ir-passes",
562 cl::desc("Enable scalar IR passes"),
563 cl::init(true),
564 cl::Hidden);
565
567 "amdgpu-enable-lower-exec-sync",
568 cl::desc("Enable lowering of execution synchronization."), cl::init(true),
569 cl::Hidden);
570
571static cl::opt<bool>
572 EnableSwLowerLDS("amdgpu-enable-sw-lower-lds",
573 cl::desc("Enable lowering of lds to global memory pass "
574 "and asan instrument resulting IR."),
575 cl::init(true), cl::Hidden);
576
578 "amdgpu-enable-object-linking",
579 cl::desc("Enable object linking for cross-TU LDS and ABI support"),
581 cl::Hidden);
582
584 "amdgpu-enable-lower-module-lds", cl::desc("Enable lower module lds pass"),
586 cl::Hidden);
587
589 "amdgpu-enable-pre-ra-optimizations",
590 cl::desc("Enable Pre-RA optimizations pass"), cl::init(true),
591 cl::Hidden);
592
594 "amdgpu-enable-promote-kernel-arguments",
595 cl::desc("Enable promotion of flat kernel pointer arguments to global"),
596 cl::Hidden, cl::init(true));
597
599 "amdgpu-enable-image-intrinsic-optimizer",
600 cl::desc("Enable image intrinsic optimizer pass"), cl::init(true),
601 cl::Hidden);
602
603static cl::opt<bool>
604 EnableLoopPrefetch("amdgpu-loop-prefetch",
605 cl::desc("Enable loop data prefetch on AMDGPU"),
606 cl::Hidden, cl::init(false));
607
609 AMDGPUSchedStrategy("amdgpu-sched-strategy",
610 cl::desc("Select custom AMDGPU scheduling strategy."),
611 cl::Hidden, cl::init(""));
612
613// Scheduler selection is consulted both when creating the scheduler and from
614// overrideSchedPolicy(), so keep the attribute and global command line handling
615// in one helper.
617 Attribute SchedStrategyAttr = F.getFnAttribute("amdgpu-sched-strategy");
618 if (SchedStrategyAttr.isValid())
619 return SchedStrategyAttr.getValueAsString();
620
621 if (!AMDGPUSchedStrategy.empty())
622 return AMDGPUSchedStrategy;
623
624 return "";
625}
626
627static void
629 const GCNSubtarget &ST) {
630 if (ST.hasGFX1250Insts())
631 return;
632
633 F.getContext().diagnose(DiagnosticInfoUnsupported(
634 F, "'amdgpu-sched-strategy'='coexec' is only supported for gfx1250",
636}
637
638static bool useNoopPostScheduler(const Function &F) {
639 Attribute PostSchedStrategyAttr =
640 F.getFnAttribute("amdgpu-post-sched-strategy");
641 return PostSchedStrategyAttr.isValid() &&
642 PostSchedStrategyAttr.getValueAsString() == "nop";
643}
644
646 "amdgpu-enable-rewrite-partial-reg-uses",
647 cl::desc("Enable rewrite partial reg uses pass"), cl::init(true),
648 cl::Hidden);
649
651 "amdgpu-enable-hipstdpar",
652 cl::desc("Enable HIP Standard Parallelism Offload support"), cl::init(false),
653 cl::Hidden);
654
655static cl::opt<bool>
656 EnableAMDGPUAttributor("amdgpu-attributor-enable",
657 cl::desc("Enable AMDGPUAttributorPass"),
658 cl::init(true), cl::Hidden);
659
661 "amdgpu-link-time-closed-world",
662 cl::desc("Whether has closed-world assumption at link time"),
663 cl::init(false), cl::Hidden);
664
666 "amdgpu-enable-uniform-intrinsic-combine",
667 cl::desc("Enable/Disable the Uniform Intrinsic Combine Pass"),
668 cl::init(true), cl::Hidden);
669
671 // Register the target
675
761}
762
763static std::unique_ptr<TargetLoweringObjectFile> createTLOF(const Triple &TT) {
764 return std::make_unique<AMDGPUTargetObjectFile>();
765}
766
770
771static ScheduleDAGInstrs *
773 const GCNSubtarget &ST = C->MF->getSubtarget<GCNSubtarget>();
774 ScheduleDAGMILive *DAG =
775 new GCNScheduleDAGMILive(C, std::make_unique<GCNMaxOccupancySchedStrategy>(C));
776 DAG->addMutation(createLoadClusterDAGMutation(DAG->TII, DAG->TRI));
777 if (ST.shouldClusterStores())
778 DAG->addMutation(createStoreClusterDAGMutation(DAG->TII, DAG->TRI));
780 DAG->addMutation(createAMDGPUMacroFusionDAGMutation());
781 DAG->addMutation(createAMDGPUExportClusteringDAGMutation());
782 DAG->addMutation(createAMDGPUBarrierLatencyDAGMutation(C->MF));
783 DAG->addMutation(createAMDGPUHazardLatencyDAGMutation(C->MF));
784 return DAG;
785}
786
787static ScheduleDAGInstrs *
789 ScheduleDAGMILive *DAG =
790 new GCNScheduleDAGMILive(C, std::make_unique<GCNMaxILPSchedStrategy>(C));
792 return DAG;
793}
794
795static ScheduleDAGInstrs *
797 const GCNSubtarget &ST = C->MF->getSubtarget<GCNSubtarget>();
799 C, std::make_unique<GCNMaxMemoryClauseSchedStrategy>(C));
800 DAG->addMutation(createLoadClusterDAGMutation(DAG->TII, DAG->TRI));
801 if (ST.shouldClusterStores())
802 DAG->addMutation(createStoreClusterDAGMutation(DAG->TII, DAG->TRI));
803 DAG->addMutation(createAMDGPUExportClusteringDAGMutation());
804 DAG->addMutation(createAMDGPUBarrierLatencyDAGMutation(C->MF));
805 DAG->addMutation(createAMDGPUHazardLatencyDAGMutation(C->MF));
806 return DAG;
807}
808
809static ScheduleDAGInstrs *
811 const GCNSubtarget &ST = C->MF->getSubtarget<GCNSubtarget>();
812 auto *DAG = new GCNIterativeScheduler(
814 DAG->addMutation(createLoadClusterDAGMutation(DAG->TII, DAG->TRI));
815 if (ST.shouldClusterStores())
816 DAG->addMutation(createStoreClusterDAGMutation(DAG->TII, DAG->TRI));
818 return DAG;
819}
820
827
828static ScheduleDAGInstrs *
830 const GCNSubtarget &ST = C->MF->getSubtarget<GCNSubtarget>();
832 DAG->addMutation(createLoadClusterDAGMutation(DAG->TII, DAG->TRI));
833 if (ST.shouldClusterStores())
834 DAG->addMutation(createStoreClusterDAGMutation(DAG->TII, DAG->TRI));
835 DAG->addMutation(createAMDGPUMacroFusionDAGMutation());
837 return DAG;
838}
839
840static MachineSchedRegistry
841SISchedRegistry("si", "Run SI's custom scheduler",
843
846 "Run GCN scheduler to maximize occupancy",
848
850 GCNMaxILPSchedRegistry("gcn-max-ilp", "Run GCN scheduler to maximize ilp",
852
854 "gcn-max-memory-clause", "Run GCN scheduler to maximize memory clause",
856
858 "gcn-iterative-max-occupancy-experimental",
859 "Run GCN scheduler to maximize occupancy (experimental)",
861
863 "gcn-iterative-minreg",
864 "Run GCN iterative scheduler for minimal register usage (experimental)",
866
868 "gcn-iterative-ilp",
869 "Run GCN iterative scheduler for ILP scheduling (experimental)",
871
874 if (!GPU.empty())
875 return GPU;
876
877 if (StringRef Name = AMDGPU::getArchNameFromSubArch(TT.getSubArch());
878 !Name.empty())
879 return Name;
880
881 // Need to default to a target with flat support for HSA.
882 if (TT.isAMDGCN())
883 return TT.getOS() == Triple::AMDHSA ? "generic-hsa" : "generic";
884
885 return "r600";
886}
887
889 // The AMDGPU toolchain only supports generating shared objects, so we
890 // must always use PIC.
891 return Reloc::PIC_;
892}
893
895 StringRef CPU, StringRef FS,
896 const TargetOptions &Options,
897 std::optional<Reloc::Model> RM,
898 std::optional<CodeModel::Model> CM,
901 T, TT.computeDataLayout(), TT, getGPUOrDefault(TT, CPU), FS, Options,
903 OptLevel),
905 initAsmInfo();
906 if (TT.isAMDGCN()) {
907 // Triple is missing a representation for non-empty, but unrecognized
908 // subarches. Only permit no subarch for any subtarget if it was really
909 // empty.
910 bool IsUnknownSubArch =
911 TT.getSubArch() == Triple::NoSubArch && TT.getArchName().size() != 6;
912 if (IsUnknownSubArch)
913 reportFatalUsageError("unknown subarch " + TT.getArchName());
914
915 if (TT.getSubArch() != Triple::NoSubArch) {
917 Triple::SubArchType GPUSubArch = AMDGPU::getSubArch(Kind);
918 if (Kind != AMDGPU::GK_NONE && GPUSubArch != TT.getSubArch() &&
919 TT.getSubArch() != AMDGPU::getMajorSubArch(GPUSubArch)) {
920 reportFatalUsageError("invalid cpu '" + CPU + "' for subarch " +
921 TT.getArchName());
922 }
923 }
924
925 if (getMCSubtargetInfo().checkFeatures("+wavefrontsize64"))
927 else if (getMCSubtargetInfo().checkFeatures("+wavefrontsize32"))
929 }
931}
932
936
938
940 Attribute GPUAttr = F.getFnAttribute("target-cpu");
941 return GPUAttr.isValid() ? GPUAttr.getValueAsString() : getTargetCPU();
942}
943
945 Attribute FSAttr = F.getFnAttribute("target-features");
946
947 return FSAttr.isValid() ? FSAttr.getValueAsString()
949}
950
953 const GCNSubtarget &ST = C->MF->getSubtarget<GCNSubtarget>();
955 DAG->addMutation(createLoadClusterDAGMutation(DAG->TII, DAG->TRI));
956 if (ST.shouldClusterStores())
957 DAG->addMutation(createStoreClusterDAGMutation(DAG->TII, DAG->TRI));
958 return DAG;
959}
960
961/// Predicate for Internalize pass.
962static bool mustPreserveGV(const GlobalValue &GV) {
963 if (const Function *F = dyn_cast<Function>(&GV))
964 return F->isDeclaration() || F->getName().starts_with("__asan_") ||
965 F->getName().starts_with("__sanitizer_") ||
966 AMDGPU::isEntryFunctionCC(F->getCallingConv());
967
969 return !GV.use_empty();
970}
971
976
979 if (Params.empty())
981 Params.consume_front("strategy=");
982 auto Result = StringSwitch<std::optional<ScanOptions>>(Params)
983 .Case("dpp", ScanOptions::DPP)
984 .Cases({"iterative", ""}, ScanOptions::Iterative)
985 .Case("none", ScanOptions::None)
986 .Default(std::nullopt);
987 if (Result)
988 return *Result;
989 return make_error<StringError>("invalid parameter", inconvertibleErrorCode());
990}
991
995 while (!Params.empty()) {
996 StringRef ParamName;
997 std::tie(ParamName, Params) = Params.split(';');
998 if (ParamName == "closed-world") {
999 Result.IsClosedWorld = true;
1000 } else {
1002 formatv("invalid AMDGPUAttributor pass parameter '{0}' ", ParamName)
1003 .str(),
1005 }
1006 }
1007 return Result;
1008}
1009
1011
1012#define GET_PASS_REGISTRY "AMDGPUPassRegistry.def"
1014
1015 PB.registerPipelineParsingCallback(
1016 [this](StringRef Name, CGSCCPassManager &PM,
1018 if (Name == "amdgpu-attributor-cgscc" && getTargetTriple().isAMDGCN()) {
1020 *static_cast<GCNTargetMachine *>(this)));
1021 return true;
1022 }
1023 return false;
1024 });
1025
1026 PB.registerScalarOptimizerLateEPCallback(
1027 [](FunctionPassManager &FPM, OptimizationLevel Level) {
1028 if (Level == OptimizationLevel::O0)
1029 return;
1030
1032 });
1033
1034 PB.registerVectorizerEndEPCallback(
1035 [](FunctionPassManager &FPM, OptimizationLevel Level) {
1036 if (Level == OptimizationLevel::O0)
1037 return;
1038
1040 });
1041
1042 PB.registerPipelineEarlySimplificationEPCallback(
1043 [this](ModulePassManager &PM, OptimizationLevel Level,
1045 if (!isLTOPreLink(Phase) && getTargetTriple().isAMDGCN()) {
1046 // When we are not using -fgpu-rdc, we can run accelerator code
1047 // selection relatively early, but still after linking to prevent
1048 // eager removal of potentially reachable symbols.
1049 if (EnableHipStdPar) {
1052 }
1053
1055 }
1056
1057 if (Level == OptimizationLevel::O0)
1058 return;
1059
1060 // We don't want to run internalization at per-module stage.
1063 PM.addPass(GlobalDCEPass());
1064 }
1065
1068 });
1069
1070 PB.registerPeepholeEPCallback(
1071 [](FunctionPassManager &FPM, OptimizationLevel Level) {
1072 if (Level == OptimizationLevel::O0)
1073 return;
1074
1078
1081 });
1082
1083 PB.registerCGSCCOptimizerLateEPCallback(
1084 [this](CGSCCPassManager &PM, OptimizationLevel Level) {
1085 if (Level == OptimizationLevel::O0)
1086 return;
1087
1089
1090 // Add promote kernel arguments pass to the opt pipeline right before
1091 // infer address spaces which is needed to do actual address space
1092 // rewriting.
1095
1096 // Add infer address spaces pass to the opt pipeline after inlining
1097 // but before SROA to increase SROA opportunities.
1099
1100 // This should run after inlining to have any chance of doing
1101 // anything, and before other cleanup optimizations.
1103
1104 // Promote alloca to vector before SROA and loop unroll. If we
1105 // manage to eliminate allocas before unroll we may choose to unroll
1106 // less.
1108
1109 PM.addPass(createCGSCCToFunctionPassAdaptor(std::move(FPM)));
1110 });
1111
1112 // FIXME: Why is AMDGPUAttributor not in CGSCC?
1113 PB.registerOptimizerLastEPCallback([this](ModulePassManager &MPM,
1114 OptimizationLevel Level,
1116 if (Level != OptimizationLevel::O0) {
1117 if (!isLTOPreLink(Phase)) {
1118 if (EnableAMDGPUAttributor && getTargetTriple().isAMDGCN()) {
1120 MPM.addPass(AMDGPUAttributorPass(*this, Opts, Phase));
1121 }
1122 }
1123 }
1124 });
1125
1126 PB.registerFullLinkTimeOptimizationLastEPCallback(
1127 [this](ModulePassManager &PM, OptimizationLevel Level) {
1128 // Clean up redundant memory round-trips that the full-LTO pipeline,
1129 // unlike the non-LTO/ThinLTO ones, otherwise leaves for codegen.
1130 if (Level != OptimizationLevel::O0) {
1132 EarlyCSEPass(/*UseMemorySSA=*/true)));
1133 }
1134
1135 // When we are using -fgpu-rdc, we can only run accelerator code
1136 // selection after linking to prevent, otherwise we end up removing
1137 // potentially reachable symbols that were exported as external in other
1138 // modules.
1139 if (EnableHipStdPar) {
1142 }
1143 // We want to support the -lto-partitions=N option as "best effort".
1144 // For that, we need to lower LDS earlier in the pipeline before the
1145 // module is partitioned for codegen.
1148 if (EnableSwLowerLDS)
1152 if (Level != OptimizationLevel::O0) {
1153 // We only want to run this with O2 or higher since inliner and SROA
1154 // don't run in O1.
1155 if (Level != OptimizationLevel::O1) {
1156 PM.addPass(
1158 }
1159 // Do we really need internalization in LTO?
1160 if (InternalizeSymbols) {
1162 PM.addPass(GlobalDCEPass());
1163 }
1164 if (EnableAMDGPUAttributor && getTargetTriple().isAMDGCN()) {
1167 Opt.IsClosedWorld = true;
1170 }
1171 }
1172 if (!NoKernelInfoEndLTO) {
1174 FPM.addPass(KernelInfoPrinter(this));
1175 PM.addPass(createModuleToFunctionPassAdaptor(std::move(FPM)));
1176 }
1177 });
1178
1179 PB.registerRegClassFilterParsingCallback(
1180 [](StringRef FilterName) -> RegAllocFilterFunc {
1181 if (FilterName == "sgpr")
1182 return onlyAllocateSGPRs;
1183 if (FilterName == "vgpr")
1184 return onlyAllocateVGPRs;
1185 if (FilterName == "wwm")
1186 return onlyAllocateWWMRegs;
1187 return nullptr;
1188 });
1189}
1190
1192 unsigned DestAS) const {
1193 return AMDGPU::isFlatGlobalAddrSpace(SrcAS) &&
1195}
1196
1198 if (auto *Arg = dyn_cast<Argument>(V);
1199 Arg &&
1200 AMDGPU::isModuleEntryFunctionCC(Arg->getParent()->getCallingConv()) &&
1201 !Arg->hasByRefAttr())
1203
1204 const auto *LD = dyn_cast<LoadInst>(V);
1205 if (!LD) // TODO: Handle invariant load like constant.
1207
1208 // It must be a generic pointer loaded.
1209 assert(V->getType()->getPointerAddressSpace() == AMDGPUAS::FLAT_ADDRESS);
1210
1211 const auto *Ptr = LD->getPointerOperand();
1212 if (Ptr->getType()->getPointerAddressSpace() != AMDGPUAS::CONSTANT_ADDRESS)
1214 // For a generic pointer loaded from the constant memory, it could be assumed
1215 // as a global pointer since the constant memory is only populated on the
1216 // host side. As implied by the offload programming model, only global
1217 // pointers could be referenced on the host side.
1219}
1220
1221std::pair<const Value *, unsigned>
1223 if (auto *II = dyn_cast<IntrinsicInst>(V)) {
1224 switch (II->getIntrinsicID()) {
1225 case Intrinsic::amdgcn_is_shared:
1226 return std::pair(II->getArgOperand(0), AMDGPUAS::LOCAL_ADDRESS);
1227 case Intrinsic::amdgcn_is_private:
1228 return std::pair(II->getArgOperand(0), AMDGPUAS::PRIVATE_ADDRESS);
1229 default:
1230 break;
1231 }
1232 return std::pair(nullptr, -1);
1233 }
1234 // Check the global pointer predication based on
1235 // (!is_share(p) && !is_private(p)). Note that logic 'and' is commutative and
1236 // the order of 'is_shared' and 'is_private' is not significant.
1237 Value *Ptr;
1238 if (match(
1239 const_cast<Value *>(V),
1242 m_Deferred(Ptr))))))
1243 return std::pair(Ptr, AMDGPUAS::GLOBAL_ADDRESS);
1244
1245 return std::pair(nullptr, -1);
1246}
1247
1248unsigned
1263
1265 Module &M, unsigned NumParts,
1266 function_ref<void(std::unique_ptr<Module> MPart)> ModuleCallback) {
1267 // FIXME(?): Would be better to use an already existing Analysis/PassManager,
1268 // but all current users of this API don't have one ready and would need to
1269 // create one anyway. Let's hide the boilerplate for now to keep it simple.
1270
1275
1276 PassBuilder PB(this);
1277 PB.registerModuleAnalyses(MAM);
1278 PB.registerFunctionAnalyses(FAM);
1279 PB.crossRegisterProxies(LAM, FAM, CGAM, MAM);
1280
1282 MPM.addPass(AMDGPUSplitModulePass(NumParts, ModuleCallback));
1283 MPM.run(M, MAM);
1284 return true;
1285}
1286
1287//===----------------------------------------------------------------------===//
1288// GCN Target Machine (SI+)
1289//===----------------------------------------------------------------------===//
1290
1292 StringRef CPU, StringRef FS,
1293 const TargetOptions &Options,
1294 std::optional<Reloc::Model> RM,
1295 std::optional<CodeModel::Model> CM,
1296 CodeGenOptLevel OL, bool JIT)
1297 : AMDGPUTargetMachine(T, TT, CPU, FS, Options, RM, CM, OL) {
1299}
1300
1301enum class OOBFlagValue {
1302 Any = 0,
1305};
1306
1307/// Returns the OOB mode encoded by a module flag.
1308/// An absent flag defaults to Any.
1309static OOBFlagValue getOOBFlagValue(const Module &M, StringRef FlagName) {
1310 const auto *Flag =
1311 mdconst::dyn_extract_or_null<ConstantInt>(M.getModuleFlag(FlagName));
1312 if (!Flag)
1313 return OOBFlagValue::Any;
1314 return static_cast<OOBFlagValue>(Flag->getZExtValue());
1315}
1316
1317/// Returns the xnack/sramecc setting encoded by a module flag.
1318/// Module flag values: 0 = disabled, 1 = enabled.
1319/// An absent flag defaults to Any.
1322 StringRef FlagName) {
1324
1325 if (XnackSetting.getNumOccurrences() > 0 && FlagName == "amdgpu.xnack")
1326 return XnackSetting ? TargetIDSetting::On : TargetIDSetting::Off;
1327 if (SramEccSetting.getNumOccurrences() > 0 && FlagName == "amdgpu.sramecc")
1328 return SramEccSetting ? TargetIDSetting::On : TargetIDSetting::Off;
1329
1330 const auto *Flag =
1331 mdconst::dyn_extract_or_null<ConstantInt>(M.getModuleFlag(FlagName));
1332 if (!Flag)
1333 return TargetIDSetting::Any;
1334 return Flag->getZExtValue() == 0 ? TargetIDSetting::Off : TargetIDSetting::On;
1335}
1336
1337const TargetSubtargetInfo *
1339 StringRef GPU = getGPUName(F);
1341
1342 const Module &M = *F.getParent();
1345 bool BufRelaxed = BufOOB == OOBFlagValue::Relaxed;
1346 bool TBufRelaxed = TBufOOB == OOBFlagValue::Relaxed;
1347
1349 TargetIDSetting Xnack = getTargetIDSettingFromModuleFlag(M, "amdgpu.xnack");
1350 TargetIDSetting SramEcc =
1351 getTargetIDSettingFromModuleFlag(M, "amdgpu.sramecc");
1352
1353 SmallString<128> SubtargetKey(GPU);
1354 SubtargetKey.append(FS);
1355 if (BufRelaxed)
1356 SubtargetKey.append(",buf-oob=1");
1357 if (TBufRelaxed)
1358 SubtargetKey.append(",tbuf-oob=1");
1359 if (Xnack != TargetIDSetting::Any) {
1360 SubtargetKey.append(",xnack=");
1361 SubtargetKey.push_back(Xnack == TargetIDSetting::On ? '1' : '0');
1362 }
1363 if (SramEcc != TargetIDSetting::Any) {
1364 SubtargetKey.append(",sramecc=");
1365 SubtargetKey.push_back(Xnack == TargetIDSetting::On ? '1' : '0');
1366 }
1367
1368 auto &I = SubtargetMap[SubtargetKey];
1369 if (!I) {
1371 Triple::SubArchType GPUSubArch = AMDGPU::getSubArch(Kind);
1372
1373 // Enforce the subtarget is covered by the subarch. Tolerate no subarch for
1374 // legacy compatibility.
1375 const Triple &TT = M.getTargetTriple();
1376 if (GPUSubArch != TT.getSubArch() && Kind != AMDGPU::GK_NONE) {
1377 // Check if this is a generic subarch which has subtargets. Ignore
1378 // unknown subtargets with a known subarch, since for whatever reason
1379 // the convention is to just print a warning and ignore unrecognized
1380 // subtargets.
1381 bool IsLegacyEmptySubArch = TT.getSubArch() == Triple::NoSubArch;
1382 if (!IsLegacyEmptySubArch &&
1383 AMDGPU::getMajorSubArch(GPUSubArch) != TT.getSubArch()) {
1384 F.getContext().emitError("invalid subtarget '" + Twine(GPU) +
1385 "' for subarch " + TT.getArchName());
1386 }
1387 }
1388
1389 I = std::make_unique<GCNSubtarget>(TargetTriple, GPU, FS, *this, BufRelaxed,
1390 TBufRelaxed, Xnack, SramEcc);
1391 }
1392
1393 I->setScalarizeGlobalBehavior(ScalarizeGlobal);
1394
1395 return I.get();
1396}
1397
1400 return TargetTransformInfo(std::make_unique<GCNTTIImpl>(this, F));
1401}
1402
1405 raw_pwrite_stream *DwoOut, CodeGenFileType FileType,
1406 const CGPassBuilderOption &Opts, MCContext &Ctx,
1408 AMDGPUCodeGenPassBuilder CGPB(*this, Opts, PIC);
1409 return CGPB.buildPipeline(MPM, MAM, Out, DwoOut, FileType, Ctx);
1410}
1411
1414 const GCNSubtarget &ST = C->MF->getSubtarget<GCNSubtarget>();
1415 if (ST.enableSIScheduler())
1417
1418 StringRef SchedStrategy = AMDGPU::getSchedStrategy(C->MF->getFunction());
1419
1420 if (SchedStrategy == "max-ilp")
1422
1423 if (SchedStrategy == "max-memory-clause")
1425
1426 if (SchedStrategy == "iterative-ilp")
1428
1429 if (SchedStrategy == "iterative-minreg")
1430 return createMinRegScheduler(C);
1431
1432 if (SchedStrategy == "iterative-maxocc")
1434
1435 if (SchedStrategy == "coexec") {
1436 diagnoseUnsupportedCoExecSchedulerSelection(C->MF->getFunction(), ST);
1438 }
1439
1441}
1442
1445 if (useNoopPostScheduler(C->MF->getFunction()))
1447
1448 ScheduleDAGMI *DAG =
1449 new GCNPostScheduleDAGMILive(C, std::make_unique<PostGenericScheduler>(C),
1450 /*RemoveKillFlags=*/true);
1451 const GCNSubtarget &ST = C->MF->getSubtarget<GCNSubtarget>();
1453 if (ST.shouldClusterStores())
1456 if ((EnableVOPD.getNumOccurrences() ||
1458 EnableVOPD)
1463 return DAG;
1464}
1465//===----------------------------------------------------------------------===//
1466// AMDGPU Legacy Pass Setup
1467//===----------------------------------------------------------------------===//
1468
1469std::unique_ptr<CSEConfigBase> llvm::AMDGPUPassConfig::getCSEConfig() const {
1470 return getStandardCSEConfigForOpt(TM->getOptLevel());
1471}
1472
1473namespace {
1474
1475class GCNPassConfig final : public AMDGPUPassConfig {
1476public:
1477 GCNPassConfig(TargetMachine &TM, PassManagerBase &PM)
1478 : AMDGPUPassConfig(TM, PM) {
1479 substitutePass(&PostRASchedulerID, &PostMachineSchedulerID);
1480 }
1481
1482 GCNTargetMachine &getGCNTargetMachine() const {
1483 return getTM<GCNTargetMachine>();
1484 }
1485
1486 bool addPreISel() override;
1487 void addMachineSSAOptimization() override;
1488 bool addILPOpts() override;
1489 bool addInstSelector() override;
1490 bool addIRTranslator() override;
1491 void addPreLegalizeMachineIR() override;
1492 bool addLegalizeMachineIR() override;
1493 void addPreRegBankSelect() override;
1494 bool addRegBankSelect() override;
1495 void addPreGlobalInstructionSelect() override;
1496 bool addGlobalInstructionSelect() override;
1497 void addPreRegAlloc() override;
1498 void addFastRegAlloc() override;
1499 void addOptimizedRegAlloc() override;
1500
1501 FunctionPass *createSGPRAllocPass(bool Optimized);
1502 FunctionPass *createVGPRAllocPass(bool Optimized);
1503 FunctionPass *createWWMRegAllocPass(bool Optimized);
1504 FunctionPass *createRegAllocPass(bool Optimized) override;
1505
1506 bool addRegAssignAndRewriteFast() override;
1507 bool addRegAssignAndRewriteOptimized() override;
1508
1509 bool addPreRewrite() override;
1510 void addPostRegAlloc() override;
1511 void addPreSched2() override;
1512 void addPreEmitPass() override;
1513 void addPostBBSections() override;
1514};
1515
1516} // end anonymous namespace
1517
1519 : TargetPassConfig(TM, PM) {
1520 // Exceptions and StackMaps are not supported, so these passes will never do
1521 // anything.
1524 // Garbage collection is not supported.
1527}
1528
1535
1540 // ReassociateGEPs exposes more opportunities for SLSR. See
1541 // the example in reassociate-geps-and-slsr.ll.
1543 // SeparateConstOffsetFromGEP and SLSR creates common expressions which GVN or
1544 // EarlyCSE can reuse.
1546 // Run NaryReassociate after EarlyCSE/GVN to be more effective.
1548 // NaryReassociate on GEPs creates redundant common expressions, so run
1549 // EarlyCSE after it.
1551}
1552
1555
1556 if (RemoveIncompatibleFunctions && TM.getTargetTriple().isAMDGCN())
1558
1559 // There is no reason to run these.
1563
1564 if (TM.getTargetTriple().isAMDGCN())
1566
1567 if (LowerCtorDtor)
1569
1570 if (TM.getTargetTriple().isAMDGCN() &&
1573
1576
1577 // This can be disabled by passing ::Disable here or on the command line
1578 // with --expand-variadics-override=disable.
1580
1581 // Function calls are not supported, so make sure we inline everything.
1584
1585 // Handle uses of OpenCL image2d_t, image3d_t and sampler_t arguments.
1586 if (TM.getTargetTriple().getArch() == Triple::r600)
1588
1589 // Make enqueued block runtime handles externally visible.
1591
1592 // Lower special LDS accesses.
1595
1596 // Lower LDS accesses to global memory pass if address sanitizer is enabled.
1597 if (EnableSwLowerLDS)
1599
1600 // Runs before PromoteAlloca so the latter can account for function uses
1603 }
1604
1605 // Run atomic optimizer before Atomic Expand
1606 if ((TM.getTargetTriple().isAMDGCN()) &&
1607 (TM.getOptLevel() >= CodeGenOptLevel::Less) &&
1610 }
1611
1613
1614 if (TM.getOptLevel() > CodeGenOptLevel::None) {
1616
1619
1623 AAResults &AAR) {
1624 if (auto *WrapperPass = P.getAnalysisIfAvailable<AMDGPUAAWrapperPass>())
1625 AAR.addAAResult(WrapperPass->getResult());
1626 }));
1627 }
1628
1629 if (TM.getTargetTriple().isAMDGCN()) {
1630 // TODO: May want to move later or split into an early and late one.
1632 }
1633
1634 // Try to hoist loop invariant parts of divisions AMDGPUCodeGenPrepare may
1635 // have expanded.
1636 if (TM.getOptLevel() > CodeGenOptLevel::Less)
1638 }
1639
1641
1642 // EarlyCSE is not always strong enough to clean up what LSR produces. For
1643 // example, GVN can combine
1644 //
1645 // %0 = add %a, %b
1646 // %1 = add %b, %a
1647 //
1648 // and
1649 //
1650 // %0 = shl nsw %a, 2
1651 // %1 = shl %a, 2
1652 //
1653 // but EarlyCSE can do neither of them.
1656}
1657
1659 if (TM->getTargetTriple().isAMDGCN() &&
1660 TM->getOptLevel() > CodeGenOptLevel::None)
1662
1663 if (TM->getTargetTriple().isAMDGCN() && EnableLowerKernelArguments)
1665
1667
1670
1671 if (TM->getTargetTriple().isAMDGCN()) {
1672 // This lowering has been placed after codegenprepare to take advantage of
1673 // address mode matching (which is why it isn't put with the LDS lowerings).
1674 // It could be placed anywhere before uniformity annotations (an analysis
1675 // that it changes by splitting up fat pointers into their components)
1676 // but has been put before switch lowering and CFG flattening so that those
1677 // passes can run on the more optimized control flow this pass creates in
1678 // many cases.
1681 }
1682
1683 // LowerSwitch pass may introduce unreachable blocks that can
1684 // cause unexpected behavior for subsequent passes. Placing it
1685 // here seems better that these blocks would get cleaned up by
1686 // UnreachableBlockElim inserted next in the pass flow.
1688}
1689
1691 if (TM->getOptLevel() > CodeGenOptLevel::None)
1693 return false;
1694}
1695
1700
1702 // Do nothing. GC is not supported.
1703 return false;
1704}
1705
1706//===----------------------------------------------------------------------===//
1707// GCN Legacy Pass Setup
1708//===----------------------------------------------------------------------===//
1709
1710bool GCNPassConfig::addPreISel() {
1712
1713 if (TM->getOptLevel() > CodeGenOptLevel::None) {
1714 addPass(createSinkingPass());
1716 }
1717
1718 // Merge divergent exit nodes. StructurizeCFG won't recognize the multi-exit
1719 // regions formed by them.
1721 addPass(createFixIrreduciblePass());
1722 addPass(createUnifyLoopExitsPass());
1723 addPass(createStructurizeCFGPass(false)); // true -> SkipUniformRegions
1724
1727 // TODO: Move this right after structurizeCFG to avoid extra divergence
1728 // analysis. This depends on stopping SIAnnotateControlFlow from making
1729 // control flow modifications.
1731
1732 // SDAG requires LCSSA, GlobalISel does not. Disable LCSSA for -global-isel
1733 // without any of the fallback options.
1736 !isGlobalISelAbortEnabled())
1737 addPass(createLCSSAPass());
1738
1739 if (TM->getOptLevel() > CodeGenOptLevel::Less)
1741
1742 return false;
1743}
1744
1745void GCNPassConfig::addMachineSSAOptimization() {
1747
1748 // We want to fold operands after PeepholeOptimizer has run (or as part of
1749 // it), because it will eliminate extra copies making it easier to fold the
1750 // real source operand. We want to eliminate dead instructions after, so that
1751 // we see fewer uses of the copies. We then need to clean up the dead
1752 // instructions leftover after the operands are folded as well.
1753 //
1754 // XXX - Can we get away without running DeadMachineInstructionElim again?
1755 addPass(&SIFoldOperandsLegacyID);
1756 if (EnableDPPCombine)
1757 addPass(&GCNDPPCombineLegacyID);
1759 if (isPassEnabled(EnableSDWAPeephole)) {
1760 addPass(&SIPeepholeSDWALegacyID);
1761 addPass(&EarlyMachineLICMID);
1762 addPass(&MachineCSELegacyID);
1763 addPass(&SIFoldOperandsLegacyID);
1764 }
1767}
1768
1769bool GCNPassConfig::addILPOpts() {
1771 addPass(&EarlyIfConverterLegacyID);
1772
1774 return false;
1775}
1776
1777bool GCNPassConfig::addInstSelector() {
1779 addPass(&SIFixSGPRCopiesLegacyID);
1781 return false;
1782}
1783
1784bool GCNPassConfig::addIRTranslator() {
1785 addPass(new IRTranslatorLegacy(getOptLevel()));
1786 return false;
1787}
1788
1789void GCNPassConfig::addPreLegalizeMachineIR() {
1790 bool IsOptNone = getOptLevel() == CodeGenOptLevel::None;
1791 addPass(createAMDGPUPreLegalizeCombiner(IsOptNone));
1792 addPass(new LocalizerLegacy());
1793}
1794
1795bool GCNPassConfig::addLegalizeMachineIR() {
1796 addPass(new LegalizerLegacy());
1797 return false;
1798}
1799
1800void GCNPassConfig::addPreRegBankSelect() {
1801 bool IsOptNone = getOptLevel() == CodeGenOptLevel::None;
1802 addPass(createAMDGPUPostLegalizeCombiner(IsOptNone));
1804}
1805
1806bool GCNPassConfig::addRegBankSelect() {
1809 return false;
1810}
1811
1812void GCNPassConfig::addPreGlobalInstructionSelect() {
1813 bool IsOptNone = getOptLevel() == CodeGenOptLevel::None;
1814 addPass(createAMDGPURegBankCombiner(IsOptNone));
1815}
1816
1817bool GCNPassConfig::addGlobalInstructionSelect() {
1818 addPass(new InstructionSelectLegacy(getOptLevel()));
1819 return false;
1820}
1821
1822void GCNPassConfig::addFastRegAlloc() {
1823 // FIXME: We have to disable the verifier here because of PHIElimination +
1824 // TwoAddressInstructions disabling it.
1825
1826 // This must be run immediately after phi elimination and before
1827 // TwoAddressInstructions, otherwise the processing of the tied operand of
1828 // SI_ELSE will introduce a copy of the tied operand source after the else.
1830
1832
1834}
1835
1836void GCNPassConfig::addPreRegAlloc() {
1837 if (getOptLevel() != CodeGenOptLevel::None)
1839}
1840
1841void GCNPassConfig::addOptimizedRegAlloc() {
1842 if (EnableDCEInRA)
1844
1845 // FIXME: when an instruction has a Killed operand, and the instruction is
1846 // inside a bundle, seems only the BUNDLE instruction appears as the Kills of
1847 // the register in LiveVariables, this would trigger a failure in verifier,
1848 // we should fix it and enable the verifier.
1849 if (OptVGPRLiveRange)
1851
1852 // This must be run immediately after phi elimination and before
1853 // TwoAddressInstructions, otherwise the processing of the tied operand of
1854 // SI_ELSE will introduce a copy of the tied operand source after the else.
1856
1859
1860 if (isPassEnabled(EnablePreRAOptimizations))
1862
1863 // Allow the scheduler to run before SIWholeQuadMode inserts exec manipulation
1864 // instructions that cause scheduling barriers.
1866
1867 if (OptExecMaskPreRA)
1869
1870 // This is not an essential optimization and it has a noticeable impact on
1871 // compilation time, so we only enable it from O2.
1872 if (TM->getOptLevel() > CodeGenOptLevel::Less)
1874
1876}
1877
1878bool GCNPassConfig::addPreRewrite() {
1880 addPass(&GCNNSAReassignID);
1881
1883 return true;
1884}
1885
1886FunctionPass *GCNPassConfig::createSGPRAllocPass(bool Optimized) {
1887 // Initialize the global default.
1888 llvm::call_once(InitializeDefaultSGPRRegisterAllocatorFlag,
1889 initializeDefaultSGPRRegisterAllocatorOnce);
1890
1891 RegisterRegAlloc::FunctionPassCtor Ctor = SGPRRegisterRegAlloc::getDefault();
1892 if (Ctor != useDefaultRegisterAllocator)
1893 return Ctor();
1894
1895 if (Optimized)
1896 return createGreedyRegisterAllocator(onlyAllocateSGPRs);
1897
1898 return createFastRegisterAllocator(onlyAllocateSGPRs, false);
1899}
1900
1901FunctionPass *GCNPassConfig::createVGPRAllocPass(bool Optimized) {
1902 // Initialize the global default.
1903 llvm::call_once(InitializeDefaultVGPRRegisterAllocatorFlag,
1904 initializeDefaultVGPRRegisterAllocatorOnce);
1905
1906 RegisterRegAlloc::FunctionPassCtor Ctor = VGPRRegisterRegAlloc::getDefault();
1907 if (Ctor != useDefaultRegisterAllocator)
1908 return Ctor();
1909
1910 if (Optimized)
1911 return createGreedyVGPRRegisterAllocator();
1912
1913 return createFastVGPRRegisterAllocator();
1914}
1915
1916FunctionPass *GCNPassConfig::createWWMRegAllocPass(bool Optimized) {
1917 // Initialize the global default.
1918 llvm::call_once(InitializeDefaultWWMRegisterAllocatorFlag,
1919 initializeDefaultWWMRegisterAllocatorOnce);
1920
1921 RegisterRegAlloc::FunctionPassCtor Ctor = WWMRegisterRegAlloc::getDefault();
1922 if (Ctor != useDefaultRegisterAllocator)
1923 return Ctor();
1924
1925 if (Optimized)
1926 return createGreedyWWMRegisterAllocator();
1927
1928 return createFastWWMRegisterAllocator();
1929}
1930
1931FunctionPass *GCNPassConfig::createRegAllocPass(bool Optimized) {
1932 llvm_unreachable("should not be used");
1933}
1934
1936 "-regalloc not supported with amdgcn. Use -sgpr-regalloc, -wwm-regalloc, "
1937 "and -vgpr-regalloc";
1938
1939bool GCNPassConfig::addRegAssignAndRewriteFast() {
1940 if (!usingDefaultRegAlloc())
1942
1943 addPass(&GCNPreRALongBranchRegID);
1944
1945 addPass(createSGPRAllocPass(false));
1946
1947 // Equivalent of PEI for SGPRs.
1948 addPass(&SILowerSGPRSpillsLegacyID);
1949
1950 // To Allocate wwm registers used in whole quad mode operations (for shaders).
1952
1953 // For allocating other wwm register operands.
1954 addPass(createWWMRegAllocPass(false));
1955
1956 addPass(&SILowerWWMCopiesLegacyID);
1958
1959 // For allocating per-thread VGPRs.
1960 addPass(createVGPRAllocPass(false));
1961
1962 return true;
1963}
1964
1965bool GCNPassConfig::addRegAssignAndRewriteOptimized() {
1966 if (!usingDefaultRegAlloc())
1968
1969 addPass(&GCNPreRALongBranchRegID);
1970
1971 addPass(createSGPRAllocPass(true));
1972
1973 // Commit allocated register changes. This is mostly necessary because too
1974 // many things rely on the use lists of the physical registers, such as the
1975 // verifier. This is only necessary with allocators which use LiveIntervals,
1976 // since FastRegAlloc does the replacements itself.
1977 addPass(createVirtRegRewriter(false));
1978
1979 // At this point, the sgpr-regalloc has been done and it is good to have the
1980 // stack slot coloring to try to optimize the SGPR spill stack indices before
1981 // attempting the custom SGPR spill lowering.
1982 addPass(&StackSlotColoringID);
1983
1984 // Equivalent of PEI for SGPRs.
1985 addPass(&SILowerSGPRSpillsLegacyID);
1986
1987 // To Allocate wwm registers used in whole quad mode operations (for shaders).
1989
1990 // For allocating other whole wave mode registers.
1991 addPass(createWWMRegAllocPass(true));
1992 addPass(&SILowerWWMCopiesLegacyID);
1993 addPass(createVirtRegRewriter(false));
1995
1996 // For allocating per-thread VGPRs.
1997 addPass(createVGPRAllocPass(true));
1998
1999 addPreRewrite();
2000 addPass(&VirtRegRewriterID);
2001
2003
2004 return true;
2005}
2006
2007void GCNPassConfig::addPostRegAlloc() {
2008 addPass(&SIFixVGPRCopiesID);
2009 if (getOptLevel() > CodeGenOptLevel::None)
2012}
2013
2014void GCNPassConfig::addPreSched2() {
2015 if (TM->getOptLevel() > CodeGenOptLevel::None)
2017 addPass(&SIPostRABundlerLegacyID);
2018}
2019
2020void GCNPassConfig::addPreEmitPass() {
2021 if (isPassEnabled(EnableVOPD, CodeGenOptLevel::Less))
2022 addPass(&GCNCreateVOPDID);
2023 addPass(createSIMemoryLegalizerPass());
2024 addPass(createSIInsertWaitcntsPass());
2025
2026 addPass(createSIModeRegisterPass());
2027
2028 if (getOptLevel() > CodeGenOptLevel::None)
2029 addPass(&SIInsertHardClausesID);
2030
2032 if (isPassEnabled(EnableSetWavePriority, CodeGenOptLevel::Less))
2034 if (getOptLevel() > CodeGenOptLevel::None)
2035 addPass(&SIPreEmitPeepholeID);
2036 // The hazard recognizer that runs as part of the post-ra scheduler does not
2037 // guarantee to be able handle all hazards correctly. This is because if there
2038 // are multiple scheduling regions in a basic block, the regions are scheduled
2039 // bottom up, so when we begin to schedule a region we don't know what
2040 // instructions were emitted directly before it.
2041 //
2042 // Here we add a stand-alone hazard recognizer pass which can handle all
2043 // cases.
2044 addPass(&PostRAHazardRecognizerID);
2045
2047
2049
2050 if (isPassEnabled(EnableInsertDelayAlu, CodeGenOptLevel::Less))
2051 addPass(&AMDGPUInsertDelayAluID);
2052
2053 addPass(&BranchRelaxationPassID);
2054}
2055
2056void GCNPassConfig::addPostBBSections() {
2057 // We run this later to avoid passes like livedebugvalues and BBSections
2058 // having to deal with the apparent multi-entry functions we may generate.
2060}
2061
2063 return new GCNPassConfig(*this, PM);
2064}
2065
2071
2078
2082
2089
2092 SMDiagnostic &Error, SMRange &SourceRange) const {
2093 const yaml::SIMachineFunctionInfo &YamlMFI =
2094 static_cast<const yaml::SIMachineFunctionInfo &>(MFI_);
2095 MachineFunction &MF = PFS.MF;
2097 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
2098
2099 if (MFI->initializeBaseYamlFields(YamlMFI, MF, PFS, Error, SourceRange))
2100 return true;
2101
2102 if (MFI->Occupancy == 0) {
2103 // Fixup the subtarget dependent default value.
2104 MFI->Occupancy = ST.getOccupancyWithWorkGroupSizes(MF).second;
2105 }
2106
2107 auto parseRegister = [&](const yaml::StringValue &RegName, Register &RegVal) {
2108 Register TempReg;
2109 if (parseNamedRegisterReference(PFS, TempReg, RegName.Value, Error)) {
2110 SourceRange = RegName.SourceRange;
2111 return true;
2112 }
2113 RegVal = TempReg;
2114
2115 return false;
2116 };
2117
2118 auto parseOptionalRegister = [&](const yaml::StringValue &RegName,
2119 Register &RegVal) {
2120 return !RegName.Value.empty() && parseRegister(RegName, RegVal);
2121 };
2122
2123 if (parseOptionalRegister(YamlMFI.VGPRForAGPRCopy, MFI->VGPRForAGPRCopy))
2124 return true;
2125
2126 if (parseOptionalRegister(YamlMFI.SGPRForEXECCopy, MFI->SGPRForEXECCopy))
2127 return true;
2128
2129 if (parseOptionalRegister(YamlMFI.LongBranchReservedReg,
2130 MFI->LongBranchReservedReg))
2131 return true;
2132
2133 auto diagnoseRegisterClass = [&](const yaml::StringValue &RegName) {
2134 // Create a diagnostic for a the register string literal.
2135 const MemoryBuffer &Buffer =
2136 *PFS.SM->getMemoryBuffer(PFS.SM->getMainFileID());
2137 Error = SMDiagnostic(*PFS.SM, SMLoc(), Buffer.getBufferIdentifier(), 1,
2138 RegName.Value.size(), SourceMgr::DK_Error,
2139 "incorrect register class for field", RegName.Value,
2140 {}, {});
2141 SourceRange = RegName.SourceRange;
2142 return true;
2143 };
2144
2145 if (parseRegister(YamlMFI.ScratchRSrcReg, MFI->ScratchRSrcReg) ||
2146 parseRegister(YamlMFI.FrameOffsetReg, MFI->FrameOffsetReg) ||
2147 parseRegister(YamlMFI.StackPtrOffsetReg, MFI->StackPtrOffsetReg))
2148 return true;
2149
2150 if (MFI->ScratchRSrcReg != AMDGPU::PRIVATE_RSRC_REG &&
2151 !AMDGPU::SGPR_128RegClass.contains(MFI->ScratchRSrcReg)) {
2152 return diagnoseRegisterClass(YamlMFI.ScratchRSrcReg);
2153 }
2154
2155 if (MFI->FrameOffsetReg != AMDGPU::FP_REG &&
2156 !AMDGPU::SGPR_32RegClass.contains(MFI->FrameOffsetReg)) {
2157 return diagnoseRegisterClass(YamlMFI.FrameOffsetReg);
2158 }
2159
2160 if (MFI->StackPtrOffsetReg != AMDGPU::SP_REG &&
2161 !AMDGPU::SGPR_32RegClass.contains(MFI->StackPtrOffsetReg)) {
2162 return diagnoseRegisterClass(YamlMFI.StackPtrOffsetReg);
2163 }
2164
2165 for (const auto &YamlReg : YamlMFI.WWMReservedRegs) {
2166 Register ParsedReg;
2167 if (parseRegister(YamlReg, ParsedReg))
2168 return true;
2169
2170 MFI->reserveWWMRegister(ParsedReg);
2171 }
2172
2173 for (const auto &[_, Info] : PFS.VRegInfosNamed) {
2174 MFI->setFlag(Info->VReg, Info->Flags);
2175 }
2176 for (const auto &[_, Info] : PFS.VRegInfos) {
2177 MFI->setFlag(Info->VReg, Info->Flags);
2178 }
2179
2180 for (const auto &YamlRegStr : YamlMFI.SpillPhysVGPRS) {
2181 Register ParsedReg;
2182 if (parseRegister(YamlRegStr, ParsedReg))
2183 return true;
2184 MFI->SpillPhysVGPRs.push_back(ParsedReg);
2185 }
2186
2187 auto parseAndCheckArgument = [&](const std::optional<yaml::SIArgument> &A,
2188 const TargetRegisterClass &RC,
2189 ArgDescriptor &Arg, unsigned UserSGPRs,
2190 unsigned SystemSGPRs) {
2191 // Skip parsing if it's not present.
2192 if (!A)
2193 return false;
2194
2195 if (A->IsRegister) {
2196 Register Reg;
2197 if (parseNamedRegisterReference(PFS, Reg, A->RegisterName.Value, Error)) {
2198 SourceRange = A->RegisterName.SourceRange;
2199 return true;
2200 }
2201 if (!RC.contains(Reg))
2202 return diagnoseRegisterClass(A->RegisterName);
2204 } else
2205 Arg = ArgDescriptor::createStack(A->StackOffset);
2206 // Check and apply the optional mask.
2207 if (A->Mask)
2208 Arg = ArgDescriptor::createArg(Arg, *A->Mask);
2209
2210 MFI->NumUserSGPRs += UserSGPRs;
2211 MFI->NumSystemSGPRs += SystemSGPRs;
2212 return false;
2213 };
2214
2215 if (YamlMFI.ArgInfo &&
2216 (parseAndCheckArgument(YamlMFI.ArgInfo->PrivateSegmentBuffer,
2217 AMDGPU::SGPR_128RegClass,
2218 MFI->ArgInfo.PrivateSegmentBuffer, 4, 0) ||
2219 parseAndCheckArgument(YamlMFI.ArgInfo->DispatchPtr,
2220 AMDGPU::SReg_64RegClass, MFI->ArgInfo.DispatchPtr,
2221 2, 0) ||
2222 parseAndCheckArgument(YamlMFI.ArgInfo->QueuePtr, AMDGPU::SReg_64RegClass,
2223 MFI->ArgInfo.QueuePtr, 2, 0) ||
2224 parseAndCheckArgument(YamlMFI.ArgInfo->KernargSegmentPtr,
2225 AMDGPU::SReg_64RegClass,
2226 MFI->ArgInfo.KernargSegmentPtr, 2, 0) ||
2227 parseAndCheckArgument(YamlMFI.ArgInfo->DispatchID,
2228 AMDGPU::SReg_64RegClass, MFI->ArgInfo.DispatchID,
2229 2, 0) ||
2230 parseAndCheckArgument(YamlMFI.ArgInfo->FlatScratchInit,
2231 AMDGPU::SReg_64RegClass,
2232 MFI->ArgInfo.FlatScratchInit, 2, 0) ||
2233 parseAndCheckArgument(YamlMFI.ArgInfo->PrivateSegmentSize,
2234 AMDGPU::SGPR_32RegClass,
2235 MFI->ArgInfo.PrivateSegmentSize, 0, 0) ||
2236 parseAndCheckArgument(YamlMFI.ArgInfo->LDSKernelId,
2237 AMDGPU::SGPR_32RegClass,
2238 MFI->ArgInfo.LDSKernelId, 0, 1) ||
2239 parseAndCheckArgument(YamlMFI.ArgInfo->WorkGroupIDX,
2240 AMDGPU::SGPR_32RegClass, MFI->ArgInfo.WorkGroupIDX,
2241 0, 1) ||
2242 parseAndCheckArgument(YamlMFI.ArgInfo->WorkGroupIDY,
2243 AMDGPU::SGPR_32RegClass, MFI->ArgInfo.WorkGroupIDY,
2244 0, 1) ||
2245 parseAndCheckArgument(YamlMFI.ArgInfo->WorkGroupIDZ,
2246 AMDGPU::SGPR_32RegClass, MFI->ArgInfo.WorkGroupIDZ,
2247 0, 1) ||
2248 parseAndCheckArgument(YamlMFI.ArgInfo->WorkGroupInfo,
2249 AMDGPU::SGPR_32RegClass,
2250 MFI->ArgInfo.WorkGroupInfo, 0, 1) ||
2251 parseAndCheckArgument(YamlMFI.ArgInfo->PrivateSegmentWaveByteOffset,
2252 AMDGPU::SGPR_32RegClass,
2253 MFI->ArgInfo.PrivateSegmentWaveByteOffset, 0, 1) ||
2254 parseAndCheckArgument(YamlMFI.ArgInfo->ImplicitArgPtr,
2255 AMDGPU::SReg_64RegClass,
2256 MFI->ArgInfo.ImplicitArgPtr, 0, 0) ||
2257 parseAndCheckArgument(YamlMFI.ArgInfo->ImplicitBufferPtr,
2258 AMDGPU::SReg_64RegClass,
2259 MFI->ArgInfo.ImplicitBufferPtr, 2, 0) ||
2260 parseAndCheckArgument(YamlMFI.ArgInfo->WorkItemIDX,
2261 AMDGPU::VGPR_32RegClass,
2262 MFI->ArgInfo.WorkItemIDX, 0, 0) ||
2263 parseAndCheckArgument(YamlMFI.ArgInfo->WorkItemIDY,
2264 AMDGPU::VGPR_32RegClass,
2265 MFI->ArgInfo.WorkItemIDY, 0, 0) ||
2266 parseAndCheckArgument(YamlMFI.ArgInfo->WorkItemIDZ,
2267 AMDGPU::VGPR_32RegClass,
2268 MFI->ArgInfo.WorkItemIDZ, 0, 0)))
2269 return true;
2270
2271 // Parse FirstKernArgPreloadReg separately, since it's a Register,
2272 // not ArgDescriptor.
2273 if (YamlMFI.ArgInfo && YamlMFI.ArgInfo->FirstKernArgPreloadReg) {
2274 const yaml::SIArgument &A = *YamlMFI.ArgInfo->FirstKernArgPreloadReg;
2275
2276 if (!A.IsRegister) {
2277 // For stack arguments, we don't have RegisterName.SourceRange,
2278 // but we should have some location info from the YAML parser
2279 const MemoryBuffer &Buffer =
2280 *PFS.SM->getMemoryBuffer(PFS.SM->getMainFileID());
2281 // Create a minimal valid source range
2283 SMRange Range(Loc, Loc);
2284
2286 *PFS.SM, Loc, Buffer.getBufferIdentifier(), 1, 0, SourceMgr::DK_Error,
2287 "firstKernArgPreloadReg must be a register, not a stack location", "",
2288 {}, {});
2289
2290 SourceRange = Range;
2291 return true;
2292 }
2293
2294 Register Reg;
2295 if (parseNamedRegisterReference(PFS, Reg, A.RegisterName.Value, Error)) {
2296 SourceRange = A.RegisterName.SourceRange;
2297 return true;
2298 }
2299
2300 if (!AMDGPU::SGPR_32RegClass.contains(Reg))
2301 return diagnoseRegisterClass(A.RegisterName);
2302
2303 MFI->ArgInfo.FirstKernArgPreloadReg = Reg;
2304 MFI->NumUserSGPRs += YamlMFI.NumKernargPreloadSGPRs;
2305 }
2306
2307 if (ST.hasFeature(AMDGPU::FeatureDX10ClampAndIEEEMode)) {
2308 MFI->Mode.IEEE = YamlMFI.Mode.IEEE;
2309 MFI->Mode.DX10Clamp = YamlMFI.Mode.DX10Clamp;
2310 }
2311
2312 // FIXME: Move proper support for denormal-fp-math into base MachineFunction
2313 MFI->Mode.FP32Denormals.Input = YamlMFI.Mode.FP32InputDenormals
2316 MFI->Mode.FP32Denormals.Output = YamlMFI.Mode.FP32OutputDenormals
2319
2326
2327 if (YamlMFI.HasInitWholeWave)
2328 MFI->setInitWholeWave();
2329
2330 return false;
2331}
2332
2333//===----------------------------------------------------------------------===//
2334// AMDGPU CodeGen Pass Builder interface.
2335//===----------------------------------------------------------------------===//
2336
2337AMDGPUCodeGenPassBuilder::AMDGPUCodeGenPassBuilder(
2338 GCNTargetMachine &TM, const CGPassBuilderOption &Opts,
2340 : CodeGenPassBuilder(TM, Opts, PIC) {
2341 Opt.MISchedPostRA = true;
2342 Opt.RequiresCodeGenSCCOrder = true;
2343 // Exceptions and StackMaps are not supported, so these passes will never do
2344 // anything.
2345 // Garbage collection is not supported.
2346 disablePass<StackMapLivenessPass, FuncletLayoutPass, PatchableFunctionPass,
2348}
2349
2350void AMDGPUCodeGenPassBuilder::addIRPasses(PassManagerWrapper &PMW) {
2351 if (RemoveIncompatibleFunctions && TM.getTargetTriple().isAMDGCN()) {
2352 flushFPMsToMPM(PMW);
2353 addModulePass(AMDGPURemoveIncompatibleFunctionsPass(TM), PMW);
2354 }
2355
2356 flushFPMsToMPM(PMW);
2357
2358 if (TM.getTargetTriple().isAMDGCN())
2359 addModulePass(AMDGPUPrintfRuntimeBindingPass(), PMW);
2360
2361 if (LowerCtorDtor)
2362 addModulePass(AMDGPUCtorDtorLoweringPass(), PMW);
2363
2364 if (isPassEnabled(EnableImageIntrinsicOptimizer))
2365 addFunctionPass(AMDGPUImageIntrinsicOptimizerPass(TM), PMW);
2366
2368 addFunctionPass(AMDGPUUniformIntrinsicCombinePass(), PMW);
2369 // This can be disabled by passing ::Disable here or on the command line
2370 // with --expand-variadics-override=disable.
2371 flushFPMsToMPM(PMW);
2373
2374 addModulePass(AMDGPUAlwaysInlinePass(), PMW);
2375 addModulePass(AlwaysInlinerPass(), PMW);
2376
2377 addModulePass(AMDGPUExportKernelRuntimeHandlesPass(), PMW);
2378
2380 addModulePass(AMDGPULowerExecSyncPass(), PMW);
2381
2382 if (EnableSwLowerLDS)
2383 addModulePass(AMDGPUSwLowerLDSPass(), PMW);
2384
2385 // Runs before PromoteAlloca so the latter can account for function uses
2387 addModulePass(AMDGPULowerModuleLDSPass(getTM()), PMW);
2388
2389 // Run atomic optimizer before Atomic Expand
2390 if (TM.getOptLevel() >= CodeGenOptLevel::Less &&
2392 addFunctionPass(
2394
2395 addFunctionPass(AtomicExpandPass(TM), PMW);
2396
2397 if (TM.getOptLevel() > CodeGenOptLevel::None) {
2398 addFunctionPass(AMDGPUPromoteAllocaPass(TM), PMW);
2399 if (isPassEnabled(EnableScalarIRPasses))
2400 addStraightLineScalarOptimizationPasses(PMW);
2401
2402 // TODO: Handle EnableAMDGPUAliasAnalysis
2403
2404 // TODO: May want to move later or split into an early and late one.
2405 addFunctionPass(AMDGPUCodeGenPreparePass(TM), PMW);
2406
2407 // Try to hoist loop invariant parts of divisions AMDGPUCodeGenPrepare may
2408 // have expanded.
2409 if (TM.getOptLevel() > CodeGenOptLevel::Less) {
2411 /*UseMemorySSA=*/true),
2412 PMW);
2413 }
2414 }
2415
2416 Base::addIRPasses(PMW);
2417
2418 // EarlyCSE is not always strong enough to clean up what LSR produces. For
2419 // example, GVN can combine
2420 //
2421 // %0 = add %a, %b
2422 // %1 = add %b, %a
2423 //
2424 // and
2425 //
2426 // %0 = shl nsw %a, 2
2427 // %1 = shl %a, 2
2428 //
2429 // but EarlyCSE can do neither of them.
2430 if (isPassEnabled(EnableScalarIRPasses))
2431 addEarlyCSEOrGVNPass(PMW);
2432}
2433
2434void AMDGPUCodeGenPassBuilder::addCodeGenPrepare(PassManagerWrapper &PMW) {
2435 if (TM.getOptLevel() > CodeGenOptLevel::None) {
2436 flushFPMsToMPM(PMW);
2437 addModulePass(AMDGPUPreloadKernelArgumentsPass(TM), PMW);
2438 }
2439
2441 addFunctionPass(AMDGPULowerKernelArgumentsPass(TM), PMW);
2442
2443 Base::addCodeGenPrepare(PMW);
2444
2445 if (isPassEnabled(EnableLoadStoreVectorizer))
2446 addFunctionPass(LoadStoreVectorizerPass(), PMW);
2447
2448 // This lowering has been placed after codegenprepare to take advantage of
2449 // address mode matching (which is why it isn't put with the LDS lowerings).
2450 // It could be placed anywhere before uniformity annotations (an analysis
2451 // that it changes by splitting up fat pointers into their components)
2452 // but has been put before switch lowering and CFG flattening so that those
2453 // passes can run on the more optimized control flow this pass creates in
2454 // many cases.
2455 flushFPMsToMPM(PMW);
2456 addModulePass(AMDGPULowerBufferFatPointersPass(TM), PMW);
2457 flushFPMsToMPM(PMW);
2458 requireCGSCCOrder(PMW);
2459
2460 addModulePass(AMDGPULowerIntrinsicsPass(getTM()), PMW);
2461
2462 // LowerSwitch pass may introduce unreachable blocks that can cause unexpected
2463 // behavior for subsequent passes. Placing it here seems better that these
2464 // blocks would get cleaned up by UnreachableBlockElim inserted next in the
2465 // pass flow.
2466 addFunctionPass(LowerSwitchPass(), PMW);
2467}
2468
2469void AMDGPUCodeGenPassBuilder::addPreISel(PassManagerWrapper &PMW) {
2470
2471 if (TM.getOptLevel() > CodeGenOptLevel::None) {
2472 addFunctionPass(FlattenCFGPass(), PMW);
2473 addFunctionPass(SinkingPass(), PMW);
2474 addFunctionPass(AMDGPULateCodeGenPreparePass(getTM()), PMW);
2475 }
2476
2477 // Merge divergent exit nodes. StructurizeCFG won't recognize the multi-exit
2478 // regions formed by them.
2479
2480 addFunctionPass(AMDGPUUnifyDivergentExitNodesPass(), PMW);
2481 addFunctionPass(FixIrreduciblePass(), PMW);
2482 addFunctionPass(UnifyLoopExitsPass(), PMW);
2483 addFunctionPass(StructurizeCFGPass(/*SkipUniformRegions=*/false), PMW);
2484
2485 addFunctionPass(AMDGPUAnnotateUniformValuesPass(), PMW);
2486
2487 addFunctionPass(SIAnnotateControlFlowPass(getTM()), PMW);
2488
2489 // TODO: Move this right after structurizeCFG to avoid extra divergence
2490 // analysis. This depends on stopping SIAnnotateControlFlow from making
2491 // control flow modifications.
2492 addFunctionPass(AMDGPURewriteUndefForPHIPass(), PMW);
2493
2496 !isGlobalISelAbortEnabled())
2497 addFunctionPass(LCSSAPass(), PMW);
2498
2499 if (TM.getOptLevel() > CodeGenOptLevel::Less) {
2500 flushFPMsToMPM(PMW);
2501 addModulePass(AMDGPUPerfHintAnalysisPass(getTM()), PMW);
2502 }
2503}
2504
2505void AMDGPUCodeGenPassBuilder::addILPOpts(PassManagerWrapper &PMW) {
2507 addMachineFunctionPass(EarlyIfConverterPass(), PMW);
2508
2509 Base::addILPOpts(PMW);
2510}
2511
2512void AMDGPUCodeGenPassBuilder::addAsmPrinterBegin(PassManagerWrapper &PMW) {
2513 addModulePass(AMDGPUAsmPrinterBeginPass(), PMW,
2514 /*Force=*/true);
2515}
2516
2517void AMDGPUCodeGenPassBuilder::addAsmPrinter(PassManagerWrapper &PMW) {
2518 addMachineFunctionPass(AMDGPUAsmPrinterPass(), PMW);
2519}
2520
2521void AMDGPUCodeGenPassBuilder::addAsmPrinterEnd(PassManagerWrapper &PMW) {
2522 addModulePass(AMDGPUAsmPrinterEndPass(), PMW);
2523}
2524
2525Error AMDGPUCodeGenPassBuilder::addInstSelector(PassManagerWrapper &PMW) {
2526 addMachineFunctionPass(AMDGPUISelDAGToDAGPass(TM), PMW);
2527 addMachineFunctionPass(SIFixSGPRCopiesPass(), PMW);
2528 addMachineFunctionPass(SILowerI1CopiesPass(), PMW);
2529 return Error::success();
2530}
2531
2532void AMDGPUCodeGenPassBuilder::addPreRewrite(PassManagerWrapper &PMW) {
2533 if (EnableRegReassign) {
2534 addMachineFunctionPass(GCNNSAReassignPass(), PMW);
2535 }
2536
2537 addMachineFunctionPass(AMDGPURewriteAGPRCopyMFMAPass(), PMW);
2538}
2539
2540void AMDGPUCodeGenPassBuilder::addMachineSSAOptimization(
2541 PassManagerWrapper &PMW) {
2542 Base::addMachineSSAOptimization(PMW);
2543
2544 addMachineFunctionPass(SIFoldOperandsPass(), PMW);
2545 if (EnableDPPCombine) {
2546 addMachineFunctionPass(GCNDPPCombinePass(), PMW);
2547 }
2548 addMachineFunctionPass(SILoadStoreOptimizerPass(), PMW);
2549 if (isPassEnabled(EnableSDWAPeephole)) {
2550 addMachineFunctionPass(SIPeepholeSDWAPass(), PMW);
2551 addMachineFunctionPass(EarlyMachineLICMPass(), PMW);
2552 addMachineFunctionPass(MachineCSEPass(), PMW);
2553 addMachineFunctionPass(SIFoldOperandsPass(), PMW);
2554 }
2555 addMachineFunctionPass(DeadMachineInstructionElimPass(), PMW);
2556 addMachineFunctionPass(SIShrinkInstructionsPass(), PMW);
2557}
2558
2559Error AMDGPUCodeGenPassBuilder::addFastRegAlloc(PassManagerWrapper &PMW) {
2560 insertPass<PHIEliminationPass>(SILowerControlFlowPass());
2561
2562 insertPass<TwoAddressInstructionPass>(SIWholeQuadModePass());
2563
2564 return Base::addFastRegAlloc(PMW);
2565}
2566
2567Error AMDGPUCodeGenPassBuilder::addRegAssignAndRewriteFast(
2568 PassManagerWrapper &PMW) {
2569 if (auto Err = validateRegAllocOptions())
2570 return Err;
2571
2572 addMachineFunctionPass(GCNPreRALongBranchRegPass(), PMW);
2573
2574 // SGPR allocation - default to fast at -O0.
2575 if (SGPRRegAllocNPM == RegAllocType::Greedy)
2576 addMachineFunctionPass(RAGreedyPass({onlyAllocateSGPRs, "sgpr"}), PMW);
2577 else
2578 addMachineFunctionPass(RegAllocFastPass({onlyAllocateSGPRs, "sgpr", false}),
2579 PMW);
2580
2581 // Equivalent of PEI for SGPRs.
2582 addMachineFunctionPass(SILowerSGPRSpillsPass(), PMW);
2583
2584 // To Allocate wwm registers used in whole quad mode operations (for shaders).
2585 addMachineFunctionPass(SIPreAllocateWWMRegsPass(), PMW);
2586
2587 // WWM allocation - default to fast at -O0.
2588 if (WWMRegAllocNPM == RegAllocType::Greedy)
2589 addMachineFunctionPass(RAGreedyPass({onlyAllocateWWMRegs, "wwm"}), PMW);
2590 else
2591 addMachineFunctionPass(
2592 RegAllocFastPass({onlyAllocateWWMRegs, "wwm", false}), PMW);
2593
2594 addMachineFunctionPass(SILowerWWMCopiesPass(), PMW);
2595 addMachineFunctionPass(AMDGPUReserveWWMRegsPass(), PMW);
2596
2597 // VGPR allocation - default to fast at -O0.
2598 if (VGPRRegAllocNPM == RegAllocType::Greedy)
2599 addMachineFunctionPass(RAGreedyPass({onlyAllocateVGPRs, "vgpr"}), PMW);
2600 else
2601 addMachineFunctionPass(RegAllocFastPass({onlyAllocateVGPRs, "vgpr"}), PMW);
2602
2603 return Error::success();
2604}
2605
2606Error AMDGPUCodeGenPassBuilder::addOptimizedRegAlloc(PassManagerWrapper &PMW) {
2607 if (EnableDCEInRA)
2608 insertPass<DetectDeadLanesPass>(DeadMachineInstructionElimPass());
2609
2610 // FIXME: when an instruction has a Killed operand, and the instruction is
2611 // inside a bundle, seems only the BUNDLE instruction appears as the Kills of
2612 // the register in LiveVariables, this would trigger a failure in verifier,
2613 // we should fix it and enable the verifier.
2614 if (OptVGPRLiveRange)
2615 insertPass<RequireAnalysisPass<LiveVariablesAnalysis, MachineFunction>>(
2617
2618 // This must be run immediately after phi elimination and before
2619 // TwoAddressInstructions, otherwise the processing of the tied operand of
2620 // SI_ELSE will introduce a copy of the tied operand source after the else.
2621 insertPass<PHIEliminationPass>(SILowerControlFlowPass());
2622
2624 insertPass<RenameIndependentSubregsPass>(GCNRewritePartialRegUsesPass());
2625
2626 if (isPassEnabled(EnablePreRAOptimizations))
2627 insertPass<MachineSchedulerPass>(GCNPreRAOptimizationsPass());
2628
2629 // Allow the scheduler to run before SIWholeQuadMode inserts exec manipulation
2630 // instructions that cause scheduling barriers.
2631 insertPass<MachineSchedulerPass>(SIWholeQuadModePass());
2632
2633 if (OptExecMaskPreRA)
2634 insertPass<MachineSchedulerPass>(SIOptimizeExecMaskingPreRAPass());
2635
2636 // This is not an essential optimization and it has a noticeable impact on
2637 // compilation time, so we only enable it from O2.
2638 if (TM.getOptLevel() > CodeGenOptLevel::Less)
2639 insertPass<MachineSchedulerPass>(SIFormMemoryClausesPass());
2640
2641 return Base::addOptimizedRegAlloc(PMW);
2642}
2643
2644void AMDGPUCodeGenPassBuilder::addPreRegAlloc(PassManagerWrapper &PMW) {
2645 if (getOptLevel() != CodeGenOptLevel::None)
2646 addMachineFunctionPass(AMDGPUPrepareAGPRAllocPass(), PMW);
2647}
2648
2649Expected<bool> AMDGPUCodeGenPassBuilder::addRegAssignAndRewriteOptimized(
2650 PassManagerWrapper &PMW) {
2651 if (auto Err = validateRegAllocOptions())
2652 return Err;
2653
2654 addMachineFunctionPass(GCNPreRALongBranchRegPass(), PMW);
2655
2656 // SGPR allocation - default to greedy at -O1 and above.
2657 if (SGPRRegAllocNPM == RegAllocType::Fast)
2658 addMachineFunctionPass(RegAllocFastPass({onlyAllocateSGPRs, "sgpr", false}),
2659 PMW);
2660 else
2661 addMachineFunctionPass(RAGreedyPass({onlyAllocateSGPRs, "sgpr"}), PMW);
2662
2663 // Commit allocated register changes. This is mostly necessary because too
2664 // many things rely on the use lists of the physical registers, such as the
2665 // verifier. This is only necessary with allocators which use LiveIntervals,
2666 // since FastRegAlloc does the replacements itself.
2667 addMachineFunctionPass(VirtRegRewriterPass(false), PMW);
2668
2669 // At this point, the sgpr-regalloc has been done and it is good to have the
2670 // stack slot coloring to try to optimize the SGPR spill stack indices before
2671 // attempting the custom SGPR spill lowering.
2672 addMachineFunctionPass(StackSlotColoringPass(), PMW);
2673
2674 // Equivalent of PEI for SGPRs.
2675 addMachineFunctionPass(SILowerSGPRSpillsPass(), PMW);
2676
2677 // To Allocate wwm registers used in whole quad mode operations (for shaders).
2678 addMachineFunctionPass(SIPreAllocateWWMRegsPass(), PMW);
2679
2680 // WWM allocation - default to greedy at -O1 and above.
2681 if (WWMRegAllocNPM == RegAllocType::Fast)
2682 addMachineFunctionPass(
2683 RegAllocFastPass({onlyAllocateWWMRegs, "wwm", false}), PMW);
2684 else
2685 addMachineFunctionPass(RAGreedyPass({onlyAllocateWWMRegs, "wwm"}), PMW);
2686 addMachineFunctionPass(SILowerWWMCopiesPass(), PMW);
2687 addMachineFunctionPass(VirtRegRewriterPass(false), PMW);
2688 addMachineFunctionPass(AMDGPUReserveWWMRegsPass(), PMW);
2689
2690 // VGPR allocation - default to greedy at -O1 and above.
2691 if (VGPRRegAllocNPM == RegAllocType::Fast)
2692 addMachineFunctionPass(RegAllocFastPass({onlyAllocateVGPRs, "vgpr"}), PMW);
2693 else
2694 addMachineFunctionPass(RAGreedyPass({onlyAllocateVGPRs, "vgpr"}), PMW);
2695
2696 addPreRewrite(PMW);
2697 addMachineFunctionPass(VirtRegRewriterPass(true), PMW);
2698
2699 addMachineFunctionPass(AMDGPUMarkLastScratchLoadPass(), PMW);
2700 return true;
2701}
2702
2703void AMDGPUCodeGenPassBuilder::addPostRegAlloc(PassManagerWrapper &PMW) {
2704 addMachineFunctionPass(SIFixVGPRCopiesPass(), PMW);
2705 if (TM.getOptLevel() > CodeGenOptLevel::None)
2706 addMachineFunctionPass(SIOptimizeExecMaskingPass(), PMW);
2707 Base::addPostRegAlloc(PMW);
2708}
2709
2710void AMDGPUCodeGenPassBuilder::addPreSched2(PassManagerWrapper &PMW) {
2711 if (TM.getOptLevel() > CodeGenOptLevel::None)
2712 addMachineFunctionPass(SIShrinkInstructionsPass(), PMW);
2713 addMachineFunctionPass(SIPostRABundlerPass(), PMW);
2714}
2715
2716void AMDGPUCodeGenPassBuilder::addPostBBSections(PassManagerWrapper &PMW) {
2717 // We run this later to avoid passes like livedebugvalues and BBSections
2718 // having to deal with the apparent multi-entry functions we may generate.
2719 addMachineFunctionPass(AMDGPUPreloadKernArgPrologPass(), PMW);
2720}
2721
2722void AMDGPUCodeGenPassBuilder::addPreEmitPass(PassManagerWrapper &PMW) {
2723 if (isPassEnabled(EnableVOPD, CodeGenOptLevel::Less)) {
2724 addMachineFunctionPass(GCNCreateVOPDPass(), PMW);
2725 }
2726
2727 addMachineFunctionPass(SIMemoryLegalizerPass(), PMW);
2728 addMachineFunctionPass(SIInsertWaitcntsPass(), PMW);
2729
2730 addMachineFunctionPass(SIModeRegisterPass(), PMW);
2731
2732 if (TM.getOptLevel() > CodeGenOptLevel::None)
2733 addMachineFunctionPass(SIInsertHardClausesPass(), PMW);
2734
2735 addMachineFunctionPass(SILateBranchLoweringPass(), PMW);
2736
2737 if (isPassEnabled(EnableSetWavePriority, CodeGenOptLevel::Less))
2738 addMachineFunctionPass(AMDGPUSetWavePriorityPass(), PMW);
2739
2740 if (TM.getOptLevel() > CodeGenOptLevel::None)
2741 addMachineFunctionPass(SIPreEmitPeepholePass(), PMW);
2742
2743 // The hazard recognizer that runs as part of the post-ra scheduler does not
2744 // guarantee to be able handle all hazards correctly. This is because if there
2745 // are multiple scheduling regions in a basic block, the regions are scheduled
2746 // bottom up, so when we begin to schedule a region we don't know what
2747 // instructions were emitted directly before it.
2748 //
2749 // Here we add a stand-alone hazard recognizer pass which can handle all
2750 // cases.
2751 addMachineFunctionPass(PostRAHazardRecognizerPass(), PMW);
2752 addMachineFunctionPass(AMDGPUWaitSGPRHazardsPass(), PMW);
2753 addMachineFunctionPass(AMDGPULowerVGPREncodingPass(), PMW);
2754
2755 if (isPassEnabled(EnableInsertDelayAlu, CodeGenOptLevel::Less)) {
2756 addMachineFunctionPass(AMDGPUInsertDelayAluPass(), PMW);
2757 }
2758
2759 addMachineFunctionPass(BranchRelaxationPass(), PMW);
2760}
2761
2762bool AMDGPUCodeGenPassBuilder::isPassEnabled(const cl::opt<bool> &Opt,
2763 CodeGenOptLevel Level) const {
2764 if (Opt.getNumOccurrences())
2765 return Opt;
2766 if (TM.getOptLevel() < Level)
2767 return false;
2768 return Opt;
2769}
2770
2771void AMDGPUCodeGenPassBuilder::addEarlyCSEOrGVNPass(PassManagerWrapper &PMW) {
2772 if (TM.getOptLevel() == CodeGenOptLevel::Aggressive)
2773 addFunctionPass(GVNPass(), PMW);
2774 else
2775 addFunctionPass(EarlyCSEPass(), PMW);
2776}
2777
2778void AMDGPUCodeGenPassBuilder::addStraightLineScalarOptimizationPasses(
2779 PassManagerWrapper &PMW) {
2781 addFunctionPass(LoopDataPrefetchPass(), PMW);
2782
2783 addFunctionPass(SeparateConstOffsetFromGEPPass(), PMW);
2784
2785 // ReassociateGEPs exposes more opportunities for SLSR. See
2786 // the example in reassociate-geps-and-slsr.ll.
2787 addFunctionPass(StraightLineStrengthReducePass(), PMW);
2788
2789 // SeparateConstOffsetFromGEP and SLSR creates common expressions which GVN or
2790 // EarlyCSE can reuse.
2791 addEarlyCSEOrGVNPass(PMW);
2792
2793 // Run NaryReassociate after EarlyCSE/GVN to be more effective.
2794 addFunctionPass(NaryReassociatePass(), PMW);
2795
2796 // NaryReassociate on GEPs creates redundant common expressions, so run
2797 // EarlyCSE after it.
2798 addFunctionPass(EarlyCSEPass(), PMW);
2799}
aarch64 falkor hwpf fix Falkor HW Prefetch Fix Late Phase
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static cl::opt< bool > EnableEarlyIfConversion("aarch64-enable-early-ifcvt", cl::Hidden, cl::desc("Run early if-conversion"), cl::init(true))
static std::unique_ptr< TargetLoweringObjectFile > createTLOF(const Triple &TT)
This is the AMGPU address space based alias analysis pass.
AMDGPU Assembly printer class.
Coexecution-focused scheduling strategy for AMDGPU.
Defines an instruction selector for the AMDGPU target.
Analyzes if a function potentially memory bound and if a kernel kernel may benefit from limiting numb...
Analyzes how many registers and other resources are used by functions.
static cl::opt< bool > EnableDCEInRA("amdgpu-dce-in-ra", cl::init(true), cl::Hidden, cl::desc("Enable machine DCE inside regalloc"))
static cl::opt< bool, true > EnableLowerModuleLDS("amdgpu-enable-lower-module-lds", cl::desc("Enable lower module lds pass"), cl::location(AMDGPUTargetMachine::EnableLowerModuleLDS), cl::init(true), cl::Hidden)
static MachineSchedRegistry GCNMaxMemoryClauseSchedRegistry("gcn-max-memory-clause", "Run GCN scheduler to maximize memory clause", createGCNMaxMemoryClauseMachineScheduler)
static Reloc::Model getEffectiveRelocModel()
static cl::opt< bool > EnableUniformIntrinsicCombine("amdgpu-enable-uniform-intrinsic-combine", cl::desc("Enable/Disable the Uniform Intrinsic Combine Pass"), cl::init(true), cl::Hidden)
static MachineSchedRegistry SISchedRegistry("si", "Run SI's custom scheduler", createSIMachineScheduler)
static ScheduleDAGInstrs * createIterativeILPMachineScheduler(MachineSchedContext *C)
static cl::opt< bool > EarlyInlineAll("amdgpu-early-inline-all", cl::desc("Inline all functions early"), cl::init(false), cl::Hidden)
static OOBFlagValue getOOBFlagValue(const Module &M, StringRef FlagName)
Returns the OOB mode encoded by a module flag.
static cl::opt< bool > EnableSwLowerLDS("amdgpu-enable-sw-lower-lds", cl::desc("Enable lowering of lds to global memory pass " "and asan instrument resulting IR."), cl::init(true), cl::Hidden)
static cl::opt< bool > EnableLowerKernelArguments("amdgpu-ir-lower-kernel-arguments", cl::desc("Lower kernel argument loads in IR pass"), cl::init(true), cl::Hidden)
static cl::opt< bool, true > EnableObjectLinking("amdgpu-enable-object-linking", cl::desc("Enable object linking for cross-TU LDS and ABI support"), cl::location(AMDGPUTargetMachine::EnableObjectLinking), cl::init(false), cl::Hidden)
static ScheduleDAGInstrs * createGCNMaxILPMachineScheduler(MachineSchedContext *C)
static cl::opt< bool > EnableSDWAPeephole("amdgpu-sdwa-peephole", cl::desc("Enable SDWA peepholer"), cl::init(true))
static MachineSchedRegistry GCNMinRegSchedRegistry("gcn-iterative-minreg", "Run GCN iterative scheduler for minimal register usage (experimental)", createMinRegScheduler)
static cl::opt< bool > SramEccSetting("amdgpu-sramecc", cl::desc("Force amdgpu.sramecc for testing"), cl::ReallyHidden)
static void diagnoseUnsupportedCoExecSchedulerSelection(const Function &F, const GCNSubtarget &ST)
static cl::opt< bool > EnableImageIntrinsicOptimizer("amdgpu-enable-image-intrinsic-optimizer", cl::desc("Enable image intrinsic optimizer pass"), cl::init(true), cl::Hidden)
static cl::opt< bool > HasClosedWorldAssumption("amdgpu-link-time-closed-world", cl::desc("Whether has closed-world assumption at link time"), cl::init(false), cl::Hidden)
static bool useNoopPostScheduler(const Function &F)
static ScheduleDAGInstrs * createGCNMaxMemoryClauseMachineScheduler(MachineSchedContext *C)
static cl::opt< bool > EnableSIModeRegisterPass("amdgpu-mode-register", cl::desc("Enable mode register pass"), cl::init(true), cl::Hidden)
static cl::opt< std::string > AMDGPUSchedStrategy("amdgpu-sched-strategy", cl::desc("Select custom AMDGPU scheduling strategy."), cl::Hidden, cl::init(""))
static cl::opt< bool > EnableDPPCombine("amdgpu-dpp-combine", cl::desc("Enable DPP combiner"), cl::init(true))
static MachineSchedRegistry IterativeGCNMaxOccupancySchedRegistry("gcn-iterative-max-occupancy-experimental", "Run GCN scheduler to maximize occupancy (experimental)", createIterativeGCNMaxOccupancyMachineScheduler)
static cl::opt< bool > EnableSetWavePriority("amdgpu-set-wave-priority", cl::desc("Adjust wave priority"), cl::init(false), cl::Hidden)
static cl::opt< bool > LowerCtorDtor("amdgpu-lower-global-ctor-dtor", cl::desc("Lower GPU ctor / dtors to globals on the device."), cl::init(true), cl::Hidden)
static cl::opt< bool > XnackSetting("amdgpu-xnack", cl::desc("Force amdgpu.xnack value for testing"), cl::ReallyHidden)
static cl::opt< bool > OptExecMaskPreRA("amdgpu-opt-exec-mask-pre-ra", cl::Hidden, cl::desc("Run pre-RA exec mask optimizations"), cl::init(true))
static cl::opt< bool > EnablePromoteKernelArguments("amdgpu-enable-promote-kernel-arguments", cl::desc("Enable promotion of flat kernel pointer arguments to global"), cl::Hidden, cl::init(true))
LLVM_ABI LLVM_EXTERNAL_VISIBILITY void LLVMInitializeAMDGPUTarget()
static cl::opt< bool > EnableRewritePartialRegUses("amdgpu-enable-rewrite-partial-reg-uses", cl::desc("Enable rewrite partial reg uses pass"), cl::init(true), cl::Hidden)
static cl::opt< bool > EnableLibCallSimplify("amdgpu-simplify-libcall", cl::desc("Enable amdgpu library simplifications"), cl::init(true), cl::Hidden)
static MachineSchedRegistry GCNMaxILPSchedRegistry("gcn-max-ilp", "Run GCN scheduler to maximize ilp", createGCNMaxILPMachineScheduler)
static cl::opt< bool > InternalizeSymbols("amdgpu-internalize-symbols", cl::desc("Enable elimination of non-kernel functions and unused globals"), cl::init(false), cl::Hidden)
static cl::opt< bool > EnableAMDGPUAttributor("amdgpu-attributor-enable", cl::desc("Enable AMDGPUAttributorPass"), cl::init(true), cl::Hidden)
static LLVM_READNONE StringRef getGPUOrDefault(const Triple &TT, StringRef GPU)
Expected< AMDGPUAttributorOptions > parseAMDGPUAttributorPassOptions(StringRef Params)
static cl::opt< bool > EnableAMDGPUAliasAnalysis("enable-amdgpu-aa", cl::Hidden, cl::desc("Enable AMDGPU Alias Analysis"), cl::init(true))
static Expected< ScanOptions > parseAMDGPUAtomicOptimizerStrategy(StringRef Params)
static ScheduleDAGInstrs * createMinRegScheduler(MachineSchedContext *C)
static cl::opt< bool > EnableHipStdPar("amdgpu-enable-hipstdpar", cl::desc("Enable HIP Standard Parallelism Offload support"), cl::init(false), cl::Hidden)
static cl::opt< bool > EnableInsertDelayAlu("amdgpu-enable-delay-alu", cl::desc("Enable s_delay_alu insertion"), cl::init(true), cl::Hidden)
static ScheduleDAGInstrs * createIterativeGCNMaxOccupancyMachineScheduler(MachineSchedContext *C)
static cl::opt< bool > EnableLoadStoreVectorizer("amdgpu-load-store-vectorizer", cl::desc("Enable load store vectorizer"), cl::init(true), cl::Hidden)
static bool mustPreserveGV(const GlobalValue &GV)
Predicate for Internalize pass.
static cl::opt< bool > EnableLoopPrefetch("amdgpu-loop-prefetch", cl::desc("Enable loop data prefetch on AMDGPU"), cl::Hidden, cl::init(false))
static cl::opt< bool > RemoveIncompatibleFunctions("amdgpu-enable-remove-incompatible-functions", cl::Hidden, cl::desc("Enable removal of functions when they" "use features not supported by the target GPU"), cl::init(true))
static cl::opt< bool > EnableScalarIRPasses("amdgpu-scalar-ir-passes", cl::desc("Enable scalar IR passes"), cl::init(true), cl::Hidden)
static cl::opt< bool > EnableRegReassign("amdgpu-reassign-regs", cl::desc("Enable register reassign optimizations on gfx10+"), cl::init(true), cl::Hidden)
static cl::opt< bool > OptVGPRLiveRange("amdgpu-opt-vgpr-liverange", cl::desc("Enable VGPR liverange optimizations for if-else structure"), cl::init(true), cl::Hidden)
static ScheduleDAGInstrs * createSIMachineScheduler(MachineSchedContext *C)
static cl::opt< bool > EnablePreRAOptimizations("amdgpu-enable-pre-ra-optimizations", cl::desc("Enable Pre-RA optimizations pass"), cl::init(true), cl::Hidden)
static cl::opt< ScanOptions > AMDGPUAtomicOptimizerStrategy("amdgpu-atomic-optimizer-strategy", cl::desc("Select DPP or Iterative strategy for scan"), cl::init(ScanOptions::Iterative), cl::values(clEnumValN(ScanOptions::DPP, "DPP", "Use DPP operations for scan"), clEnumValN(ScanOptions::Iterative, "Iterative", "Use Iterative approach for scan"), clEnumValN(ScanOptions::None, "None", "Disable atomic optimizer")))
static cl::opt< bool > EnableVOPD("amdgpu-enable-vopd", cl::desc("Enable VOPD, dual issue of VALU in wave32"), cl::init(true), cl::Hidden)
static ScheduleDAGInstrs * createGCNMaxOccupancyMachineScheduler(MachineSchedContext *C)
static cl::opt< bool > EnableLowerExecSync("amdgpu-enable-lower-exec-sync", cl::desc("Enable lowering of execution synchronization."), cl::init(true), cl::Hidden)
static MachineSchedRegistry GCNILPSchedRegistry("gcn-iterative-ilp", "Run GCN iterative scheduler for ILP scheduling (experimental)", createIterativeILPMachineScheduler)
static cl::opt< bool > ScalarizeGlobal("amdgpu-scalarize-global-loads", cl::desc("Enable global load scalarization"), cl::init(true), cl::Hidden)
static const char RegAllocOptNotSupportedMessage[]
static MachineSchedRegistry GCNMaxOccupancySchedRegistry("gcn-max-occupancy", "Run GCN scheduler to maximize occupancy", createGCNMaxOccupancyMachineScheduler)
The AMDGPU TargetMachine interface definition for hw codegen targets.
This file declares the AMDGPU-specific subclass of TargetLoweringObjectFile.
This file a TargetTransformInfoImplBase conforming object specific to the AMDGPU target machine.
Provides passes to inlining "always_inline" functions.
#define X(NUM, ENUM, NAME)
Definition ELF.h:856
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
This header provides classes for managing passes over SCCs of the call graph.
Provides analysis for continuously CSEing during GISel passes.
Interfaces for producing common pass manager configurations.
#define clEnumValN(ENUMVAL, FLAGNAME, DESC)
#define LLVM_READNONE
Definition Compiler.h:323
#define LLVM_ABI
Definition Compiler.h:215
#define LLVM_EXTERNAL_VISIBILITY
Definition Compiler.h:132
Analysis that tracks defined/used subregister lanes across COPY instructions and instructions that ge...
This file provides the interface for a simple, fast CSE pass.
This file defines the class GCNIterativeScheduler, which uses an iterative approach to find a best sc...
This file provides the interface for LLVM's Global Value Numbering pass which eliminates fully redund...
#define _
AcceleratorCodeSelection - Identify all functions reachable from a kernel, removing those that are un...
This file declares the IRTranslator pass.
Module.h This file contains the declarations for the Module class.
This header defines various interfaces for pass management in LLVM.
#define RegName(no)
This file provides the interface for LLVM's Loop Data Prefetching Pass.
This header provides classes for managing a pipeline of passes over loops in LLVM IR.
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
#define T
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t IntrinsicInst * II
#define P(N)
CGSCCAnalysisManager CGAM
LoopAnalysisManager LAM
FunctionAnalysisManager FAM
ModuleAnalysisManager MAM
PassInstrumentationCallbacks PIC
PassBuilder PB(Machine, PassOpts->PTO, std::nullopt, &PIC)
static bool isLTOPreLink(ThinOrFullLTOPhase Phase)
The AMDGPU TargetMachine interface definition for hw codegen targets.
This file describes the interface of the MachineFunctionPass responsible for assigning the generic vi...
const GCNTargetMachine & getTM(const GCNSubtarget *STI)
SI Machine Scheduler interface.
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static FunctionPass * useDefaultRegisterAllocator()
-regalloc=... command line option.
static cl::opt< cl::boolOrDefault > EnableGlobalISelOption("global-isel", cl::Hidden, cl::desc("Enable the \"global\" instruction selector"))
Target-Independent Code Generator Pass Configuration Options pass.
LLVM IR instance of the generic uniformity analysis.
static std::unique_ptr< TargetLoweringObjectFile > createTLOF()
A manager for alias analyses.
void registerFunctionAnalysis()
Register a specific AA result.
void addAAResult(AAResultT &AAResult)
Register a specific AA result.
Legacy wrapper pass to provide the AMDGPUAAResult object.
Analysis pass providing a never-invalidated alias analysis result.
Lower llvm.global_ctors and llvm.global_dtors to special kernels.
AMDGPUTargetMachine & getAMDGPUTargetMachine() const
std::unique_ptr< CSEConfigBase > getCSEConfig() const override
Returns the CSEConfig object to use for the current optimization level.
bool isPassEnabled(const cl::opt< bool > &Opt, CodeGenOptLevel Level=CodeGenOptLevel::Default) const
Check if a pass is enabled given Opt option.
bool addPreISel() override
Methods with trivial inline returns are convenient points in the common codegen pass pipeline where t...
bool addInstSelector() override
addInstSelector - This method should install an instruction selector pass, which converts from LLVM c...
bool addGCPasses() override
addGCPasses - Add late codegen passes that analyze code for garbage collection.
AMDGPUPassConfig(TargetMachine &TM, PassManagerBase &PM)
void addIRPasses() override
Add common target configurable passes that perform LLVM IR to IR transforms following machine indepen...
void addCodeGenPrepare() override
Add pass to prepare the LLVM IR for code generation.
Splits the module M into N linkable partitions.
std::unique_ptr< TargetLoweringObjectFile > TLOF
unsigned getAddressSpaceForPseudoSourceKind(unsigned Kind) const override
getAddressSpaceForPseudoSourceKind - Given the kind of memory (e.g.
const TargetSubtargetInfo * getSubtargetImpl() const
void registerDefaultAliasAnalyses(AAManager &) override
Allow the target to register alias analyses with the AAManager for use with the new pass manager.
std::pair< const Value *, unsigned > getPredicatedAddrSpace(const Value *V) const override
If the specified predicate checks whether a generic pointer falls within a specified address space,...
StringRef getFeatureString(const Function &F) const
ScheduleDAGInstrs * createMachineScheduler(MachineSchedContext *C) const override
Create an instance of ScheduleDAGInstrs to be run within the standard MachineScheduler pass for this ...
AMDGPUTargetMachine(const Target &T, const Triple &TT, StringRef CPU, StringRef FS, const TargetOptions &Options, std::optional< Reloc::Model > RM, std::optional< CodeModel::Model > CM, CodeGenOptLevel OL)
bool isNoopAddrSpaceCast(unsigned SrcAS, unsigned DestAS) const override
Returns true if a cast between SrcAS and DestAS is a noop.
void registerPassBuilderCallbacks(PassBuilder &PB) override
Allow the target to modify the pass pipeline.
StringRef getGPUName(const Function &F) const
unsigned getAssumedAddrSpace(const Value *V) const override
If the specified generic pointer could be assumed as a pointer to a specific address space,...
bool splitModule(Module &M, unsigned NumParts, function_ref< void(std::unique_ptr< Module > MPart)> ModuleCallback) override
Entry point for module splitting.
Inlines functions marked as "always_inline".
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:105
LLVM_ABI StringRef getValueAsString() const
Return the attribute's value as a string.
bool isValid() const
Return true if the attribute is any kind of attribute.
Definition Attributes.h:261
This class provides access to building LLVM's passes.
CodeGenTargetMachineImpl(const Target &T, StringRef DataLayoutString, const Triple &TT, StringRef CPU, StringRef FS, const TargetOptions &Options, Reloc::Model RM, CodeModel::Model CM, CodeGenOptLevel OL)
LLVM_ABI void removeDeadConstantUsers() const
If there are any dead constant users dangling off of this constant, remove them.
Diagnostic information for unsupported feature in backend.
Lightweight error class with error context and mandatory checking.
Definition Error.h:159
static ErrorSuccess success()
Create a success value.
Definition Error.h:336
Tagged union holding either a T or a Error.
Definition Error.h:485
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
LowerIntrinsics - This pass rewrites calls to the llvm.gcread or llvm.gcwrite intrinsics,...
Definition GCMetadata.h:229
const SIRegisterInfo * getRegisterInfo() const override
TargetTransformInfo getTargetTransformInfo(const Function &F) const override
Get a TargetTransformInfo implementation for the target.
ScheduleDAGInstrs * createPostMachineScheduler(MachineSchedContext *C) const override
Similar to createMachineScheduler but used when postRA machine scheduling is enabled.
static AMDGPU::TargetIDSetting getTargetIDSettingFromModuleFlag(const Module &M, StringRef FlagName)
Get xnack/sramecc setting from module flag or cl::opt (for testing).
ScheduleDAGInstrs * createMachineScheduler(MachineSchedContext *C) const override
Create an instance of ScheduleDAGInstrs to be run within the standard MachineScheduler pass for this ...
void registerMachineRegisterInfoCallback(MachineFunction &MF) const override
bool parseMachineFunctionInfo(const yaml::MachineFunctionInfo &, PerFunctionMIParsingState &PFS, SMDiagnostic &Error, SMRange &SourceRange) const override
Parse out the target's MachineFunctionInfo from the YAML reprsentation.
yaml::MachineFunctionInfo * convertFuncInfoToYAML(const MachineFunction &MF) const override
Allocate and initialize an instance of the YAML representation of the MachineFunctionInfo.
yaml::MachineFunctionInfo * createDefaultFuncInfoYAML() const override
Allocate and return a default initialized instance of the YAML representation for the MachineFunction...
Error buildCodeGenPipeline(ModulePassManager &MPM, ModuleAnalysisManager &MAM, raw_pwrite_stream &Out, raw_pwrite_stream *DwoOut, CodeGenFileType FileType, const CGPassBuilderOption &Opts, MCContext &Ctx, PassInstrumentationCallbacks *PIC) override
TargetPassConfig * createPassConfig(PassManagerBase &PM) override
Create a pass configuration object to be used by addPassToEmitX methods for generating a pipeline of ...
GCNTargetMachine(const Target &T, const Triple &TT, StringRef CPU, StringRef FS, const TargetOptions &Options, std::optional< Reloc::Model > RM, std::optional< CodeModel::Model > CM, CodeGenOptLevel OL, bool JIT)
MachineFunctionInfo * createMachineFunctionInfo(BumpPtrAllocator &Allocator, const Function &F, const TargetSubtargetInfo *STI) const override
Create the target's instance of MachineFunctionInfo.
The core GVN pass object.
Definition GVN.h:123
Pass to remove unused function declarations.
Definition GlobalDCE.h:38
This pass is responsible for selecting generic machine instructions to target-specific instructions.
A pass that internalizes all functions and variables other than those that must be preserved accordin...
Definition Internalize.h:37
Converts loops into loop-closed SSA form.
Definition LCSSA.h:38
Performs Loop Invariant Code Motion Pass.
Definition LICM.h:66
static void setUseExtended(bool Enable)
This pass implements the localization mechanism described at the top of this file.
Definition Localizer.h:40
An optimization pass inserting data prefetches in loops.
Context object for machine code objects.
Definition MCContext.h:83
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
void addDelegate(Delegate *delegate)
const MachineFunction & getMF() const
MachineSchedRegistry provides a selection of available machine instruction schedulers.
This interface provides simple read-only access to a block of memory, and provides simple methods for...
virtual StringRef getBufferIdentifier() const
Return an identifier for this buffer, typically the filename it was read from.
const char * getBufferStart() const
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:67
This class provides access to building LLVM's passes.
This class manages callbacks registration, as well as provides a way for PassInstrumentation to pass ...
LLVM_ATTRIBUTE_MINSIZE std::enable_if_t<!std::is_same_v< PassT, PassManager > > addPass(PassT &&Pass)
PreservedAnalyses run(IRUnitT &IR, AnalysisManagerT &AM, ExtraArgTs... ExtraArgs)
Run all of the passes in this manager over the given unit of IR.
PassRegistry - This class manages the registration and intitialization of the pass subsystem as appli...
static LLVM_ABI PassRegistry * getPassRegistry()
getPassRegistry - Access the global registry object, which is automatically initialized at applicatio...
Pass interface - Implemented by all 'passes'.
Definition Pass.h:99
RegisterPassParser class - Handle the addition of new machine passes.
RegisterRegAllocBase class - Track the registration of register allocators.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
bool initializeBaseYamlFields(const yaml::SIMachineFunctionInfo &YamlMFI, const MachineFunction &MF, PerFunctionMIParsingState &PFS, SMDiagnostic &Error, SMRange &SourceRange)
void setFlag(Register Reg, uint8_t Flag)
bool checkFlag(Register Reg, uint8_t Flag) const
Instances of this class encapsulate one diagnostic report, allowing printing to a raw_ostream as a ca...
Definition SourceMgr.h:308
Represents a location in source code.
Definition SMLoc.h:22
static SMLoc getFromPointer(const char *Ptr)
Definition SMLoc.h:35
Represents a range in source code.
Definition SMLoc.h:47
A ScheduleDAG for scheduling lists of MachineInstr.
ScheduleDAGMILive is an implementation of ScheduleDAGInstrs that schedules machine instructions while...
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
void addMutation(std::unique_ptr< ScheduleDAGMutation > Mutation)
Add a postprocessing step to the DAG builder.
const TargetInstrInfo * TII
Target instruction information.
const TargetRegisterInfo * TRI
Target processor register info.
Move instructions into successor blocks when possible.
Definition Sink.h:24
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
void append(StringRef RHS)
Append from a StringRef.
Definition SmallString.h:68
void push_back(const T &Elt)
unsigned getMainFileID() const
Definition SourceMgr.h:151
const MemoryBuffer * getMemoryBuffer(unsigned i) const
Definition SourceMgr.h:144
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
std::pair< StringRef, StringRef > split(char Separator) const
Split into two substrings around the first occurrence of a separator character.
Definition StringRef.h:736
constexpr bool empty() const
Check if the string is empty.
Definition StringRef.h:141
bool consume_front(char Prefix)
Returns true if this StringRef has the given prefix and removes that prefix.
Definition StringRef.h:661
A switch()-like statement whose cases are string literals.
StringSwitch & Cases(std::initializer_list< StringLiteral > CaseStrings, T Value)
Primary interface to the complete machine description for the target machine.
CodeGenOptLevel getOptLevel() const
Returns the optimization level: None, Less, Default, or Aggressive.
Triple TargetTriple
Triple string, CPU name, and target feature strings the TargetMachine instance is created with.
const Triple & getTargetTriple() const
const MCSubtargetInfo & getMCSubtargetInfo() const
StringRef getTargetFeatureString() const
StringRef getTargetCPU() const
std::unique_ptr< const MCSubtargetInfo > STI
TargetOptions Options
std::unique_ptr< const MCRegisterInfo > MRI
CodeGenOptLevel OptLevel
void setEnableDefaultMachineVerifier(bool Enable)
Target-Independent Code Generator Pass Configuration Options.
virtual void addCodeGenPrepare()
Add pass to prepare the LLVM IR for code generation.
virtual bool addILPOpts()
Add passes that optimize instruction level parallelism for out-of-order targets.
virtual void addPostRegAlloc()
This method may be implemented by targets that want to run passes after register allocation pass pipe...
CodeGenOptLevel getOptLevel() const
virtual void addOptimizedRegAlloc()
addOptimizedRegAlloc - Add passes related to register allocation.
virtual void addIRPasses()
Add common target configurable passes that perform LLVM IR to IR transforms following machine indepen...
virtual void addFastRegAlloc()
addFastRegAlloc - Add the minimum set of target-independent passes that are required for fast registe...
virtual void addMachineSSAOptimization()
addMachineSSAOptimization - Add standard passes that optimize machine instructions in SSA form.
void disablePass(AnalysisID PassID)
Allow the target to disable a specific standard pass by default.
AnalysisID addPass(AnalysisID PassID)
Utilities for targets to add passes to the pass manager.
TargetPassConfig(TargetMachine &TM, PassManagerBase &PM)
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
TargetSubtargetInfo - Generic base class for all target subtargets.
This pass provides access to the codegen interfaces that are needed for IR-level transformations.
Target - Wrapper for Target specific information.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
LLVM Value Representation.
Definition Value.h:75
bool use_empty() const
Definition Value.h:346
int getNumOccurrences() const
An efficient, type-erasing, non-owning reference to a callable.
PassManagerBase - An abstract interface to allow code to add passes to a pass manager without having ...
An abstract base class for streams implementations that also support a pwrite operation.
Interfaces for registering analysis passes, producing common pass manager configurations,...
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ LOCAL_ADDRESS
Address space for local memory.
@ CONSTANT_ADDRESS
Address space for constant memory (VTX2).
@ FLAT_ADDRESS
Address space for flat memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
@ PRIVATE_ADDRESS
Address space for private memory.
constexpr StringLiteral BufferFlag("amdgpu.buffer.oob.mode")
constexpr StringLiteral TBufferFlag("amdgpu.tbuffer.oob.mode")
StringRef getSchedStrategy(const Function &F)
bool isFlatGlobalAddrSpace(unsigned AS)
LLVM_READNONE constexpr bool isModuleEntryFunctionCC(CallingConv::ID CC)
GPUKind
GPU kinds supported by the AMDGPU target.
LLVM_READNONE constexpr bool isEntryFunctionCC(CallingConv::ID CC)
LLVM_ABI Triple::SubArchType getSubArch(GPUKind AK)
LLVM_ABI StringRef getArchNameFromSubArch(Triple::SubArchType SubArch)
Returns the canonical GPU name for an AMDGPU subarch, e.g.
LLVM_ABI GPUKind parseArchAMDGCN(StringRef CPU)
LLVM_ABI Triple::SubArchType getMajorSubArch(Triple::SubArchType SubArch)
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
BinaryOp_match< LHS, RHS, Instruction::And, true > m_c_And(const LHS &L, const RHS &R)
Matches an And with LHS and RHS in either order.
bool match(Val *V, const Pattern &P)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
auto m_Value()
Match an arbitrary value and ignore it.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
template class LLVM_TEMPLATE_ABI opt< bool >
ValuesClass values(OptsTy... Options)
Helper to build a ValuesClass by forwarding a variable number of arguments as an initializer list to ...
initializer< Ty > init(const Ty &Val)
LocationClass< Ty > location(Ty &L)
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > dyn_extract_or_null(Y &&MD)
Extract a Value from Metadata, if any, allowing null.
Definition Metadata.h:709
This is an optimization pass for GlobalISel generic memory operations.
ScheduleDAGMILive * createSchedLive(MachineSchedContext *C)
Create the standard converging machine scheduler.
LLVM_ABI FunctionPass * createFlattenCFGPass()
ModulePass * createAMDGPUSwLowerLDSLegacyPass()
std::unique_ptr< ScheduleDAGMutation > createAMDGPUBarrierLatencyDAGMutation(MachineFunction *MF)
LLVM_ABI FunctionPass * createFastRegisterAllocator()
FastRegisterAllocation Pass - This pass register allocates as fast as possible.
LLVM_ABI char & EarlyMachineLICMID
This pass performs loop invariant code motion on machine instructions.
ImmutablePass * createAMDGPUAAWrapperPass()
LLVM_ABI char & PostRAHazardRecognizerID
PostRAHazardRecognizer - This pass runs the post-ra hazard recognizer.
std::function< bool(const TargetRegisterInfo &TRI, const MachineRegisterInfo &MRI, const Register Reg)> RegAllocFilterFunc
Filter function for register classes during regalloc.
FunctionPass * createAMDGPUSetWavePriorityPass()
LLVM_ABI Pass * createLCSSAPass()
Definition LCSSA.cpp:544
void initializeAMDGPUMarkLastScratchLoadLegacyPass(PassRegistry &)
void initializeAMDGPUInsertDelayAluLegacyPass(PassRegistry &)
void initializeSIOptimizeExecMaskingPreRALegacyPass(PassRegistry &)
char & GCNPreRAOptimizationsID
LLVM_ABI char & GCLoweringID
GCLowering Pass - Used by gc.root to perform its default lowering operations.
void initializeSIInsertHardClausesLegacyPass(PassRegistry &)
FunctionPass * createSIAnnotateControlFlowLegacyPass()
Create the annotation pass.
FunctionPass * createSIModeRegisterPass()
void initializeGCNPreRAOptimizationsLegacyPass(PassRegistry &)
void initializeSILowerWWMCopiesLegacyPass(PassRegistry &)
LLVM_ABI FunctionPass * createGreedyRegisterAllocator()
Greedy register allocation pass - This pass implements a global register allocator for optimized buil...
void initializeAMDGPUAAWrapperPassPass(PassRegistry &)
void initializeSIShrinkInstructionsLegacyPass(PassRegistry &)
ModulePass * createAMDGPULowerBufferFatPointersPass()
void initializeR600ClauseMergePassPass(PassRegistry &)
ModulePass * createAMDGPUCtorDtorLoweringLegacyPass()
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
ModuleToFunctionPassAdaptor createModuleToFunctionPassAdaptor(FunctionPassT &&Pass, bool EagerlyInvalidate=false)
A function to deduce a function pass type and wrap it in the templated adaptor.
void initializeGCNRewritePartialRegUsesLegacyPass(llvm::PassRegistry &)
void initializeAMDGPURewriteUndefForPHILegacyPass(PassRegistry &)
char & GCNRewritePartialRegUsesID
void initializeAMDGPUSwLowerLDSLegacyPass(PassRegistry &)
LLVM_ABI std::error_code inconvertibleErrorCode()
The value returned by this function can be returned from convertToErrorCode for Error values where no...
Definition Error.cpp:94
void initializeAMDGPULowerVGPREncodingLegacyPass(PassRegistry &)
char & AMDGPUWaitSGPRHazardsLegacyID
void initializeSILowerSGPRSpillsLegacyPass(PassRegistry &)
LLVM_ABI Pass * createLoadStoreVectorizerPass()
Create a legacy pass manager instance of the LoadStoreVectorizer pass.
std::unique_ptr< ScheduleDAGMutation > createIGroupLPDAGMutation(AMDGPU::SchedulingPhase Phase)
Phase specifes whether or not this is a reentry into the IGroupLPDAGMutation.
void initializeAMDGPUDAGToDAGISelLegacyPass(PassRegistry &)
FunctionPass * createAMDGPURegBankCombiner(bool IsOptNone)
LLVM_ABI FunctionPass * createNaryReassociatePass()
char & AMDGPUReserveWWMRegsLegacyID
void initializeAMDGPUWaitSGPRHazardsLegacyPass(PassRegistry &)
LLVM_ABI char & PatchableFunctionID
This pass implements the "patchable-function" attribute.
char & SIOptimizeExecMaskingLegacyID
LLVM_ABI char & PostRASchedulerID
PostRAScheduler - This pass performs post register allocation scheduling.
void initializeAMDGPUNextUseAnalysisLegacyPassPass(PassRegistry &)
void initializeR600ExpandSpecialInstrsPassPass(PassRegistry &)
void initializeR600PacketizerPass(PassRegistry &)
@ O1
Optimize quickly without destroying debuggability.
@ O0
Disable as many optimizations as possible.
std::unique_ptr< ScheduleDAGMutation > createVOPDPairingMutation()
ModulePass * createAMDGPUExportKernelRuntimeHandlesLegacyPass()
ModulePass * createAMDGPUAlwaysInlinePass(bool GlobalOpt=true)
void initializeAMDGPUAsmPrinterPass(PassRegistry &)
void initializeSIFoldOperandsLegacyPass(PassRegistry &)
char & SILoadStoreOptimizerLegacyID
PassManager< LazyCallGraph::SCC, CGSCCAnalysisManager, LazyCallGraph &, CGSCCUpdateResult & > CGSCCPassManager
The CGSCC pass manager.
LLVM_ABI std::unique_ptr< CSEConfigBase > getStandardCSEConfigForOpt(CodeGenOptLevel Level)
Definition CSEInfo.cpp:85
Target & getTheR600Target()
The target for R600 GPUs.
LLVM_ABI char & MachineSchedulerID
MachineScheduler - This pass schedules machine instructions.
LLVM_ABI Pass * createStructurizeCFGPass(bool SkipUniformRegions=false)
When SkipUniformRegions is true the structizer will not structurize regions that only contain uniform...
LLVM_ABI char & PostMachineSchedulerID
PostMachineScheduler - This pass schedules machine instructions postRA.
LLVM_ABI Pass * createLICMPass()
Definition LICM.cpp:392
char & SIFormMemoryClausesID
void initializeSILoadStoreOptimizerLegacyPass(PassRegistry &)
void initializeAMDGPULowerModuleLDSLegacyPass(PassRegistry &)
AnalysisManager< LazyCallGraph::SCC, LazyCallGraph & > CGSCCAnalysisManager
The CGSCC analysis manager.
void initializeAMDGPUCtorDtorLoweringLegacyPass(PassRegistry &)
LLVM_ABI char & EarlyIfConverterLegacyID
EarlyIfConverter - This pass performs if-conversion on SSA form by inserting cmov instructions.
AnalysisManager< Loop, LoopStandardAnalysisResults & > LoopAnalysisManager
The loop analysis manager.
FunctionPass * createAMDGPUUniformIntrinsicCombineLegacyPass()
void initializeAMDGPURegBankCombinerPass(PassRegistry &)
ThinOrFullLTOPhase
This enumerates the LLVM full LTO or ThinLTO optimization phases.
Definition Pass.h:77
@ FullLTOPostLink
Full LTO postlink (backend compile) phase.
Definition Pass.h:87
char & AMDGPUUnifyDivergentExitNodesID
void initializeAMDGPUPrepareAGPRAllocLegacyPass(PassRegistry &)
FunctionPass * createAMDGPUAtomicOptimizerPass(ScanOptions ScanStrategy)
FunctionPass * createAMDGPUPreloadKernArgPrologLegacyPass()
char & SIOptimizeVGPRLiveRangeLegacyID
LLVM_ABI char & ShadowStackGCLoweringID
ShadowStackGCLowering - Implements the custom lowering mechanism used by the shadow stack GC.
char & GCNNSAReassignID
void initializeAMDGPURewriteOutArgumentsPass(PassRegistry &)
static Reloc::Model getEffectiveRelocModel(std::optional< Reloc::Model > RM)
void initializeAMDGPUExternalAAWrapperPass(PassRegistry &)
char & SIFoldOperandsLegacyID
auto formatv(bool Validate, const char *Fmt, Ts &&...Vals)
void initializeAMDGPULowerKernelArgumentsPass(PassRegistry &)
void initializeSIModeRegisterLegacyPass(PassRegistry &)
CodeModel::Model getEffectiveCodeModel(std::optional< CodeModel::Model > CM, CodeModel::Model Default)
Helper method for getting the code model, returning Default if CM does not have a value.
void initializeAMDGPUPreloadKernelArgumentsLegacyPass(PassRegistry &)
LLVM_ABI ModulePass * createExpandVariadicsPass(ExpandVariadicsMode)
char & SILateBranchLoweringPassID
FunctionToLoopPassAdaptor createFunctionToLoopPassAdaptor(LoopPassT &&Pass, bool UseMemorySSA=false)
A function to deduce a loop pass type and wrap it in the templated adaptor.
LLVM_ABI char & BranchRelaxationPassID
BranchRelaxation - This pass replaces branches that need to jump further than is supported by a branc...
LLVM_ABI FunctionPass * createSinkingPass()
Definition Sink.cpp:273
CGSCCToFunctionPassAdaptor createCGSCCToFunctionPassAdaptor(FunctionPassT &&Pass, bool EagerlyInvalidate=false, bool NoRerun=false)
A function to deduce a function pass type and wrap it in the templated adaptor.
void initializeSIMemoryLegalizerLegacyPass(PassRegistry &)
ModulePass * createAMDGPULowerIntrinsicsLegacyPass()
void initializeR600MachineCFGStructurizerPass(PassRegistry &)
CodeGenFileType
These enums are meant to be passed into addPassesToEmitFile to indicate what type of file to emit,...
Definition CodeGen.h:178
char & GCNDPPCombineLegacyID
PassManager< Module > ModulePassManager
Convenience typedef for a pass manager over modules.
LLVM_ABI std::unique_ptr< ScheduleDAGMutation > createStoreClusterDAGMutation(const TargetInstrInfo *TII, const TargetRegisterInfo *TRI, bool ReorderWhileClustering=false)
If ReorderWhileClustering is set to true, no attempt will be made to reduce reordering due to store c...
LLVM_ABI FunctionPass * createLoopDataPrefetchPass()
FunctionPass * createAMDGPULowerKernelArgumentsPass()
char & AMDGPUInsertDelayAluID
std::unique_ptr< ScheduleDAGMutation > createAMDGPUMacroFusionDAGMutation()
Note that you have to add: DAG.addMutation(createAMDGPUMacroFusionDAGMutation()); to AMDGPUTargetMach...
LLVM_ABI char & StackMapLivenessID
StackMapLiveness - This pass analyses the register live-out set of stackmap/patchpoint intrinsics and...
void initializeGCNPreRALongBranchRegLegacyPass(PassRegistry &)
char & SILowerWWMCopiesLegacyID
LLVM_ABI FunctionPass * createUnifyLoopExitsPass()
char & SIOptimizeExecMaskingPreRAID
LLVM_ABI FunctionPass * createFixIrreduciblePass()
void initializeR600EmitClauseMarkersPass(PassRegistry &)
LLVM_ABI char & FuncletLayoutID
This pass lays out funclets contiguously.
LLVM_ABI char & DetectDeadLanesID
This pass adds dead/undef flags after analyzing subregister lanes.
void initializeAMDGPULowerExecSyncLegacyPass(PassRegistry &)
void initializeAMDGPUPostLegalizerCombinerPass(PassRegistry &)
ScheduleDAGInstrs * createGCNNoopPostMachineScheduler(MachineSchedContext *C)
void initializeAMDGPUExportKernelRuntimeHandlesLegacyPass(PassRegistry &)
CodeGenOptLevel
Code generation optimization level.
Definition CodeGen.h:149
void initializeSIInsertWaitcntsLegacyPass(PassRegistry &)
ModulePass * createAMDGPUPreloadKernelArgumentsLegacyPass(const TargetMachine *)
ModulePass * createAMDGPUPrintfRuntimeBinding()
LLVM_ABI char & StackSlotColoringID
StackSlotColoring - This pass performs stack slot coloring.
LLVM_ABI Pass * createAlwaysInlinerLegacyPass(bool InsertLifetime=true)
Create a legacy pass manager instance of a pass to inline and remove functions marked as "always_inli...
void initializeR600ControlFlowFinalizerPass(PassRegistry &)
void initializeAMDGPUImageIntrinsicOptimizerPass(PassRegistry &)
void initializeSILateBranchLoweringLegacyPass(PassRegistry &)
void initializeSILowerControlFlowLegacyPass(PassRegistry &)
void initializeSIFormMemoryClausesLegacyPass(PassRegistry &)
char & SIPreAllocateWWMRegsLegacyID
Error make_error(ArgTs &&... Args)
Make a Error instance representing failure using the given error info type.
Definition Error.h:340
ModulePass * createAMDGPULowerModuleLDSLegacyPass(const AMDGPUTargetMachine *TM=nullptr)
void initializeAMDGPUPreLegalizerCombinerPass(PassRegistry &)
FunctionPass * createAMDGPUPromoteAlloca()
LLVM_ABI FunctionPass * createSeparateConstOffsetFromGEPPass(bool LowerGEP=false)
void initializeAMDGPUReserveWWMRegsLegacyPass(PassRegistry &)
char & SIPreEmitPeepholeID
char & SIPostRABundlerLegacyID
ModulePass * createAMDGPURemoveIncompatibleFunctionsPass(const TargetMachine *)
void initializeGCNRegPressurePrinterPass(PassRegistry &)
void initializeSILowerI1CopiesLegacyPass(PassRegistry &)
LLVM_ABI ImmutablePass * createExternalAAWrapperPass(std::function< void(Pass &, Function &, AAResults &)> Callback, bool RunEarly=false)
A wrapper pass around a callback which can be used to populate the AAResults in the AAResultsWrapperP...
char & SILowerSGPRSpillsLegacyID
LLVM_ABI FunctionPass * createBasicRegisterAllocator()
BasicRegisterAllocation Pass - This pass implements a degenerate global register allocator using the ...
LLVM_ABI void initializeGlobalISel(PassRegistry &)
Initialize all passes linked into the GlobalISel library.
char & SILowerControlFlowLegacyID
ModulePass * createR600OpenCLImageTypeLoweringPass()
FunctionPass * createAMDGPUCodeGenPreparePass()
void initializeSIAnnotateControlFlowLegacyPass(PassRegistry &)
FunctionPass * createAMDGPUISelDag(TargetMachine &TM, CodeGenOptLevel OptLevel)
This pass converts a legalized DAG into a AMDGPU-specific.
void initializeGCNCreateVOPDLegacyPass(PassRegistry &)
void initializeAMDGPUUniformIntrinsicCombineLegacyPass(PassRegistry &)
ScheduleDAGInstrs * createGCNCoExecMachineScheduler(MachineSchedContext *C)
void initializeSIPreAllocateWWMRegsLegacyPass(PassRegistry &)
void initializeSIFixVGPRCopiesLegacyPass(PassRegistry &)
Target & getTheGCNTarget()
The target for GCN GPUs.
void initializeSIFixSGPRCopiesLegacyPass(PassRegistry &)
void initializeAMDGPUAtomicOptimizerPass(PassRegistry &)
void initializeAMDGPULowerIntrinsicsLegacyPass(PassRegistry &)
LLVM_ABI FunctionPass * createGVNPass()
Definition GVN.cpp:4065
void initializeAMDGPURewriteAGPRCopyMFMALegacyPass(PassRegistry &)
void initializeAMDGPUNextUseAnalysisPrinterLegacyPassPass(PassRegistry &)
void initializeSIPostRABundlerLegacyPass(PassRegistry &)
FunctionPass * createAMDGPURegBankSelectPass()
FunctionPass * createAMDGPURegBankLegalizePass()
LLVM_ABI char & MachineCSELegacyID
MachineCSE - This pass performs global CSE on machine instructions.
char & SIWholeQuadModeID
LLVM_ABI std::unique_ptr< ScheduleDAGMutation > createLoadClusterDAGMutation(const TargetInstrInfo *TII, const TargetRegisterInfo *TRI, bool ReorderWhileClustering=false)
If ReorderWhileClustering is set to true, no attempt will be made to reduce reordering due to store c...
PassManager< Function > FunctionPassManager
Convenience typedef for a pass manager over functions.
LLVM_ABI char & LiveVariablesID
LiveVariables pass - This pass computes the set of blocks in which each variable is life and sets mac...
void initializeAMDGPUCodeGenPreparePass(PassRegistry &)
FunctionPass * createAMDGPURewriteUndefForPHILegacyPass()
void initializeSIOptimizeExecMaskingLegacyPass(PassRegistry &)
void call_once(once_flag &flag, Function &&F, Args &&... ArgList)
Execute the function specified as a parameter once.
Definition Threading.h:86
FunctionPass * createSILowerI1CopiesLegacyPass()
FunctionPass * createAMDGPUPostLegalizeCombiner(bool IsOptNone)
void initializeAMDGPULowerKernelAttributesPass(PassRegistry &)
char & SIInsertHardClausesID
char & SIFixSGPRCopiesLegacyID
void initializeGCNDPPCombineLegacyPass(PassRegistry &)
char & GCNCreateVOPDID
char & SIPeepholeSDWALegacyID
LLVM_ABI char & VirtRegRewriterID
VirtRegRewriter pass.
char & SIFixVGPRCopiesID
void initializeGCNNSAReassignLegacyPass(PassRegistry &)
LLVM_ABI FunctionPass * createLowerSwitchPass()
void initializeAMDGPUPreloadKernArgPrologLegacyPass(PassRegistry &)
Target & getTheGCNLegacyTarget()
The target for GCN GPUs, registered under the legacy "amdgcn" architecture name for use with -march.
LLVM_ABI FunctionPass * createVirtRegRewriter(bool ClearVirtRegs=true)
void initializeR600VectorRegMergerPass(PassRegistry &)
char & AMDGPURewriteAGPRCopyMFMALegacyID
ModulePass * createAMDGPULowerExecSyncLegacyPass()
char & AMDGPULowerVGPREncodingLegacyID
FunctionPass * createAMDGPUGlobalISelDivergenceLoweringPass()
FunctionPass * createSIMemoryLegalizerPass()
void initializeAMDGPULateCodeGenPrepareLegacyPass(PassRegistry &)
void initializeSIOptimizeVGPRLiveRangeLegacyPass(PassRegistry &)
void initializeSIPeepholeSDWALegacyPass(PassRegistry &)
void initializeAMDGPURegBankLegalizePass(PassRegistry &)
LLVM_ABI char & TwoAddressInstructionPassID
TwoAddressInstruction - This pass reduces two-address instructions to use two operands.
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
FunctionPass * createAMDGPUPreLegalizeCombiner(bool IsOptNone)
void initializeAMDGPURegBankSelectPass(PassRegistry &)
FunctionPass * createAMDGPULateCodeGenPrepareLegacyPass()
LLVM_ABI FunctionPass * createAtomicExpandLegacyPass()
AtomicExpandPass - At IR level this pass replace atomic instructions with __atomic_* library calls,...
void initializeAMDGPUGlobalISelDivergenceLoweringLegacyPass(PassRegistry &)
MCRegisterInfo * createGCNMCRegisterInfo(AMDGPUDwarfFlavour DwarfFlavour)
LLVM_ABI FunctionPass * createStraightLineStrengthReducePass()
BumpPtrAllocatorImpl<> BumpPtrAllocator
The standard BumpPtrAllocator which just uses the default template parameters.
Definition Allocator.h:390
FunctionPass * createAMDGPUImageIntrinsicOptimizerPass(const TargetMachine *)
void initializeAMDGPULowerBufferFatPointersPass(PassRegistry &)
void initializeAMDGPUUnifyDivergentExitNodesLegacyPass(PassRegistry &)
FunctionPass * createSIInsertWaitcntsPass()
FunctionPass * createAMDGPUAnnotateUniformValuesLegacy()
LLVM_ABI FunctionPass * createEarlyCSEPass(bool UseMemorySSA=false)
void initializeSIWholeQuadModeLegacyPass(PassRegistry &)
LLVM_ABI char & PHIEliminationID
PHIElimination - This pass eliminates machine instruction PHI nodes by inserting copy instructions.
LLVM_ABI llvm::cl::opt< bool > NoKernelInfoEndLTO
LLVM_ABI bool parseNamedRegisterReference(PerFunctionMIParsingState &PFS, Register &Reg, StringRef Src, SMDiagnostic &Error)
void initializeAMDGPUResourceUsageAnalysisWrapperPassPass(PassRegistry &)
FunctionPass * createSIShrinkInstructionsLegacyPass()
char & AMDGPUPrepareAGPRAllocLegacyID
char & AMDGPUMarkLastScratchLoadID
LLVM_ABI char & RenameIndependentSubregsID
This pass detects subregister lanes in a virtual register that are used independently of other lanes ...
void initializeAMDGPUAnnotateUniformValuesLegacyPass(PassRegistry &)
std::unique_ptr< ScheduleDAGMutation > createAMDGPUExportClusteringDAGMutation()
void initializeAMDGPUPrintfRuntimeBindingPass(PassRegistry &)
void initializeAMDGPUPromoteAllocaPass(PassRegistry &)
void initializeAMDGPURemoveIncompatibleFunctionsLegacyPass(PassRegistry &)
std::unique_ptr< ScheduleDAGMutation > createAMDGPUHazardLatencyDAGMutation(MachineFunction *MF)
void initializeAMDGPUAlwaysInlinePass(PassRegistry &)
LLVM_ABI char & DeadMachineInstructionElimID
DeadMachineInstructionElim - This pass removes dead machine instructions.
void initializeSIPreEmitPeepholeLegacyPass(PassRegistry &)
AnalysisManager< Module > ModuleAnalysisManager
Convenience typedef for the Module analysis manager.
Definition MIRParser.h:39
char & AMDGPUPerfHintAnalysisLegacyID
char & GCNPreRALongBranchRegID
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
LLVM_ABI CGPassBuilderOption getCGPassBuilderOption()
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
void initializeAMDGPUPromoteKernelArgumentsPass(PassRegistry &)
#define N
static ArgDescriptor createStack(unsigned Offset, unsigned Mask=~0u)
static ArgDescriptor createArg(const ArgDescriptor &Arg, unsigned Mask)
static ArgDescriptor createRegister(Register Reg, unsigned Mask=~0u)
DenormalModeKind Input
Denormal treatment kind for floating point instruction inputs in the default floating-point environme...
@ PreserveSign
The sign of a flushed-to-zero number is preserved in the sign of 0.
@ IEEE
IEEE-754 denormal numbers preserved.
DenormalModeKind Output
Denormal flushing mode for floating point instruction results in the default floating point environme...
A simple and fast domtree-based CSE pass.
Definition EarlyCSE.h:31
MachineFunctionInfo - This class can be derived from and used by targets to hold private target-speci...
static FuncInfoTy * create(BumpPtrAllocator &Allocator, const Function &F, const SubtargetTy *STI)
Factory function: default behavior is to call new using the supplied allocator.
MachineSchedContext provides enough context from the MachineScheduler pass for the target to instanti...
StringMap< VRegInfo * > VRegInfosNamed
Definition MIParser.h:180
DenseMap< Register, VRegInfo * > VRegInfos
Definition MIParser.h:179
RegisterTargetMachine - Helper template for registering a target machine implementation,...
bool DX10Clamp
Used by the vector ALU to force DX10-style treatment of NaNs: when set, clamp NaN to zero; otherwise,...
DenormalMode FP64FP16Denormals
If this is set, neither input or output denormals are flushed for both f64 and f16/v2f16 instructions...
bool IEEE
Floating point opcodes that support exception flag gathering quiet and propagate signaling NaN inputs...
DenormalMode FP32Denormals
If this is set, neither input or output denormals are flushed for most f32 instructions.
The llvm::once_flag structure.
Definition Threading.h:67
Targets should override this in a way that mirrors the implementation of llvm::MachineFunctionInfo.
SmallVector< StringValue > WWMReservedRegs
std::optional< SIArgumentInfo > ArgInfo
SmallVector< StringValue, 2 > SpillPhysVGPRS
A wrapper around std::string which contains a source range that's being set during parsing.