LLVM 24.0.0git
PassBuilderPipelines.cpp
Go to the documentation of this file.
1//===- Construction of pass pipelines -------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9///
10/// This file provides the implementation of the PassBuilder based on our
11/// static pass registry as well as related functionality. It also provides
12/// helpers to aid in analyzing, debugging, and testing passes and pass
13/// pipelines.
14///
15//===----------------------------------------------------------------------===//
16
17#include "llvm/ADT/Statistic.h"
28#include "llvm/IR/PassManager.h"
29#include "llvm/IR/Verifier.h"
30#include "llvm/Pass.h"
159
160using namespace llvm;
161
162namespace llvm {
163
165 "enable-ml-inliner", cl::init(InliningAdvisorMode::Default), cl::Hidden,
166 cl::desc("Enable ML policy for inliner. Currently trained for -Oz only"),
168 "Heuristics-based inliner version"),
170 "Use development mode (runtime-loadable model)"),
172 "Use release mode (AOT-compiled model)")));
173
174/// Flag to enable inline deferral during PGO.
175static cl::opt<bool>
176 EnablePGOInlineDeferral("enable-npm-pgo-inline-deferral", cl::init(true),
178 cl::desc("Enable inline deferral during PGO"));
179
180static cl::opt<bool> EnableModuleInliner("enable-module-inliner",
181 cl::init(false), cl::Hidden,
182 cl::desc("Enable module inliner"));
183
185 "mandatory-inlining-first", cl::init(false), cl::Hidden,
186 cl::desc("Perform mandatory inlinings module-wide, before performing "
187 "inlining"));
188
190 "eagerly-invalidate-analyses", cl::init(true), cl::Hidden,
191 cl::desc("Eagerly invalidate more analyses in default pipelines"));
192
194 "enable-merge-functions", cl::init(false), cl::Hidden,
195 cl::desc("Enable function merging as part of the optimization pipeline"));
196
198 "enable-post-pgo-loop-rotation", cl::init(true), cl::Hidden,
199 cl::desc("Run the loop rotation transformation after PGO instrumentation"));
200
201static cl::opt<bool>
202 TriggerCrash("opt-pipeline-trigger-crash", cl::init(false), cl::Hidden,
203 cl::desc("Trigger crash in optimization pipeline"));
204
206 "enable-global-analyses", cl::init(true), cl::Hidden,
207 cl::desc("Enable inter-procedural analyses"));
208
209static cl::opt<bool> RunPartialInlining("enable-partial-inlining",
210 cl::init(false), cl::Hidden,
211 cl::desc("Run Partial inlining pass"));
212
214 "extra-vectorizer-passes", cl::init(false), cl::Hidden,
215 cl::desc("Run cleanup optimization passes after vectorization"));
216
217static cl::opt<bool> RunNewGVN("enable-newgvn", cl::init(false), cl::Hidden,
218 cl::desc("Run the NewGVN pass"));
219
220static cl::opt<bool>
221 EnableLoopInterchange("enable-loopinterchange", cl::init(true), cl::Hidden,
222 cl::desc("Enable the LoopInterchange Pass"));
223
224static cl::opt<bool> EnableUnrollAndJam("enable-unroll-and-jam",
225 cl::init(false), cl::Hidden,
226 cl::desc("Enable Unroll And Jam Pass"));
227
228static cl::opt<bool> EnableLoopFlatten("enable-loop-flatten", cl::init(false),
230 cl::desc("Enable the LoopFlatten Pass"));
231
232static cl::opt<bool>
233 EnableInstrumentor("enable-instrumentor", cl::init(false), cl::Hidden,
234 cl::desc("Enable the Instrumentor Pass"));
235
236static cl::opt<bool>
237 EnableDFAJumpThreading("enable-dfa-jump-thread",
238 cl::desc("Enable DFA jump threading"),
239 cl::init(true), cl::Hidden);
240
241static cl::opt<bool>
242 EnableHotColdSplit("hot-cold-split",
243 cl::desc("Enable hot-cold splitting pass"));
244
245static cl::opt<bool>
246 DisablePreInliner("disable-preinline", cl::init(false), cl::Hidden,
247 cl::desc("Disable pre-instrumentation inliner"));
248
250 "preinline-threshold", cl::Hidden, cl::init(75),
251 cl::desc("Control the amount of inlining in pre-instrumentation inliner "
252 "(default = 75)"));
253
254static cl::opt<bool>
255 EnableGVNHoist("enable-gvn-hoist",
256 cl::desc("Enable the GVN hoisting pass (default = off)"));
257
258static cl::opt<bool>
259 EnableGVNSink("enable-gvn-sink",
260 cl::desc("Enable the GVN sinking pass (default = off)"));
261
263 "enable-jump-table-to-switch", cl::init(true),
264 cl::desc("Enable JumpTableToSwitch pass (default = true)"));
265
266// This option is used in simplifying testing SampleFDO optimizations for
267// profile loading.
268static cl::opt<bool>
269 EnableCHR("enable-chr", cl::init(true), cl::Hidden,
270 cl::desc("Enable control height reduction optimization (CHR)"));
271
273 "flattened-profile-used", cl::init(false), cl::Hidden,
274 cl::desc("Indicate the sample profile being used is flattened, i.e., "
275 "no inline hierarchy exists in the profile"));
276
277static cl::opt<bool>
278 EnableMatrix("enable-matrix", cl::init(false), cl::Hidden,
279 cl::desc("Enable lowering of the matrix intrinsics"));
280
282 "enable-mergeicmps", cl::init(true), cl::Hidden,
283 cl::desc("Enable MergeICmps pass in the optimization pipeline"));
284
286 "enable-constraint-elimination", cl::init(true), cl::Hidden,
287 cl::desc(
288 "Enable pass to eliminate conditions based on linear constraints"));
289
291 "attributor-enable", cl::Hidden, cl::init(AttributorRunOption::NONE),
292 cl::desc("Enable the attributor inter-procedural deduction pass"),
294 "enable all full attributor runs"),
296 "enable all attributor-light runs"),
298 "enable module-wide attributor runs"),
300 "enable module-wide attributor-light runs"),
302 "enable call graph SCC attributor runs"),
304 "enable call graph SCC attributor-light runs"),
305 clEnumValN(AttributorRunOption::NONE, "none",
306 "disable attributor runs")));
307
309 "enable-sampled-instrumentation", cl::init(false), cl::Hidden,
310 cl::desc("Enable profile instrumentation sampling (default = off)"));
312 "enable-loop-versioning-licm", cl::init(false), cl::Hidden,
313 cl::desc("Enable the experimental Loop Versioning LICM pass"));
314
316 "instrument-cold-function-only-path", cl::init(""),
317 cl::desc("File path for cold function only instrumentation(requires use "
318 "with --pgo-instrument-cold-function-only)"),
319 cl::Hidden);
320
321// TODO: There is a similar flag in WPD pass, we should consolidate them by
322// parsing the option only once in PassBuilder and share it across both places.
324 "enable-devirtualize-speculatively",
325 cl::desc("Enable speculative devirtualization optimization"),
326 cl::init(false));
327
330
332} // namespace llvm
333
351
352namespace llvm {
354} // namespace llvm
355
357 OptimizationLevel Level) {
358 for (auto &C : PeepholeEPCallbacks)
359 C(FPM, Level);
360}
363 for (auto &C : LateLoopOptimizationsEPCallbacks)
364 C(LPM, Level);
365}
367 OptimizationLevel Level) {
368 for (auto &C : LoopOptimizerEndEPCallbacks)
369 C(LPM, Level);
370}
373 for (auto &C : ScalarOptimizerLateEPCallbacks)
374 C(FPM, Level);
375}
377 OptimizationLevel Level) {
378 for (auto &C : CGSCCOptimizerLateEPCallbacks)
379 C(CGPM, Level);
380}
382 OptimizationLevel Level) {
383 for (auto &C : VectorizerStartEPCallbacks)
384 C(FPM, Level);
385}
387 OptimizationLevel Level) {
388 for (auto &C : VectorizerEndEPCallbacks)
389 C(FPM, Level);
390}
392 OptimizationLevel Level,
394 for (auto &C : OptimizerEarlyEPCallbacks)
395 C(MPM, Level, Phase);
396}
398 OptimizationLevel Level,
400 for (auto &C : OptimizerLastEPCallbacks)
401 C(MPM, Level, Phase);
402}
405 for (auto &C : FullLinkTimeOptimizationEarlyEPCallbacks)
406 C(MPM, Level);
407}
410 for (auto &C : FullLinkTimeOptimizationLastEPCallbacks)
411 C(MPM, Level);
412}
415 for (auto &C : ThinLinkTimeOptimizationEarlyEPCallbacks)
416 C(MPM, Level);
417}
420 for (auto &C : ThinLinkTimeOptimizationLastEPCallbacks)
421 C(MPM, Level);
422}
424 OptimizationLevel Level) {
425 for (auto &C : PipelineStartEPCallbacks)
426 C(MPM, Level);
427}
430 for (auto &C : PipelineEarlySimplificationEPCallbacks)
431 C(MPM, Level, Phase);
432}
433
434// Get IR stats with InstCount before/after the optimization pipeline
436 bool IsPreOptimization) {
437 if (AreStatisticsEnabled()) {
438 MPM.addPass(
441 FunctionPropertiesStatisticsPass(IsPreOptimization)));
442 }
443}
444
445// Helper to add AnnotationRemarksPass.
449
450// Helper to check if the current compilation phase is preparing for LTO
455
456// Helper to check if the current compilation phase is preparing for FullLTO
457[[maybe_unused]] static bool isFullLTOPreLink(ThinOrFullLTOPhase Phase) {
459}
460
461// Helper to check if the current compilation phase is preparing for ThinLTO
465
466// Helper to check if the current compilation phase is LTO backend
471
472// Helper to check if the current compilation phase is FullLTO backend
476
477// Helper to check if the current compilation phase is ThinLTO backend
481
482// Helper to wrap conditionally Coro passes.
484 // TODO: Skip passes according to Phase.
485 ModulePassManager CoroPM;
486 CoroPM.addPass(CoroEarlyPass());
487 CGSCCPassManager CGPM;
488 CGPM.addPass(CoroSplitPass());
489 CoroPM.addPass(createModuleToPostOrderCGSCCPassAdaptor(std::move(CGPM)));
490 CoroPM.addPass(CoroCleanupPass());
491 CoroPM.addPass(GlobalDCEPass());
492 return CoroConditionalWrapper(std::move(CoroPM));
493}
494
495// TODO: Investigate the cost/benefit of tail call elimination on debugging.
497PassBuilder::buildO1FunctionSimplificationPipeline(OptimizationLevel Level,
499
501
503 FPM.addPass(CountVisitsPass());
504
505 // Form SSA out of local memory accesses after breaking apart aggregates into
506 // scalars.
507 FPM.addPass(SROAPass(SROAOptions::ModifyCFG));
508
509 // Catch trivial redundancies
510 FPM.addPass(EarlyCSEPass(true /* Enable mem-ssa. */));
511
512 // Hoisting of scalars and load expressions.
513 FPM.addPass(
514 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
515 FPM.addPass(InstCombinePass());
516
517 FPM.addPass(LibCallsShrinkWrapPass());
518
519 invokePeepholeEPCallbacks(FPM, Level);
520
521 FPM.addPass(
522 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
523
524 // Form canonically associated expression trees, and simplify the trees using
525 // basic mathematical properties. For example, this will form (nearly)
526 // minimal multiplication trees.
527 FPM.addPass(ReassociatePass());
528
529 // Add the primary loop simplification pipeline.
530 // FIXME: Currently this is split into two loop pass pipelines because we run
531 // some function passes in between them. These can and should be removed
532 // and/or replaced by scheduling the loop pass equivalents in the correct
533 // positions. But those equivalent passes aren't powerful enough yet.
534 // Specifically, `SimplifyCFGPass` and `InstCombinePass` are currently still
535 // used. We have `LoopSimplifyCFGPass` which isn't yet powerful enough yet to
536 // fully replace `SimplifyCFGPass`, and the closest to the other we have is
537 // `LoopInstSimplify`.
538 LoopPassManager LPM1, LPM2;
539
540 // Simplify the loop body. We do this initially to clean up after other loop
541 // passes run, either when iterating on a loop or on inner loops with
542 // implications on the outer loop.
543 LPM1.addPass(LoopInstSimplifyPass());
544 LPM1.addPass(LoopSimplifyCFGPass());
545
546 // Try to remove as much code from the loop header as possible,
547 // to reduce amount of IR that will have to be duplicated. However,
548 // do not perform speculative hoisting the first time as LICM
549 // will destroy metadata that may not need to be destroyed if run
550 // after loop rotation.
551 // TODO: Investigate promotion cap for O1.
552 LPM1.addPass(LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
553 /*AllowSpeculation=*/false));
554
555 LPM1.addPass(
556 LoopRotatePass(/*EnableHeaderDuplication=*/true, isLTOPreLink(Phase)));
557 // TODO: Investigate promotion cap for O1.
558 LPM1.addPass(LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
559 /*AllowSpeculation=*/true));
560 LPM1.addPass(SimpleLoopUnswitchPass());
562 LPM1.addPass(LoopFlattenPass());
563
564 LPM2.addPass(LoopIdiomRecognizePass());
565 LPM2.addPass(IndVarSimplifyPass());
566
568
569 LPM2.addPass(LoopDeletionPass());
570
571 // Do not enable unrolling in PreLinkThinLTO phase during sample PGO
572 // because it changes IR to makes profile annotation in back compile
573 // inaccurate. The normal unroller doesn't pay attention to forced full unroll
574 // attributes so we need to make sure and allow the full unroll pass to pay
575 // attention to it.
576 if (!isThinLTOPreLink(Phase) || !PGOOpt ||
577 PGOOpt->Action != PGOOptions::SampleUse)
578 LPM2.addPass(LoopFullUnrollPass(static_cast<int>(Level),
579 /* OnlyWhenForced= */ !PTO.LoopUnrolling,
580 PTO.ForgetAllSCEVInLoopUnroll));
581
583
584 FPM.addPass(createFunctionToLoopPassAdaptor(std::move(LPM1),
585 /*UseMemorySSA=*/true));
586 FPM.addPass(
587 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
588 FPM.addPass(InstCombinePass());
589 // The loop passes in LPM2 (LoopFullUnrollPass) do not preserve MemorySSA.
590 // *All* loop passes must preserve it, in order to be able to use it.
591 FPM.addPass(createFunctionToLoopPassAdaptor(std::move(LPM2),
592 /*UseMemorySSA=*/false));
593
594 // Delete small array after loop unroll.
595 FPM.addPass(SROAPass(SROAOptions::ModifyCFG));
596
597 // Specially optimize memory movement as it doesn't look like dataflow in SSA.
598 FPM.addPass(MemCpyOptPass());
599
600 // Sparse conditional constant propagation.
601 // FIXME: It isn't clear why we do this *after* loop passes rather than
602 // before...
603 FPM.addPass(SCCPPass());
604
605 // Delete dead bit computations (instcombine runs after to fold away the dead
606 // computations, and then ADCE will run later to exploit any new DCE
607 // opportunities that creates).
608 FPM.addPass(BDCEPass());
609
610 // Run instcombine after redundancy and dead bit elimination to exploit
611 // opportunities opened up by them.
612 FPM.addPass(InstCombinePass());
613 invokePeepholeEPCallbacks(FPM, Level);
614
615 FPM.addPass(CoroElidePass());
616
618
619 // Finally, do an expensive DCE pass to catch all the dead code exposed by
620 // the simplifications and basic cleanup after all the simplifications.
621 // TODO: Investigate if this is too expensive.
622 FPM.addPass(ADCEPass());
623 FPM.addPass(
624 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
625 FPM.addPass(InstCombinePass());
626 invokePeepholeEPCallbacks(FPM, Level);
627
628 return FPM;
629}
630
634 assert(Level != OptimizationLevel::O0 && "Must request optimizations!");
635
636 // The O1 pipeline has a separate pipeline creation function to simplify
637 // construction readability.
638 if (Level == OptimizationLevel::O1)
639 return buildO1FunctionSimplificationPipeline(Level, Phase);
640
642
645
646 // Form SSA out of local memory accesses after breaking apart aggregates into
647 // scalars.
649
650 // Catch trivial redundancies
651 FPM.addPass(EarlyCSEPass(true /* Enable mem-ssa. */));
654
655 // Hoisting of scalars and load expressions.
656 if (EnableGVNHoist)
657 FPM.addPass(GVNHoistPass());
658
659 // Global value numbering based sinking.
660 if (EnableGVNSink) {
661 FPM.addPass(GVNSinkPass());
662 FPM.addPass(
663 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
664 }
665
666 // Speculative execution if the target has divergent branches; otherwise nop.
667 FPM.addPass(SpeculativeExecutionPass(/* OnlyIfDivergentTarget =*/true));
668
669 // Optimize based on known information about branches, and cleanup afterward.
672
673 // Jump table to switch conversion.
676
677 FPM.addPass(
678 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
682
683 invokePeepholeEPCallbacks(FPM, Level);
684
685 // For PGO use pipeline, try to optimize memory intrinsics such as memcpy
686 // using the size value profile. Don't perform this when optimizing for size.
687 if (PGOOpt && PGOOpt->Action == PGOOptions::IRUse)
689
690 FPM.addPass(TailCallElimPass(/*UpdateFunctionEntryCount=*/
691 isInstrumentedPGOUse()));
692 FPM.addPass(
693 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
694
695 // Form canonically associated expression trees, and simplify the trees using
696 // basic mathematical properties. For example, this will form (nearly)
697 // minimal multiplication trees.
699
702
703 // Add the primary loop simplification pipeline.
704 // FIXME: Currently this is split into two loop pass pipelines because we run
705 // some function passes in between them. These can and should be removed
706 // and/or replaced by scheduling the loop pass equivalents in the correct
707 // positions. But those equivalent passes aren't powerful enough yet.
708 // Specifically, `SimplifyCFGPass` and `InstCombinePass` are currently still
709 // used. We have `LoopSimplifyCFGPass` which isn't yet powerful enough yet to
710 // fully replace `SimplifyCFGPass`, and the closest to the other we have is
711 // `LoopInstSimplify`.
712 LoopPassManager LPM1, LPM2;
713
714 // Simplify the loop body. We do this initially to clean up after other loop
715 // passes run, either when iterating on a loop or on inner loops with
716 // implications on the outer loop.
717 LPM1.addPass(LoopInstSimplifyPass());
718 LPM1.addPass(LoopSimplifyCFGPass());
719
720 // Try to remove as much code from the loop header as possible,
721 // to reduce amount of IR that will have to be duplicated. However,
722 // do not perform speculative hoisting the first time as LICM
723 // will destroy metadata that may not need to be destroyed if run
724 // after loop rotation.
725 // TODO: Investigate promotion cap for O1.
726 LPM1.addPass(LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
727 /*AllowSpeculation=*/false));
728
729 LPM1.addPass(
730 LoopRotatePass(/*EnableHeaderDuplication=*/true, isLTOPreLink(Phase)));
731 // TODO: Investigate promotion cap for O1.
732 LPM1.addPass(LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
733 /*AllowSpeculation=*/true));
734 LPM1.addPass(
735 SimpleLoopUnswitchPass(/* NonTrivial */ Level == OptimizationLevel::O3));
737 LPM1.addPass(LoopFlattenPass());
738
739 LPM2.addPass(LoopIdiomRecognizePass());
740 LPM2.addPass(IndVarSimplifyPass());
741
742 {
744 ExtraPasses.addPass(SimpleLoopUnswitchPass(/* NonTrivial */ Level ==
746 LPM2.addPass(std::move(ExtraPasses));
747 }
748
750
751 LPM2.addPass(LoopDeletionPass());
752
753 // Do not enable unrolling in PreLinkThinLTO phase during sample PGO
754 // because it changes IR to makes profile annotation in back compile
755 // inaccurate. The normal unroller doesn't pay attention to forced full unroll
756 // attributes so we need to make sure and allow the full unroll pass to pay
757 // attention to it.
758 if (!isThinLTOPreLink(Phase) || !PGOOpt ||
759 PGOOpt->Action != PGOOptions::SampleUse)
760 LPM2.addPass(LoopFullUnrollPass(static_cast<int>(Level),
761 /* OnlyWhenForced= */ !PTO.LoopUnrolling,
762 PTO.ForgetAllSCEVInLoopUnroll));
763
765
766 FPM.addPass(createFunctionToLoopPassAdaptor(std::move(LPM1),
767 /*UseMemorySSA=*/true));
768 FPM.addPass(
769 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
771 // The loop passes in LPM2 (LoopIdiomRecognizePass, IndVarSimplifyPass,
772 // LoopDeletionPass and LoopFullUnrollPass) do not preserve MemorySSA.
773 // *All* loop passes must preserve it, in order to be able to use it.
774 FPM.addPass(createFunctionToLoopPassAdaptor(std::move(LPM2),
775 /*UseMemorySSA=*/false));
776
777 // Delete small array after loop unroll.
779
780 // Try vectorization/scalarization transforms that are both improvements
781 // themselves and can allow further folds with GVN and InstCombine.
782 FPM.addPass(VectorCombinePass(/*TryEarlyFoldsOnly=*/true));
783
784 // Eliminate redundancies.
786 if (RunNewGVN)
787 FPM.addPass(NewGVNPass());
788 else
789 FPM.addPass(GVNPass());
790
791 // Sparse conditional constant propagation.
792 // FIXME: It isn't clear why we do this *after* loop passes rather than
793 // before...
794 FPM.addPass(SCCPPass());
795
796 // Delete dead bit computations (instcombine runs after to fold away the dead
797 // computations, and then ADCE will run later to exploit any new DCE
798 // opportunities that creates).
799 FPM.addPass(BDCEPass());
800
801 // Run instcombine after redundancy and dead bit elimination to exploit
802 // opportunities opened up by them.
804 invokePeepholeEPCallbacks(FPM, Level);
805
806 // Re-consider control flow based optimizations after redundancy elimination,
807 // redo DCE, etc.
810
813
814 // Finally, do an expensive DCE pass to catch all the dead code exposed by
815 // the simplifications and basic cleanup after all the simplifications.
816 // TODO: Investigate if this is too expensive.
817 FPM.addPass(ADCEPass());
818
819 // Specially optimize memory movement as it doesn't look like dataflow in SSA.
820 FPM.addPass(MemCpyOptPass());
821
822 FPM.addPass(DSEPass());
824
826 LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
827 /*AllowSpeculation=*/true),
828 /*UseMemorySSA=*/true));
829
830 FPM.addPass(CoroElidePass());
831
833
835 .convertSwitchRangeToICmp(true)
836 .convertSwitchToArithmetic(true)
837 .hoistCommonInsts(true)
838 .sinkCommonInsts(true)));
840 invokePeepholeEPCallbacks(FPM, Level);
841
842 return FPM;
843}
844
845void PassBuilder::addRequiredLTOPreLinkPasses(ModulePassManager &MPM) {
848 MPM.addPass(AssignGUIDPass());
849}
850
851void PassBuilder::addPreInlinerPasses(ModulePassManager &MPM,
852 OptimizationLevel Level,
853 ThinOrFullLTOPhase LTOPhase) {
854 assert(Level != OptimizationLevel::O0 && "Not expecting O0 here!");
856 return;
857 InlineParams IP;
858
860
861 // FIXME: The hint threshold has the same value used by the regular inliner
862 // when not optimzing for size. This should probably be lowered after
863 // performance testing.
864 // FIXME: this comment is cargo culted from the old pass manager, revisit).
865 IP.HintThreshold = 325;
868 IP, /* MandatoryFirst */ true,
870 CGSCCPassManager &CGPipeline = MIWP.getPM();
871
873 FPM.addPass(SROAPass(SROAOptions::ModifyCFG));
874 FPM.addPass(EarlyCSEPass()); // Catch trivial redundancies.
875 FPM.addPass(SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(
876 true))); // Merge & remove basic blocks.
877 FPM.addPass(InstCombinePass()); // Combine silly sequences.
878 invokePeepholeEPCallbacks(FPM, Level);
879
880 CGPipeline.addPass(createCGSCCToFunctionPassAdaptor(
881 std::move(FPM), PTO.EagerlyInvalidateAnalyses));
882
883 MPM.addPass(std::move(MIWP));
884
885 // Delete anything that is now dead to make sure that we don't instrument
886 // dead code. Instrumentation can end up keeping dead code around and
887 // dramatically increase code size.
888 MPM.addPass(GlobalDCEPass());
889}
890
891void PassBuilder::addPostPGOLoopRotation(ModulePassManager &MPM,
892 OptimizationLevel Level) {
894 // Disable header duplication in loop rotation at -Oz.
896 createFunctionToLoopPassAdaptor(LoopRotatePass(),
897 /*UseMemorySSA=*/false),
898 PTO.EagerlyInvalidateAnalyses));
899 }
900}
901
902void PassBuilder::addPGOInstrPasses(ModulePassManager &MPM,
903 OptimizationLevel Level, bool RunProfileGen,
904 bool IsCS, bool AtomicCounterUpdate,
905 std::string ProfileFile,
906 std::string ProfileRemappingFile) {
907 assert(Level != OptimizationLevel::O0 && "Not expecting O0 here!");
908
909 if (!RunProfileGen) {
910 assert(!ProfileFile.empty() && "Profile use expecting a profile file!");
911 MPM.addPass(
912 PGOInstrumentationUse(ProfileFile, ProfileRemappingFile, IsCS, FS));
913 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
914 // RequireAnalysisPass for PSI before subsequent non-module passes.
915 MPM.addPass(RequireAnalysisPass<ProfileSummaryAnalysis, Module>());
916 return;
917 }
918
919 // Perform PGO instrumentation.
920 MPM.addPass(PGOInstrumentationGen(IsCS ? PGOInstrumentationType::CSFDO
922
923 addPostPGOLoopRotation(MPM, Level);
924 // Add the profile lowering pass.
925 InstrProfOptions Options;
926 if (!ProfileFile.empty())
927 Options.InstrProfileOutput = ProfileFile;
928 // Do counter promotion at Level greater than O0.
929 Options.DoCounterPromotion = true;
930 Options.UseBFIInPromotion = IsCS;
931 if (EnableSampledInstr) {
932 Options.Sampling = true;
933 // With sampling, there is little beneifit to enable counter promotion.
934 // But note that sampling does work with counter promotion.
935 Options.DoCounterPromotion = false;
936 }
937 Options.Atomic = AtomicCounterUpdate;
938 MPM.addPass(InstrProfilingLoweringPass(Options, IsCS));
939}
940
942 bool RunProfileGen, bool IsCS,
943 bool AtomicCounterUpdate,
944 std::string ProfileFile,
945 std::string ProfileRemappingFile) {
946 if (!RunProfileGen) {
947 assert(!ProfileFile.empty() && "Profile use expecting a profile file!");
948 MPM.addPass(
949 PGOInstrumentationUse(ProfileFile, ProfileRemappingFile, IsCS, FS));
950 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
951 // RequireAnalysisPass for PSI before subsequent non-module passes.
953 return;
954 }
955
956 // Perform PGO instrumentation.
959 // Add the profile lowering pass.
961 if (!ProfileFile.empty())
962 Options.InstrProfileOutput = ProfileFile;
963 // Do not do counter promotion at O0.
964 Options.DoCounterPromotion = false;
965 Options.UseBFIInPromotion = IsCS;
966 Options.Atomic = AtomicCounterUpdate;
968}
969
971 return getInlineParamsFromOptLevel(static_cast<unsigned>(Level));
972}
973
977 InlineParams IP;
978 if (PTO.InlinerThreshold == -1)
980 else
981 IP = getInlineParams(PTO.InlinerThreshold);
982 // For PreLinkThinLTO + SamplePGO or PreLinkFullLTO + SamplePGO,
983 // set hot-caller threshold to 0 to disable hot
984 // callsite inline (as much as possible [1]) because it makes
985 // profile annotation in the backend inaccurate.
986 //
987 // [1] Note the cost of a function could be below zero due to erased
988 // prologue / epilogue.
989 if (isLTOPreLink(Phase) && PGOOpt && PGOOpt->Action == PGOOptions::SampleUse)
991
992 if (PGOOpt)
994
998
999 // Require the GlobalsAA analysis for the module so we can query it within
1000 // the CGSCC pipeline.
1002 MIWP.addModulePass(RequireAnalysisPass<GlobalsAA, Module>());
1003 // Invalidate AAManager so it can be recreated and pick up the newly
1004 // available GlobalsAA.
1005 MIWP.addModulePass(
1007 }
1008
1009 // Require the ProfileSummaryAnalysis for the module so we can query it within
1010 // the inliner pass.
1012
1013 // Now begin the main postorder CGSCC pipeline.
1014 // FIXME: The current CGSCC pipeline has its origins in the legacy pass
1015 // manager and trying to emulate its precise behavior. Much of this doesn't
1016 // make a lot of sense and we should revisit the core CGSCC structure.
1017 CGSCCPassManager &MainCGPipeline = MIWP.getPM();
1018
1019 // Note: historically, the PruneEH pass was run first to deduce nounwind and
1020 // generally clean up exception handling overhead. It isn't clear this is
1021 // valuable as the inliner doesn't currently care whether it is inlining an
1022 // invoke or a call.
1023
1025 MainCGPipeline.addPass(AttributorCGSCCPass());
1027 MainCGPipeline.addPass(AttributorLightCGSCCPass());
1028
1029 // Deduce function attributes. We do another run of this after the function
1030 // simplification pipeline, so this only needs to run when it could affect the
1031 // function simplification pipeline, which is only the case with recursive
1032 // functions.
1033 MainCGPipeline.addPass(PostOrderFunctionAttrsPass(/*SkipNonRecursive*/ true));
1034
1035 // When at O3 add argument promotion to the pass pipeline.
1036 // FIXME: It isn't at all clear why this should be limited to O3.
1037 if (Level == OptimizationLevel::O3)
1038 MainCGPipeline.addPass(ArgumentPromotionPass());
1039
1040 // Try to perform OpenMP specific optimizations. This is a (quick!) no-op if
1041 // there are no OpenMP runtime calls present in the module.
1042 if (Level == OptimizationLevel::O2 || Level == OptimizationLevel::O3)
1043 MainCGPipeline.addPass(OpenMPOptCGSCCPass(Phase));
1044
1045 invokeCGSCCOptimizerLateEPCallbacks(MainCGPipeline, Level);
1046
1047 // Add the core function simplification pipeline nested inside the
1048 // CGSCC walk.
1051 PTO.EagerlyInvalidateAnalyses, /*NoRerun=*/true));
1052
1053 // Finally, deduce any function attributes based on the fully simplified
1054 // function.
1055 MainCGPipeline.addPass(PostOrderFunctionAttrsPass());
1056
1057 // Mark that the function is fully simplified and that it shouldn't be
1058 // simplified again if we somehow revisit it due to CGSCC mutations unless
1059 // it's been modified since.
1062
1063 if (!isThinLTOPreLink(Phase)) {
1064 MainCGPipeline.addPass(CoroSplitPass(Level != OptimizationLevel::O0));
1065 MainCGPipeline.addPass(CoroAnnotationElidePass());
1066 }
1067
1068 // Make sure we don't affect potential future NoRerun CGSCC adaptors.
1069 MIWP.addLateModulePass(createModuleToFunctionPassAdaptor(
1071
1072 return MIWP;
1073}
1074
1079
1081 // For PreLinkThinLTO + SamplePGO or PreLinkFullLTO + SamplePGO,
1082 // set hot-caller threshold to 0 to disable hot
1083 // callsite inline (as much as possible [1]) because it makes
1084 // profile annotation in the backend inaccurate.
1085 //
1086 // [1] Note the cost of a function could be below zero due to erased
1087 // prologue / epilogue.
1088 if (isLTOPreLink(Phase) && PGOOpt && PGOOpt->Action == PGOOptions::SampleUse)
1089 IP.HotCallSiteThreshold = 0;
1090
1091 if (PGOOpt)
1093
1094 // The inline deferral logic is used to avoid losing some
1095 // inlining chance in future. It is helpful in SCC inliner, in which
1096 // inlining is processed in bottom-up order.
1097 // While in module inliner, the inlining order is a priority-based order
1098 // by default. The inline deferral is unnecessary there. So we disable the
1099 // inline deferral logic in module inliner.
1100 IP.EnableDeferral = false;
1101
1104 MPM.addPass(GlobalOptPass());
1105 MPM.addPass(GlobalDCEPass());
1106 MPM.addPass(AssignGUIDPass());
1107 MPM.addPass(PGOCtxProfFlatteningPass(/*IsPreThinlink=*/false));
1108 }
1109
1112 PTO.EagerlyInvalidateAnalyses));
1113
1114 if (!isThinLTOPreLink(Phase)) {
1117 MPM.addPass(
1119 }
1120
1121 return MPM;
1122}
1123
1127 assert(Level != OptimizationLevel::O0 &&
1128 "Should not be used for O0 pipeline");
1129
1131 "FullLTOPostLink shouldn't call buildModuleSimplificationPipeline!");
1132
1134
1135 // Place pseudo probe instrumentation as the first pass of the pipeline to
1136 // minimize the impact of optimization changes.
1137 if (PGOOpt && PGOOpt->PseudoProbeForProfiling && !isThinLTOPostLink(Phase))
1139
1140 bool HasSampleProfile = PGOOpt && (PGOOpt->Action == PGOOptions::SampleUse);
1141
1142 // In ThinLTO mode, when flattened profile is used, all the available
1143 // profile information will be annotated in PreLink phase so there is
1144 // no need to load the profile again in PostLink.
1145 bool LoadSampleProfile =
1146 HasSampleProfile && !(FlattenedProfileUsed && isThinLTOPostLink(Phase));
1147
1148 // During the ThinLTO backend phase we perform early indirect call promotion
1149 // here, before globalopt. Otherwise imported available_externally functions
1150 // look unreferenced and are removed. If we are going to load the sample
1151 // profile then defer until later.
1152 // TODO: See if we can move later and consolidate with the location where
1153 // we perform ICP when we are loading a sample profile.
1154 // TODO: We pass HasSampleProfile (whether there was a sample profile file
1155 // passed to the compile) to the SamplePGO flag of ICP. This is used to
1156 // determine whether the new direct calls are annotated with prof metadata.
1157 // Ideally this should be determined from whether the IR is annotated with
1158 // sample profile, and not whether the a sample profile was provided on the
1159 // command line. E.g. for flattened profiles where we will not be reloading
1160 // the sample profile in the ThinLTO backend, we ideally shouldn't have to
1161 // provide the sample profile file.
1162 if (isThinLTOPostLink(Phase) && !LoadSampleProfile)
1163 MPM.addPass(PGOIndirectCallPromotion(true /* InLTO */, HasSampleProfile));
1164
1165 // Create an early function pass manager to cleanup the output of the
1166 // frontend. Not necessary with LTO post link pipelines since the pre link
1167 // pipeline already cleaned up the frontend output.
1168 if (!isThinLTOPostLink(Phase)) {
1169 // Do basic inference of function attributes from known properties of system
1170 // libraries and other oracles.
1172 MPM.addPass(CoroEarlyPass());
1173
1174 FunctionPassManager EarlyFPM;
1175 EarlyFPM.addPass(EntryExitInstrumenterPass(/*PostInlining=*/false));
1176 // Lower llvm.expect to metadata before attempting transforms.
1177 // Compare/branch metadata may alter the behavior of passes like
1178 // SimplifyCFG.
1180 EarlyFPM.addPass(SimplifyCFGPass());
1182 EarlyFPM.addPass(EarlyCSEPass());
1183 if (Level == OptimizationLevel::O3)
1184 EarlyFPM.addPass(CallSiteSplittingPass());
1186 std::move(EarlyFPM), PTO.EagerlyInvalidateAnalyses));
1187 }
1188
1189 if (LoadSampleProfile) {
1190 // Annotate sample profile right after early FPM to ensure freshness of
1191 // the debug info.
1193 PGOOpt->ProfileFile, PGOOpt->ProfileRemappingFile, Phase, FS));
1194 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
1195 // RequireAnalysisPass for PSI before subsequent non-module passes.
1197 // Do not invoke ICP in the LTOPrelink phase as it makes it hard
1198 // for the profile annotation to be accurate in the LTO backend.
1199 if (!isLTOPreLink(Phase))
1200 // We perform early indirect call promotion here, before globalopt.
1201 // This is important for the ThinLTO backend phase because otherwise
1202 // imported available_externally functions look unreferenced and are
1203 // removed.
1204 MPM.addPass(
1205 PGOIndirectCallPromotion(true /* IsInLTO */, true /* SamplePGO */));
1206 }
1207
1208 // Try to perform OpenMP specific optimizations on the module. This is a
1209 // (quick!) no-op if there are no OpenMP runtime calls present in the module.
1211
1213 MPM.addPass(AttributorPass());
1216
1217 // Lower type metadata and the type.test intrinsic in the ThinLTO
1218 // post link pipeline after ICP. This is to enable usage of the type
1219 // tests in ICP sequences.
1222
1224
1225 // Interprocedural constant propagation now that basic cleanup has occurred
1226 // and prior to optimizing globals.
1227 // FIXME: This position in the pipeline hasn't been carefully considered in
1228 // years, it should be re-analyzed.
1229 MPM.addPass(
1230 IPSCCPPass(IPSCCPOptions(/*AllowFuncSpec=*/!isLTOPreLink(Phase))));
1231
1232 // Attach metadata to indirect call sites indicating the set of functions
1233 // they may target at run-time. This should follow IPSCCP.
1235
1236 // Optimize globals to try and fold them into constants.
1237 MPM.addPass(GlobalOptPass());
1238
1239 // Create a small function pass pipeline to cleanup after all the global
1240 // optimizations.
1241 FunctionPassManager GlobalCleanupPM;
1242 // FIXME: Should this instead by a run of SROA?
1243 GlobalCleanupPM.addPass(PromotePass());
1244 GlobalCleanupPM.addPass(InstCombinePass());
1245 invokePeepholeEPCallbacks(GlobalCleanupPM, Level);
1246 GlobalCleanupPM.addPass(
1247 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
1248 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(GlobalCleanupPM),
1249 PTO.EagerlyInvalidateAnalyses));
1250
1251 // We already asserted this happens in non-FullLTOPostLink earlier.
1252 const bool IsPreLink = !isThinLTOPostLink(Phase);
1253 // Enable contextual profiling instrumentation.
1254 const bool IsCtxProfGen =
1256 const bool IsPGOPreLink = !IsCtxProfGen && PGOOpt && IsPreLink;
1257 const bool IsPGOInstrGen =
1258 IsPGOPreLink && PGOOpt->Action == PGOOptions::IRInstr;
1259 const bool IsPGOInstrUse =
1260 IsPGOPreLink && PGOOpt->Action == PGOOptions::IRUse;
1261 const bool IsMemprofUse = IsPGOPreLink && !PGOOpt->MemoryProfile.empty();
1262 // We don't want to mix pgo ctx gen and pgo gen; we also don't currently
1263 // enable ctx profiling from the frontend.
1265 "Enabling both instrumented PGO and contextual instrumentation is not "
1266 "supported.");
1267 const bool IsCtxProfUse = !UseCtxProfile.empty() && isThinLTOPreLink(Phase);
1268
1269 assert(
1271 "--instrument-cold-function-only-path is provided but "
1272 "--pgo-instrument-cold-function-only is not enabled");
1273 const bool IsColdFuncOnlyInstrGen = PGOInstrumentColdFunctionOnly &&
1274 IsPGOPreLink &&
1276
1277 if (IsPGOInstrGen || IsPGOInstrUse || IsMemprofUse || IsCtxProfGen ||
1278 IsCtxProfUse || IsColdFuncOnlyInstrGen)
1279 addPreInlinerPasses(MPM, Level, Phase);
1280
1281 // Add all the requested passes for instrumentation PGO, if requested.
1282 if (IsPGOInstrGen || IsPGOInstrUse) {
1283 addPGOInstrPasses(MPM, Level,
1284 /*RunProfileGen=*/IsPGOInstrGen,
1285 /*IsCS=*/false, PGOOpt->AtomicCounterUpdate,
1286 PGOOpt->ProfileFile, PGOOpt->ProfileRemappingFile);
1287 } else if (IsCtxProfGen || IsCtxProfUse) {
1289 // In pre-link, we just want the instrumented IR. We use the contextual
1290 // profile in the post-thinlink phase.
1291 // The instrumentation will be removed in post-thinlink after IPO.
1292 if (IsCtxProfUse) {
1293 MPM.addPass(AssignGUIDPass());
1294 MPM.addPass(PGOCtxProfFlatteningPass(/*IsPreThinlink=*/true));
1295 return MPM;
1296 }
1297 // Block further inlining in the instrumented ctxprof case. This avoids
1298 // confusingly collecting profiles for the same GUID corresponding to
1299 // different variants of the function. We could do like PGO and identify
1300 // functions by a (GUID, Hash) tuple, but since the ctxprof "use" waits for
1301 // thinlto to happen before performing any further optimizations, it's
1302 // unnecessary to collect profiles for non-prevailing copies.
1304 addPostPGOLoopRotation(MPM, Level);
1305 MPM.addPass(AssignGUIDPass());
1307 } else if (IsColdFuncOnlyInstrGen) {
1308 addPGOInstrPasses(MPM, Level, /* RunProfileGen */ true, /* IsCS */ false,
1309 /* AtomicCounterUpdate */ false,
1311 /* ProfileRemappingFile */ "");
1312 }
1313
1314 if (IsPGOInstrGen || IsPGOInstrUse || IsCtxProfGen)
1315 MPM.addPass(PGOIndirectCallPromotion(false, false));
1316
1317 if (IsPGOPreLink && PGOOpt->CSAction == PGOOptions::CSIRInstr)
1318 MPM.addPass(PGOInstrumentationGenCreateVar(PGOOpt->CSProfileGenFile,
1320
1321 if (IsMemprofUse)
1322 MPM.addPass(MemProfUsePass(PGOOpt->MemoryProfile, FS));
1323
1324 if (PGOOpt && (PGOOpt->Action == PGOOptions::IRUse ||
1325 PGOOpt->Action == PGOOptions::SampleUse))
1326 MPM.addPass(PGOForceFunctionAttrsPass(PGOOpt->ColdOptType));
1327
1328 MPM.addPass(AlwaysInlinerPass(/*InsertLifetimeIntrinsics=*/true));
1329
1332 else
1333 MPM.addPass(buildInlinerPipeline(Level, Phase));
1334
1335 // Remove any dead arguments exposed by cleanups, constant folding globals,
1336 // and argument promotion.
1338
1341
1342 if (!isThinLTOPreLink(Phase))
1343 MPM.addPass(CoroCleanupPass());
1344
1345 // Optimize globals now that functions are fully simplified.
1346 MPM.addPass(GlobalOptPass());
1347 MPM.addPass(GlobalDCEPass());
1348
1349 return MPM;
1350}
1351
1352/// TODO: Should LTO cause any differences to this set of passes?
1353void PassBuilder::addVectorPasses(OptimizationLevel Level,
1355 ThinOrFullLTOPhase LTOPhase) {
1358
1359 // Drop dereferenceable assumes after vectorization, as they are no longer
1360 // needed and can inhibit further optimization.
1361 if (!isLTOPreLink(LTOPhase))
1362 FPM.addPass(DropUnnecessaryAssumesPass(/*DropDereferenceable=*/true));
1363
1365 if (isFullLTOPostLink(LTOPhase)) {
1366 // The vectorizer may have significantly shortened a loop body; unroll
1367 // again. Unroll small loops to hide loop backedge latency and saturate any
1368 // parallel execution resources of an out-of-order processor. We also then
1369 // need to clean up redundancies and loop invariant code.
1370 // FIXME: It would be really good to use a loop-integrated instruction
1371 // combiner for cleanup here so that the unrolling and LICM can be pipelined
1372 // across the loop nests.
1373 // We do UnrollAndJam in a separate LPM to ensure it happens before unroll
1376 LoopUnrollAndJamPass(static_cast<int>(Level))));
1378 static_cast<int>(Level), /*OnlyWhenForced=*/!PTO.LoopUnrolling,
1381 // Now that we are done with loop unrolling, be it either by LoopVectorizer,
1382 // or LoopUnroll passes, some variable-offset GEP's into alloca's could have
1383 // become constant-offset, thus enabling SROA and alloca promotion. Do so.
1384 // NOTE: we are very late in the pipeline, and we don't have any LICM
1385 // or SimplifyCFG passes scheduled after us, that would cleanup
1386 // the CFG mess this may created if allowed to modify CFG, so forbid that.
1387
1388 // We also turn on struct to vector canonicalization here, which allows
1389 // converting allocas of homogeneous structs into vector allocas when the
1390 // allocas' users are all memory intrinsics. This allows promotion in some
1391 // cases because structs cannot promote to SSA values, but vectors can. We
1392 // only turn this on after memcpyopt runs because this might hinder
1393 // memcpyopt's optimizations if done before. Look at the documentation for
1394 // `tryCanonicalizeStructToVector` in SROA.cpp to see why.
1396 /*AggregateToVector=*/true)));
1397 }
1398
1399 if (!isFullLTOPostLink(LTOPhase)) {
1400 // Eliminate loads by forwarding stores from the previous iteration to loads
1401 // of the current iteration.
1403 }
1404 // Cleanup after the loop optimization passes.
1405 FPM.addPass(InstCombinePass());
1406
1408 ExtraFunctionPassManager<ShouldRunExtraVectorPasses> ExtraPasses;
1409 // At higher optimization levels, try to clean up any runtime overlap and
1410 // alignment checks inserted by the vectorizer. We want to track correlated
1411 // runtime checks for two inner loops in the same outer loop, fold any
1412 // common computations, hoist loop-invariant aspects out of any outer loop,
1413 // and unswitch the runtime checks if possible. Once hoisted, we may have
1414 // dead (or speculatable) control flows or more combining opportunities.
1415 ExtraPasses.addPass(EarlyCSEPass());
1416 ExtraPasses.addPass(CorrelatedValuePropagationPass());
1417 ExtraPasses.addPass(InstCombinePass());
1418 LoopPassManager LPM;
1419 LPM.addPass(LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
1420 /*AllowSpeculation=*/true));
1421 LPM.addPass(SimpleLoopUnswitchPass(/* NonTrivial */ Level ==
1423 ExtraPasses.addPass(
1424 createFunctionToLoopPassAdaptor(std::move(LPM), /*UseMemorySSA=*/true));
1425 ExtraPasses.addPass(
1426 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
1427 ExtraPasses.addPass(InstCombinePass());
1428 FPM.addPass(std::move(ExtraPasses));
1429 }
1430
1431 // Now that we've formed fast to execute loop structures, we do further
1432 // optimizations. These are run afterward as they might block doing complex
1433 // analyses and transforms such as what are needed for loop vectorization.
1434
1435 // Cleanup after loop vectorization, etc. Simplification passes like CVP and
1436 // GVN, loop transforms, and others have already run, so it's now better to
1437 // convert to more optimized IR using more aggressive simplify CFG options.
1438 // The extra sinking transform can create larger basic blocks, so do this
1439 // before SLP vectorization.
1440 FPM.addPass(SimplifyCFGPass(SimplifyCFGOptions()
1441 .forwardSwitchCondToPhi(true)
1442 .convertSwitchRangeToICmp(true)
1443 .convertSwitchToArithmetic(true)
1444 .convertSwitchToLookupTable(true)
1445 .needCanonicalLoops(false)
1446 .hoistCommonInsts(true)
1447 .sinkCommonInsts(true)));
1448
1449 if (isFullLTOPostLink(LTOPhase)) {
1450 FPM.addPass(SCCPPass());
1451 FPM.addPass(InstCombinePass());
1452 FPM.addPass(BDCEPass());
1453 }
1454
1455 // Optimize parallel scalar instruction chains into SIMD instructions.
1456 if (PTO.SLPVectorization) {
1457 FPM.addPass(SLPVectorizerPass());
1459 FPM.addPass(EarlyCSEPass());
1460 }
1461 }
1462 // Enhance/cleanup vector code.
1463 FPM.addPass(VectorCombinePass());
1464
1465 if (!isFullLTOPostLink(LTOPhase)) {
1466 FPM.addPass(InstCombinePass());
1467 // Unroll small loops to hide loop backedge latency and saturate any
1468 // parallel execution resources of an out-of-order processor. We also then
1469 // need to clean up redundancies and loop invariant code.
1470 // FIXME: It would be really good to use a loop-integrated instruction
1471 // combiner for cleanup here so that the unrolling and LICM can be pipelined
1472 // across the loop nests.
1473 // We do UnrollAndJam in a separate LPM to ensure it happens before unroll
1474 if (EnableUnrollAndJam && PTO.LoopUnrolling) {
1476 LoopUnrollAndJamPass(static_cast<int>(Level))));
1477 }
1478 FPM.addPass(LoopUnrollPass(LoopUnrollOptions(
1479 static_cast<int>(Level), /*OnlyWhenForced=*/!PTO.LoopUnrolling,
1480 PTO.ForgetAllSCEVInLoopUnroll)));
1481 FPM.addPass(WarnMissedTransformationsPass());
1482 // Now that we are done with loop unrolling, be it either by LoopVectorizer,
1483 // or LoopUnroll passes, some variable-offset GEP's into alloca's could have
1484 // become constant-offset, thus enabling SROA and alloca promotion. Do so.
1485 // NOTE: we are very late in the pipeline, and we don't have any LICM
1486 // or SimplifyCFG passes scheduled after us, that would cleanup
1487 // the CFG mess this may created if allowed to modify CFG, so forbid that.
1488
1489 // We also turn on struct to vector canonicalization here, which allows
1490 // converting allocas of homogeneous structs into vector allocas when the
1491 // allocas' users are all memory intrinsics. This allows promotion in some
1492 // cases because structs cannot promote to SSA values, but vectors can. We
1493 // only turn this on after memcpyopt runs because this might hinder
1494 // memcpyopt's optimizations if done before. Look at the documentation for
1495 // `tryCanonicalizeStructToVector` in SROA.cpp to see why.
1496 FPM.addPass(SROAPass(SROAOptions(SROAOptions::PreserveCFG,
1497 /*AggregateToVector=*/true)));
1498 }
1499
1500 FPM.addPass(InferAlignmentPass());
1501 FPM.addPass(InstCombinePass());
1502
1503 // This is needed for two reasons:
1504 // 1. It works around problems that instcombine introduces, such as sinking
1505 // expensive FP divides into loops containing multiplications using the
1506 // divide result.
1507 // 2. It helps to clean up some loop-invariant code created by the loop
1508 // unroll pass when IsFullLTO=false.
1510 LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
1511 /*AllowSpeculation=*/true),
1512 /*UseMemorySSA=*/true));
1513
1514 // Now that we've vectorized and unrolled loops, we may have more refined
1515 // alignment information, try to re-derive it here.
1516 FPM.addPass(AlignmentFromAssumptionsPass());
1517}
1518
1521 ThinOrFullLTOPhase LTOPhase) {
1523
1524 // Run partial inlining pass to partially inline functions that have
1525 // large bodies.
1528
1529 // Remove avail extern fns and globals definitions since we aren't compiling
1530 // an object file for later LTO. For LTO we want to preserve these so they
1531 // are eligible for inlining at link-time. Note if they are unreferenced they
1532 // will be removed by GlobalDCE later, so this only impacts referenced
1533 // available externally globals. Eventually they will be suppressed during
1534 // codegen, but eliminating here enables more opportunity for GlobalDCE as it
1535 // may make globals referenced by available external functions dead and saves
1536 // running remaining passes on the eliminated functions. These should be
1537 // preserved during prelinking for link-time inlining decisions.
1538 if (!isLTOPreLink(LTOPhase))
1540
1541 // Do RPO function attribute inference across the module to forward-propagate
1542 // attributes where applicable.
1543 // FIXME: Is this really an optimization rather than a canonicalization?
1545
1546 // Do a post inline PGO instrumentation and use pass. This is a context
1547 // sensitive PGO pass. We don't want to do this in LTOPreLink phrase as
1548 // cross-module inline has not been done yet. The context sensitive
1549 // instrumentation is after all the inlines are done.
1550 if (!isLTOPreLink(LTOPhase) && PGOOpt) {
1551 if (PGOOpt->CSAction == PGOOptions::CSIRInstr)
1552 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/true,
1553 /*IsCS=*/true, PGOOpt->AtomicCounterUpdate,
1554 PGOOpt->CSProfileGenFile, PGOOpt->ProfileRemappingFile);
1555 else if (PGOOpt->CSAction == PGOOptions::CSIRUse)
1556 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/false,
1557 /*IsCS=*/true, PGOOpt->AtomicCounterUpdate,
1558 PGOOpt->ProfileFile, PGOOpt->ProfileRemappingFile);
1559 }
1560
1561 // Re-compute GlobalsAA here prior to function passes. This is particularly
1562 // useful as the above will have inlined, DCE'ed, and function-attr
1563 // propagated everything. We should at this point have a reasonably minimal
1564 // and richly annotated call graph. By computing aliasing and mod/ref
1565 // information for all local globals here, the late loop passes and notably
1566 // the vectorizer will be able to use them to help recognize vectorizable
1567 // memory operations.
1570
1571 invokeOptimizerEarlyEPCallbacks(MPM, Level, LTOPhase);
1572
1573 FunctionPassManager OptimizePM;
1574
1575 // Only drop unnecessary assumes post-inline and post-link, as otherwise
1576 // additional uses of the affected value may be introduced through inlining
1577 // and CSE.
1578 if (!isLTOPreLink(LTOPhase))
1579 OptimizePM.addPass(DropUnnecessaryAssumesPass());
1580
1581 // Scheduling LoopVersioningLICM when inlining is over, because after that
1582 // we may see more accurate aliasing. Reason to run this late is that too
1583 // early versioning may prevent further inlining due to increase of code
1584 // size. Other optimizations which runs later might get benefit of no-alias
1585 // assumption in clone loop.
1587 OptimizePM.addPass(
1589 // LoopVersioningLICM pass might increase new LICM opportunities.
1591 LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
1592 /*AllowSpeculation=*/true),
1593 /*USeMemorySSA=*/true));
1594 }
1595
1596 OptimizePM.addPass(Float2IntPass());
1597 // Defer until LTO post-link where some constants may become known.
1598 if (!isLTOPreLink(LTOPhase))
1600
1601 if (EnableMatrix) {
1602 OptimizePM.addPass(LowerMatrixIntrinsicsPass());
1603 OptimizePM.addPass(EarlyCSEPass());
1604 }
1605
1606 // CHR pass should only be applied with the profile information.
1607 // The check is to check the profile summary information in CHR.
1608 if (EnableCHR && Level == OptimizationLevel::O3)
1609 OptimizePM.addPass(ControlHeightReductionPass());
1610
1611 // FIXME: We need to run some loop optimizations to re-rotate loops after
1612 // simplifycfg and others undo their rotation.
1613
1614 // Optimize the loop execution. These passes operate on entire loop nests
1615 // rather than on each loop in an inside-out manner, and so they are actually
1616 // function passes.
1617
1618 invokeVectorizerStartEPCallbacks(OptimizePM, Level);
1619
1620 LoopPassManager LPM;
1621 // First rotate loops that may have been un-rotated by prior passes.
1622 // Disable header duplication at -Oz.
1623 LPM.addPass(LoopRotatePass(/*EnableLoopHeaderDuplication=*/true,
1624 isLTOPreLink(LTOPhase),
1625 /*CheckExitCount=*/true));
1626 // Some loops may have become dead by now. Try to delete them.
1627 // FIXME: see discussion in https://reviews.llvm.org/D112851,
1628 // this may need to be revisited once we run GVN before loop deletion
1629 // in the simplification pipeline.
1630 LPM.addPass(LoopDeletionPass());
1631
1632 if (PTO.LoopInterchange)
1633 LPM.addPass(LoopInterchangePass());
1634
1635 OptimizePM.addPass(
1636 createFunctionToLoopPassAdaptor(std::move(LPM), /*UseMemorySSA=*/false));
1637
1638 // FIXME: This may not be the right place in the pipeline.
1639 // We need to have the data to support the right place.
1640 if (PTO.LoopFusion)
1641 OptimizePM.addPass(LoopFusePass());
1642
1643 // Distribute loops to allow partial vectorization. I.e. isolate dependences
1644 // into separate loop that would otherwise inhibit vectorization. This is
1645 // currently only performed for loops marked with the metadata
1646 // llvm.loop.distribute=true or when -enable-loop-distribute is specified.
1647 OptimizePM.addPass(LoopDistributePass());
1648
1649 // Populates the VFABI attribute with the scalar-to-vector mappings
1650 // from the TargetLibraryInfo.
1651 OptimizePM.addPass(InjectTLIMappings());
1652
1653 addVectorPasses(Level, OptimizePM, LTOPhase);
1654
1655 invokeVectorizerEndEPCallbacks(OptimizePM, Level);
1656
1657 // LoopSink pass sinks instructions hoisted by LICM, which serves as a
1658 // canonicalization pass that enables other optimizations. As a result,
1659 // LoopSink pass needs to be a very late IR pass to avoid undoing LICM
1660 // result too early.
1661 OptimizePM.addPass(LoopSinkPass());
1662
1663 // And finally clean up LCSSA form before generating code.
1664 OptimizePM.addPass(InstSimplifyPass());
1665
1666 // This hoists/decomposes div/rem ops. It should run after other sink/hoist
1667 // passes to avoid re-sinking, but before SimplifyCFG because it can allow
1668 // flattening of blocks.
1669 OptimizePM.addPass(DivRemPairsPass());
1670
1671 // Merge adjacent icmps into memcmp, then expand memcmp to loads/compares.
1672 // TODO: move this furter up so that it can be optimized by GVN, etc.
1673 if (EnableMergeICmps)
1674 OptimizePM.addPass(MergeICmpsPass());
1675 OptimizePM.addPass(ExpandMemCmpPass());
1676
1677 // Try to annotate calls that were created during optimization.
1678 OptimizePM.addPass(
1679 TailCallElimPass(/*UpdateFunctionEntryCount=*/isInstrumentedPGOUse()));
1680
1681 // LoopSink (and other loop passes since the last simplifyCFG) might have
1682 // resulted in single-entry-single-exit or empty blocks. Clean up the CFG.
1683 OptimizePM.addPass(
1685 .convertSwitchRangeToICmp(true)
1686 .convertSwitchToArithmetic(true)
1687 .speculateUnpredictables(true)
1688 .hoistLoadsStoresWithCondFaulting(true)));
1689
1690 // Add the core optimizing pipeline.
1691 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(OptimizePM),
1692 PTO.EagerlyInvalidateAnalyses));
1693
1694 // AllocToken transforms heap allocation calls; this needs to run late after
1695 // other allocation call transformations (such as those in InstCombine).
1696 if (!isLTOPreLink(LTOPhase))
1697 MPM.addPass(AllocTokenPass());
1698
1699 invokeOptimizerLastEPCallbacks(MPM, Level, LTOPhase);
1700
1701 // Run the Instrumentor pass late.
1703 MPM.addPass(InstrumentorPass(FS));
1704
1705 // Split out cold code. Splitting is done late to avoid hiding context from
1706 // other optimizations and inadvertently regressing performance. The tradeoff
1707 // is that this has a higher code size cost than splitting early.
1708 if (EnableHotColdSplit && !isLTOPreLink(LTOPhase))
1710
1711 // Now we need to do some global optimization transforms.
1712 // FIXME: It would seem like these should come first in the optimization
1713 // pipeline and maybe be the bottom of the canonicalization pipeline? Weird
1714 // ordering here.
1715 MPM.addPass(GlobalDCEPass());
1717
1718 // Merge functions if requested. It has a better chance to merge functions
1719 // after ConstantMerge folded jump tables.
1720 if (PTO.MergeFunctions)
1722
1723 if (PTO.CallGraphProfile && !isLTOPreLink(LTOPhase))
1724 MPM.addPass(CGProfilePass(isLTOPostLink(LTOPhase)));
1725
1726 // RelLookupTableConverterPass runs later in LTO post-link pipeline.
1727 if (!isLTOPreLink(LTOPhase))
1729
1730 // Add devirtualization pass only when LTO is not enabled, as otherwise
1731 // the pass is already enabled in the LTO pipeline.
1732 if (PTO.DevirtualizeSpeculatively && LTOPhase == ThinOrFullLTOPhase::None) {
1733 // TODO: explore a better pipeline configuration that can improve
1734 // compilation time overhead.
1735 // FIXME: move this earlier (lots of pass ordering tests will need fixing)
1736 MPM.addPass(AssignGUIDPass());
1738 /*ExportSummary*/ nullptr,
1739 /*ImportSummary*/ nullptr,
1740 /*DevirtSpeculatively*/ PTO.DevirtualizeSpeculatively));
1742 // Given that the devirtualization creates more opportunities for inlining,
1743 // we run the Inliner again here to maximize the optimization gain we
1744 // get from devirtualization.
1745 // Also, we can't run devirtualization before inlining because the
1746 // devirtualization depends on the passes optimizing/eliminating vtable GVs
1747 // and those passes are only effective after inlining.
1748 if (EnableModuleInliner) {
1752 } else {
1755 /* MandatoryFirst */ true,
1757 }
1758 }
1759
1760 // Attach !implicit.ref metadata from all functions to copyright strings.
1762
1763 return MPM;
1764}
1765
1769 if (Level == OptimizationLevel::O0)
1770 return buildO0DefaultPipeline(Level, Phase);
1771
1773 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1774 // Currently this pipeline is only invoked in an LTO pre link pass or when we
1775 // are not running LTO. If that changes the below checks may need updating.
1777
1778 // If we are invoking this in non-LTO mode, remove any MemProf related
1779 // attributes and metadata, as we don't know whether we are linking with
1780 // a library containing the necessary interfaces.
1783
1784 // Convert @llvm.global.annotations to !annotation metadata.
1786
1787 // Force any function attributes we want the rest of the pipeline to observe.
1789
1790 if (TriggerCrash)
1792
1793 if (PGOOpt && PGOOpt->DebugInfoForProfiling)
1795
1796 // Apply module pipeline start EP callback.
1798
1799 // Add the core simplification pipeline.
1801
1802 // Now add the optimization pipeline.
1804
1805 if (PGOOpt && PGOOpt->PseudoProbeForProfiling &&
1806 PGOOpt->Action == PGOOptions::SampleUse)
1808
1809 // Emit annotation remarks.
1811
1812 if (isLTOPreLink(Phase))
1813 addRequiredLTOPreLinkPasses(MPM);
1814
1815 instructionCountersPass(MPM, /* IsPreOptimization */ false);
1816 return MPM;
1817}
1818
1821 bool EmitSummary, bool Verify) {
1823
1824 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1825
1826 if (ThinLTO)
1828 else
1830 // AssignGUIDPass attaches !guid metadata (MD_unique_id) to global objects,
1831 // triggering the bitcode writer to emit a METADATA_KIND_BLOCK. Standard LTO
1832 // bitcode emission runs VerifierPass by default, which registers metadata
1833 // kind IDs in LLVMContext. Running VerifierPass here before EmbedBitcodePass
1834 // to get the same behavior.
1835 if (Verify)
1836 MPM.addPass(VerifierPass());
1837 MPM.addPass(EmbedBitcodePass(ThinLTO, EmitSummary));
1838
1839 // Perform any cleanups to the IR that aren't suitable for per TU compilation,
1840 // like removing CFI/WPD related instructions. Note, we reuse
1841 // DropTypeTestsPass to clean up type tests rather than duplicate that logic
1842 // in FatLtoCleanup.
1843 MPM.addPass(FatLtoCleanup());
1844
1845 // If we're doing FatLTO w/ CFI enabled, we don't want the type tests in the
1846 // object code, only in the bitcode section, so drop it before we run
1847 // module optimization and generate machine code. If llvm.type.test() isn't in
1848 // the IR, this won't do anything.
1850
1851 // Use the ThinLTO post-link pipeline with sample profiling
1852 if (ThinLTO && PGOOpt && PGOOpt->Action == PGOOptions::SampleUse)
1853 MPM.addPass(buildThinLTODefaultPipeline(Level, /*ImportSummary=*/nullptr));
1854 else {
1855 // ModuleSimplification does not run the coroutine passes for
1856 // ThinLTOPreLink, so we need the coroutine passes to run for ThinLTO
1857 // builds, otherwise they will miscompile.
1858 if (ThinLTO) {
1859 // TODO: replace w/ buildCoroWrapper() when it takes phase and level into
1860 // consideration.
1861 CGSCCPassManager CGPM;
1865 MPM.addPass(CoroCleanupPass());
1866 }
1867
1868 // otherwise, just use module optimization
1869 MPM.addPass(
1871 // Emit annotation remarks.
1873 }
1874
1875 instructionCountersPass(MPM, /* IsPreOptimization */ false);
1876
1877 return MPM;
1878}
1879
1882 if (Level == OptimizationLevel::O0)
1884
1886
1887 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1888
1889 // Convert @llvm.global.annotations to !annotation metadata.
1891
1892 // Force any function attributes we want the rest of the pipeline to observe.
1894
1895 if (PGOOpt && PGOOpt->DebugInfoForProfiling)
1897
1898 // Apply module pipeline start EP callback.
1900
1901 // If we are planning to perform ThinLTO later, we don't bloat the code with
1902 // unrolling/vectorization/... now. Just simplify the module as much as we
1903 // can.
1906 // In pre-link, for ctx prof use, we stop here with an instrumented IR. We let
1907 // thinlto use the contextual info to perform imports; then use the contextual
1908 // profile in the post-thinlink phase.
1909 if (!UseCtxProfile.empty()) {
1910 addRequiredLTOPreLinkPasses(MPM);
1911 return MPM;
1912 }
1913
1914 // Run partial inlining pass to partially inline functions that have
1915 // large bodies.
1916 // FIXME: It isn't clear whether this is really the right place to run this
1917 // in ThinLTO. Because there is another canonicalization and simplification
1918 // phase that will run after the thin link, running this here ends up with
1919 // less information than will be available later and it may grow functions in
1920 // ways that aren't beneficial.
1923
1924 if (PGOOpt && PGOOpt->PseudoProbeForProfiling &&
1925 PGOOpt->Action == PGOOptions::SampleUse)
1927
1928 // Handle Optimizer{Early,Last}EPCallbacks added by clang on PreLink. Actual
1929 // optimization is going to be done in PostLink stage, but clang can't add
1930 // callbacks there in case of in-process ThinLTO called by linker.
1935
1936 // Emit annotation remarks.
1938
1939 // Attach !implicit.ref metadata from all functions to copyright strings.
1941
1942 addRequiredLTOPreLinkPasses(MPM);
1943
1944 instructionCountersPass(MPM, /* IsPreOptimization */ false);
1945
1946 return MPM;
1947}
1948
1950 OptimizationLevel Level, const ModuleSummaryIndex *ImportSummary) {
1952
1953 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1954
1956
1957 // If we are invoking this without a summary index noting that we are linking
1958 // with a library containing the necessary APIs, remove any MemProf related
1959 // attributes and metadata.
1960 if (!ImportSummary || !ImportSummary->withSupportsHotColdNew())
1962
1963 if (ImportSummary) {
1964 // For ThinLTO we must apply the context disambiguation decisions early, to
1965 // ensure we can correctly match the callsites to summary data.
1968 ImportSummary, PGOOpt && PGOOpt->Action == PGOOptions::SampleUse));
1969
1970 // These passes import type identifier resolutions for whole-program
1971 // devirtualization and CFI. They must run early because other passes may
1972 // disturb the specific instruction patterns that these passes look for,
1973 // creating dependencies on resolutions that may not appear in the summary.
1974 //
1975 // For example, GVN may transform the pattern assume(type.test) appearing in
1976 // two basic blocks into assume(phi(type.test, type.test)), which would
1977 // transform a dependency on a WPD resolution into a dependency on a type
1978 // identifier resolution for CFI.
1979 //
1980 // Also, WPD has access to more precise information than ICP and can
1981 // devirtualize more effectively, so it should operate on the IR first.
1982 //
1983 // The WPD and LowerTypeTest passes need to run at -O0 to lower type
1984 // metadata and intrinsics.
1985 MPM.addPass(WholeProgramDevirtPass(nullptr, ImportSummary));
1986 MPM.addPass(LowerTypeTestsPass(nullptr, ImportSummary));
1987 }
1988
1989 if (Level == OptimizationLevel::O0) {
1990 // Run a second time to clean up any type tests left behind by WPD for use
1991 // in ICP.
1994
1995 // AllocToken transforms heap allocation calls; this needs to run late after
1996 // other allocation call transformations (such as those in InstCombine).
1997 MPM.addPass(AllocTokenPass());
1998
1999 // Drop available_externally and unreferenced globals. This is necessary
2000 // with ThinLTO in order to avoid leaving undefined references to dead
2001 // globals in the object file.
2003 MPM.addPass(GlobalDCEPass());
2004
2006
2007 return MPM;
2008 }
2009 if (!UseCtxProfile.empty()) {
2010 MPM.addPass(
2012 } else {
2013 // Add the core simplification pipeline.
2016 }
2017 // Now add the optimization pipeline.
2020
2022
2023 // Emit annotation remarks.
2025
2026 instructionCountersPass(MPM, /* IsPreOptimization */ false);
2027
2028 return MPM;
2029}
2030
2033 // FIXME: We should use a customized pre-link pipeline!
2034 return buildPerModuleDefaultPipeline(Level,
2036}
2037
2040 ModuleSummaryIndex *ExportSummary) {
2042
2043 instructionCountersPass(MPM, /* IsPreOptimization */ true);
2044
2046
2047 // If we are invoking this without a summary index noting that we are linking
2048 // with a library containing the necessary APIs, remove any MemProf related
2049 // attributes and metadata.
2050 if (!ExportSummary || !ExportSummary->withSupportsHotColdNew())
2052
2053 // Create a function that performs CFI checks for cross-DSO calls with targets
2054 // in the current module.
2055 MPM.addPass(CrossDSOCFIPass());
2056
2057 if (Level == OptimizationLevel::O0) {
2058 // The WPD and LowerTypeTest passes need to run at -O0 to lower type
2059 // metadata and intrinsics.
2060 MPM.addPass(WholeProgramDevirtPass(ExportSummary, nullptr));
2061 MPM.addPass(LowerTypeTestsPass(ExportSummary, nullptr));
2062 // Run a second time to clean up any type tests left behind by WPD for use
2063 // in ICP.
2065
2067
2068 // AllocToken transforms heap allocation calls; this needs to run late after
2069 // other allocation call transformations (such as those in InstCombine).
2070 MPM.addPass(AllocTokenPass());
2071
2073
2074 // Emit annotation remarks.
2076
2077 return MPM;
2078 }
2079
2080 if (PGOOpt && PGOOpt->Action == PGOOptions::SampleUse) {
2081 // Load sample profile before running the LTO optimization pipeline.
2082 MPM.addPass(SampleProfileLoaderPass(PGOOpt->ProfileFile,
2083 PGOOpt->ProfileRemappingFile,
2085 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
2086 // RequireAnalysisPass for PSI before subsequent non-module passes.
2088 }
2089
2090 // Try to run OpenMP optimizations, quick no-op if no OpenMP metadata present.
2092
2093 // Remove unused virtual tables to improve the quality of code generated by
2094 // whole-program devirtualization and bitset lowering.
2095 MPM.addPass(GlobalDCEPass(/*InLTOPostLink=*/true));
2096
2097 // Do basic inference of function attributes from known properties of system
2098 // libraries and other oracles.
2100
2101 if (Level >= OptimizationLevel::O2) {
2103 CallSiteSplittingPass(), PTO.EagerlyInvalidateAnalyses));
2104
2105 // Indirect call promotion. This should promote all the targets that are
2106 // left by the earlier promotion pass that promotes intra-module targets.
2107 // This two-step promotion is to save the compile time. For LTO, it should
2108 // produce the same result as if we only do promotion here.
2110 true /* InLTO */, PGOOpt && PGOOpt->Action == PGOOptions::SampleUse));
2111
2112 // Promoting by-reference arguments to by-value exposes more constants to
2113 // IPSCCP.
2114 CGSCCPassManager CGPM;
2117 CGPM.addPass(
2120
2121 // Propagate constants at call sites into the functions they call. This
2122 // opens opportunities for globalopt (and inlining) by substituting function
2123 // pointers passed as arguments to direct uses of functions.
2124 MPM.addPass(IPSCCPPass(IPSCCPOptions(/*AllowFuncSpec=*/true)));
2125
2126 // Attach metadata to indirect call sites indicating the set of functions
2127 // they may target at run-time. This should follow IPSCCP.
2129 }
2130
2131 // Do RPO function attribute inference across the module to forward-propagate
2132 // attributes where applicable.
2133 // FIXME: Is this really an optimization rather than a canonicalization?
2135
2136 // Use in-range annotations on GEP indices to split globals where beneficial.
2137 MPM.addPass(GlobalSplitPass());
2138
2139 // Run whole program optimization of virtual call when the list of callees
2140 // is fixed.
2141 MPM.addPass(WholeProgramDevirtPass(ExportSummary, nullptr));
2142
2144 // Stop here at -O1.
2145 if (Level == OptimizationLevel::O1) {
2147 LowerConstantIntrinsicsPass(), PTO.EagerlyInvalidateAnalyses));
2148
2149 // The LowerTypeTestsPass needs to run to lower type metadata and the
2150 // type.test intrinsics. The pass does nothing if CFI is disabled.
2151 MPM.addPass(LowerTypeTestsPass(ExportSummary, nullptr));
2152 // Run a second time to clean up any type tests left behind by WPD for use
2153 // in ICP (which is performed earlier than this in the regular LTO
2154 // pipeline).
2156
2158
2159 // AllocToken transforms heap allocation calls; this needs to run late after
2160 // other allocation call transformations (such as those in InstCombine).
2161 MPM.addPass(AllocTokenPass());
2162
2164
2165 // Emit annotation remarks.
2167
2168 instructionCountersPass(MPM, /* IsPreOptimization */ false);
2169
2170 return MPM;
2171 }
2172
2173 // TODO: Skip to match buildCoroWrapper.
2174 MPM.addPass(CoroEarlyPass());
2175
2176 // Optimize globals to try and fold them into constants.
2177 MPM.addPass(GlobalOptPass());
2178
2179 // Promote any localized globals to SSA registers.
2181
2182 // Linking modules together can lead to duplicate global constant, only
2183 // keep one copy of each constant.
2185
2186 // Remove unused arguments from functions.
2188
2189 // Reduce the code after globalopt and ipsccp. Both can open up significant
2190 // simplification opportunities, and both can propagate functions through
2191 // function pointers. When this happens, we often have to resolve varargs
2192 // calls, etc, so let instcombine do this.
2193 FunctionPassManager PeepholeFPM;
2194 PeepholeFPM.addPass(InstCombinePass());
2195 if (Level >= OptimizationLevel::O2)
2196 PeepholeFPM.addPass(AggressiveInstCombinePass());
2197 invokePeepholeEPCallbacks(PeepholeFPM, Level);
2198
2199 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(PeepholeFPM),
2200 PTO.EagerlyInvalidateAnalyses));
2201
2202 // Lower variadic functions for supported targets prior to inlining.
2204
2205 // Note: historically, the PruneEH pass was run first to deduce nounwind and
2206 // generally clean up exception handling overhead. It isn't clear this is
2207 // valuable as the inliner doesn't currently care whether it is inlining an
2208 // invoke or a call.
2209 // Run the inliner now.
2210 if (EnableModuleInliner) {
2214 } else {
2217 /* MandatoryFirst */ true,
2220 }
2221
2222 // Perform context disambiguation after inlining, since that would reduce the
2223 // amount of additional cloning required to distinguish the allocation
2224 // contexts.
2227 /*Summary=*/nullptr,
2228 PGOOpt && PGOOpt->Action == PGOOptions::SampleUse));
2229
2230 // Optimize globals again after we ran the inliner.
2231 MPM.addPass(GlobalOptPass());
2232
2233 // Run the OpenMPOpt pass again after global optimizations.
2235
2236 // Garbage collect dead functions.
2237 MPM.addPass(GlobalDCEPass(/*InLTOPostLink=*/true));
2238
2239 // If we didn't decide to inline a function, check to see if we can
2240 // transform it to pass arguments by value instead of by reference.
2241 CGSCCPassManager CGPM;
2247
2249 // The IPO Passes may leave cruft around. Clean up after them.
2250 FPM.addPass(InstCombinePass());
2251 invokePeepholeEPCallbacks(FPM, Level);
2252
2255
2257
2258 // Do a post inline PGO instrumentation and use pass. This is a context
2259 // sensitive PGO pass.
2260 if (PGOOpt) {
2261 if (PGOOpt->CSAction == PGOOptions::CSIRInstr)
2262 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/true,
2263 /*IsCS=*/true, PGOOpt->AtomicCounterUpdate,
2264 PGOOpt->CSProfileGenFile, PGOOpt->ProfileRemappingFile);
2265 else if (PGOOpt->CSAction == PGOOptions::CSIRUse)
2266 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/false,
2267 /*IsCS=*/true, PGOOpt->AtomicCounterUpdate,
2268 PGOOpt->ProfileFile, PGOOpt->ProfileRemappingFile);
2269 }
2270
2271 // Break up allocas
2273
2274 // LTO provides additional opportunities for tailcall elimination due to
2275 // link-time inlining, and visibility of nocapture attribute.
2276 FPM.addPass(
2277 TailCallElimPass(/*UpdateFunctionEntryCount=*/isInstrumentedPGOUse()));
2278
2279 // Run a few AA driver optimizations here and now to cleanup the code.
2280 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(FPM),
2281 PTO.EagerlyInvalidateAnalyses));
2282
2283 MPM.addPass(
2285
2286 // Require the GlobalsAA analysis for the module so we can query it within
2287 // MainFPM.
2290 // Invalidate AAManager so it can be recreated and pick up the newly
2291 // available GlobalsAA.
2292 MPM.addPass(
2294 }
2295
2296 FunctionPassManager MainFPM;
2298 LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
2299 /*AllowSpeculation=*/true),
2300 /*USeMemorySSA=*/true));
2301
2302 if (RunNewGVN)
2303 MainFPM.addPass(NewGVNPass());
2304 else
2305 MainFPM.addPass(GVNPass());
2306
2307 // Remove dead memcpy()'s.
2308 MainFPM.addPass(MemCpyOptPass());
2309
2310 // Nuke dead stores.
2311 MainFPM.addPass(DSEPass());
2312 MainFPM.addPass(MoveAutoInitPass());
2314
2316
2317 invokeVectorizerStartEPCallbacks(MainFPM, Level);
2318
2319 LoopPassManager LPM;
2321 LPM.addPass(LoopFlattenPass());
2322 LPM.addPass(IndVarSimplifyPass());
2323 LPM.addPass(LoopDeletionPass());
2324 // FIXME: Add loop interchange.
2325
2326 // Unroll small loops and perform peeling.
2327 LPM.addPass(LoopFullUnrollPass(static_cast<int>(Level),
2328 /* OnlyWhenForced= */ !PTO.LoopUnrolling,
2329 PTO.ForgetAllSCEVInLoopUnroll));
2330 // The loop passes in LPM (LoopFullUnrollPass) do not preserve MemorySSA.
2331 // *All* loop passes must preserve it, in order to be able to use it.
2332 MainFPM.addPass(
2333 createFunctionToLoopPassAdaptor(std::move(LPM), /*UseMemorySSA=*/false));
2334
2335 MainFPM.addPass(LoopDistributePass());
2336
2337 addVectorPasses(Level, MainFPM, ThinOrFullLTOPhase::FullLTOPostLink);
2338
2339 invokeVectorizerEndEPCallbacks(MainFPM, Level);
2340
2341 // Run the OpenMPOpt CGSCC pass again late.
2344
2345 invokePeepholeEPCallbacks(MainFPM, Level);
2346 MainFPM.addPass(JumpThreadingPass());
2347 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(MainFPM),
2348 PTO.EagerlyInvalidateAnalyses));
2349
2350 // Lower type metadata and the type.test intrinsic. This pass supports
2351 // clang's control flow integrity mechanisms (-fsanitize=cfi*) and needs
2352 // to be run at link time if CFI is enabled. This pass does nothing if
2353 // CFI is disabled.
2354 MPM.addPass(LowerTypeTestsPass(ExportSummary, nullptr));
2355 // Run a second time to clean up any type tests left behind by WPD for use
2356 // in ICP (which is performed earlier than this in the regular LTO pipeline).
2358
2359 // Enable splitting late in the FullLTO post-link pipeline.
2362
2363 // Add late LTO optimization passes.
2364 FunctionPassManager LateFPM;
2365
2366 // LoopSink pass sinks instructions hoisted by LICM, which serves as a
2367 // canonicalization pass that enables other optimizations. As a result,
2368 // LoopSink pass needs to be a very late IR pass to avoid undoing LICM
2369 // result too early.
2370 LateFPM.addPass(LoopSinkPass());
2371
2372 // This hoists/decomposes div/rem ops. It should run after other sink/hoist
2373 // passes to avoid re-sinking, but before SimplifyCFG because it can allow
2374 // flattening of blocks.
2375 LateFPM.addPass(DivRemPairsPass());
2376
2377 // Delete basic blocks, which optimization passes may have killed.
2379 .convertSwitchRangeToICmp(true)
2380 .convertSwitchToArithmetic(true)
2381 .hoistCommonInsts(true)
2382 .speculateUnpredictables(true)));
2383 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(LateFPM)));
2384
2385 // Drop bodies of available eternally objects to improve GlobalDCE.
2387
2388 // Now that we have optimized the program, discard unreachable functions.
2389 MPM.addPass(GlobalDCEPass(/*InLTOPostLink=*/true));
2390
2391 if (PTO.MergeFunctions)
2393
2395
2396 if (PTO.CallGraphProfile)
2397 MPM.addPass(CGProfilePass(/*InLTOPostLink=*/true));
2398
2399 MPM.addPass(CoroCleanupPass());
2400
2401 // AllocToken transforms heap allocation calls; this needs to run late after
2402 // other allocation call transformations (such as those in InstCombine).
2403 MPM.addPass(AllocTokenPass());
2404
2406
2407 // Emit annotation remarks.
2409
2410 instructionCountersPass(MPM, /* IsPreOptimization */ false);
2411
2412 return MPM;
2413}
2414
2418 assert(Level == OptimizationLevel::O0 &&
2419 "buildO0DefaultPipeline should only be used with O0");
2420
2422
2423 instructionCountersPass(MPM, /* IsPreOptimization */ true);
2424
2425 // Perform pseudo probe instrumentation in O0 mode. This is for the
2426 // consistency between different build modes. For example, a LTO build can be
2427 // mixed with an O0 prelink and an O2 postlink. Loading a sample profile in
2428 // the postlink will require pseudo probe instrumentation in the prelink.
2429 if (PGOOpt && PGOOpt->PseudoProbeForProfiling)
2431
2432 if (PGOOpt && (PGOOpt->Action == PGOOptions::IRInstr ||
2433 PGOOpt->Action == PGOOptions::IRUse))
2435 MPM,
2436 /*RunProfileGen=*/(PGOOpt->Action == PGOOptions::IRInstr),
2437 /*IsCS=*/false, PGOOpt->AtomicCounterUpdate, PGOOpt->ProfileFile,
2438 PGOOpt->ProfileRemappingFile);
2439
2440 // Instrument function entry and exit before all inlining.
2442 EntryExitInstrumenterPass(/*PostInlining=*/false)));
2443
2445
2446 if (PGOOpt && PGOOpt->DebugInfoForProfiling)
2448
2449 if (PGOOpt && PGOOpt->Action == PGOOptions::SampleUse) {
2450 // Explicitly disable sample loader inlining and use flattened profile in O0
2451 // pipeline.
2452 MPM.addPass(SampleProfileLoaderPass(PGOOpt->ProfileFile,
2453 PGOOpt->ProfileRemappingFile,
2455 /*DisableSampleProfileInlining=*/true,
2456 /*UseFlattenedProfile=*/true));
2457 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
2458 // RequireAnalysisPass for PSI before subsequent non-module passes.
2460 }
2461
2463
2464 // Build a minimal pipeline based on the semantics required by LLVM,
2465 // which is just that always inlining occurs. Further, disable generating
2466 // lifetime intrinsics to avoid enabling further optimizations during
2467 // code generation.
2469 /*InsertLifetimeIntrinsics=*/false));
2470
2471 if (PTO.MergeFunctions)
2473
2474 if (EnableMatrix)
2475 MPM.addPass(
2477
2478 if (!CGSCCOptimizerLateEPCallbacks.empty()) {
2479 CGSCCPassManager CGPM;
2481 if (!CGPM.isEmpty())
2483 }
2484 if (!LateLoopOptimizationsEPCallbacks.empty()) {
2485 LoopPassManager LPM;
2487 if (!LPM.isEmpty()) {
2489 createFunctionToLoopPassAdaptor(std::move(LPM))));
2490 }
2491 }
2492 if (!LoopOptimizerEndEPCallbacks.empty()) {
2493 LoopPassManager LPM;
2495 if (!LPM.isEmpty()) {
2497 createFunctionToLoopPassAdaptor(std::move(LPM))));
2498 }
2499 }
2500 if (!ScalarOptimizerLateEPCallbacks.empty()) {
2503 if (!FPM.isEmpty())
2504 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(FPM)));
2505 }
2506
2508
2509 if (!VectorizerStartEPCallbacks.empty()) {
2512 if (!FPM.isEmpty())
2513 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(FPM)));
2514 }
2515
2516 if (!VectorizerEndEPCallbacks.empty()) {
2519 if (!FPM.isEmpty())
2520 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(FPM)));
2521 }
2522
2524
2525 // AllocToken transforms heap allocation calls; this needs to run late after
2526 // other allocation call transformations (such as those in InstCombine).
2527 if (!isLTOPreLink(Phase))
2528 MPM.addPass(AllocTokenPass());
2529
2531
2533 MPM.addPass(InstrumentorPass(FS));
2534
2535 // Attach !implicit.ref metadata from all functions to copyright strings.
2537
2538 if (isLTOPreLink(Phase))
2539 addRequiredLTOPreLinkPasses(MPM);
2540
2541 // Emit annotation remarks.
2543
2544 instructionCountersPass(MPM, /* IsPreOptimization */ false);
2545
2546 return MPM;
2547}
2548
2550 AAManager AA;
2551
2552 // The order in which these are registered determines their priority when
2553 // being queried.
2554
2555 // Add any target-specific alias analyses that should be run early.
2556 if (TM)
2557 TM->registerEarlyDefaultAliasAnalyses(AA);
2558
2559 // First we register the basic alias analysis that provides the majority of
2560 // per-function local AA logic. This is a stateless, on-demand local set of
2561 // AA techniques.
2562 AA.registerFunctionAnalysis<BasicAA>();
2563
2564 // Next we query fast, specialized alias analyses that wrap IR-embedded
2565 // information about aliasing.
2566 AA.registerFunctionAnalysis<ScopedNoAliasAA>();
2567 AA.registerFunctionAnalysis<TypeBasedAA>();
2568
2569 // Add support for querying global aliasing information when available.
2570 // Because the `AAManager` is a function analysis and `GlobalsAA` is a module
2571 // analysis, all that the `AAManager` can do is query for any *cached*
2572 // results from `GlobalsAA` through a readonly proxy.
2574 AA.registerModuleAnalysis<GlobalsAA>();
2575
2576 // Add target-specific alias analyses.
2577 if (TM)
2578 TM->registerDefaultAliasAnalyses(AA);
2579
2580 return AA;
2581}
2582
2583bool PassBuilder::isInstrumentedPGOUse() const {
2584 return (PGOOpt && PGOOpt->Action == PGOOptions::IRUse) ||
2585 !UseCtxProfile.empty();
2586}
aarch64 falkor hwpf fix Falkor HW Prefetch Fix Late Phase
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AggressiveInstCombiner - Combine expression patterns to form expressions with fewer,...
Provides passes to inlining "always_inline" functions.
This is the interface for LLVM's primary stateless and local alias analysis.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
This file provides the interface for LLVM's Call Graph Profile pass.
This header provides classes for managing passes over SCCs of the call graph.
#define clEnumValN(ENUMVAL, FLAGNAME, DESC)
This file provides the interface for a simple, fast CSE pass.
This file provides a pass which clones the current module and runs the provided pass pipeline on the ...
This file provides a pass manager that only runs its passes if the provided marker analysis has been ...
Super simple passes to force specific function attrs from the commandline into the IR for debugging p...
Provides passes for computing function attributes based on interprocedural analyses.
This file provides the interface for the GVNHoist pass.
This file provides the interface for the GVNSink pass.
This file provides the interface for LLVM's Global Value Numbering pass which eliminates fully redund...
This is the interface for a simple mod/ref and alias analysis over globals.
AcceleratorCodeSelection - Identify all functions reachable from a kernel, removing those that are un...
This header defines various interfaces for pass management in LLVM.
Interfaces for passes which infer implicit function attributes from the name and signature of functio...
This file provides the primary interface to the instcombine pass.
Defines passes for running instruction simplification across chunks of IR.
This file provides the interface for LLVM's PGO Instrumentation lowering pass.
See the comments on JumpThreadingPass.
static LVOptions Options
Definition LVOptions.cpp:25
This file implements the Loop Fusion pass.
This header defines the LoopLoadEliminationPass object.
This header provides classes for managing a pipeline of passes over loops in LLVM IR.
The header file for the LowerConstantIntrinsics pass as used by the new pass manager.
The header file for the LowerExpectIntrinsic pass as used by the new pass manager.
This pass performs merges of loads and stores on both sides of a.
This file provides the interface for LLVM's Global Value Numbering pass.
This header enumerates the LLVM-provided high-level optimization levels.
This file provides the interface for IR based instrumentation passes ( (profile-gen,...
Define option tunables for PGO.
ppc ctr loops PowerPC CTR Loops Verify
static bool isThinLTOPostLink(ThinOrFullLTOPhase Phase)
static void addAnnotationRemarksPass(ModulePassManager &MPM)
static CoroConditionalWrapper buildCoroWrapper(ThinOrFullLTOPhase Phase)
static bool isFullLTOPostLink(ThinOrFullLTOPhase Phase)
static bool isThinLTOPreLink(ThinOrFullLTOPhase Phase)
static bool isLTOPreLink(ThinOrFullLTOPhase Phase)
static void instructionCountersPass(ModulePassManager &MPM, bool IsPreOptimization)
static bool isFullLTOPreLink(ThinOrFullLTOPhase Phase)
static bool isLTOPostLink(ThinOrFullLTOPhase Phase)
This file implements relative lookup table converter that converts lookup tables to relative lookup t...
This file provides the interface for LLVM's Scalar Replacement of Aggregates pass.
This file provides the interface for the pseudo probe implementation for AutoFDO.
This file provides the interface for the sampled PGO loader pass.
This is the interface for a metadata-based scoped no-alias analysis.
This file provides the interface for the pass responsible for both simplifying and canonicalizing the...
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
This is the interface for a metadata-based TBAA.
A manager for alias analyses.
A module pass that rewrites heap allocations to use token-enabled allocation functions based on vario...
Definition AllocToken.h:36
Inlines functions marked as "always_inline".
Argument promotion pass.
Analysis pass providing a never-invalidated alias analysis result.
Simple pass that canonicalizes aliases.
A pass that merges duplicate global constants into a single constant.
This class implements a trivial dead store elimination.
Eliminate dead arguments (and return values) from functions.
A pass that transforms external global definitions into declarations.
Pass embeds a copy of the module optimized with the provided pass pipeline into a global variable.
A pass manager to run a set of extra loop passes if the MarkerTy analysis is present.
Statistics pass for the FunctionPropertiesAnalysis results.
Pass to remove unused function declarations.
Definition GlobalDCE.h:38
Optimize globals that never have their address taken.
Definition GlobalOpt.h:25
Pass to perform split of global variables.
Definition GlobalSplit.h:26
Analysis pass providing a never-invalidated alias analysis result.
Pass to outline cold regions.
Pass to perform interprocedural constant propagation.
Definition SCCP.h:48
Run instruction simplification across each instruction in the function.
Instrumentation based profiling lowering pass.
The Instrumentor pass.
This pass performs 'jump threading', which looks at blocks that have multiple predecessors and multip...
Performs Loop Invariant Code Motion Pass.
Definition LICM.h:66
Loop unroll pass that only does full loop unrolling and peeling.
Performs Loop Idiom Recognize Pass.
Performs Loop Inst Simplify Pass.
A simple loop rotation transformation.
Performs basic CFG simplifications to assist other loop passes.
A pass that does profile-guided sinking of instructions into loops.
Definition LoopSink.h:33
A simple loop rotation transformation.
Loop unroll pass that will support both full and partial unrolling.
Strips MemProf attributes and metadata.
Merge identical functions.
The module inliner pass for the new pass manager.
Module pass, wrapping the inliner pass.
Definition Inliner.h:65
void addModulePass(T Pass)
Add a module pass that runs before the CGSCC passes.
Definition Inliner.h:81
Class to hold module path string table and global value map, and encapsulate methods for operating on...
Simple pass that provides a name to every anonymous globals.
Additional 'norecurse' attribute deduction during postlink LTO phase.
OpenMP optimizations pass.
Definition OpenMPOpt.h:42
static LLVM_ABI bool isCtxIRPGOInstrEnabled()
The indirect function call promotion pass.
The instrumentation (profile-instr-gen) pass for IR based PGO.
The instrumentation (profile-instr-gen) pass for IR based PGO.
The profile annotation (profile-instr-use) pass for IR based PGO.
The profile size based optimization pass for memory intrinsics.
Pass to remove unused function declarations.
LLVM_ABI void invokeFullLinkTimeOptimizationLastEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level)
LLVM_ABI ModuleInlinerWrapperPass buildInlinerPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase)
Construct the module pipeline that performs inlining as well as the inlining-driven cleanups.
LLVM_ABI void invokeOptimizerEarlyEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level, ThinOrFullLTOPhase Phase)
LLVM_ABI ModulePassManager buildFatLTODefaultPipeline(OptimizationLevel Level, bool ThinLTO, bool EmitSummary, bool Verify=true)
Build a fat object default optimization pipeline.
LLVM_ABI void invokeVectorizerStartEPCallbacks(FunctionPassManager &FPM, OptimizationLevel Level)
LLVM_ABI AAManager buildDefaultAAPipeline()
Build the default AAManager with the default alias analysis pipeline registered.
LLVM_ABI void invokeCGSCCOptimizerLateEPCallbacks(CGSCCPassManager &CGPM, OptimizationLevel Level)
LLVM_ABI ModulePassManager buildThinLTOPreLinkDefaultPipeline(OptimizationLevel Level)
Build a pre-link, ThinLTO-targeting default optimization pipeline to a pass manager.
LLVM_ABI void addPGOInstrPassesForO0(ModulePassManager &MPM, bool RunProfileGen, bool IsCS, bool AtomicCounterUpdate, std::string ProfileFile, std::string ProfileRemappingFile)
Add PGOInstrumenation passes for O0 only.
LLVM_ABI void invokeScalarOptimizerLateEPCallbacks(FunctionPassManager &FPM, OptimizationLevel Level)
LLVM_ABI ModulePassManager buildPerModuleDefaultPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase=ThinOrFullLTOPhase::None)
Build a per-module default optimization pipeline.
LLVM_ABI void invokePipelineStartEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level)
LLVM_ABI void invokeVectorizerEndEPCallbacks(FunctionPassManager &FPM, OptimizationLevel Level)
LLVM_ABI void invokeThinLinkTimeOptimizationLastEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level)
LLVM_ABI ModulePassManager buildO0DefaultPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase=ThinOrFullLTOPhase::None)
Build an O0 pipeline with the minimal semantically required passes.
LLVM_ABI FunctionPassManager buildFunctionSimplificationPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase)
Construct the core LLVM function canonicalization and simplification pipeline.
LLVM_ABI void invokePeepholeEPCallbacks(FunctionPassManager &FPM, OptimizationLevel Level)
LLVM_ABI void invokePipelineEarlySimplificationEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level, ThinOrFullLTOPhase Phase)
LLVM_ABI void invokeLoopOptimizerEndEPCallbacks(LoopPassManager &LPM, OptimizationLevel Level)
LLVM_ABI ModulePassManager buildLTODefaultPipeline(OptimizationLevel Level, ModuleSummaryIndex *ExportSummary)
Build an LTO default optimization pipeline to a pass manager.
LLVM_ABI ModulePassManager buildModuleInlinerPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase)
Construct the module pipeline that performs inlining with module inliner pass.
LLVM_ABI ModulePassManager buildThinLTODefaultPipeline(OptimizationLevel Level, const ModuleSummaryIndex *ImportSummary)
Build a ThinLTO default optimization pipeline to a pass manager.
LLVM_ABI void invokeLateLoopOptimizationsEPCallbacks(LoopPassManager &LPM, OptimizationLevel Level)
LLVM_ABI void invokeFullLinkTimeOptimizationEarlyEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level)
LLVM_ABI ModulePassManager buildModuleSimplificationPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase)
Construct the core LLVM module canonicalization and simplification pipeline.
LLVM_ABI ModulePassManager buildModuleOptimizationPipeline(OptimizationLevel Level, ThinOrFullLTOPhase LTOPhase)
Construct the core LLVM module optimization pipeline.
LLVM_ABI void invokeThinLinkTimeOptimizationEarlyEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level)
LLVM_ABI void invokeOptimizerLastEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level, ThinOrFullLTOPhase Phase)
LLVM_ABI ModulePassManager buildLTOPreLinkDefaultPipeline(OptimizationLevel Level)
Build a pre-link, LTO-targeting default optimization pipeline to a pass manager.
LLVM_ATTRIBUTE_MINSIZE std::enable_if_t<!std::is_same_v< PassT, PassManager > > addPass(PassT &&Pass)
bool isEmpty() const
Returns if the pass manager contains any passes.
unsigned LicmMssaNoAccForPromotionCap
Tuning option to disable promotion to scalars in LICM with MemorySSA, if the number of access is too ...
Definition PassBuilder.h:78
bool SLPVectorization
Tuning option to enable/disable slp loop vectorization, set based on opt level.
Definition PassBuilder.h:56
int InlinerThreshold
Tuning option to override the default inliner threshold.
Definition PassBuilder.h:92
bool LoopFusion
Tuning option to enable/disable loop fusion. Its default value is false.
Definition PassBuilder.h:66
bool CallGraphProfile
Tuning option to enable/disable call graph profile.
Definition PassBuilder.h:82
bool MergeFunctions
Tuning option to enable/disable function merging.
Definition PassBuilder.h:89
bool ForgetAllSCEVInLoopUnroll
Tuning option to forget all SCEV loops in LoopUnroll.
Definition PassBuilder.h:70
unsigned LicmMssaOptCap
Tuning option to cap the number of calls to retrive clobbering accesses in MemorySSA,...
Definition PassBuilder.h:74
bool LoopInterleaving
Tuning option to set loop interleaving on/off, set based on opt level.
Definition PassBuilder.h:48
LLVM_ABI PipelineTuningOptions()
Constructor sets pipeline tuning defaults based on cl::opts.
bool LoopUnrolling
Tuning option to enable/disable loop unrolling. Its default value is true.
Definition PassBuilder.h:59
bool LoopInterchange
Tuning option to enable/disable loop interchange.
Definition PassBuilder.h:63
bool LoopVectorization
Tuning option to enable/disable loop vectorization, set based on opt level.
Definition PassBuilder.h:52
Reassociate commutative expressions.
Definition Reassociate.h:75
A pass to do RPO deduction and propagation of function attributes.
This pass performs function-level constant propagation and merging.
Definition SCCP.h:30
The sample profiler data loader pass.
Analysis pass providing a never-invalidated alias analysis result.
This pass transforms loops that contain branches or switches on loop- invariant conditions to have mu...
A pass to simplify and canonicalize the CFG of a function.
Definition SimplifyCFG.h:30
Analysis pass providing a never-invalidated alias analysis result.
Optimize scalar/vector interactions in IR using target cost models.
Create a verifier pass.
Definition Verifier.h:134
Interfaces for registering analysis passes, producing common pass manager configurations,...
Abstract Attribute helper functions.
Definition Attributor.h:165
ValuesClass values(OptsTy... Options)
Helper to build a ValuesClass by forwarding a variable number of arguments as an initializer list to ...
initializer< Ty > init(const Ty &Val)
@ All
Drop only llvm.assumes using type test value.
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI cl::opt< bool > EnableKnowledgeRetention
static cl::opt< bool > RunNewGVN("enable-newgvn", cl::init(false), cl::Hidden, cl::desc("Run the NewGVN pass"))
static cl::opt< bool > DisablePreInliner("disable-preinline", cl::init(false), cl::Hidden, cl::desc("Disable pre-instrumentation inliner"))
static cl::opt< bool > EnableDFAJumpThreading("enable-dfa-jump-thread", cl::desc("Enable DFA jump threading"), cl::init(true), cl::Hidden)
static cl::opt< bool > PerformMandatoryInliningsFirst("mandatory-inlining-first", cl::init(false), cl::Hidden, cl::desc("Perform mandatory inlinings module-wide, before performing " "inlining"))
static cl::opt< bool > RunPartialInlining("enable-partial-inlining", cl::init(false), cl::Hidden, cl::desc("Run Partial inlining pass"))
static cl::opt< bool > EnableGVNSink("enable-gvn-sink", cl::desc("Enable the GVN sinking pass (default = off)"))
static cl::opt< bool > EnableModuleInliner("enable-module-inliner", cl::init(false), cl::Hidden, cl::desc("Enable module inliner"))
static cl::opt< bool > EnableEagerlyInvalidateAnalyses("eagerly-invalidate-analyses", cl::init(true), cl::Hidden, cl::desc("Eagerly invalidate more analyses in default pipelines"))
static cl::opt< bool > EnableMatrix("enable-matrix", cl::init(false), cl::Hidden, cl::desc("Enable lowering of the matrix intrinsics"))
ModuleToFunctionPassAdaptor createModuleToFunctionPassAdaptor(FunctionPassT &&Pass, bool EagerlyInvalidate=false)
A function to deduce a function pass type and wrap it in the templated adaptor.
cl::opt< std::string > UseCtxProfile("use-ctx-profile", cl::init(""), cl::Hidden, cl::desc("Use the specified contextual profile file"))
static cl::opt< bool > EnableSampledInstr("enable-sampled-instrumentation", cl::init(false), cl::Hidden, cl::desc("Enable profile instrumentation sampling (default = off)"))
static cl::opt< bool > EnableLoopFlatten("enable-loop-flatten", cl::init(false), cl::Hidden, cl::desc("Enable the LoopFlatten Pass"))
@ O1
Optimize quickly without destroying debuggability.
@ O0
Disable as many optimizations as possible.
@ O3
Optimize for fast execution as much as possible.
@ O2
Optimize for fast execution as much as possible without triggering significant incremental compile ti...
static cl::opt< InliningAdvisorMode > UseInlineAdvisor("enable-ml-inliner", cl::init(InliningAdvisorMode::Default), cl::Hidden, cl::desc("Enable ML policy for inliner. Currently trained for -Oz only"), cl::values(clEnumValN(InliningAdvisorMode::Default, "default", "Heuristics-based inliner version"), clEnumValN(InliningAdvisorMode::Development, "development", "Use development mode (runtime-loadable model)"), clEnumValN(InliningAdvisorMode::Release, "release", "Use release mode (AOT-compiled model)")))
static cl::opt< bool > EnableJumpTableToSwitch("enable-jump-table-to-switch", cl::init(true), cl::desc("Enable JumpTableToSwitch pass (default = true)"))
PassManager< LazyCallGraph::SCC, CGSCCAnalysisManager, LazyCallGraph &, CGSCCUpdateResult & > CGSCCPassManager
The CGSCC pass manager.
static cl::opt< bool > EnableUnrollAndJam("enable-unroll-and-jam", cl::init(false), cl::Hidden, cl::desc("Enable Unroll And Jam Pass"))
@ CGSCC_LIGHT
@ MODULE_LIGHT
ThinOrFullLTOPhase
This enumerates the LLVM full LTO or ThinLTO optimization phases.
Definition Pass.h:77
@ FullLTOPreLink
Full LTO prelink phase.
Definition Pass.h:85
@ ThinLTOPostLink
ThinLTO postlink (backend compile) phase.
Definition Pass.h:83
@ None
No LTO/ThinLTO behavior needed.
Definition Pass.h:79
@ FullLTOPostLink
Full LTO postlink (backend compile) phase.
Definition Pass.h:87
@ ThinLTOPreLink
ThinLTO prelink (summary) phase.
Definition Pass.h:81
PassManager< Loop, LoopAnalysisManager, LoopStandardAnalysisResults &, LPMUpdater & > LoopPassManager
The Loop pass manager.
static cl::opt< bool > EnableConstraintElimination("enable-constraint-elimination", cl::init(true), cl::Hidden, cl::desc("Enable pass to eliminate conditions based on linear constraints"))
ModuleToPostOrderCGSCCPassAdaptor createModuleToPostOrderCGSCCPassAdaptor(CGSCCPassT &&Pass)
A function to deduce a function pass type and wrap it in the templated adaptor.
static cl::opt< bool > EnablePGOInlineDeferral("enable-npm-pgo-inline-deferral", cl::init(true), cl::Hidden, cl::desc("Enable inline deferral during PGO"))
Flag to enable inline deferral during PGO.
FunctionToLoopPassAdaptor createFunctionToLoopPassAdaptor(LoopPassT &&Pass, bool UseMemorySSA=false)
A function to deduce a loop pass type and wrap it in the templated adaptor.
CGSCCToFunctionPassAdaptor createCGSCCToFunctionPassAdaptor(FunctionPassT &&Pass, bool EagerlyInvalidate=false, bool NoRerun=false)
A function to deduce a function pass type and wrap it in the templated adaptor.
LLVM_ABI cl::opt< bool > ForgetSCEVInLoopUnroll
PassManager< Module > ModulePassManager
Convenience typedef for a pass manager over modules.
static cl::opt< bool > EnablePostPGOLoopRotation("enable-post-pgo-loop-rotation", cl::init(true), cl::Hidden, cl::desc("Run the loop rotation transformation after PGO instrumentation"))
LLVM_ABI bool AreStatisticsEnabled()
Check if statistics are enabled.
static cl::opt< std::string > InstrumentColdFuncOnlyPath("instrument-cold-function-only-path", cl::init(""), cl::desc("File path for cold function only instrumentation(requires use " "with --pgo-instrument-cold-function-only)"), cl::Hidden)
static cl::opt< bool > EnableGlobalAnalyses("enable-global-analyses", cl::init(true), cl::Hidden, cl::desc("Enable inter-procedural analyses"))
static cl::opt< bool > FlattenedProfileUsed("flattened-profile-used", cl::init(false), cl::Hidden, cl::desc("Indicate the sample profile being used is flattened, i.e., " "no inline hierarchy exists in the profile"))
static cl::opt< AttributorRunOption > AttributorRun("attributor-enable", cl::Hidden, cl::init(AttributorRunOption::NONE), cl::desc("Enable the attributor inter-procedural deduction pass"), cl::values(clEnumValN(AttributorRunOption::FULL, "full", "enable all full attributor runs"), clEnumValN(AttributorRunOption::LIGHT, "light", "enable all attributor-light runs"), clEnumValN(AttributorRunOption::MODULE, "module", "enable module-wide attributor runs"), clEnumValN(AttributorRunOption::MODULE_LIGHT, "module-light", "enable module-wide attributor-light runs"), clEnumValN(AttributorRunOption::CGSCC, "cgscc", "enable call graph SCC attributor runs"), clEnumValN(AttributorRunOption::CGSCC_LIGHT, "cgscc-light", "enable call graph SCC attributor-light runs"), clEnumValN(AttributorRunOption::NONE, "none", "disable attributor runs")))
static cl::opt< bool > EnableLoopInterchange("enable-loopinterchange", cl::init(true), cl::Hidden, cl::desc("Enable the LoopInterchange Pass"))
static cl::opt< bool > ExtraVectorizerPasses("extra-vectorizer-passes", cl::init(false), cl::Hidden, cl::desc("Run cleanup optimization passes after vectorization"))
static cl::opt< bool > EnableHotColdSplit("hot-cold-split", cl::desc("Enable hot-cold splitting pass"))
cl::opt< bool > EnableMemProfContextDisambiguation
Enable MemProf context disambiguation for thin link.
static cl::opt< bool > TriggerCrash("opt-pipeline-trigger-crash", cl::init(false), cl::Hidden, cl::desc("Trigger crash in optimization pipeline"))
PassManager< Function > FunctionPassManager
Convenience typedef for a pass manager over functions.
LLVM_ABI InlineParams getInlineParams()
Generate the parameters to tune the inline cost analysis based only on the commandline options.
cl::opt< bool > PGOInstrumentColdFunctionOnly
static cl::opt< bool > EnableCHR("enable-chr", cl::init(true), cl::Hidden, cl::desc("Enable control height reduction optimization (CHR)"))
static cl::opt< bool > EnableMergeFunctions("enable-merge-functions", cl::init(false), cl::Hidden, cl::desc("Enable function merging as part of the optimization pipeline"))
static cl::opt< bool > EnableDevirtualizeSpeculatively("enable-devirtualize-speculatively", cl::desc("Enable speculative devirtualization optimization"), cl::init(false))
static cl::opt< bool > EnableGVNHoist("enable-gvn-hoist", cl::desc("Enable the GVN hoisting pass (default = off)"))
LLVM_ABI cl::opt< unsigned > SetLicmMssaNoAccForPromotionCap
LLVM_ABI InlineParams getInlineParamsFromOptLevel(unsigned OptLevel)
Generate the parameters to tune the inline cost analysis based on command line options.
static cl::opt< int > PreInlineThreshold("preinline-threshold", cl::Hidden, cl::init(75), cl::desc("Control the amount of inlining in pre-instrumentation inliner " "(default = 75)"))
static cl::opt< bool > UseLoopVersioningLICM("enable-loop-versioning-licm", cl::init(false), cl::Hidden, cl::desc("Enable the experimental Loop Versioning LICM pass"))
cl::opt< unsigned > MaxDevirtIterations("max-devirt-iterations", cl::ReallyHidden, cl::init(4))
LLVM_ABI cl::opt< unsigned > SetLicmMssaOptCap
static cl::opt< bool > EnableInstrumentor("enable-instrumentor", cl::init(false), cl::Hidden, cl::desc("Enable the Instrumentor Pass"))
static cl::opt< bool > EnableMergeICmps("enable-mergeicmps", cl::init(true), cl::Hidden, cl::desc("Enable MergeICmps pass in the optimization pipeline"))
A DCE pass that assumes instructions are dead until proven otherwise.
Definition ADCE.h:31
Pass to convert @llvm.global.annotations to !annotation metadata.
This pass attempts to minimize the number of assume without loosing any information.
A more lightweight version of the Attributor which only runs attribute inference but no simplificatio...
A more lightweight version of the Attributor which only runs attribute inference but no simplificatio...
Hoist/decompose integer division and remainder instructions to enable CFG improvements and better cod...
Definition DivRemPairs.h:23
A simple and fast domtree-based CSE pass.
Definition EarlyCSE.h:31
Pass which forces specific function attributes into the IR, primarily as a debugging tool.
A simple and fast domtree-based GVN pass to hoist common expressions from sibling branches.
Definition GVNHoist.h:23
Uses an "inverted" value numbering to decide the similarity of expressions and sinks similar expressi...
Definition GVNSink.h:23
A set of parameters to control various transforms performed by IPSCCP pass.
Definition SCCP.h:35
A pass which infers function attributes from the names and signatures of function declarations in a m...
Provides context on when an inline advisor is constructed in the pipeline (e.g., link phase,...
Thresholds to tune inline cost analysis.
Definition InlineCost.h:207
std::optional< int > OptSizeHintThreshold
Threshold to use for callees with inline hint, when the caller is optimized for size.
Definition InlineCost.h:216
std::optional< int > HotCallSiteThreshold
Threshold to use when the callsite is considered hot.
Definition InlineCost.h:228
int DefaultThreshold
The default threshold to start with for a callee.
Definition InlineCost.h:209
std::optional< bool > EnableDeferral
Indicate whether we should allow inline deferral.
Definition InlineCost.h:241
std::optional< int > HintThreshold
Threshold to use for callees with inline hint.
Definition InlineCost.h:212
Options for the frontend instrumentation based profiling pass.
A no-op pass template which simply forces a specific analysis result to be invalidated.
Pass to forward loads in a loop around the backedge to subsequent iterations.
A set of parameters used to control various transforms performed by the LoopUnroll pass.
The LoopVectorize Pass.
Computes function attributes in post-order over the call graph.
A utility pass template to force an analysis result to be available.