LLVM 24.0.0git
PassBuilderPipelines.cpp
Go to the documentation of this file.
1//===- Construction of pass pipelines -------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9///
10/// This file provides the implementation of the PassBuilder based on our
11/// static pass registry as well as related functionality. It also provides
12/// helpers to aid in analyzing, debugging, and testing passes and pass
13/// pipelines.
14///
15//===----------------------------------------------------------------------===//
16
17#include "llvm/ADT/Statistic.h"
29#include "llvm/IR/PassManager.h"
30#include "llvm/IR/Verifier.h"
31#include "llvm/Pass.h"
159
160using namespace llvm;
161
162namespace llvm {
163
165 "enable-ml-inliner", cl::init(InliningAdvisorMode::Default), cl::Hidden,
166 cl::desc("Enable ML policy for inliner. Currently trained for -Oz only"),
168 "Heuristics-based inliner version"),
170 "Use development mode (runtime-loadable model)"),
172 "Use release mode (AOT-compiled model)")));
173
174/// Flag to enable inline deferral during PGO.
175static cl::opt<bool>
176 EnablePGOInlineDeferral("enable-npm-pgo-inline-deferral", cl::init(true),
178 cl::desc("Enable inline deferral during PGO"));
179
180static cl::opt<bool> EnableModuleInliner("enable-module-inliner",
181 cl::init(false), cl::Hidden,
182 cl::desc("Enable module inliner"));
183
185 "mandatory-inlining-first", cl::init(false), cl::Hidden,
186 cl::desc("Perform mandatory inlinings module-wide, before performing "
187 "inlining"));
188
190 "eagerly-invalidate-analyses", cl::init(true), cl::Hidden,
191 cl::desc("Eagerly invalidate more analyses in default pipelines"));
192
194 "enable-merge-functions", cl::init(false), cl::Hidden,
195 cl::desc("Enable function merging as part of the optimization pipeline"));
196
198 "enable-post-pgo-loop-rotation", cl::init(true), cl::Hidden,
199 cl::desc("Run the loop rotation transformation after PGO instrumentation"));
200
201static cl::opt<bool>
202 TriggerCrash("opt-pipeline-trigger-crash", cl::init(false), cl::Hidden,
203 cl::desc("Trigger crash in optimization pipeline"));
204
206 "enable-global-analyses", cl::init(true), cl::Hidden,
207 cl::desc("Enable inter-procedural analyses"));
208
209static cl::opt<bool> RunPartialInlining("enable-partial-inlining",
210 cl::init(false), cl::Hidden,
211 cl::desc("Run Partial inlining pass"));
212
214 "extra-vectorizer-passes", cl::init(false), cl::Hidden,
215 cl::desc("Run cleanup optimization passes after vectorization"));
216
217static cl::opt<bool> RunNewGVN("enable-newgvn", cl::init(false), cl::Hidden,
218 cl::desc("Run the NewGVN pass"));
219
220static cl::opt<bool>
221 EnableLoopInterchange("enable-loopinterchange", cl::init(true), cl::Hidden,
222 cl::desc("Enable the LoopInterchange Pass"));
223
224static cl::opt<bool> EnableUnrollAndJam("enable-unroll-and-jam",
225 cl::init(false), cl::Hidden,
226 cl::desc("Enable Unroll And Jam Pass"));
227
228static cl::opt<bool> EnableLoopFlatten("enable-loop-flatten", cl::init(false),
230 cl::desc("Enable the LoopFlatten Pass"));
231
232static cl::opt<bool>
233 EnableInstrumentor("enable-instrumentor", cl::init(false), cl::Hidden,
234 cl::desc("Enable the Instrumentor Pass"));
235
236static cl::opt<bool>
237 EnableDFAJumpThreading("enable-dfa-jump-thread",
238 cl::desc("Enable DFA jump threading"),
239 cl::init(true), cl::Hidden);
240
241static cl::opt<bool>
242 EnableHotColdSplit("hot-cold-split",
243 cl::desc("Enable hot-cold splitting pass"));
244
245static cl::opt<bool>
246 DisablePreInliner("disable-preinline", cl::init(false), cl::Hidden,
247 cl::desc("Disable pre-instrumentation inliner"));
248
250 "preinline-threshold", cl::Hidden, cl::init(75),
251 cl::desc("Control the amount of inlining in pre-instrumentation inliner "
252 "(default = 75)"));
253
254static cl::opt<bool>
255 EnableGVNHoist("enable-gvn-hoist",
256 cl::desc("Enable the GVN hoisting pass (default = off)"));
257
258static cl::opt<bool>
259 EnableGVNSink("enable-gvn-sink",
260 cl::desc("Enable the GVN sinking pass (default = off)"));
261
263 "enable-jump-table-to-switch", cl::init(true),
264 cl::desc("Enable JumpTableToSwitch pass (default = true)"));
265
266// This option is used in simplifying testing SampleFDO optimizations for
267// profile loading.
268static cl::opt<bool>
269 EnableCHR("enable-chr", cl::init(true), cl::Hidden,
270 cl::desc("Enable control height reduction optimization (CHR)"));
271
273 "flattened-profile-used", cl::init(false), cl::Hidden,
274 cl::desc("Indicate the sample profile being used is flattened, i.e., "
275 "no inline hierarchy exists in the profile"));
276
277static cl::opt<bool>
278 EnableMatrix("enable-matrix", cl::init(false), cl::Hidden,
279 cl::desc("Enable lowering of the matrix intrinsics"));
280
282 "enable-mergeicmps", cl::init(true), cl::Hidden,
283 cl::desc("Enable MergeICmps pass in the optimization pipeline"));
284
286 "enable-constraint-elimination", cl::init(true), cl::Hidden,
287 cl::desc(
288 "Enable pass to eliminate conditions based on linear constraints"));
289
291 "attributor-enable", cl::Hidden, cl::init(AttributorRunOption::NONE),
292 cl::desc("Enable the attributor inter-procedural deduction pass"),
294 "enable all full attributor runs"),
296 "enable all attributor-light runs"),
298 "enable module-wide attributor runs"),
300 "enable module-wide attributor-light runs"),
302 "enable call graph SCC attributor runs"),
304 "enable call graph SCC attributor-light runs"),
305 clEnumValN(AttributorRunOption::NONE, "none",
306 "disable attributor runs")));
307
309 "enable-sampled-instrumentation", cl::init(false), cl::Hidden,
310 cl::desc("Enable profile instrumentation sampling (default = off)"));
312 "enable-loop-versioning-licm", cl::init(false), cl::Hidden,
313 cl::desc("Enable the experimental Loop Versioning LICM pass"));
314
316 "instrument-cold-function-only-path", cl::init(""),
317 cl::desc("File path for cold function only instrumentation(requires use "
318 "with --pgo-instrument-cold-function-only)"),
319 cl::Hidden);
320
321// TODO: There is a similar flag in WPD pass, we should consolidate them by
322// parsing the option only once in PassBuilder and share it across both places.
324 "enable-devirtualize-speculatively",
325 cl::desc("Enable speculative devirtualization optimization"),
326 cl::init(false));
327
330
332} // namespace llvm
333
351
352namespace llvm {
354} // namespace llvm
355
357 OptimizationLevel Level) {
358 for (auto &C : PeepholeEPCallbacks)
359 C(FPM, Level);
360}
363 for (auto &C : LateLoopOptimizationsEPCallbacks)
364 C(LPM, Level);
365}
367 OptimizationLevel Level) {
368 for (auto &C : LoopOptimizerEndEPCallbacks)
369 C(LPM, Level);
370}
373 for (auto &C : ScalarOptimizerLateEPCallbacks)
374 C(FPM, Level);
375}
377 OptimizationLevel Level) {
378 for (auto &C : CGSCCOptimizerLateEPCallbacks)
379 C(CGPM, Level);
380}
382 OptimizationLevel Level) {
383 for (auto &C : VectorizerStartEPCallbacks)
384 C(FPM, Level);
385}
387 OptimizationLevel Level) {
388 for (auto &C : VectorizerEndEPCallbacks)
389 C(FPM, Level);
390}
392 OptimizationLevel Level,
394 for (auto &C : OptimizerEarlyEPCallbacks)
395 C(MPM, Level, Phase);
396}
398 OptimizationLevel Level,
400 for (auto &C : OptimizerLastEPCallbacks)
401 C(MPM, Level, Phase);
402}
405 for (auto &C : FullLinkTimeOptimizationEarlyEPCallbacks)
406 C(MPM, Level);
407}
410 for (auto &C : FullLinkTimeOptimizationLastEPCallbacks)
411 C(MPM, Level);
412}
414 OptimizationLevel Level) {
415 for (auto &C : PipelineStartEPCallbacks)
416 C(MPM, Level);
417}
420 for (auto &C : PipelineEarlySimplificationEPCallbacks)
421 C(MPM, Level, Phase);
422}
423
424// Get IR stats with InstCount before/after the optimization pipeline
426 bool IsPreOptimization) {
427 if (AreStatisticsEnabled()) {
428 MPM.addPass(
431 FunctionPropertiesStatisticsPass(IsPreOptimization)));
432 }
433}
434
435// Helper to add AnnotationRemarksPass.
439
440// Helper to check if the current compilation phase is preparing for LTO
445
446// Helper to check if the current compilation phase is preparing for FullLTO
447[[maybe_unused]] static bool isFullLTOPreLink(ThinOrFullLTOPhase Phase) {
449}
450
451// Helper to check if the current compilation phase is preparing for ThinLTO
455
456// Helper to check if the current compilation phase is LTO backend
461
462// Helper to check if the current compilation phase is FullLTO backend
466
467// Helper to check if the current compilation phase is ThinLTO backend
471
472// Helper to wrap conditionally Coro passes.
474 // TODO: Skip passes according to Phase.
475 ModulePassManager CoroPM;
476 CoroPM.addPass(CoroEarlyPass());
477 CGSCCPassManager CGPM;
478 CGPM.addPass(CoroSplitPass());
479 CoroPM.addPass(createModuleToPostOrderCGSCCPassAdaptor(std::move(CGPM)));
480 CoroPM.addPass(CoroCleanupPass());
481 CoroPM.addPass(GlobalDCEPass());
482 return CoroConditionalWrapper(std::move(CoroPM));
483}
484
485// TODO: Investigate the cost/benefit of tail call elimination on debugging.
487PassBuilder::buildO1FunctionSimplificationPipeline(OptimizationLevel Level,
489
491
493 FPM.addPass(CountVisitsPass());
494
495 // Form SSA out of local memory accesses after breaking apart aggregates into
496 // scalars.
497 FPM.addPass(SROAPass(SROAOptions::ModifyCFG));
498
499 // Catch trivial redundancies
500 FPM.addPass(EarlyCSEPass(true /* Enable mem-ssa. */));
501
502 // Hoisting of scalars and load expressions.
503 FPM.addPass(
504 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
505 FPM.addPass(InstCombinePass());
506
507 FPM.addPass(LibCallsShrinkWrapPass());
508
509 invokePeepholeEPCallbacks(FPM, Level);
510
511 FPM.addPass(
512 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
513
514 // Form canonically associated expression trees, and simplify the trees using
515 // basic mathematical properties. For example, this will form (nearly)
516 // minimal multiplication trees.
517 FPM.addPass(ReassociatePass());
518
519 // Add the primary loop simplification pipeline.
520 // FIXME: Currently this is split into two loop pass pipelines because we run
521 // some function passes in between them. These can and should be removed
522 // and/or replaced by scheduling the loop pass equivalents in the correct
523 // positions. But those equivalent passes aren't powerful enough yet.
524 // Specifically, `SimplifyCFGPass` and `InstCombinePass` are currently still
525 // used. We have `LoopSimplifyCFGPass` which isn't yet powerful enough yet to
526 // fully replace `SimplifyCFGPass`, and the closest to the other we have is
527 // `LoopInstSimplify`.
528 LoopPassManager LPM1, LPM2;
529
530 // Simplify the loop body. We do this initially to clean up after other loop
531 // passes run, either when iterating on a loop or on inner loops with
532 // implications on the outer loop.
533 LPM1.addPass(LoopInstSimplifyPass());
534 LPM1.addPass(LoopSimplifyCFGPass());
535
536 // Try to remove as much code from the loop header as possible,
537 // to reduce amount of IR that will have to be duplicated. However,
538 // do not perform speculative hoisting the first time as LICM
539 // will destroy metadata that may not need to be destroyed if run
540 // after loop rotation.
541 // TODO: Investigate promotion cap for O1.
542 LPM1.addPass(LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
543 /*AllowSpeculation=*/false));
544
545 LPM1.addPass(
546 LoopRotatePass(/*EnableHeaderDuplication=*/true, isLTOPreLink(Phase)));
547 // TODO: Investigate promotion cap for O1.
548 LPM1.addPass(LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
549 /*AllowSpeculation=*/true));
550 LPM1.addPass(SimpleLoopUnswitchPass());
552 LPM1.addPass(LoopFlattenPass());
553
554 LPM2.addPass(LoopIdiomRecognizePass());
555 LPM2.addPass(IndVarSimplifyPass());
556
558
559 LPM2.addPass(LoopDeletionPass());
560
561 // Do not enable unrolling in PreLinkThinLTO phase during sample PGO
562 // because it changes IR to makes profile annotation in back compile
563 // inaccurate. The normal unroller doesn't pay attention to forced full unroll
564 // attributes so we need to make sure and allow the full unroll pass to pay
565 // attention to it.
566 if (!isThinLTOPreLink(Phase) || !PGOOpt ||
567 PGOOpt->Action != PGOOptions::SampleUse)
568 LPM2.addPass(LoopFullUnrollPass(static_cast<int>(Level),
569 /* OnlyWhenForced= */ !PTO.LoopUnrolling,
570 PTO.ForgetAllSCEVInLoopUnroll,
571 /* PrepareForLTO= */ isLTOPreLink(Phase)));
572
574
575 FPM.addPass(createFunctionToLoopPassAdaptor(std::move(LPM1),
576 /*UseMemorySSA=*/true));
577 FPM.addPass(
578 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
579 FPM.addPass(InstCombinePass());
580 // The loop passes in LPM2 (LoopFullUnrollPass) do not preserve MemorySSA.
581 // *All* loop passes must preserve it, in order to be able to use it.
582 FPM.addPass(createFunctionToLoopPassAdaptor(std::move(LPM2),
583 /*UseMemorySSA=*/false));
584
585 // Delete small array after loop unroll.
586 FPM.addPass(SROAPass(SROAOptions::ModifyCFG));
587
588 // Specially optimize memory movement as it doesn't look like dataflow in SSA.
589 FPM.addPass(MemCpyOptPass());
590
591 // Sparse conditional constant propagation.
592 // FIXME: It isn't clear why we do this *after* loop passes rather than
593 // before...
594 FPM.addPass(SCCPPass());
595
596 // Delete dead bit computations (instcombine runs after to fold away the dead
597 // computations, and then ADCE will run later to exploit any new DCE
598 // opportunities that creates).
599 FPM.addPass(BDCEPass());
600
601 // Run instcombine after redundancy and dead bit elimination to exploit
602 // opportunities opened up by them.
603 FPM.addPass(InstCombinePass());
604 invokePeepholeEPCallbacks(FPM, Level);
605
606 FPM.addPass(CoroElidePass());
607
609
610 // Finally, do an expensive DCE pass to catch all the dead code exposed by
611 // the simplifications and basic cleanup after all the simplifications.
612 // TODO: Investigate if this is too expensive.
613 FPM.addPass(ADCEPass());
614 FPM.addPass(
615 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
616 FPM.addPass(InstCombinePass());
617 invokePeepholeEPCallbacks(FPM, Level);
618
619 return FPM;
620}
621
625 assert(Level != OptimizationLevel::O0 && "Must request optimizations!");
626
627 // The O1 pipeline has a separate pipeline creation function to simplify
628 // construction readability.
629 if (Level == OptimizationLevel::O1)
630 return buildO1FunctionSimplificationPipeline(Level, Phase);
631
633
636
637 // Form SSA out of local memory accesses after breaking apart aggregates into
638 // scalars.
640
641 // Catch trivial redundancies
642 FPM.addPass(EarlyCSEPass(true /* Enable mem-ssa. */));
645
646 // Hoisting of scalars and load expressions.
647 if (EnableGVNHoist)
648 FPM.addPass(GVNHoistPass());
649
650 // Global value numbering based sinking.
651 if (EnableGVNSink) {
652 FPM.addPass(GVNSinkPass());
653 FPM.addPass(
654 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
655 }
656
657 // Speculative execution if the target has divergent branches; otherwise nop.
658 FPM.addPass(SpeculativeExecutionPass(/* OnlyIfDivergentTarget =*/true));
659
660 // Optimize based on known information about branches, and cleanup afterward.
663
664 // Jump table to switch conversion.
667
668 FPM.addPass(
669 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
673
674 invokePeepholeEPCallbacks(FPM, Level);
675
676 // For PGO use pipeline, try to optimize memory intrinsics such as memcpy
677 // using the size value profile. Don't perform this when optimizing for size.
678 if (PGOOpt && PGOOpt->Action == PGOOptions::IRUse)
680
681 FPM.addPass(TailCallElimPass(/*UpdateFunctionEntryCount=*/
682 isInstrumentedPGOUse()));
683 FPM.addPass(
684 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
685
686 // Form canonically associated expression trees, and simplify the trees using
687 // basic mathematical properties. For example, this will form (nearly)
688 // minimal multiplication trees.
690
693
694 // Add the primary loop simplification pipeline.
695 // FIXME: Currently this is split into two loop pass pipelines because we run
696 // some function passes in between them. These can and should be removed
697 // and/or replaced by scheduling the loop pass equivalents in the correct
698 // positions. But those equivalent passes aren't powerful enough yet.
699 // Specifically, `SimplifyCFGPass` and `InstCombinePass` are currently still
700 // used. We have `LoopSimplifyCFGPass` which isn't yet powerful enough yet to
701 // fully replace `SimplifyCFGPass`, and the closest to the other we have is
702 // `LoopInstSimplify`.
703 LoopPassManager LPM1, LPM2;
704
705 // Simplify the loop body. We do this initially to clean up after other loop
706 // passes run, either when iterating on a loop or on inner loops with
707 // implications on the outer loop.
708 LPM1.addPass(LoopInstSimplifyPass());
709 LPM1.addPass(LoopSimplifyCFGPass());
710
711 // Try to remove as much code from the loop header as possible,
712 // to reduce amount of IR that will have to be duplicated. However,
713 // do not perform speculative hoisting the first time as LICM
714 // will destroy metadata that may not need to be destroyed if run
715 // after loop rotation.
716 // TODO: Investigate promotion cap for O1.
717 LPM1.addPass(LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
718 /*AllowSpeculation=*/false));
719
720 LPM1.addPass(
721 LoopRotatePass(/*EnableHeaderDuplication=*/true, isLTOPreLink(Phase)));
722 // TODO: Investigate promotion cap for O1.
723 LPM1.addPass(LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
724 /*AllowSpeculation=*/true));
725 LPM1.addPass(
726 SimpleLoopUnswitchPass(/* NonTrivial */ Level == OptimizationLevel::O3));
728 LPM1.addPass(LoopFlattenPass());
729
730 LPM2.addPass(LoopIdiomRecognizePass());
731 LPM2.addPass(IndVarSimplifyPass());
732
733 {
735 ExtraPasses.addPass(SimpleLoopUnswitchPass(/* NonTrivial */ Level ==
737 LPM2.addPass(std::move(ExtraPasses));
738 }
739
741
742 LPM2.addPass(LoopDeletionPass());
743
744 // Do not enable unrolling in PreLinkThinLTO phase during sample PGO
745 // because it changes IR to makes profile annotation in back compile
746 // inaccurate. The normal unroller doesn't pay attention to forced full unroll
747 // attributes so we need to make sure and allow the full unroll pass to pay
748 // attention to it.
749 if (!isThinLTOPreLink(Phase) || !PGOOpt ||
750 PGOOpt->Action != PGOOptions::SampleUse)
751 LPM2.addPass(LoopFullUnrollPass(static_cast<int>(Level),
752 /* OnlyWhenForced= */ !PTO.LoopUnrolling,
753 PTO.ForgetAllSCEVInLoopUnroll,
754 /* PrepareForLTO= */ isLTOPreLink(Phase)));
755
757
758 FPM.addPass(createFunctionToLoopPassAdaptor(std::move(LPM1),
759 /*UseMemorySSA=*/true));
760 FPM.addPass(
761 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
763 // The loop passes in LPM2 (LoopIdiomRecognizePass, IndVarSimplifyPass,
764 // LoopDeletionPass and LoopFullUnrollPass) do not preserve MemorySSA.
765 // *All* loop passes must preserve it, in order to be able to use it.
766 FPM.addPass(createFunctionToLoopPassAdaptor(std::move(LPM2),
767 /*UseMemorySSA=*/false));
768
769 // Delete small array after loop unroll.
771
772 // Try vectorization/scalarization transforms that are both improvements
773 // themselves and can allow further folds with GVN and InstCombine.
774 FPM.addPass(VectorCombinePass(/*TryEarlyFoldsOnly=*/true));
775
776 // Eliminate redundancies.
778 if (RunNewGVN)
779 FPM.addPass(NewGVNPass());
780 else
781 FPM.addPass(GVNPass());
782
783 // Sparse conditional constant propagation.
784 // FIXME: It isn't clear why we do this *after* loop passes rather than
785 // before...
786 FPM.addPass(SCCPPass());
787
788 // Delete dead bit computations (instcombine runs after to fold away the dead
789 // computations, and then ADCE will run later to exploit any new DCE
790 // opportunities that creates).
791 FPM.addPass(BDCEPass());
792
793 // Run instcombine after redundancy and dead bit elimination to exploit
794 // opportunities opened up by them.
796 invokePeepholeEPCallbacks(FPM, Level);
797
798 // Re-consider control flow based optimizations after redundancy elimination,
799 // redo DCE, etc.
802
805
806 // Finally, do an expensive DCE pass to catch all the dead code exposed by
807 // the simplifications and basic cleanup after all the simplifications.
808 // TODO: Investigate if this is too expensive.
809 FPM.addPass(ADCEPass());
810
811 // Specially optimize memory movement as it doesn't look like dataflow in SSA.
812 FPM.addPass(MemCpyOptPass());
813
814 FPM.addPass(DSEPass());
816
818 LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
819 /*AllowSpeculation=*/true),
820 /*UseMemorySSA=*/true));
821
822 FPM.addPass(CoroElidePass());
823
825
827 .convertSwitchRangeToICmp(true)
828 .convertSwitchToArithmetic(true)
829 .hoistCommonInsts(true)
830 .sinkCommonInsts(true)));
832 invokePeepholeEPCallbacks(FPM, Level);
833
834 return FPM;
835}
836
837void PassBuilder::addRequiredLTOPreLinkPasses(ModulePassManager &MPM) {
840 MPM.addPass(AssignGUIDPass());
841}
842
843void PassBuilder::addPreInlinerPasses(ModulePassManager &MPM,
844 OptimizationLevel Level,
845 ThinOrFullLTOPhase LTOPhase) {
846 assert(Level != OptimizationLevel::O0 && "Not expecting O0 here!");
848 return;
849 InlineParams IP;
850
852
853 // FIXME: The hint threshold has the same value used by the regular inliner
854 // when not optimzing for size. This should probably be lowered after
855 // performance testing.
856 // FIXME: this comment is cargo culted from the old pass manager, revisit).
857 IP.HintThreshold = 325;
860 IP, /* MandatoryFirst */ true,
862 CGSCCPassManager &CGPipeline = MIWP.getPM();
863
865 FPM.addPass(SROAPass(SROAOptions::ModifyCFG));
866 FPM.addPass(EarlyCSEPass()); // Catch trivial redundancies.
867 FPM.addPass(SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(
868 true))); // Merge & remove basic blocks.
869 FPM.addPass(InstCombinePass()); // Combine silly sequences.
870 invokePeepholeEPCallbacks(FPM, Level);
871
872 CGPipeline.addPass(createCGSCCToFunctionPassAdaptor(
873 std::move(FPM), PTO.EagerlyInvalidateAnalyses));
874
875 MPM.addPass(std::move(MIWP));
876
877 // Delete anything that is now dead to make sure that we don't instrument
878 // dead code. Instrumentation can end up keeping dead code around and
879 // dramatically increase code size.
880 MPM.addPass(GlobalDCEPass());
881}
882
883void PassBuilder::addPostPGOLoopRotation(ModulePassManager &MPM,
884 OptimizationLevel Level) {
886 // Disable header duplication in loop rotation at -Oz.
888 createFunctionToLoopPassAdaptor(LoopRotatePass(),
889 /*UseMemorySSA=*/false),
890 PTO.EagerlyInvalidateAnalyses));
891 }
892}
893
894void PassBuilder::addPGOInstrPasses(ModulePassManager &MPM,
895 OptimizationLevel Level, bool RunProfileGen,
896 bool IsCS, bool AtomicCounterUpdate,
897 std::string ProfileFile,
898 std::string ProfileRemappingFile) {
899 assert(Level != OptimizationLevel::O0 && "Not expecting O0 here!");
900
901 if (!RunProfileGen) {
902 assert(!ProfileFile.empty() && "Profile use expecting a profile file!");
903 MPM.addPass(
904 PGOInstrumentationUse(ProfileFile, ProfileRemappingFile, IsCS, FS));
905 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
906 // RequireAnalysisPass for PSI before subsequent non-module passes.
907 MPM.addPass(RequireAnalysisPass<ProfileSummaryAnalysis, Module>());
908 return;
909 }
910
911 // Perform PGO instrumentation.
912 MPM.addPass(PGOInstrumentationGen(IsCS ? PGOInstrumentationType::CSFDO
914
915 addPostPGOLoopRotation(MPM, Level);
916 // Add the profile lowering pass.
917 InstrProfOptions Options;
918 if (!ProfileFile.empty())
919 Options.InstrProfileOutput = ProfileFile;
920 // Do counter promotion at Level greater than O0.
921 Options.DoCounterPromotion = true;
922 Options.UseBFIInPromotion = IsCS;
923 if (EnableSampledInstr) {
924 Options.Sampling = true;
925 // With sampling, there is little beneifit to enable counter promotion.
926 // But note that sampling does work with counter promotion.
927 Options.DoCounterPromotion = false;
928 }
929 Options.Atomic = AtomicCounterUpdate;
930 MPM.addPass(InstrProfilingLoweringPass(Options, IsCS));
931}
932
934 bool RunProfileGen, bool IsCS,
935 bool AtomicCounterUpdate,
936 std::string ProfileFile,
937 std::string ProfileRemappingFile) {
938 if (!RunProfileGen) {
939 assert(!ProfileFile.empty() && "Profile use expecting a profile file!");
940 MPM.addPass(
941 PGOInstrumentationUse(ProfileFile, ProfileRemappingFile, IsCS, FS));
942 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
943 // RequireAnalysisPass for PSI before subsequent non-module passes.
945 return;
946 }
947
948 // Perform PGO instrumentation.
951 // Add the profile lowering pass.
953 if (!ProfileFile.empty())
954 Options.InstrProfileOutput = ProfileFile;
955 // Do not do counter promotion at O0.
956 Options.DoCounterPromotion = false;
957 Options.UseBFIInPromotion = IsCS;
958 Options.Atomic = AtomicCounterUpdate;
960}
961
963 return getInlineParamsFromOptLevel(static_cast<unsigned>(Level));
964}
965
969 InlineParams IP;
970 if (PTO.InlinerThreshold == -1)
972 else
973 IP = getInlineParams(PTO.InlinerThreshold);
974 // For PreLinkThinLTO + SamplePGO or PreLinkFullLTO + SamplePGO,
975 // set hot-caller threshold to 0 to disable hot
976 // callsite inline (as much as possible [1]) because it makes
977 // profile annotation in the backend inaccurate.
978 //
979 // [1] Note the cost of a function could be below zero due to erased
980 // prologue / epilogue.
981 if (isLTOPreLink(Phase) && PGOOpt && PGOOpt->Action == PGOOptions::SampleUse)
983
984 if (PGOOpt)
986
990
991 // Require the GlobalsAA analysis for the module so we can query it within
992 // the CGSCC pipeline.
994 MIWP.addModulePass(RequireAnalysisPass<GlobalsAA, Module>());
995 // Invalidate AAManager so it can be recreated and pick up the newly
996 // available GlobalsAA.
997 MIWP.addModulePass(
999 }
1000
1001 // Require the ProfileSummaryAnalysis for the module so we can query it within
1002 // the inliner pass.
1004
1005 // Now begin the main postorder CGSCC pipeline.
1006 // FIXME: The current CGSCC pipeline has its origins in the legacy pass
1007 // manager and trying to emulate its precise behavior. Much of this doesn't
1008 // make a lot of sense and we should revisit the core CGSCC structure.
1009 CGSCCPassManager &MainCGPipeline = MIWP.getPM();
1010
1011 // Note: historically, the PruneEH pass was run first to deduce nounwind and
1012 // generally clean up exception handling overhead. It isn't clear this is
1013 // valuable as the inliner doesn't currently care whether it is inlining an
1014 // invoke or a call.
1015
1017 MainCGPipeline.addPass(AttributorCGSCCPass());
1019 MainCGPipeline.addPass(AttributorLightCGSCCPass());
1020
1021 // Deduce function attributes. We do another run of this after the function
1022 // simplification pipeline, so this only needs to run when it could affect the
1023 // function simplification pipeline, which is only the case with recursive
1024 // functions.
1025 MainCGPipeline.addPass(PostOrderFunctionAttrsPass(/*SkipNonRecursive*/ true));
1026
1027 // When at O3 add argument promotion to the pass pipeline.
1028 // FIXME: It isn't at all clear why this should be limited to O3.
1029 if (Level == OptimizationLevel::O3)
1030 MainCGPipeline.addPass(ArgumentPromotionPass());
1031
1032 // Try to perform OpenMP specific optimizations. This is a (quick!) no-op if
1033 // there are no OpenMP runtime calls present in the module.
1034 if (Level == OptimizationLevel::O2 || Level == OptimizationLevel::O3)
1035 MainCGPipeline.addPass(OpenMPOptCGSCCPass(Phase));
1036
1037 invokeCGSCCOptimizerLateEPCallbacks(MainCGPipeline, Level);
1038
1039 // Add the core function simplification pipeline nested inside the
1040 // CGSCC walk.
1043 PTO.EagerlyInvalidateAnalyses, /*NoRerun=*/true));
1044
1045 // Finally, deduce any function attributes based on the fully simplified
1046 // function.
1047 MainCGPipeline.addPass(PostOrderFunctionAttrsPass());
1048
1049 // Mark that the function is fully simplified and that it shouldn't be
1050 // simplified again if we somehow revisit it due to CGSCC mutations unless
1051 // it's been modified since.
1054
1055 if (!isThinLTOPreLink(Phase)) {
1056 MainCGPipeline.addPass(CoroSplitPass(Level != OptimizationLevel::O0));
1057 MainCGPipeline.addPass(CoroAnnotationElidePass());
1058 }
1059
1060 // Make sure we don't affect potential future NoRerun CGSCC adaptors.
1061 MIWP.addLateModulePass(createModuleToFunctionPassAdaptor(
1063
1064 return MIWP;
1065}
1066
1071
1073 // For PreLinkThinLTO + SamplePGO or PreLinkFullLTO + SamplePGO,
1074 // set hot-caller threshold to 0 to disable hot
1075 // callsite inline (as much as possible [1]) because it makes
1076 // profile annotation in the backend inaccurate.
1077 //
1078 // [1] Note the cost of a function could be below zero due to erased
1079 // prologue / epilogue.
1080 if (isLTOPreLink(Phase) && PGOOpt && PGOOpt->Action == PGOOptions::SampleUse)
1081 IP.HotCallSiteThreshold = 0;
1082
1083 if (PGOOpt)
1085
1086 // The inline deferral logic is used to avoid losing some
1087 // inlining chance in future. It is helpful in SCC inliner, in which
1088 // inlining is processed in bottom-up order.
1089 // While in module inliner, the inlining order is a priority-based order
1090 // by default. The inline deferral is unnecessary there. So we disable the
1091 // inline deferral logic in module inliner.
1092 IP.EnableDeferral = false;
1093
1096 MPM.addPass(GlobalOptPass());
1097 MPM.addPass(GlobalDCEPass());
1098 MPM.addPass(AssignGUIDPass());
1099 MPM.addPass(PGOCtxProfFlatteningPass(/*IsPreThinlink=*/false));
1100 }
1101
1104 PTO.EagerlyInvalidateAnalyses));
1105
1106 if (!isThinLTOPreLink(Phase)) {
1109 MPM.addPass(
1111 }
1112
1113 return MPM;
1114}
1115
1119 assert(Level != OptimizationLevel::O0 &&
1120 "Should not be used for O0 pipeline");
1121
1123 "FullLTOPostLink shouldn't call buildModuleSimplificationPipeline!");
1124
1126
1127 // Place pseudo probe instrumentation as the first pass of the pipeline to
1128 // minimize the impact of optimization changes.
1129 if (PGOOpt && PGOOpt->PseudoProbeForProfiling && !isThinLTOPostLink(Phase))
1131
1132 bool HasSampleProfile = PGOOpt && (PGOOpt->Action == PGOOptions::SampleUse);
1133
1134 // In ThinLTO mode, when flattened profile is used, all the available
1135 // profile information will be annotated in PreLink phase so there is
1136 // no need to load the profile again in PostLink.
1137 bool LoadSampleProfile =
1138 HasSampleProfile && !(FlattenedProfileUsed && isThinLTOPostLink(Phase));
1139
1140 // During the ThinLTO backend phase we perform early indirect call promotion
1141 // here, before globalopt. Otherwise imported available_externally functions
1142 // look unreferenced and are removed. If we are going to load the sample
1143 // profile then defer until later.
1144 // TODO: See if we can move later and consolidate with the location where
1145 // we perform ICP when we are loading a sample profile.
1146 // TODO: We pass HasSampleProfile (whether there was a sample profile file
1147 // passed to the compile) to the SamplePGO flag of ICP. This is used to
1148 // determine whether the new direct calls are annotated with prof metadata.
1149 // Ideally this should be determined from whether the IR is annotated with
1150 // sample profile, and not whether the a sample profile was provided on the
1151 // command line. E.g. for flattened profiles where we will not be reloading
1152 // the sample profile in the ThinLTO backend, we ideally shouldn't have to
1153 // provide the sample profile file.
1154 if (isThinLTOPostLink(Phase) && !LoadSampleProfile)
1155 MPM.addPass(PGOIndirectCallPromotion(true /* InLTO */, HasSampleProfile));
1156
1157 // Create an early function pass manager to cleanup the output of the
1158 // frontend. Not necessary with LTO post link pipelines since the pre link
1159 // pipeline already cleaned up the frontend output.
1160 if (!isThinLTOPostLink(Phase)) {
1161 // Do basic inference of function attributes from known properties of system
1162 // libraries and other oracles.
1164 MPM.addPass(CoroEarlyPass());
1165
1166 FunctionPassManager EarlyFPM;
1167 EarlyFPM.addPass(EntryExitInstrumenterPass(/*PostInlining=*/false));
1168 // Lower llvm.expect to metadata before attempting transforms.
1169 // Compare/branch metadata may alter the behavior of passes like
1170 // SimplifyCFG.
1172 EarlyFPM.addPass(SimplifyCFGPass());
1174 EarlyFPM.addPass(EarlyCSEPass());
1175 if (Level == OptimizationLevel::O3)
1176 EarlyFPM.addPass(CallSiteSplittingPass());
1178 std::move(EarlyFPM), PTO.EagerlyInvalidateAnalyses));
1179 }
1180
1181 if (LoadSampleProfile) {
1182 // Annotate sample profile right after early FPM to ensure freshness of
1183 // the debug info.
1185 PGOOpt->ProfileFile, PGOOpt->ProfileRemappingFile, Phase, FS));
1186 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
1187 // RequireAnalysisPass for PSI before subsequent non-module passes.
1189 // Do not invoke ICP in the LTOPrelink phase as it makes it hard
1190 // for the profile annotation to be accurate in the LTO backend.
1191 if (!isLTOPreLink(Phase))
1192 // We perform early indirect call promotion here, before globalopt.
1193 // This is important for the ThinLTO backend phase because otherwise
1194 // imported available_externally functions look unreferenced and are
1195 // removed.
1196 MPM.addPass(
1197 PGOIndirectCallPromotion(true /* IsInLTO */, true /* SamplePGO */));
1198 }
1199
1200 // Try to perform OpenMP specific optimizations on the module. This is a
1201 // (quick!) no-op if there are no OpenMP runtime calls present in the module.
1203
1205 MPM.addPass(AttributorPass());
1208
1209 // Lower type metadata and the type.test intrinsic in the ThinLTO
1210 // post link pipeline after ICP. This is to enable usage of the type
1211 // tests in ICP sequences.
1214
1216
1217 // Interprocedural constant propagation now that basic cleanup has occurred
1218 // and prior to optimizing globals.
1219 // FIXME: This position in the pipeline hasn't been carefully considered in
1220 // years, it should be re-analyzed.
1221 MPM.addPass(
1222 IPSCCPPass(IPSCCPOptions(/*AllowFuncSpec=*/!isLTOPreLink(Phase))));
1223
1224 // Attach metadata to indirect call sites indicating the set of functions
1225 // they may target at run-time. This should follow IPSCCP.
1227
1228 // Optimize globals to try and fold them into constants.
1229 MPM.addPass(GlobalOptPass());
1230
1231 // Create a small function pass pipeline to cleanup after all the global
1232 // optimizations.
1233 FunctionPassManager GlobalCleanupPM;
1234 // FIXME: Should this instead by a run of SROA?
1235 GlobalCleanupPM.addPass(PromotePass());
1236 GlobalCleanupPM.addPass(InstCombinePass());
1237 invokePeepholeEPCallbacks(GlobalCleanupPM, Level);
1238 GlobalCleanupPM.addPass(
1239 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
1240 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(GlobalCleanupPM),
1241 PTO.EagerlyInvalidateAnalyses));
1242
1243 // We already asserted this happens in non-FullLTOPostLink earlier.
1244 const bool IsPreLink = !isThinLTOPostLink(Phase);
1245 // Enable contextual profiling instrumentation.
1246 const bool IsCtxProfGen =
1248 const bool IsPGOPreLink = !IsCtxProfGen && PGOOpt && IsPreLink;
1249 const bool IsPGOInstrGen =
1250 IsPGOPreLink && PGOOpt->Action == PGOOptions::IRInstr;
1251 const bool IsPGOInstrUse =
1252 IsPGOPreLink && PGOOpt->Action == PGOOptions::IRUse;
1253 const bool IsMemprofUse = IsPGOPreLink && !PGOOpt->MemoryProfile.empty();
1254 // We don't want to mix pgo ctx gen and pgo gen; we also don't currently
1255 // enable ctx profiling from the frontend.
1257 "Enabling both instrumented PGO and contextual instrumentation is not "
1258 "supported.");
1259 const bool IsCtxProfUse = !UseCtxProfile.empty() && isThinLTOPreLink(Phase);
1260
1261 assert(
1263 "--instrument-cold-function-only-path is provided but "
1264 "--pgo-instrument-cold-function-only is not enabled");
1265 const bool IsColdFuncOnlyInstrGen = PGOInstrumentColdFunctionOnly &&
1266 IsPGOPreLink &&
1268
1269 if (IsPGOInstrGen || IsPGOInstrUse || IsMemprofUse || IsCtxProfGen ||
1270 IsCtxProfUse || IsColdFuncOnlyInstrGen)
1271 addPreInlinerPasses(MPM, Level, Phase);
1272
1273 // Add all the requested passes for instrumentation PGO, if requested.
1274 if (IsPGOInstrGen || IsPGOInstrUse) {
1275 addPGOInstrPasses(MPM, Level,
1276 /*RunProfileGen=*/IsPGOInstrGen,
1277 /*IsCS=*/false, PGOOpt->AtomicCounterUpdate,
1278 PGOOpt->ProfileFile, PGOOpt->ProfileRemappingFile);
1279 } else if (IsCtxProfGen || IsCtxProfUse) {
1281 // In pre-link, we just want the instrumented IR. We use the contextual
1282 // profile in the post-thinlink phase.
1283 // The instrumentation will be removed in post-thinlink after IPO.
1284 if (IsCtxProfUse) {
1285 MPM.addPass(AssignGUIDPass());
1286 MPM.addPass(PGOCtxProfFlatteningPass(/*IsPreThinlink=*/true));
1287 return MPM;
1288 }
1289 // Block further inlining in the instrumented ctxprof case. This avoids
1290 // confusingly collecting profiles for the same GUID corresponding to
1291 // different variants of the function. We could do like PGO and identify
1292 // functions by a (GUID, Hash) tuple, but since the ctxprof "use" waits for
1293 // thinlto to happen before performing any further optimizations, it's
1294 // unnecessary to collect profiles for non-prevailing copies.
1296 addPostPGOLoopRotation(MPM, Level);
1297 MPM.addPass(AssignGUIDPass());
1299 } else if (IsColdFuncOnlyInstrGen) {
1300 addPGOInstrPasses(MPM, Level, /* RunProfileGen */ true, /* IsCS */ false,
1301 /* AtomicCounterUpdate */ false,
1303 /* ProfileRemappingFile */ "");
1304 }
1305
1306 if (IsPGOInstrGen || IsPGOInstrUse || IsCtxProfGen)
1307 MPM.addPass(PGOIndirectCallPromotion(false, false));
1308
1309 if (IsPGOPreLink && PGOOpt->CSAction == PGOOptions::CSIRInstr)
1310 MPM.addPass(PGOInstrumentationGenCreateVar(PGOOpt->CSProfileGenFile,
1312
1313 if (IsMemprofUse)
1314 MPM.addPass(MemProfUsePass(PGOOpt->MemoryProfile, FS));
1315
1316 if (PGOOpt && (PGOOpt->Action == PGOOptions::IRUse ||
1317 PGOOpt->Action == PGOOptions::SampleUse))
1318 MPM.addPass(PGOForceFunctionAttrsPass(PGOOpt->ColdOptType));
1319
1320 MPM.addPass(AlwaysInlinerPass(/*InsertLifetimeIntrinsics=*/true));
1321
1324 else
1325 MPM.addPass(buildInlinerPipeline(Level, Phase));
1326
1327 // Remove any dead arguments exposed by cleanups, constant folding globals,
1328 // and argument promotion.
1330
1333
1334 if (!isThinLTOPreLink(Phase))
1335 MPM.addPass(CoroCleanupPass());
1336
1337 // Optimize globals now that functions are fully simplified.
1338 MPM.addPass(GlobalOptPass());
1339 MPM.addPass(GlobalDCEPass());
1340
1341 return MPM;
1342}
1343
1344/// TODO: Should LTO cause any differences to this set of passes?
1345void PassBuilder::addVectorPasses(OptimizationLevel Level,
1347 ThinOrFullLTOPhase LTOPhase) {
1350
1351 // Drop dereferenceable assumes after vectorization, as they are no longer
1352 // needed and can inhibit further optimization.
1353 if (!isLTOPreLink(LTOPhase))
1354 FPM.addPass(DropUnnecessaryAssumesPass(/*DropDereferenceable=*/true));
1355
1357 if (isFullLTOPostLink(LTOPhase)) {
1358 // The vectorizer may have significantly shortened a loop body; unroll
1359 // again. Unroll small loops to hide loop backedge latency and saturate any
1360 // parallel execution resources of an out-of-order processor. We also then
1361 // need to clean up redundancies and loop invariant code.
1362 // FIXME: It would be really good to use a loop-integrated instruction
1363 // combiner for cleanup here so that the unrolling and LICM can be pipelined
1364 // across the loop nests.
1365 // We do UnrollAndJam in a separate LPM to ensure it happens before unroll
1368 LoopUnrollAndJamPass(static_cast<int>(Level))));
1370 static_cast<int>(Level), /*OnlyWhenForced=*/!PTO.LoopUnrolling,
1373 // Now that we are done with loop unrolling, be it either by LoopVectorizer,
1374 // or LoopUnroll passes, some variable-offset GEP's into alloca's could have
1375 // become constant-offset, thus enabling SROA and alloca promotion. Do so.
1376 // NOTE: we are very late in the pipeline, and we don't have any LICM
1377 // or SimplifyCFG passes scheduled after us, that would cleanup
1378 // the CFG mess this may created if allowed to modify CFG, so forbid that.
1379
1380 // We also turn on struct to vector canonicalization here, which allows
1381 // converting allocas of homogeneous structs into vector allocas when the
1382 // allocas' users are all memory intrinsics. This allows promotion in some
1383 // cases because structs cannot promote to SSA values, but vectors can. We
1384 // only turn this on after memcpyopt runs because this might hinder
1385 // memcpyopt's optimizations if done before. Look at the documentation for
1386 // `tryCanonicalizeStructToVector` in SROA.cpp to see why.
1388 /*AggregateToVector=*/true)));
1389 }
1390
1391 if (!isFullLTOPostLink(LTOPhase)) {
1392 // Eliminate loads by forwarding stores from the previous iteration to loads
1393 // of the current iteration.
1395 }
1396 // Cleanup after the loop optimization passes.
1397 FPM.addPass(InstCombinePass());
1398
1400 ExtraFunctionPassManager<ShouldRunExtraVectorPasses> ExtraPasses;
1401 // At higher optimization levels, try to clean up any runtime overlap and
1402 // alignment checks inserted by the vectorizer. We want to track correlated
1403 // runtime checks for two inner loops in the same outer loop, fold any
1404 // common computations, hoist loop-invariant aspects out of any outer loop,
1405 // and unswitch the runtime checks if possible. Once hoisted, we may have
1406 // dead (or speculatable) control flows or more combining opportunities.
1407 ExtraPasses.addPass(EarlyCSEPass());
1408 ExtraPasses.addPass(CorrelatedValuePropagationPass());
1409 ExtraPasses.addPass(InstCombinePass());
1410 LoopPassManager LPM;
1411 LPM.addPass(LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
1412 /*AllowSpeculation=*/true));
1413 LPM.addPass(SimpleLoopUnswitchPass(/* NonTrivial */ Level ==
1415 ExtraPasses.addPass(
1416 createFunctionToLoopPassAdaptor(std::move(LPM), /*UseMemorySSA=*/true));
1417 ExtraPasses.addPass(
1418 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
1419 ExtraPasses.addPass(InstCombinePass());
1420 FPM.addPass(std::move(ExtraPasses));
1421 }
1422
1423 // Now that we've formed fast to execute loop structures, we do further
1424 // optimizations. These are run afterward as they might block doing complex
1425 // analyses and transforms such as what are needed for loop vectorization.
1426
1427 // Cleanup after loop vectorization, etc. Simplification passes like CVP and
1428 // GVN, loop transforms, and others have already run, so it's now better to
1429 // convert to more optimized IR using more aggressive simplify CFG options.
1430 // The extra sinking transform can create larger basic blocks, so do this
1431 // before SLP vectorization.
1432 FPM.addPass(SimplifyCFGPass(SimplifyCFGOptions()
1433 .forwardSwitchCondToPhi(true)
1434 .convertSwitchRangeToICmp(true)
1435 .convertSwitchToArithmetic(true)
1436 .convertSwitchToLookupTable(true)
1437 .needCanonicalLoops(false)
1438 .hoistCommonInsts(true)
1439 .sinkCommonInsts(true)));
1440
1441 if (isFullLTOPostLink(LTOPhase)) {
1442 FPM.addPass(SCCPPass());
1443 FPM.addPass(InstCombinePass());
1444 FPM.addPass(BDCEPass());
1445 }
1446
1447 // Optimize parallel scalar instruction chains into SIMD instructions.
1448 if (PTO.SLPVectorization) {
1449 FPM.addPass(SLPVectorizerPass());
1451 FPM.addPass(EarlyCSEPass());
1452 }
1453 }
1454 // Enhance/cleanup vector code.
1455 FPM.addPass(VectorCombinePass());
1456
1457 if (!isFullLTOPostLink(LTOPhase)) {
1458 FPM.addPass(InstCombinePass());
1459 // Unroll small loops to hide loop backedge latency and saturate any
1460 // parallel execution resources of an out-of-order processor. We also then
1461 // need to clean up redundancies and loop invariant code.
1462 // FIXME: It would be really good to use a loop-integrated instruction
1463 // combiner for cleanup here so that the unrolling and LICM can be pipelined
1464 // across the loop nests.
1465 // We do UnrollAndJam in a separate LPM to ensure it happens before unroll
1466 if (EnableUnrollAndJam && PTO.LoopUnrolling) {
1468 LoopUnrollAndJamPass(static_cast<int>(Level))));
1469 }
1470 FPM.addPass(LoopUnrollPass(LoopUnrollOptions(
1471 static_cast<int>(Level), /*OnlyWhenForced=*/!PTO.LoopUnrolling,
1472 PTO.ForgetAllSCEVInLoopUnroll)));
1473 FPM.addPass(WarnMissedTransformationsPass());
1474 // Now that we are done with loop unrolling, be it either by LoopVectorizer,
1475 // or LoopUnroll passes, some variable-offset GEP's into alloca's could have
1476 // become constant-offset, thus enabling SROA and alloca promotion. Do so.
1477 // NOTE: we are very late in the pipeline, and we don't have any LICM
1478 // or SimplifyCFG passes scheduled after us, that would cleanup
1479 // the CFG mess this may created if allowed to modify CFG, so forbid that.
1480
1481 // We also turn on struct to vector canonicalization here, which allows
1482 // converting allocas of homogeneous structs into vector allocas when the
1483 // allocas' users are all memory intrinsics. This allows promotion in some
1484 // cases because structs cannot promote to SSA values, but vectors can. We
1485 // only turn this on after memcpyopt runs because this might hinder
1486 // memcpyopt's optimizations if done before. Look at the documentation for
1487 // `tryCanonicalizeStructToVector` in SROA.cpp to see why.
1488 FPM.addPass(SROAPass(SROAOptions(SROAOptions::PreserveCFG,
1489 /*AggregateToVector=*/true)));
1490 }
1491
1492 FPM.addPass(InferAlignmentPass());
1493 FPM.addPass(InstCombinePass());
1494
1495 // This is needed for two reasons:
1496 // 1. It works around problems that instcombine introduces, such as sinking
1497 // expensive FP divides into loops containing multiplications using the
1498 // divide result.
1499 // 2. It helps to clean up some loop-invariant code created by the loop
1500 // unroll pass when IsFullLTO=false.
1502 LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
1503 /*AllowSpeculation=*/true),
1504 /*UseMemorySSA=*/true));
1505
1506 // Now that we've vectorized and unrolled loops, we may have more refined
1507 // alignment information, try to re-derive it here.
1508 FPM.addPass(AlignmentFromAssumptionsPass());
1509}
1510
1513 ThinOrFullLTOPhase LTOPhase) {
1515
1516 // Run partial inlining pass to partially inline functions that have
1517 // large bodies.
1520
1521 // Remove avail extern fns and globals definitions since we aren't compiling
1522 // an object file for later LTO. For LTO we want to preserve these so they
1523 // are eligible for inlining at link-time. Note if they are unreferenced they
1524 // will be removed by GlobalDCE later, so this only impacts referenced
1525 // available externally globals. Eventually they will be suppressed during
1526 // codegen, but eliminating here enables more opportunity for GlobalDCE as it
1527 // may make globals referenced by available external functions dead and saves
1528 // running remaining passes on the eliminated functions. These should be
1529 // preserved during prelinking for link-time inlining decisions.
1530 if (!isLTOPreLink(LTOPhase))
1532
1533 // Do RPO function attribute inference across the module to forward-propagate
1534 // attributes where applicable.
1535 // FIXME: Is this really an optimization rather than a canonicalization?
1537
1538 // Do a post inline PGO instrumentation and use pass. This is a context
1539 // sensitive PGO pass. We don't want to do this in LTOPreLink phrase as
1540 // cross-module inline has not been done yet. The context sensitive
1541 // instrumentation is after all the inlines are done.
1542 if (!isLTOPreLink(LTOPhase) && PGOOpt) {
1543 if (PGOOpt->CSAction == PGOOptions::CSIRInstr)
1544 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/true,
1545 /*IsCS=*/true, PGOOpt->AtomicCounterUpdate,
1546 PGOOpt->CSProfileGenFile, PGOOpt->ProfileRemappingFile);
1547 else if (PGOOpt->CSAction == PGOOptions::CSIRUse)
1548 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/false,
1549 /*IsCS=*/true, PGOOpt->AtomicCounterUpdate,
1550 PGOOpt->ProfileFile, PGOOpt->ProfileRemappingFile);
1551 }
1552
1553 // Re-compute GlobalsAA here prior to function passes. This is particularly
1554 // useful as the above will have inlined, DCE'ed, and function-attr
1555 // propagated everything. We should at this point have a reasonably minimal
1556 // and richly annotated call graph. By computing aliasing and mod/ref
1557 // information for all local globals here, the late loop passes and notably
1558 // the vectorizer will be able to use them to help recognize vectorizable
1559 // memory operations.
1562
1563 invokeOptimizerEarlyEPCallbacks(MPM, Level, LTOPhase);
1564
1565 FunctionPassManager OptimizePM;
1566
1567 // Only drop unnecessary assumes post-inline and post-link, as otherwise
1568 // additional uses of the affected value may be introduced through inlining
1569 // and CSE.
1570 if (!isLTOPreLink(LTOPhase))
1571 OptimizePM.addPass(DropUnnecessaryAssumesPass());
1572
1573 // Scheduling LoopVersioningLICM when inlining is over, because after that
1574 // we may see more accurate aliasing. Reason to run this late is that too
1575 // early versioning may prevent further inlining due to increase of code
1576 // size. Other optimizations which runs later might get benefit of no-alias
1577 // assumption in clone loop.
1579 OptimizePM.addPass(
1581 // LoopVersioningLICM pass might increase new LICM opportunities.
1583 LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
1584 /*AllowSpeculation=*/true),
1585 /*USeMemorySSA=*/true));
1586 }
1587
1588 OptimizePM.addPass(Float2IntPass());
1590
1591 if (EnableMatrix) {
1592 OptimizePM.addPass(LowerMatrixIntrinsicsPass());
1593 OptimizePM.addPass(EarlyCSEPass());
1594 }
1595
1596 // CHR pass should only be applied with the profile information.
1597 // The check is to check the profile summary information in CHR.
1598 if (EnableCHR && Level == OptimizationLevel::O3)
1599 OptimizePM.addPass(ControlHeightReductionPass());
1600
1601 // FIXME: We need to run some loop optimizations to re-rotate loops after
1602 // simplifycfg and others undo their rotation.
1603
1604 // Optimize the loop execution. These passes operate on entire loop nests
1605 // rather than on each loop in an inside-out manner, and so they are actually
1606 // function passes.
1607
1608 invokeVectorizerStartEPCallbacks(OptimizePM, Level);
1609
1610 LoopPassManager LPM;
1611 // First rotate loops that may have been un-rotated by prior passes.
1612 // Disable header duplication at -Oz.
1613 LPM.addPass(LoopRotatePass(/*EnableLoopHeaderDuplication=*/true,
1614 isLTOPreLink(LTOPhase),
1615 /*CheckExitCount=*/true));
1616 // Some loops may have become dead by now. Try to delete them.
1617 // FIXME: see discussion in https://reviews.llvm.org/D112851,
1618 // this may need to be revisited once we run GVN before loop deletion
1619 // in the simplification pipeline.
1620 LPM.addPass(LoopDeletionPass());
1621
1622 if (PTO.LoopInterchange)
1623 LPM.addPass(LoopInterchangePass());
1624
1625 OptimizePM.addPass(
1626 createFunctionToLoopPassAdaptor(std::move(LPM), /*UseMemorySSA=*/false));
1627
1628 // FIXME: This may not be the right place in the pipeline.
1629 // We need to have the data to support the right place.
1630 if (PTO.LoopFusion)
1631 OptimizePM.addPass(LoopFusePass());
1632
1633 // Distribute loops to allow partial vectorization. I.e. isolate dependences
1634 // into separate loop that would otherwise inhibit vectorization. This is
1635 // currently only performed for loops marked with the metadata
1636 // llvm.loop.distribute=true or when -enable-loop-distribute is specified.
1637 OptimizePM.addPass(LoopDistributePass());
1638
1639 // Populates the VFABI attribute with the scalar-to-vector mappings
1640 // from the TargetLibraryInfo.
1641 OptimizePM.addPass(InjectTLIMappings());
1642
1643 addVectorPasses(Level, OptimizePM, LTOPhase);
1644
1645 invokeVectorizerEndEPCallbacks(OptimizePM, Level);
1646
1647 // LoopSink pass sinks instructions hoisted by LICM, which serves as a
1648 // canonicalization pass that enables other optimizations. As a result,
1649 // LoopSink pass needs to be a very late IR pass to avoid undoing LICM
1650 // result too early.
1651 OptimizePM.addPass(LoopSinkPass());
1652
1653 // And finally clean up LCSSA form before generating code.
1654 OptimizePM.addPass(InstSimplifyPass());
1655
1656 // This hoists/decomposes div/rem ops. It should run after other sink/hoist
1657 // passes to avoid re-sinking, but before SimplifyCFG because it can allow
1658 // flattening of blocks.
1659 OptimizePM.addPass(DivRemPairsPass());
1660
1661 // Merge adjacent icmps into memcmp, then expand memcmp to loads/compares.
1662 // TODO: move this furter up so that it can be optimized by GVN, etc.
1663 if (EnableMergeICmps)
1664 OptimizePM.addPass(MergeICmpsPass());
1665 OptimizePM.addPass(ExpandMemCmpPass());
1666
1667 // Try to annotate calls that were created during optimization.
1668 OptimizePM.addPass(
1669 TailCallElimPass(/*UpdateFunctionEntryCount=*/isInstrumentedPGOUse()));
1670
1671 // LoopSink (and other loop passes since the last simplifyCFG) might have
1672 // resulted in single-entry-single-exit or empty blocks. Clean up the CFG.
1673 OptimizePM.addPass(
1675 .convertSwitchRangeToICmp(true)
1676 .convertSwitchToArithmetic(true)
1677 .speculateUnpredictables(true)
1678 .hoistLoadsStoresWithCondFaulting(true)));
1679
1680 // Add the core optimizing pipeline.
1681 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(OptimizePM),
1682 PTO.EagerlyInvalidateAnalyses));
1683
1684 // AllocToken transforms heap allocation calls; this needs to run late after
1685 // other allocation call transformations (such as those in InstCombine).
1686 if (!isLTOPreLink(LTOPhase))
1687 MPM.addPass(AllocTokenPass());
1688
1689 invokeOptimizerLastEPCallbacks(MPM, Level, LTOPhase);
1690
1691 // Run the Instrumentor pass late.
1693 MPM.addPass(InstrumentorPass(FS));
1694
1695 // Split out cold code. Splitting is done late to avoid hiding context from
1696 // other optimizations and inadvertently regressing performance. The tradeoff
1697 // is that this has a higher code size cost than splitting early.
1698 if (EnableHotColdSplit && !isLTOPreLink(LTOPhase))
1700
1701 // Now we need to do some global optimization transforms.
1702 // FIXME: It would seem like these should come first in the optimization
1703 // pipeline and maybe be the bottom of the canonicalization pipeline? Weird
1704 // ordering here.
1705 MPM.addPass(GlobalDCEPass());
1707
1708 // Merge functions if requested. It has a better chance to merge functions
1709 // after ConstantMerge folded jump tables.
1710 if (PTO.MergeFunctions)
1712
1713 if (PTO.CallGraphProfile && !isLTOPreLink(LTOPhase))
1714 MPM.addPass(CGProfilePass(isLTOPostLink(LTOPhase)));
1715
1716 // RelLookupTableConverterPass runs later in LTO post-link pipeline.
1717 if (!isLTOPreLink(LTOPhase))
1719
1720 // Add devirtualization pass only when LTO is not enabled, as otherwise
1721 // the pass is already enabled in the LTO pipeline.
1722 if (PTO.DevirtualizeSpeculatively && LTOPhase == ThinOrFullLTOPhase::None) {
1723 // TODO: explore a better pipeline configuration that can improve
1724 // compilation time overhead.
1725 // FIXME: move this earlier (lots of pass ordering tests will need fixing)
1726 MPM.addPass(AssignGUIDPass());
1728 /*ExportSummary*/ nullptr,
1729 /*ImportSummary*/ nullptr,
1730 /*DevirtSpeculatively*/ PTO.DevirtualizeSpeculatively));
1732 // Given that the devirtualization creates more opportunities for inlining,
1733 // we run the Inliner again here to maximize the optimization gain we
1734 // get from devirtualization.
1735 // Also, we can't run devirtualization before inlining because the
1736 // devirtualization depends on the passes optimizing/eliminating vtable GVs
1737 // and those passes are only effective after inlining.
1738 if (EnableModuleInliner) {
1742 } else {
1745 /* MandatoryFirst */ true,
1747 }
1748 }
1749
1750 // Attach !implicit.ref metadata from all functions to copyright strings.
1752
1753 return MPM;
1754}
1755
1759 if (Level == OptimizationLevel::O0)
1760 return buildO0DefaultPipeline(Level, Phase);
1761
1763 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1764 // Currently this pipeline is only invoked in an LTO pre link pass or when we
1765 // are not running LTO. If that changes the below checks may need updating.
1767
1768 // If we are invoking this in non-LTO mode, remove any MemProf related
1769 // attributes and metadata, as we don't know whether we are linking with
1770 // a library containing the necessary interfaces.
1773
1774 // Convert @llvm.global.annotations to !annotation metadata.
1776
1777 // Force any function attributes we want the rest of the pipeline to observe.
1779
1780 if (TriggerCrash)
1782
1783 if (PGOOpt && PGOOpt->DebugInfoForProfiling)
1785
1786 // Apply module pipeline start EP callback.
1788
1789 // Add the core simplification pipeline.
1791
1792 // Now add the optimization pipeline.
1794
1795 if (PGOOpt && PGOOpt->PseudoProbeForProfiling &&
1796 PGOOpt->Action == PGOOptions::SampleUse)
1798
1799 // Emit annotation remarks.
1801
1802 if (isLTOPreLink(Phase))
1803 addRequiredLTOPreLinkPasses(MPM);
1804
1805 instructionCountersPass(MPM, /* IsPreOptimization */ false);
1806 return MPM;
1807}
1808
1811 bool EmitSummary, bool Verify) {
1813
1814 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1815
1816 if (ThinLTO)
1818 else
1820 // AssignGUIDPass attaches !guid metadata (MD_unique_id) to global objects,
1821 // triggering the bitcode writer to emit a METADATA_KIND_BLOCK. Standard LTO
1822 // bitcode emission runs VerifierPass by default, which registers metadata
1823 // kind IDs in LLVMContext. Running VerifierPass here before EmbedBitcodePass
1824 // to get the same behavior.
1825 if (Verify)
1826 MPM.addPass(VerifierPass());
1827 MPM.addPass(EmbedBitcodePass(ThinLTO, EmitSummary));
1828
1829 // Perform any cleanups to the IR that aren't suitable for per TU compilation,
1830 // like removing CFI/WPD related instructions. Note, we reuse
1831 // DropTypeTestsPass to clean up type tests rather than duplicate that logic
1832 // in FatLtoCleanup.
1833 MPM.addPass(FatLtoCleanup());
1834
1835 // If we're doing FatLTO w/ CFI enabled, we don't want the type tests in the
1836 // object code, only in the bitcode section, so drop it before we run
1837 // module optimization and generate machine code. If llvm.type.test() isn't in
1838 // the IR, this won't do anything.
1840
1841 // Use the ThinLTO post-link pipeline with sample profiling
1842 if (ThinLTO && PGOOpt && PGOOpt->Action == PGOOptions::SampleUse)
1843 MPM.addPass(buildThinLTODefaultPipeline(Level, /*ImportSummary=*/nullptr));
1844 else {
1845 // ModuleSimplification does not run the coroutine passes for
1846 // ThinLTOPreLink, so we need the coroutine passes to run for ThinLTO
1847 // builds, otherwise they will miscompile.
1848 if (ThinLTO) {
1849 // TODO: replace w/ buildCoroWrapper() when it takes phase and level into
1850 // consideration.
1851 CGSCCPassManager CGPM;
1855 MPM.addPass(CoroCleanupPass());
1856 }
1857
1858 // otherwise, just use module optimization
1859 MPM.addPass(
1861 // Emit annotation remarks.
1863 }
1864
1865 instructionCountersPass(MPM, /* IsPreOptimization */ false);
1866
1867 return MPM;
1868}
1869
1872 if (Level == OptimizationLevel::O0)
1874
1876
1877 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1878
1879 // Convert @llvm.global.annotations to !annotation metadata.
1881
1882 // Force any function attributes we want the rest of the pipeline to observe.
1884
1885 if (PGOOpt && PGOOpt->DebugInfoForProfiling)
1887
1888 // Apply module pipeline start EP callback.
1890
1891 // If we are planning to perform ThinLTO later, we don't bloat the code with
1892 // unrolling/vectorization/... now. Just simplify the module as much as we
1893 // can.
1896 // In pre-link, for ctx prof use, we stop here with an instrumented IR. We let
1897 // thinlto use the contextual info to perform imports; then use the contextual
1898 // profile in the post-thinlink phase.
1899 if (!UseCtxProfile.empty()) {
1900 addRequiredLTOPreLinkPasses(MPM);
1901 return MPM;
1902 }
1903
1904 // Run partial inlining pass to partially inline functions that have
1905 // large bodies.
1906 // FIXME: It isn't clear whether this is really the right place to run this
1907 // in ThinLTO. Because there is another canonicalization and simplification
1908 // phase that will run after the thin link, running this here ends up with
1909 // less information than will be available later and it may grow functions in
1910 // ways that aren't beneficial.
1913
1914 if (PGOOpt && PGOOpt->PseudoProbeForProfiling &&
1915 PGOOpt->Action == PGOOptions::SampleUse)
1917
1918 // Handle Optimizer{Early,Last}EPCallbacks added by clang on PreLink. Actual
1919 // optimization is going to be done in PostLink stage, but clang can't add
1920 // callbacks there in case of in-process ThinLTO called by linker.
1925
1926 // Emit annotation remarks.
1928
1929 // Attach !implicit.ref metadata from all functions to copyright strings.
1931
1932 addRequiredLTOPreLinkPasses(MPM);
1933
1934 instructionCountersPass(MPM, /* IsPreOptimization */ false);
1935
1936 return MPM;
1937}
1938
1940 OptimizationLevel Level, const ModuleSummaryIndex *ImportSummary) {
1942
1943 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1944
1945 // If we are invoking this without a summary index noting that we are linking
1946 // with a library containing the necessary APIs, remove any MemProf related
1947 // attributes and metadata.
1948 if (!ImportSummary || !ImportSummary->withSupportsHotColdNew())
1950
1951 if (ImportSummary) {
1952 // For ThinLTO we must apply the context disambiguation decisions early, to
1953 // ensure we can correctly match the callsites to summary data.
1956 ImportSummary, PGOOpt && PGOOpt->Action == PGOOptions::SampleUse));
1957
1958 // These passes import type identifier resolutions for whole-program
1959 // devirtualization and CFI. They must run early because other passes may
1960 // disturb the specific instruction patterns that these passes look for,
1961 // creating dependencies on resolutions that may not appear in the summary.
1962 //
1963 // For example, GVN may transform the pattern assume(type.test) appearing in
1964 // two basic blocks into assume(phi(type.test, type.test)), which would
1965 // transform a dependency on a WPD resolution into a dependency on a type
1966 // identifier resolution for CFI.
1967 //
1968 // Also, WPD has access to more precise information than ICP and can
1969 // devirtualize more effectively, so it should operate on the IR first.
1970 //
1971 // The WPD and LowerTypeTest passes need to run at -O0 to lower type
1972 // metadata and intrinsics.
1973 MPM.addPass(WholeProgramDevirtPass(nullptr, ImportSummary));
1974 MPM.addPass(LowerTypeTestsPass(nullptr, ImportSummary));
1975 }
1976
1977 if (Level == OptimizationLevel::O0) {
1978 // Run a second time to clean up any type tests left behind by WPD for use
1979 // in ICP.
1982
1983 // AllocToken transforms heap allocation calls; this needs to run late after
1984 // other allocation call transformations (such as those in InstCombine).
1985 MPM.addPass(AllocTokenPass());
1986
1987 // Drop available_externally and unreferenced globals. This is necessary
1988 // with ThinLTO in order to avoid leaving undefined references to dead
1989 // globals in the object file.
1991 MPM.addPass(GlobalDCEPass());
1992 return MPM;
1993 }
1994 if (!UseCtxProfile.empty()) {
1995 MPM.addPass(
1997 } else {
1998 // Add the core simplification pipeline.
2001 }
2002 // Now add the optimization pipeline.
2005
2006 // Emit annotation remarks.
2008
2009 instructionCountersPass(MPM, /* IsPreOptimization */ false);
2010
2011 return MPM;
2012}
2013
2016 // FIXME: We should use a customized pre-link pipeline!
2017 return buildPerModuleDefaultPipeline(Level,
2019}
2020
2023 ModuleSummaryIndex *ExportSummary) {
2025
2026 instructionCountersPass(MPM, /* IsPreOptimization */ true);
2027
2029
2030 // If we are invoking this without a summary index noting that we are linking
2031 // with a library containing the necessary APIs, remove any MemProf related
2032 // attributes and metadata.
2033 if (!ExportSummary || !ExportSummary->withSupportsHotColdNew())
2035
2036 // Create a function that performs CFI checks for cross-DSO calls with targets
2037 // in the current module.
2038 MPM.addPass(CrossDSOCFIPass());
2039
2040 if (Level == OptimizationLevel::O0) {
2041 // The WPD and LowerTypeTest passes need to run at -O0 to lower type
2042 // metadata and intrinsics.
2043 MPM.addPass(WholeProgramDevirtPass(ExportSummary, nullptr));
2044 MPM.addPass(LowerTypeTestsPass(ExportSummary, nullptr));
2045 // Run a second time to clean up any type tests left behind by WPD for use
2046 // in ICP.
2048
2050
2051 // AllocToken transforms heap allocation calls; this needs to run late after
2052 // other allocation call transformations (such as those in InstCombine).
2053 MPM.addPass(AllocTokenPass());
2054
2056
2057 // Emit annotation remarks.
2059
2060 return MPM;
2061 }
2062
2063 if (PGOOpt && PGOOpt->Action == PGOOptions::SampleUse) {
2064 // Load sample profile before running the LTO optimization pipeline.
2065 MPM.addPass(SampleProfileLoaderPass(PGOOpt->ProfileFile,
2066 PGOOpt->ProfileRemappingFile,
2068 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
2069 // RequireAnalysisPass for PSI before subsequent non-module passes.
2071 }
2072
2073 // Try to run OpenMP optimizations, quick no-op if no OpenMP metadata present.
2075
2076 // Remove unused virtual tables to improve the quality of code generated by
2077 // whole-program devirtualization and bitset lowering.
2078 MPM.addPass(GlobalDCEPass(/*InLTOPostLink=*/true));
2079
2080 // Do basic inference of function attributes from known properties of system
2081 // libraries and other oracles.
2083
2084 if (Level >= OptimizationLevel::O2) {
2086 CallSiteSplittingPass(), PTO.EagerlyInvalidateAnalyses));
2087
2088 // Indirect call promotion. This should promote all the targets that are
2089 // left by the earlier promotion pass that promotes intra-module targets.
2090 // This two-step promotion is to save the compile time. For LTO, it should
2091 // produce the same result as if we only do promotion here.
2093 true /* InLTO */, PGOOpt && PGOOpt->Action == PGOOptions::SampleUse));
2094
2095 // Promoting by-reference arguments to by-value exposes more constants to
2096 // IPSCCP.
2097 CGSCCPassManager CGPM;
2100 CGPM.addPass(
2103
2104 // Propagate constants at call sites into the functions they call. This
2105 // opens opportunities for globalopt (and inlining) by substituting function
2106 // pointers passed as arguments to direct uses of functions.
2107 MPM.addPass(IPSCCPPass(IPSCCPOptions(/*AllowFuncSpec=*/true)));
2108
2109 // Attach metadata to indirect call sites indicating the set of functions
2110 // they may target at run-time. This should follow IPSCCP.
2112 }
2113
2114 // Do RPO function attribute inference across the module to forward-propagate
2115 // attributes where applicable.
2116 // FIXME: Is this really an optimization rather than a canonicalization?
2118
2119 // Use in-range annotations on GEP indices to split globals where beneficial.
2120 MPM.addPass(GlobalSplitPass());
2121
2122 // Run whole program optimization of virtual call when the list of callees
2123 // is fixed.
2124 MPM.addPass(WholeProgramDevirtPass(ExportSummary, nullptr));
2125
2127 // Stop here at -O1.
2128 if (Level == OptimizationLevel::O1) {
2129 // The LowerTypeTestsPass needs to run to lower type metadata and the
2130 // type.test intrinsics. The pass does nothing if CFI is disabled.
2131 MPM.addPass(LowerTypeTestsPass(ExportSummary, nullptr));
2132 // Run a second time to clean up any type tests left behind by WPD for use
2133 // in ICP (which is performed earlier than this in the regular LTO
2134 // pipeline).
2136
2138
2139 // AllocToken transforms heap allocation calls; this needs to run late after
2140 // other allocation call transformations (such as those in InstCombine).
2141 MPM.addPass(AllocTokenPass());
2142
2144
2145 // Emit annotation remarks.
2147
2148 instructionCountersPass(MPM, /* IsPreOptimization */ false);
2149
2150 return MPM;
2151 }
2152
2153 // TODO: Skip to match buildCoroWrapper.
2154 MPM.addPass(CoroEarlyPass());
2155
2156 // Optimize globals to try and fold them into constants.
2157 MPM.addPass(GlobalOptPass());
2158
2159 // Promote any localized globals to SSA registers.
2161
2162 // Linking modules together can lead to duplicate global constant, only
2163 // keep one copy of each constant.
2165
2166 // Remove unused arguments from functions.
2168
2169 // Reduce the code after globalopt and ipsccp. Both can open up significant
2170 // simplification opportunities, and both can propagate functions through
2171 // function pointers. When this happens, we often have to resolve varargs
2172 // calls, etc, so let instcombine do this.
2173 FunctionPassManager PeepholeFPM;
2174 PeepholeFPM.addPass(InstCombinePass());
2175 if (Level >= OptimizationLevel::O2)
2176 PeepholeFPM.addPass(AggressiveInstCombinePass());
2177 invokePeepholeEPCallbacks(PeepholeFPM, Level);
2178
2179 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(PeepholeFPM),
2180 PTO.EagerlyInvalidateAnalyses));
2181
2182 // Lower variadic functions for supported targets prior to inlining.
2184
2185 // Note: historically, the PruneEH pass was run first to deduce nounwind and
2186 // generally clean up exception handling overhead. It isn't clear this is
2187 // valuable as the inliner doesn't currently care whether it is inlining an
2188 // invoke or a call.
2189 // Run the inliner now.
2190 if (EnableModuleInliner) {
2194 } else {
2197 /* MandatoryFirst */ true,
2200 }
2201
2202 // Perform context disambiguation after inlining, since that would reduce the
2203 // amount of additional cloning required to distinguish the allocation
2204 // contexts.
2207 /*Summary=*/nullptr,
2208 PGOOpt && PGOOpt->Action == PGOOptions::SampleUse));
2209
2210 // Optimize globals again after we ran the inliner.
2211 MPM.addPass(GlobalOptPass());
2212
2213 // Run the OpenMPOpt pass again after global optimizations.
2215
2216 // Garbage collect dead functions.
2217 MPM.addPass(GlobalDCEPass(/*InLTOPostLink=*/true));
2218
2219 // If we didn't decide to inline a function, check to see if we can
2220 // transform it to pass arguments by value instead of by reference.
2221 CGSCCPassManager CGPM;
2227
2229 // The IPO Passes may leave cruft around. Clean up after them.
2230 FPM.addPass(InstCombinePass());
2231 invokePeepholeEPCallbacks(FPM, Level);
2232
2235
2237
2238 // Do a post inline PGO instrumentation and use pass. This is a context
2239 // sensitive PGO pass.
2240 if (PGOOpt) {
2241 if (PGOOpt->CSAction == PGOOptions::CSIRInstr)
2242 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/true,
2243 /*IsCS=*/true, PGOOpt->AtomicCounterUpdate,
2244 PGOOpt->CSProfileGenFile, PGOOpt->ProfileRemappingFile);
2245 else if (PGOOpt->CSAction == PGOOptions::CSIRUse)
2246 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/false,
2247 /*IsCS=*/true, PGOOpt->AtomicCounterUpdate,
2248 PGOOpt->ProfileFile, PGOOpt->ProfileRemappingFile);
2249 }
2250
2251 // Break up allocas
2253
2254 // LTO provides additional opportunities for tailcall elimination due to
2255 // link-time inlining, and visibility of nocapture attribute.
2256 FPM.addPass(
2257 TailCallElimPass(/*UpdateFunctionEntryCount=*/isInstrumentedPGOUse()));
2258
2259 // Run a few AA driver optimizations here and now to cleanup the code.
2260 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(FPM),
2261 PTO.EagerlyInvalidateAnalyses));
2262
2263 MPM.addPass(
2265
2266 // Require the GlobalsAA analysis for the module so we can query it within
2267 // MainFPM.
2270 // Invalidate AAManager so it can be recreated and pick up the newly
2271 // available GlobalsAA.
2272 MPM.addPass(
2274 }
2275
2276 FunctionPassManager MainFPM;
2278 LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
2279 /*AllowSpeculation=*/true),
2280 /*USeMemorySSA=*/true));
2281
2282 if (RunNewGVN)
2283 MainFPM.addPass(NewGVNPass());
2284 else
2285 MainFPM.addPass(GVNPass());
2286
2287 // Remove dead memcpy()'s.
2288 MainFPM.addPass(MemCpyOptPass());
2289
2290 // Nuke dead stores.
2291 MainFPM.addPass(DSEPass());
2292 MainFPM.addPass(MoveAutoInitPass());
2294
2295 invokeVectorizerStartEPCallbacks(MainFPM, Level);
2296
2297 LoopPassManager LPM;
2299 LPM.addPass(LoopFlattenPass());
2300 LPM.addPass(IndVarSimplifyPass());
2301 LPM.addPass(LoopDeletionPass());
2302 // FIXME: Add loop interchange.
2303
2304 // Unroll small loops and perform peeling.
2305 LPM.addPass(LoopFullUnrollPass(static_cast<int>(Level),
2306 /* OnlyWhenForced= */ !PTO.LoopUnrolling,
2307 PTO.ForgetAllSCEVInLoopUnroll));
2308 // The loop passes in LPM (LoopFullUnrollPass) do not preserve MemorySSA.
2309 // *All* loop passes must preserve it, in order to be able to use it.
2310 MainFPM.addPass(
2311 createFunctionToLoopPassAdaptor(std::move(LPM), /*UseMemorySSA=*/false));
2312
2313 MainFPM.addPass(LoopDistributePass());
2314
2315 addVectorPasses(Level, MainFPM, ThinOrFullLTOPhase::FullLTOPostLink);
2316
2317 invokeVectorizerEndEPCallbacks(MainFPM, Level);
2318
2319 // Run the OpenMPOpt CGSCC pass again late.
2322
2323 invokePeepholeEPCallbacks(MainFPM, Level);
2324 MainFPM.addPass(JumpThreadingPass());
2325 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(MainFPM),
2326 PTO.EagerlyInvalidateAnalyses));
2327
2328 // Lower type metadata and the type.test intrinsic. This pass supports
2329 // clang's control flow integrity mechanisms (-fsanitize=cfi*) and needs
2330 // to be run at link time if CFI is enabled. This pass does nothing if
2331 // CFI is disabled.
2332 MPM.addPass(LowerTypeTestsPass(ExportSummary, nullptr));
2333 // Run a second time to clean up any type tests left behind by WPD for use
2334 // in ICP (which is performed earlier than this in the regular LTO pipeline).
2336
2337 // Enable splitting late in the FullLTO post-link pipeline.
2340
2341 // Add late LTO optimization passes.
2342 FunctionPassManager LateFPM;
2343
2344 // LoopSink pass sinks instructions hoisted by LICM, which serves as a
2345 // canonicalization pass that enables other optimizations. As a result,
2346 // LoopSink pass needs to be a very late IR pass to avoid undoing LICM
2347 // result too early.
2348 LateFPM.addPass(LoopSinkPass());
2349
2350 // This hoists/decomposes div/rem ops. It should run after other sink/hoist
2351 // passes to avoid re-sinking, but before SimplifyCFG because it can allow
2352 // flattening of blocks.
2353 LateFPM.addPass(DivRemPairsPass());
2354
2355 // Delete basic blocks, which optimization passes may have killed.
2357 .convertSwitchRangeToICmp(true)
2358 .convertSwitchToArithmetic(true)
2359 .hoistCommonInsts(true)
2360 .speculateUnpredictables(true)));
2361 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(LateFPM)));
2362
2363 // Drop bodies of available eternally objects to improve GlobalDCE.
2365
2366 // Now that we have optimized the program, discard unreachable functions.
2367 MPM.addPass(GlobalDCEPass(/*InLTOPostLink=*/true));
2368
2369 if (PTO.MergeFunctions)
2371
2373
2374 if (PTO.CallGraphProfile)
2375 MPM.addPass(CGProfilePass(/*InLTOPostLink=*/true));
2376
2377 MPM.addPass(CoroCleanupPass());
2378
2379 // AllocToken transforms heap allocation calls; this needs to run late after
2380 // other allocation call transformations (such as those in InstCombine).
2381 MPM.addPass(AllocTokenPass());
2382
2384
2385 // Emit annotation remarks.
2387
2388 instructionCountersPass(MPM, /* IsPreOptimization */ false);
2389
2390 return MPM;
2391}
2392
2396 assert(Level == OptimizationLevel::O0 &&
2397 "buildO0DefaultPipeline should only be used with O0");
2398
2400
2401 instructionCountersPass(MPM, /* IsPreOptimization */ true);
2402
2403 // Perform pseudo probe instrumentation in O0 mode. This is for the
2404 // consistency between different build modes. For example, a LTO build can be
2405 // mixed with an O0 prelink and an O2 postlink. Loading a sample profile in
2406 // the postlink will require pseudo probe instrumentation in the prelink.
2407 if (PGOOpt && PGOOpt->PseudoProbeForProfiling)
2409
2410 if (PGOOpt && (PGOOpt->Action == PGOOptions::IRInstr ||
2411 PGOOpt->Action == PGOOptions::IRUse))
2413 MPM,
2414 /*RunProfileGen=*/(PGOOpt->Action == PGOOptions::IRInstr),
2415 /*IsCS=*/false, PGOOpt->AtomicCounterUpdate, PGOOpt->ProfileFile,
2416 PGOOpt->ProfileRemappingFile);
2417
2418 // Instrument function entry and exit before all inlining.
2420 EntryExitInstrumenterPass(/*PostInlining=*/false)));
2421
2423
2424 if (PGOOpt && PGOOpt->DebugInfoForProfiling)
2426
2427 if (PGOOpt && PGOOpt->Action == PGOOptions::SampleUse) {
2428 // Explicitly disable sample loader inlining and use flattened profile in O0
2429 // pipeline.
2430 MPM.addPass(SampleProfileLoaderPass(PGOOpt->ProfileFile,
2431 PGOOpt->ProfileRemappingFile,
2433 /*DisableSampleProfileInlining=*/true,
2434 /*UseFlattenedProfile=*/true));
2435 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
2436 // RequireAnalysisPass for PSI before subsequent non-module passes.
2438 }
2439
2441
2442 // Build a minimal pipeline based on the semantics required by LLVM,
2443 // which is just that always inlining occurs. Further, disable generating
2444 // lifetime intrinsics to avoid enabling further optimizations during
2445 // code generation.
2447 /*InsertLifetimeIntrinsics=*/false));
2448
2449 if (PTO.MergeFunctions)
2451
2452 if (EnableMatrix)
2453 MPM.addPass(
2455
2456 if (!CGSCCOptimizerLateEPCallbacks.empty()) {
2457 CGSCCPassManager CGPM;
2459 if (!CGPM.isEmpty())
2461 }
2462 if (!LateLoopOptimizationsEPCallbacks.empty()) {
2463 LoopPassManager LPM;
2465 if (!LPM.isEmpty()) {
2467 createFunctionToLoopPassAdaptor(std::move(LPM))));
2468 }
2469 }
2470 if (!LoopOptimizerEndEPCallbacks.empty()) {
2471 LoopPassManager LPM;
2473 if (!LPM.isEmpty()) {
2475 createFunctionToLoopPassAdaptor(std::move(LPM))));
2476 }
2477 }
2478 if (!ScalarOptimizerLateEPCallbacks.empty()) {
2481 if (!FPM.isEmpty())
2482 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(FPM)));
2483 }
2484
2486
2487 if (!VectorizerStartEPCallbacks.empty()) {
2490 if (!FPM.isEmpty())
2491 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(FPM)));
2492 }
2493
2494 if (!VectorizerEndEPCallbacks.empty()) {
2497 if (!FPM.isEmpty())
2498 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(FPM)));
2499 }
2500
2502
2503 // AllocToken transforms heap allocation calls; this needs to run late after
2504 // other allocation call transformations (such as those in InstCombine).
2505 if (!isLTOPreLink(Phase))
2506 MPM.addPass(AllocTokenPass());
2507
2509
2511 MPM.addPass(InstrumentorPass(FS));
2512
2513 // Attach !implicit.ref metadata from all functions to copyright strings.
2515
2516 if (isLTOPreLink(Phase))
2517 addRequiredLTOPreLinkPasses(MPM);
2518
2519 // Emit annotation remarks.
2521
2522 instructionCountersPass(MPM, /* IsPreOptimization */ false);
2523
2524 return MPM;
2525}
2526
2528 AAManager AA;
2529
2530 // The order in which these are registered determines their priority when
2531 // being queried.
2532
2533 // Add any target-specific alias analyses that should be run early.
2534 if (TM)
2535 TM->registerEarlyDefaultAliasAnalyses(AA);
2536
2537 // First we register the basic alias analysis that provides the majority of
2538 // per-function local AA logic. This is a stateless, on-demand local set of
2539 // AA techniques.
2540 AA.registerFunctionAnalysis<BasicAA>();
2541
2542 // Next we query fast, specialized alias analyses that wrap IR-embedded
2543 // information about aliasing.
2544 AA.registerFunctionAnalysis<ScopedNoAliasAA>();
2545 AA.registerFunctionAnalysis<TypeBasedAA>();
2546
2547 // Add support for querying global aliasing information when available.
2548 // Because the `AAManager` is a function analysis and `GlobalsAA` is a module
2549 // analysis, all that the `AAManager` can do is query for any *cached*
2550 // results from `GlobalsAA` through a readonly proxy.
2552 AA.registerModuleAnalysis<GlobalsAA>();
2553
2554 // Add target-specific alias analyses.
2555 if (TM)
2556 TM->registerDefaultAliasAnalyses(AA);
2557
2558 return AA;
2559}
2560
2561bool PassBuilder::isInstrumentedPGOUse() const {
2562 return (PGOOpt && PGOOpt->Action == PGOOptions::IRUse) ||
2563 !UseCtxProfile.empty();
2564}
aarch64 falkor hwpf fix Falkor HW Prefetch Fix Late Phase
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AggressiveInstCombiner - Combine expression patterns to form expressions with fewer,...
Provides passes to inlining "always_inline" functions.
This is the interface for LLVM's primary stateless and local alias analysis.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
This file provides the interface for LLVM's Call Graph Profile pass.
This header provides classes for managing passes over SCCs of the call graph.
#define clEnumValN(ENUMVAL, FLAGNAME, DESC)
This file provides the interface for a simple, fast CSE pass.
This file provides a pass which clones the current module and runs the provided pass pipeline on the ...
This file provides a pass manager that only runs its passes if the provided marker analysis has been ...
Super simple passes to force specific function attrs from the commandline into the IR for debugging p...
Provides passes for computing function attributes based on interprocedural analyses.
This file provides the interface for LLVM's Global Value Numbering pass which eliminates fully redund...
This is the interface for a simple mod/ref and alias analysis over globals.
AcceleratorCodeSelection - Identify all functions reachable from a kernel, removing those that are un...
This header defines various interfaces for pass management in LLVM.
Interfaces for passes which infer implicit function attributes from the name and signature of functio...
This file provides the primary interface to the instcombine pass.
Defines passes for running instruction simplification across chunks of IR.
This file provides the interface for LLVM's PGO Instrumentation lowering pass.
See the comments on JumpThreadingPass.
static LVOptions Options
Definition LVOptions.cpp:25
This file implements the Loop Fusion pass.
This header defines the LoopLoadEliminationPass object.
This header provides classes for managing a pipeline of passes over loops in LLVM IR.
The header file for the LowerConstantIntrinsics pass as used by the new pass manager.
The header file for the LowerExpectIntrinsic pass as used by the new pass manager.
This pass performs merges of loads and stores on both sides of a.
This file provides the interface for LLVM's Global Value Numbering pass.
This header enumerates the LLVM-provided high-level optimization levels.
This file provides the interface for IR based instrumentation passes ( (profile-gen,...
Define option tunables for PGO.
ppc ctr loops PowerPC CTR Loops Verify
static bool isThinLTOPostLink(ThinOrFullLTOPhase Phase)
static void addAnnotationRemarksPass(ModulePassManager &MPM)
static CoroConditionalWrapper buildCoroWrapper(ThinOrFullLTOPhase Phase)
static bool isFullLTOPostLink(ThinOrFullLTOPhase Phase)
static bool isThinLTOPreLink(ThinOrFullLTOPhase Phase)
static bool isLTOPreLink(ThinOrFullLTOPhase Phase)
static void instructionCountersPass(ModulePassManager &MPM, bool IsPreOptimization)
static bool isFullLTOPreLink(ThinOrFullLTOPhase Phase)
static bool isLTOPostLink(ThinOrFullLTOPhase Phase)
This file implements relative lookup table converter that converts lookup tables to relative lookup t...
This file provides the interface for LLVM's Scalar Replacement of Aggregates pass.
This file provides the interface for the pseudo probe implementation for AutoFDO.
This file provides the interface for the sampled PGO loader pass.
This is the interface for a metadata-based scoped no-alias analysis.
This file provides the interface for the pass responsible for both simplifying and canonicalizing the...
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
This is the interface for a metadata-based TBAA.
Defines the virtual file system interface vfs::FileSystem.
A manager for alias analyses.
A module pass that rewrites heap allocations to use token-enabled allocation functions based on vario...
Definition AllocToken.h:36
Inlines functions marked as "always_inline".
Argument promotion pass.
Analysis pass providing a never-invalidated alias analysis result.
Simple pass that canonicalizes aliases.
A pass that merges duplicate global constants into a single constant.
This class implements a trivial dead store elimination.
Eliminate dead arguments (and return values) from functions.
A pass that transforms external global definitions into declarations.
Pass embeds a copy of the module optimized with the provided pass pipeline into a global variable.
A pass manager to run a set of extra loop passes if the MarkerTy analysis is present.
Statistics pass for the FunctionPropertiesAnalysis results.
The core GVN pass object.
Definition GVN.h:123
Pass to remove unused function declarations.
Definition GlobalDCE.h:38
Optimize globals that never have their address taken.
Definition GlobalOpt.h:25
Pass to perform split of global variables.
Definition GlobalSplit.h:26
Analysis pass providing a never-invalidated alias analysis result.
Pass to outline cold regions.
Pass to perform interprocedural constant propagation.
Definition SCCP.h:48
Run instruction simplification across each instruction in the function.
Instrumentation based profiling lowering pass.
The Instrumentor pass.
This pass performs 'jump threading', which looks at blocks that have multiple predecessors and multip...
Performs Loop Invariant Code Motion Pass.
Definition LICM.h:66
Loop unroll pass that only does full loop unrolling and peeling.
Performs Loop Idiom Recognize Pass.
Performs Loop Inst Simplify Pass.
A simple loop rotation transformation.
Performs basic CFG simplifications to assist other loop passes.
A pass that does profile-guided sinking of instructions into loops.
Definition LoopSink.h:33
A simple loop rotation transformation.
Loop unroll pass that will support both full and partial unrolling.
Strips MemProf attributes and metadata.
Merge identical functions.
The module inliner pass for the new pass manager.
Module pass, wrapping the inliner pass.
Definition Inliner.h:65
void addModulePass(T Pass)
Add a module pass that runs before the CGSCC passes.
Definition Inliner.h:81
Class to hold module path string table and global value map, and encapsulate methods for operating on...
Simple pass that provides a name to every anonymous globals.
Additional 'norecurse' attribute deduction during postlink LTO phase.
OpenMP optimizations pass.
Definition OpenMPOpt.h:42
static LLVM_ABI bool isCtxIRPGOInstrEnabled()
The indirect function call promotion pass.
The instrumentation (profile-instr-gen) pass for IR based PGO.
The instrumentation (profile-instr-gen) pass for IR based PGO.
The profile annotation (profile-instr-use) pass for IR based PGO.
The profile size based optimization pass for memory intrinsics.
Pass to remove unused function declarations.
LLVM_ABI void invokeFullLinkTimeOptimizationLastEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level)
LLVM_ABI ModuleInlinerWrapperPass buildInlinerPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase)
Construct the module pipeline that performs inlining as well as the inlining-driven cleanups.
LLVM_ABI void invokeOptimizerEarlyEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level, ThinOrFullLTOPhase Phase)
LLVM_ABI ModulePassManager buildFatLTODefaultPipeline(OptimizationLevel Level, bool ThinLTO, bool EmitSummary, bool Verify=true)
Build a fat object default optimization pipeline.
LLVM_ABI void invokeVectorizerStartEPCallbacks(FunctionPassManager &FPM, OptimizationLevel Level)
LLVM_ABI AAManager buildDefaultAAPipeline()
Build the default AAManager with the default alias analysis pipeline registered.
LLVM_ABI void invokeCGSCCOptimizerLateEPCallbacks(CGSCCPassManager &CGPM, OptimizationLevel Level)
LLVM_ABI ModulePassManager buildThinLTOPreLinkDefaultPipeline(OptimizationLevel Level)
Build a pre-link, ThinLTO-targeting default optimization pipeline to a pass manager.
LLVM_ABI void addPGOInstrPassesForO0(ModulePassManager &MPM, bool RunProfileGen, bool IsCS, bool AtomicCounterUpdate, std::string ProfileFile, std::string ProfileRemappingFile)
Add PGOInstrumenation passes for O0 only.
LLVM_ABI void invokeScalarOptimizerLateEPCallbacks(FunctionPassManager &FPM, OptimizationLevel Level)
LLVM_ABI ModulePassManager buildPerModuleDefaultPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase=ThinOrFullLTOPhase::None)
Build a per-module default optimization pipeline.
LLVM_ABI void invokePipelineStartEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level)
LLVM_ABI void invokeVectorizerEndEPCallbacks(FunctionPassManager &FPM, OptimizationLevel Level)
LLVM_ABI ModulePassManager buildO0DefaultPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase=ThinOrFullLTOPhase::None)
Build an O0 pipeline with the minimal semantically required passes.
LLVM_ABI FunctionPassManager buildFunctionSimplificationPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase)
Construct the core LLVM function canonicalization and simplification pipeline.
LLVM_ABI void invokePeepholeEPCallbacks(FunctionPassManager &FPM, OptimizationLevel Level)
LLVM_ABI void invokePipelineEarlySimplificationEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level, ThinOrFullLTOPhase Phase)
LLVM_ABI void invokeLoopOptimizerEndEPCallbacks(LoopPassManager &LPM, OptimizationLevel Level)
LLVM_ABI ModulePassManager buildLTODefaultPipeline(OptimizationLevel Level, ModuleSummaryIndex *ExportSummary)
Build an LTO default optimization pipeline to a pass manager.
LLVM_ABI ModulePassManager buildModuleInlinerPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase)
Construct the module pipeline that performs inlining with module inliner pass.
LLVM_ABI ModulePassManager buildThinLTODefaultPipeline(OptimizationLevel Level, const ModuleSummaryIndex *ImportSummary)
Build a ThinLTO default optimization pipeline to a pass manager.
LLVM_ABI void invokeLateLoopOptimizationsEPCallbacks(LoopPassManager &LPM, OptimizationLevel Level)
LLVM_ABI void invokeFullLinkTimeOptimizationEarlyEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level)
LLVM_ABI ModulePassManager buildModuleSimplificationPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase)
Construct the core LLVM module canonicalization and simplification pipeline.
LLVM_ABI ModulePassManager buildModuleOptimizationPipeline(OptimizationLevel Level, ThinOrFullLTOPhase LTOPhase)
Construct the core LLVM module optimization pipeline.
LLVM_ABI void invokeOptimizerLastEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level, ThinOrFullLTOPhase Phase)
LLVM_ABI ModulePassManager buildLTOPreLinkDefaultPipeline(OptimizationLevel Level)
Build a pre-link, LTO-targeting default optimization pipeline to a pass manager.
LLVM_ATTRIBUTE_MINSIZE std::enable_if_t<!std::is_same_v< PassT, PassManager > > addPass(PassT &&Pass)
bool isEmpty() const
Returns if the pass manager contains any passes.
unsigned LicmMssaNoAccForPromotionCap
Tuning option to disable promotion to scalars in LICM with MemorySSA, if the number of access is too ...
Definition PassBuilder.h:78
bool SLPVectorization
Tuning option to enable/disable slp loop vectorization, set based on opt level.
Definition PassBuilder.h:56
int InlinerThreshold
Tuning option to override the default inliner threshold.
Definition PassBuilder.h:92
bool LoopFusion
Tuning option to enable/disable loop fusion. Its default value is false.
Definition PassBuilder.h:66
bool CallGraphProfile
Tuning option to enable/disable call graph profile.
Definition PassBuilder.h:82
bool MergeFunctions
Tuning option to enable/disable function merging.
Definition PassBuilder.h:89
bool ForgetAllSCEVInLoopUnroll
Tuning option to forget all SCEV loops in LoopUnroll.
Definition PassBuilder.h:70
unsigned LicmMssaOptCap
Tuning option to cap the number of calls to retrive clobbering accesses in MemorySSA,...
Definition PassBuilder.h:74
bool LoopInterleaving
Tuning option to set loop interleaving on/off, set based on opt level.
Definition PassBuilder.h:48
LLVM_ABI PipelineTuningOptions()
Constructor sets pipeline tuning defaults based on cl::opts.
bool LoopUnrolling
Tuning option to enable/disable loop unrolling. Its default value is true.
Definition PassBuilder.h:59
bool LoopInterchange
Tuning option to enable/disable loop interchange.
Definition PassBuilder.h:63
bool LoopVectorization
Tuning option to enable/disable loop vectorization, set based on opt level.
Definition PassBuilder.h:52
Reassociate commutative expressions.
Definition Reassociate.h:75
A pass to do RPO deduction and propagation of function attributes.
This pass performs function-level constant propagation and merging.
Definition SCCP.h:30
The sample profiler data loader pass.
Analysis pass providing a never-invalidated alias analysis result.
This pass transforms loops that contain branches or switches on loop- invariant conditions to have mu...
A pass to simplify and canonicalize the CFG of a function.
Definition SimplifyCFG.h:30
Analysis pass providing a never-invalidated alias analysis result.
Optimize scalar/vector interactions in IR using target cost models.
Create a verifier pass.
Definition Verifier.h:133
Interfaces for registering analysis passes, producing common pass manager configurations,...
Abstract Attribute helper functions.
Definition Attributor.h:165
ValuesClass values(OptsTy... Options)
Helper to build a ValuesClass by forwarding a variable number of arguments as an initializer list to ...
initializer< Ty > init(const Ty &Val)
@ All
Drop only llvm.assumes using type test value.
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI cl::opt< bool > EnableKnowledgeRetention
static cl::opt< bool > RunNewGVN("enable-newgvn", cl::init(false), cl::Hidden, cl::desc("Run the NewGVN pass"))
static cl::opt< bool > DisablePreInliner("disable-preinline", cl::init(false), cl::Hidden, cl::desc("Disable pre-instrumentation inliner"))
static cl::opt< bool > EnableDFAJumpThreading("enable-dfa-jump-thread", cl::desc("Enable DFA jump threading"), cl::init(true), cl::Hidden)
static cl::opt< bool > PerformMandatoryInliningsFirst("mandatory-inlining-first", cl::init(false), cl::Hidden, cl::desc("Perform mandatory inlinings module-wide, before performing " "inlining"))
static cl::opt< bool > RunPartialInlining("enable-partial-inlining", cl::init(false), cl::Hidden, cl::desc("Run Partial inlining pass"))
static cl::opt< bool > EnableGVNSink("enable-gvn-sink", cl::desc("Enable the GVN sinking pass (default = off)"))
static cl::opt< bool > EnableModuleInliner("enable-module-inliner", cl::init(false), cl::Hidden, cl::desc("Enable module inliner"))
static cl::opt< bool > EnableEagerlyInvalidateAnalyses("eagerly-invalidate-analyses", cl::init(true), cl::Hidden, cl::desc("Eagerly invalidate more analyses in default pipelines"))
static cl::opt< bool > EnableMatrix("enable-matrix", cl::init(false), cl::Hidden, cl::desc("Enable lowering of the matrix intrinsics"))
ModuleToFunctionPassAdaptor createModuleToFunctionPassAdaptor(FunctionPassT &&Pass, bool EagerlyInvalidate=false)
A function to deduce a function pass type and wrap it in the templated adaptor.
cl::opt< std::string > UseCtxProfile("use-ctx-profile", cl::init(""), cl::Hidden, cl::desc("Use the specified contextual profile file"))
static cl::opt< bool > EnableSampledInstr("enable-sampled-instrumentation", cl::init(false), cl::Hidden, cl::desc("Enable profile instrumentation sampling (default = off)"))
static cl::opt< bool > EnableLoopFlatten("enable-loop-flatten", cl::init(false), cl::Hidden, cl::desc("Enable the LoopFlatten Pass"))
@ O1
Optimize quickly without destroying debuggability.
@ O0
Disable as many optimizations as possible.
@ O3
Optimize for fast execution as much as possible.
@ O2
Optimize for fast execution as much as possible without triggering significant incremental compile ti...
static cl::opt< InliningAdvisorMode > UseInlineAdvisor("enable-ml-inliner", cl::init(InliningAdvisorMode::Default), cl::Hidden, cl::desc("Enable ML policy for inliner. Currently trained for -Oz only"), cl::values(clEnumValN(InliningAdvisorMode::Default, "default", "Heuristics-based inliner version"), clEnumValN(InliningAdvisorMode::Development, "development", "Use development mode (runtime-loadable model)"), clEnumValN(InliningAdvisorMode::Release, "release", "Use release mode (AOT-compiled model)")))
static cl::opt< bool > EnableJumpTableToSwitch("enable-jump-table-to-switch", cl::init(true), cl::desc("Enable JumpTableToSwitch pass (default = true)"))
PassManager< LazyCallGraph::SCC, CGSCCAnalysisManager, LazyCallGraph &, CGSCCUpdateResult & > CGSCCPassManager
The CGSCC pass manager.
static cl::opt< bool > EnableUnrollAndJam("enable-unroll-and-jam", cl::init(false), cl::Hidden, cl::desc("Enable Unroll And Jam Pass"))
@ CGSCC_LIGHT
@ MODULE_LIGHT
ThinOrFullLTOPhase
This enumerates the LLVM full LTO or ThinLTO optimization phases.
Definition Pass.h:77
@ FullLTOPreLink
Full LTO prelink phase.
Definition Pass.h:85
@ ThinLTOPostLink
ThinLTO postlink (backend compile) phase.
Definition Pass.h:83
@ None
No LTO/ThinLTO behavior needed.
Definition Pass.h:79
@ FullLTOPostLink
Full LTO postlink (backend compile) phase.
Definition Pass.h:87
@ ThinLTOPreLink
ThinLTO prelink (summary) phase.
Definition Pass.h:81
PassManager< Loop, LoopAnalysisManager, LoopStandardAnalysisResults &, LPMUpdater & > LoopPassManager
The Loop pass manager.
static cl::opt< bool > EnableConstraintElimination("enable-constraint-elimination", cl::init(true), cl::Hidden, cl::desc("Enable pass to eliminate conditions based on linear constraints"))
ModuleToPostOrderCGSCCPassAdaptor createModuleToPostOrderCGSCCPassAdaptor(CGSCCPassT &&Pass)
A function to deduce a function pass type and wrap it in the templated adaptor.
static cl::opt< bool > EnablePGOInlineDeferral("enable-npm-pgo-inline-deferral", cl::init(true), cl::Hidden, cl::desc("Enable inline deferral during PGO"))
Flag to enable inline deferral during PGO.
FunctionToLoopPassAdaptor createFunctionToLoopPassAdaptor(LoopPassT &&Pass, bool UseMemorySSA=false)
A function to deduce a loop pass type and wrap it in the templated adaptor.
CGSCCToFunctionPassAdaptor createCGSCCToFunctionPassAdaptor(FunctionPassT &&Pass, bool EagerlyInvalidate=false, bool NoRerun=false)
A function to deduce a function pass type and wrap it in the templated adaptor.
LLVM_ABI cl::opt< bool > ForgetSCEVInLoopUnroll
PassManager< Module > ModulePassManager
Convenience typedef for a pass manager over modules.
static cl::opt< bool > EnablePostPGOLoopRotation("enable-post-pgo-loop-rotation", cl::init(true), cl::Hidden, cl::desc("Run the loop rotation transformation after PGO instrumentation"))
LLVM_ABI bool AreStatisticsEnabled()
Check if statistics are enabled.
static cl::opt< std::string > InstrumentColdFuncOnlyPath("instrument-cold-function-only-path", cl::init(""), cl::desc("File path for cold function only instrumentation(requires use " "with --pgo-instrument-cold-function-only)"), cl::Hidden)
static cl::opt< bool > EnableGlobalAnalyses("enable-global-analyses", cl::init(true), cl::Hidden, cl::desc("Enable inter-procedural analyses"))
static cl::opt< bool > FlattenedProfileUsed("flattened-profile-used", cl::init(false), cl::Hidden, cl::desc("Indicate the sample profile being used is flattened, i.e., " "no inline hierarchy exists in the profile"))
static cl::opt< AttributorRunOption > AttributorRun("attributor-enable", cl::Hidden, cl::init(AttributorRunOption::NONE), cl::desc("Enable the attributor inter-procedural deduction pass"), cl::values(clEnumValN(AttributorRunOption::FULL, "full", "enable all full attributor runs"), clEnumValN(AttributorRunOption::LIGHT, "light", "enable all attributor-light runs"), clEnumValN(AttributorRunOption::MODULE, "module", "enable module-wide attributor runs"), clEnumValN(AttributorRunOption::MODULE_LIGHT, "module-light", "enable module-wide attributor-light runs"), clEnumValN(AttributorRunOption::CGSCC, "cgscc", "enable call graph SCC attributor runs"), clEnumValN(AttributorRunOption::CGSCC_LIGHT, "cgscc-light", "enable call graph SCC attributor-light runs"), clEnumValN(AttributorRunOption::NONE, "none", "disable attributor runs")))
static cl::opt< bool > EnableLoopInterchange("enable-loopinterchange", cl::init(true), cl::Hidden, cl::desc("Enable the LoopInterchange Pass"))
static cl::opt< bool > ExtraVectorizerPasses("extra-vectorizer-passes", cl::init(false), cl::Hidden, cl::desc("Run cleanup optimization passes after vectorization"))
static cl::opt< bool > EnableHotColdSplit("hot-cold-split", cl::desc("Enable hot-cold splitting pass"))
cl::opt< bool > EnableMemProfContextDisambiguation
Enable MemProf context disambiguation for thin link.
static cl::opt< bool > TriggerCrash("opt-pipeline-trigger-crash", cl::init(false), cl::Hidden, cl::desc("Trigger crash in optimization pipeline"))
PassManager< Function > FunctionPassManager
Convenience typedef for a pass manager over functions.
LLVM_ABI InlineParams getInlineParams()
Generate the parameters to tune the inline cost analysis based only on the commandline options.
cl::opt< bool > PGOInstrumentColdFunctionOnly
static cl::opt< bool > EnableCHR("enable-chr", cl::init(true), cl::Hidden, cl::desc("Enable control height reduction optimization (CHR)"))
static cl::opt< bool > EnableMergeFunctions("enable-merge-functions", cl::init(false), cl::Hidden, cl::desc("Enable function merging as part of the optimization pipeline"))
static cl::opt< bool > EnableDevirtualizeSpeculatively("enable-devirtualize-speculatively", cl::desc("Enable speculative devirtualization optimization"), cl::init(false))
static cl::opt< bool > EnableGVNHoist("enable-gvn-hoist", cl::desc("Enable the GVN hoisting pass (default = off)"))
LLVM_ABI cl::opt< unsigned > SetLicmMssaNoAccForPromotionCap
LLVM_ABI InlineParams getInlineParamsFromOptLevel(unsigned OptLevel)
Generate the parameters to tune the inline cost analysis based on command line options.
static cl::opt< int > PreInlineThreshold("preinline-threshold", cl::Hidden, cl::init(75), cl::desc("Control the amount of inlining in pre-instrumentation inliner " "(default = 75)"))
static cl::opt< bool > UseLoopVersioningLICM("enable-loop-versioning-licm", cl::init(false), cl::Hidden, cl::desc("Enable the experimental Loop Versioning LICM pass"))
cl::opt< unsigned > MaxDevirtIterations("max-devirt-iterations", cl::ReallyHidden, cl::init(4))
LLVM_ABI cl::opt< unsigned > SetLicmMssaOptCap
static cl::opt< bool > EnableInstrumentor("enable-instrumentor", cl::init(false), cl::Hidden, cl::desc("Enable the Instrumentor Pass"))
static cl::opt< bool > EnableMergeICmps("enable-mergeicmps", cl::init(true), cl::Hidden, cl::desc("Enable MergeICmps pass in the optimization pipeline"))
A DCE pass that assumes instructions are dead until proven otherwise.
Definition ADCE.h:31
Pass to convert @llvm.global.annotations to !annotation metadata.
This pass attempts to minimize the number of assume without loosing any information.
A more lightweight version of the Attributor which only runs attribute inference but no simplificatio...
A more lightweight version of the Attributor which only runs attribute inference but no simplificatio...
Hoist/decompose integer division and remainder instructions to enable CFG improvements and better cod...
Definition DivRemPairs.h:23
A simple and fast domtree-based CSE pass.
Definition EarlyCSE.h:31
Pass which forces specific function attributes into the IR, primarily as a debugging tool.
A simple and fast domtree-based GVN pass to hoist common expressions from sibling branches.
Definition GVN.h:521
Uses an "inverted" value numbering to decide the similarity of expressions and sinks similar expressi...
Definition GVN.h:528
A set of parameters to control various transforms performed by IPSCCP pass.
Definition SCCP.h:35
A pass which infers function attributes from the names and signatures of function declarations in a m...
Provides context on when an inline advisor is constructed in the pipeline (e.g., link phase,...
Thresholds to tune inline cost analysis.
Definition InlineCost.h:207
std::optional< int > OptSizeHintThreshold
Threshold to use for callees with inline hint, when the caller is optimized for size.
Definition InlineCost.h:216
std::optional< int > HotCallSiteThreshold
Threshold to use when the callsite is considered hot.
Definition InlineCost.h:228
int DefaultThreshold
The default threshold to start with for a callee.
Definition InlineCost.h:209
std::optional< bool > EnableDeferral
Indicate whether we should allow inline deferral.
Definition InlineCost.h:241
std::optional< int > HintThreshold
Threshold to use for callees with inline hint.
Definition InlineCost.h:212
Options for the frontend instrumentation based profiling pass.
A no-op pass template which simply forces a specific analysis result to be invalidated.
Pass to forward loads in a loop around the backedge to subsequent iterations.
A set of parameters used to control various transforms performed by the LoopUnroll pass.
The LoopVectorize Pass.
Computes function attributes in post-order over the call graph.
A utility pass template to force an analysis result to be available.