LLVM 24.0.0git
PassBuilderPipelines.cpp
Go to the documentation of this file.
1//===- Construction of pass pipelines -------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9///
10/// This file provides the implementation of the PassBuilder based on our
11/// static pass registry as well as related functionality. It also provides
12/// helpers to aid in analyzing, debugging, and testing passes and pass
13/// pipelines.
14///
15//===----------------------------------------------------------------------===//
16
17#include "PassesOptions.h"
18#include "llvm/ADT/Statistic.h"
29#include "llvm/IR/PassManager.h"
30#include "llvm/IR/Verifier.h"
31#include "llvm/Pass.h"
160
161using namespace llvm;
162
163namespace llvm {
165
167} // namespace llvm
168
170 const PassesOptions &Opts = PassesOptions::Global;
171 LoopInterleaving = true;
172 LoopVectorization = true;
173 SLPVectorization = false;
174 LoopUnrolling = true;
175 LoopInterchange = Opts.enable_loopinterchange;
176 LoopFusion = false;
180 CallGraphProfile = true;
181 UnifiedLTO = false;
182 MergeFunctions = Opts.enable_merge_functions;
183 InlinerThreshold = -1;
184 EagerlyInvalidateAnalyses = Opts.eagerly_invalidate_analyses;
185 DevirtualizeSpeculatively = Opts.enable_devirtualize_speculatively;
186}
187
188namespace llvm {
190} // namespace llvm
191
193 OptimizationLevel Level) {
194 for (auto &C : PeepholeEPCallbacks)
195 C(FPM, Level);
196}
199 for (auto &C : LateLoopOptimizationsEPCallbacks)
200 C(LPM, Level);
201}
203 OptimizationLevel Level) {
204 for (auto &C : LoopOptimizerEndEPCallbacks)
205 C(LPM, Level);
206}
209 for (auto &C : ScalarOptimizerLateEPCallbacks)
210 C(FPM, Level);
211}
213 OptimizationLevel Level) {
214 for (auto &C : CGSCCOptimizerLateEPCallbacks)
215 C(CGPM, Level);
216}
218 OptimizationLevel Level) {
219 for (auto &C : VectorizerStartEPCallbacks)
220 C(FPM, Level);
221}
223 OptimizationLevel Level) {
224 for (auto &C : VectorizerEndEPCallbacks)
225 C(FPM, Level);
226}
228 OptimizationLevel Level,
230 for (auto &C : OptimizerEarlyEPCallbacks)
231 C(MPM, Level, Phase);
232}
234 OptimizationLevel Level,
236 for (auto &C : OptimizerLastEPCallbacks)
237 C(MPM, Level, Phase);
238}
241 for (auto &C : FullLinkTimeOptimizationEarlyEPCallbacks)
242 C(MPM, Level);
243}
246 for (auto &C : FullLinkTimeOptimizationLastEPCallbacks)
247 C(MPM, Level);
248}
251 for (auto &C : ThinLinkTimeOptimizationEarlyEPCallbacks)
252 C(MPM, Level);
253}
256 for (auto &C : ThinLinkTimeOptimizationLastEPCallbacks)
257 C(MPM, Level);
258}
260 OptimizationLevel Level) {
261 for (auto &C : PipelineStartEPCallbacks)
262 C(MPM, Level);
263}
266 for (auto &C : PipelineEarlySimplificationEPCallbacks)
267 C(MPM, Level, Phase);
268}
269
270// Get IR stats with InstCount before/after the optimization pipeline
272 bool IsPreOptimization) {
273 if (AreStatisticsEnabled()) {
274 MPM.addPass(
277 FunctionPropertiesStatisticsPass(IsPreOptimization)));
278 }
279}
280
281// Helper to add AnnotationRemarksPass.
285
286// Helper to check if the current compilation phase is preparing for LTO
291
292// Helper to check if the current compilation phase is preparing for FullLTO
293[[maybe_unused]] static bool isFullLTOPreLink(ThinOrFullLTOPhase Phase) {
295}
296
297// Helper to check if the current compilation phase is preparing for ThinLTO
301
302// Helper to check if the current compilation phase is LTO backend
307
308// Helper to check if the current compilation phase is FullLTO backend
312
313// Helper to check if the current compilation phase is ThinLTO backend
317
318// Helper to wrap conditionally Coro passes.
320 // TODO: Skip passes according to Phase.
321 ModulePassManager CoroPM;
322 CoroPM.addPass(CoroEarlyPass());
323 CGSCCPassManager CGPM;
324 CGPM.addPass(CoroSplitPass());
325 CoroPM.addPass(createModuleToPostOrderCGSCCPassAdaptor(std::move(CGPM)));
326 CoroPM.addPass(CoroCleanupPass());
327 CoroPM.addPass(GlobalDCEPass());
328 return CoroConditionalWrapper(std::move(CoroPM));
329}
330
332 return getInlineParamsFromOptLevel(static_cast<unsigned>(Level));
333}
334
336 const PassesOptions &Opts,
337 OptimizationLevel Level,
340 if (Opts.enable_module_inliner)
341 MPM.addPass(ModuleInlinerPass(IP, Opts.enable_ml_inliner, Phase));
342 else
344 IP,
345 /* MandatoryFirst */ true,
347}
348
349// TODO: Investigate the cost/benefit of tail call elimination on debugging.
351PassBuilder::buildO1FunctionSimplificationPipeline(OptimizationLevel Level,
353
355
357 FPM.addPass(CountVisitsPass());
358
359 // Form SSA out of local memory accesses after breaking apart aggregates into
360 // scalars.
361 FPM.addPass(SROAPass(SROAOptions::ModifyCFG));
362
363 // Catch trivial redundancies
364 FPM.addPass(EarlyCSEPass(true /* Enable mem-ssa. */));
365
366 // Hoisting of scalars and load expressions.
367 FPM.addPass(
368 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
369 FPM.addPass(InstCombinePass());
370
371 FPM.addPass(LibCallsShrinkWrapPass());
372
373 invokePeepholeEPCallbacks(FPM, Level);
374
375 FPM.addPass(
376 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
377
378 // Form canonically associated expression trees, and simplify the trees using
379 // basic mathematical properties. For example, this will form (nearly)
380 // minimal multiplication trees.
381 FPM.addPass(ReassociatePass());
382
383 // Add the primary loop simplification pipeline.
384 // FIXME: Currently this is split into two loop pass pipelines because we run
385 // some function passes in between them. These can and should be removed
386 // and/or replaced by scheduling the loop pass equivalents in the correct
387 // positions. But those equivalent passes aren't powerful enough yet.
388 // Specifically, `SimplifyCFGPass` and `InstCombinePass` are currently still
389 // used. We have `LoopSimplifyCFGPass` which isn't yet powerful enough yet to
390 // fully replace `SimplifyCFGPass`, and the closest to the other we have is
391 // `LoopInstSimplify`.
392 LoopPassManager LPM1, LPM2;
393
394 // Simplify the loop body. We do this initially to clean up after other loop
395 // passes run, either when iterating on a loop or on inner loops with
396 // implications on the outer loop.
397 LPM1.addPass(LoopInstSimplifyPass());
398 LPM1.addPass(LoopSimplifyCFGPass());
399
400 // Try to remove as much code from the loop header as possible,
401 // to reduce amount of IR that will have to be duplicated. However,
402 // do not perform speculative hoisting the first time as LICM
403 // will destroy metadata that may not need to be destroyed if run
404 // after loop rotation.
405 // TODO: Investigate promotion cap for O1.
406 LPM1.addPass(LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
407 /*AllowSpeculation=*/false));
408
409 LPM1.addPass(
410 LoopRotatePass(/*EnableHeaderDuplication=*/true, isLTOPreLink(Phase)));
411 // TODO: Investigate promotion cap for O1.
412 LPM1.addPass(LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
413 /*AllowSpeculation=*/true));
414 LPM1.addPass(SimpleLoopUnswitchPass());
415 if (Opts.enable_loop_flatten)
416 LPM1.addPass(LoopFlattenPass());
417
418 LPM2.addPass(LoopIdiomRecognizePass());
419 LPM2.addPass(IndVarSimplifyPass());
420
422
423 LPM2.addPass(LoopDeletionPass());
424
425 // Do not enable unrolling in PreLinkThinLTO phase during sample PGO
426 // because it changes IR to makes profile annotation in back compile
427 // inaccurate. The normal unroller doesn't pay attention to forced full unroll
428 // attributes so we need to make sure and allow the full unroll pass to pay
429 // attention to it.
430 if (!isThinLTOPreLink(Phase) || !PGOOpt ||
431 PGOOpt->Action != PGOOptions::SampleUse)
432 LPM2.addPass(LoopFullUnrollPass(static_cast<int>(Level),
433 /* OnlyWhenForced= */ !PTO.LoopUnrolling,
434 PTO.ForgetAllSCEVInLoopUnroll));
435
437
438 FPM.addPass(createFunctionToLoopPassAdaptor(std::move(LPM1),
439 /*UseMemorySSA=*/true));
440 FPM.addPass(
441 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
442 FPM.addPass(InstCombinePass());
443 // The loop passes in LPM2 (LoopFullUnrollPass) do not preserve MemorySSA.
444 // *All* loop passes must preserve it, in order to be able to use it.
445 FPM.addPass(createFunctionToLoopPassAdaptor(std::move(LPM2),
446 /*UseMemorySSA=*/false));
447
448 // Delete small array after loop unroll.
449 FPM.addPass(SROAPass(SROAOptions::ModifyCFG));
450
451 // Specially optimize memory movement as it doesn't look like dataflow in SSA.
452 FPM.addPass(MemCpyOptPass());
453
454 // Sparse conditional constant propagation.
455 // FIXME: It isn't clear why we do this *after* loop passes rather than
456 // before...
457 FPM.addPass(SCCPPass());
458
459 // Delete dead bit computations (instcombine runs after to fold away the dead
460 // computations, and then ADCE will run later to exploit any new DCE
461 // opportunities that creates).
462 FPM.addPass(BDCEPass());
463
464 // Run instcombine after redundancy and dead bit elimination to exploit
465 // opportunities opened up by them.
466 FPM.addPass(InstCombinePass());
467 invokePeepholeEPCallbacks(FPM, Level);
468
469 FPM.addPass(CoroElidePass());
470
472
473 // Finally, do an expensive DCE pass to catch all the dead code exposed by
474 // the simplifications and basic cleanup after all the simplifications.
475 // TODO: Investigate if this is too expensive.
476 FPM.addPass(ADCEPass());
477 FPM.addPass(
478 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
479 FPM.addPass(InstCombinePass());
480 invokePeepholeEPCallbacks(FPM, Level);
481
482 return FPM;
483}
484
488 assert(Level != OptimizationLevel::O0 && "Must request optimizations!");
489
490 // The O1 pipeline has a separate pipeline creation function to simplify
491 // construction readability.
492 if (Level == OptimizationLevel::O1)
493 return buildO1FunctionSimplificationPipeline(Level, Phase);
494
496
499
500 // Form SSA out of local memory accesses after breaking apart aggregates into
501 // scalars.
503
504 // Catch trivial redundancies
505 FPM.addPass(EarlyCSEPass(true /* Enable mem-ssa. */));
508
509 // Hoisting of scalars and load expressions.
510 if (Opts.enable_gvn_hoist)
511 FPM.addPass(GVNHoistPass());
512
513 // Global value numbering based sinking.
514 if (Opts.enable_gvn_sink) {
515 FPM.addPass(GVNSinkPass());
516 FPM.addPass(
517 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
518 }
519
520 // Speculative execution if the target has divergent branches; otherwise nop.
521 FPM.addPass(SpeculativeExecutionPass(/* OnlyIfDivergentTarget =*/true));
522
523 // Optimize based on known information about branches, and cleanup afterward.
526
527 // Jump table to switch conversion.
528 if (Opts.enable_jump_table_to_switch)
530
531 FPM.addPass(
532 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
536
537 invokePeepholeEPCallbacks(FPM, Level);
538
539 // For PGO use pipeline, try to optimize memory intrinsics such as memcpy
540 // using the size value profile. Don't perform this when optimizing for size.
541 if (PGOOpt && PGOOpt->Action == PGOOptions::IRUse)
543
544 FPM.addPass(TailCallElimPass(/*UpdateFunctionEntryCount=*/
545 isInstrumentedPGOUse()));
546 FPM.addPass(
547 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
548
549 // Form canonically associated expression trees, and simplify the trees using
550 // basic mathematical properties. For example, this will form (nearly)
551 // minimal multiplication trees.
553
554 if (Opts.enable_constraint_elimination)
556
557 // Add the primary loop simplification pipeline.
558 // FIXME: Currently this is split into two loop pass pipelines because we run
559 // some function passes in between them. These can and should be removed
560 // and/or replaced by scheduling the loop pass equivalents in the correct
561 // positions. But those equivalent passes aren't powerful enough yet.
562 // Specifically, `SimplifyCFGPass` and `InstCombinePass` are currently still
563 // used. We have `LoopSimplifyCFGPass` which isn't yet powerful enough yet to
564 // fully replace `SimplifyCFGPass`, and the closest to the other we have is
565 // `LoopInstSimplify`.
566 LoopPassManager LPM1, LPM2;
567
568 // Simplify the loop body. We do this initially to clean up after other loop
569 // passes run, either when iterating on a loop or on inner loops with
570 // implications on the outer loop.
571 LPM1.addPass(LoopInstSimplifyPass());
572 LPM1.addPass(LoopSimplifyCFGPass());
573
574 // Try to remove as much code from the loop header as possible,
575 // to reduce amount of IR that will have to be duplicated. However,
576 // do not perform speculative hoisting the first time as LICM
577 // will destroy metadata that may not need to be destroyed if run
578 // after loop rotation.
579 // TODO: Investigate promotion cap for O1.
580 LPM1.addPass(LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
581 /*AllowSpeculation=*/false));
582
583 LPM1.addPass(
584 LoopRotatePass(/*EnableHeaderDuplication=*/true, isLTOPreLink(Phase)));
585 // TODO: Investigate promotion cap for O1.
586 LPM1.addPass(LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
587 /*AllowSpeculation=*/true));
588 LPM1.addPass(
589 SimpleLoopUnswitchPass(/* NonTrivial */ Level == OptimizationLevel::O3));
590 if (Opts.enable_loop_flatten)
591 LPM1.addPass(LoopFlattenPass());
592
593 LPM2.addPass(LoopIdiomRecognizePass());
594 LPM2.addPass(IndVarSimplifyPass());
595
596 {
598 ExtraPasses.addPass(SimpleLoopUnswitchPass(/* NonTrivial */ Level ==
600 LPM2.addPass(std::move(ExtraPasses));
601 }
602
604
605 LPM2.addPass(LoopDeletionPass());
606
607 // Do not enable unrolling in PreLinkThinLTO phase during sample PGO
608 // because it changes IR to makes profile annotation in back compile
609 // inaccurate. The normal unroller doesn't pay attention to forced full unroll
610 // attributes so we need to make sure and allow the full unroll pass to pay
611 // attention to it.
612 if (!isThinLTOPreLink(Phase) || !PGOOpt ||
613 PGOOpt->Action != PGOOptions::SampleUse)
614 LPM2.addPass(LoopFullUnrollPass(static_cast<int>(Level),
615 /* OnlyWhenForced= */ !PTO.LoopUnrolling,
616 PTO.ForgetAllSCEVInLoopUnroll));
617
619
620 FPM.addPass(createFunctionToLoopPassAdaptor(std::move(LPM1),
621 /*UseMemorySSA=*/true));
622 FPM.addPass(
623 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
625 // The loop passes in LPM2 (LoopIdiomRecognizePass, IndVarSimplifyPass,
626 // LoopDeletionPass and LoopFullUnrollPass) do not preserve MemorySSA.
627 // *All* loop passes must preserve it, in order to be able to use it.
628 FPM.addPass(createFunctionToLoopPassAdaptor(std::move(LPM2),
629 /*UseMemorySSA=*/false));
630
631 // Delete small array after loop unroll.
633
634 // Try vectorization/scalarization transforms that are both improvements
635 // themselves and can allow further folds with GVN and InstCombine.
636 FPM.addPass(VectorCombinePass(/*TryEarlyFoldsOnly=*/true));
637
638 // Eliminate redundancies.
640 if (Opts.enable_newgvn)
641 FPM.addPass(NewGVNPass());
642 else
643 FPM.addPass(GVNPass());
644
645 // Sparse conditional constant propagation.
646 // FIXME: It isn't clear why we do this *after* loop passes rather than
647 // before...
648 FPM.addPass(SCCPPass());
649
650 // Delete dead bit computations (instcombine runs after to fold away the dead
651 // computations, and then ADCE will run later to exploit any new DCE
652 // opportunities that creates).
653 FPM.addPass(BDCEPass());
654
655 // Run instcombine after redundancy and dead bit elimination to exploit
656 // opportunities opened up by them.
658 invokePeepholeEPCallbacks(FPM, Level);
659
660 // Re-consider control flow based optimizations after redundancy elimination,
661 // redo DCE, etc.
662 if (Opts.enable_dfa_jump_thread)
664
667
668 // Finally, do an expensive DCE pass to catch all the dead code exposed by
669 // the simplifications and basic cleanup after all the simplifications.
670 // TODO: Investigate if this is too expensive.
671 FPM.addPass(ADCEPass());
672
673 // Specially optimize memory movement as it doesn't look like dataflow in SSA.
674 FPM.addPass(MemCpyOptPass());
675
676 FPM.addPass(DSEPass());
678
680 LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
681 /*AllowSpeculation=*/true),
682 /*UseMemorySSA=*/true));
683
684 FPM.addPass(CoroElidePass());
685
687
689 .convertSwitchRangeToICmp(true)
690 .convertSwitchToArithmetic(true)
691 .hoistCommonInsts(true)
692 .sinkCommonInsts(true)));
694 invokePeepholeEPCallbacks(FPM, Level);
695
696 return FPM;
697}
698
699void PassBuilder::addRequiredLTOPreLinkPasses(ModulePassManager &MPM) {
702 MPM.addPass(AssignGUIDPass());
703}
704
705void PassBuilder::addPreInlinerPasses(ModulePassManager &MPM,
706 OptimizationLevel Level,
707 ThinOrFullLTOPhase LTOPhase) {
708 assert(Level != OptimizationLevel::O0 && "Not expecting O0 here!");
709 if (Opts.disable_preinline)
710 return;
711 InlineParams IP;
712
713 IP.DefaultThreshold = Opts.preinline_threshold;
714
715 // FIXME: The hint threshold has the same value used by the regular inliner
716 // when not optimzing for size. This should probably be lowered after
717 // performance testing.
718 // FIXME: this comment is cargo culted from the old pass manager, revisit).
719 IP.HintThreshold = 325;
720 IP.OptSizeHintThreshold = Opts.preinline_threshold;
722 IP, /* MandatoryFirst */ true,
724 CGSCCPassManager &CGPipeline = MIWP.getPM();
725
727 FPM.addPass(SROAPass(SROAOptions::ModifyCFG));
728 FPM.addPass(EarlyCSEPass()); // Catch trivial redundancies.
729 FPM.addPass(SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(
730 true))); // Merge & remove basic blocks.
731 FPM.addPass(InstCombinePass()); // Combine silly sequences.
732 invokePeepholeEPCallbacks(FPM, Level);
733
734 CGPipeline.addPass(createCGSCCToFunctionPassAdaptor(
735 std::move(FPM), PTO.EagerlyInvalidateAnalyses));
736
737 MPM.addPass(std::move(MIWP));
738
739 // Delete anything that is now dead to make sure that we don't instrument
740 // dead code. Instrumentation can end up keeping dead code around and
741 // dramatically increase code size.
742 MPM.addPass(GlobalDCEPass());
743}
744
745void PassBuilder::addPostPGOLoopRotation(ModulePassManager &MPM,
746 OptimizationLevel Level) {
747 if (Opts.enable_post_pgo_loop_rotation) {
748 // Disable header duplication in loop rotation at -Oz.
750 createFunctionToLoopPassAdaptor(LoopRotatePass(),
751 /*UseMemorySSA=*/false),
752 PTO.EagerlyInvalidateAnalyses));
753 }
754}
755
756void PassBuilder::addPGOInstrPasses(ModulePassManager &MPM,
757 OptimizationLevel Level, bool RunProfileGen,
758 bool IsCS, bool AtomicCounterUpdate,
759 std::string ProfileFile,
760 std::string ProfileRemappingFile) {
761 assert(Level != OptimizationLevel::O0 && "Not expecting O0 here!");
762
763 if (!RunProfileGen) {
764 assert(!ProfileFile.empty() && "Profile use expecting a profile file!");
765 MPM.addPass(
766 PGOInstrumentationUse(ProfileFile, ProfileRemappingFile, IsCS, FS));
767 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
768 // RequireAnalysisPass for PSI before subsequent non-module passes.
769 MPM.addPass(RequireAnalysisPass<ProfileSummaryAnalysis, Module>());
770 return;
771 }
772
773 // Perform PGO instrumentation.
774 MPM.addPass(PGOInstrumentationGen(IsCS ? PGOInstrumentationType::CSFDO
776
777 addPostPGOLoopRotation(MPM, Level);
778 // Add the profile lowering pass.
779 InstrProfOptions Options;
780 if (!ProfileFile.empty())
781 Options.InstrProfileOutput = ProfileFile;
782 // Do counter promotion at Level greater than O0.
783 Options.DoCounterPromotion = true;
784 Options.UseBFIInPromotion = IsCS;
785 if (Opts.enable_sampled_instrumentation) {
786 Options.Sampling = true;
787 // With sampling, there is little beneifit to enable counter promotion.
788 // But note that sampling does work with counter promotion.
789 Options.DoCounterPromotion = false;
790 }
791 Options.Atomic = AtomicCounterUpdate;
792 MPM.addPass(InstrProfilingLoweringPass(Options, IsCS));
793}
794
796 bool RunProfileGen, bool IsCS,
797 bool AtomicCounterUpdate,
798 std::string ProfileFile,
799 std::string ProfileRemappingFile) {
800 if (!RunProfileGen) {
801 assert(!ProfileFile.empty() && "Profile use expecting a profile file!");
802 MPM.addPass(
803 PGOInstrumentationUse(ProfileFile, ProfileRemappingFile, IsCS, FS));
804 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
805 // RequireAnalysisPass for PSI before subsequent non-module passes.
807 return;
808 }
809
810 // Perform PGO instrumentation.
813 // Add the profile lowering pass.
815 if (!ProfileFile.empty())
816 Options.InstrProfileOutput = ProfileFile;
817 // Do not do counter promotion at O0.
818 Options.DoCounterPromotion = false;
819 Options.UseBFIInPromotion = IsCS;
820 Options.Atomic = AtomicCounterUpdate;
822}
823
827 InlineParams IP;
828 if (PTO.InlinerThreshold == -1)
830 else
831 IP = getInlineParams(PTO.InlinerThreshold);
832 // For PreLinkThinLTO + SamplePGO or PreLinkFullLTO + SamplePGO,
833 // set hot-caller threshold to 0 to disable hot
834 // callsite inline (as much as possible [1]) because it makes
835 // profile annotation in the backend inaccurate.
836 //
837 // [1] Note the cost of a function could be below zero due to erased
838 // prologue / epilogue.
839 if (isLTOPreLink(Phase) && PGOOpt && PGOOpt->Action == PGOOptions::SampleUse)
841
842 if (PGOOpt)
843 IP.EnableDeferral = Opts.enable_npm_pgo_inline_deferral;
844
845 ModuleInlinerWrapperPass MIWP(IP, Opts.mandatory_inlining_first,
846 InlineContext{Phase, InlinePass::CGSCCInliner},
847 Opts.enable_ml_inliner, MaxDevirtIterations);
848
849 // Require the GlobalsAA analysis for the module so we can query it within
850 // the CGSCC pipeline.
851 if (Opts.enable_global_analyses) {
853 // Invalidate AAManager so it can be recreated and pick up the newly
854 // available GlobalsAA.
855 MIWP.addModulePass(
857 }
858
859 // Require the ProfileSummaryAnalysis for the module so we can query it within
860 // the inliner pass.
862
863 // Now begin the main postorder CGSCC pipeline.
864 // FIXME: The current CGSCC pipeline has its origins in the legacy pass
865 // manager and trying to emulate its precise behavior. Much of this doesn't
866 // make a lot of sense and we should revisit the core CGSCC structure.
867 CGSCCPassManager &MainCGPipeline = MIWP.getPM();
868
869 // Note: historically, the PruneEH pass was run first to deduce nounwind and
870 // generally clean up exception handling overhead. It isn't clear this is
871 // valuable as the inliner doesn't currently care whether it is inlining an
872 // invoke or a call.
873
874 if (Opts.attributor_enable & AttributorRunOption::CGSCC)
875 MainCGPipeline.addPass(AttributorCGSCCPass());
876 else if (Opts.attributor_enable & AttributorRunOption::CGSCC_LIGHT)
877 MainCGPipeline.addPass(AttributorLightCGSCCPass());
878
879 // Deduce function attributes. We do another run of this after the function
880 // simplification pipeline, so this only needs to run when it could affect the
881 // function simplification pipeline, which is only the case with recursive
882 // functions.
883 MainCGPipeline.addPass(PostOrderFunctionAttrsPass(/*SkipNonRecursive*/ true));
884
885 // When at O3 add argument promotion to the pass pipeline.
886 // FIXME: It isn't at all clear why this should be limited to O3.
887 if (Level == OptimizationLevel::O3)
888 MainCGPipeline.addPass(ArgumentPromotionPass());
889
890 // Try to perform OpenMP specific optimizations. This is a (quick!) no-op if
891 // there are no OpenMP runtime calls present in the module.
892 if (Level == OptimizationLevel::O2 || Level == OptimizationLevel::O3)
893 MainCGPipeline.addPass(OpenMPOptCGSCCPass(Phase));
894
895 invokeCGSCCOptimizerLateEPCallbacks(MainCGPipeline, Level);
896
897 // Add the core function simplification pipeline nested inside the
898 // CGSCC walk.
901 PTO.EagerlyInvalidateAnalyses, /*NoRerun=*/true));
902
903 // Finally, deduce any function attributes based on the fully simplified
904 // function.
905 MainCGPipeline.addPass(PostOrderFunctionAttrsPass());
906
907 // Mark that the function is fully simplified and that it shouldn't be
908 // simplified again if we somehow revisit it due to CGSCC mutations unless
909 // it's been modified since.
912
913 if (!isThinLTOPreLink(Phase)) {
914 MainCGPipeline.addPass(CoroSplitPass(Level != OptimizationLevel::O0));
915 MainCGPipeline.addPass(CoroAnnotationElidePass());
916 }
917
918 // Make sure we don't affect potential future NoRerun CGSCC adaptors.
921
922 return MIWP;
923}
924
929
931 // For PreLinkThinLTO + SamplePGO or PreLinkFullLTO + SamplePGO,
932 // set hot-caller threshold to 0 to disable hot
933 // callsite inline (as much as possible [1]) because it makes
934 // profile annotation in the backend inaccurate.
935 //
936 // [1] Note the cost of a function could be below zero due to erased
937 // prologue / epilogue.
938 if (isLTOPreLink(Phase) && PGOOpt && PGOOpt->Action == PGOOptions::SampleUse)
940
941 if (PGOOpt)
942 IP.EnableDeferral = Opts.enable_npm_pgo_inline_deferral;
943
944 // The inline deferral logic is used to avoid losing some
945 // inlining chance in future. It is helpful in SCC inliner, in which
946 // inlining is processed in bottom-up order.
947 // While in module inliner, the inlining order is a priority-based order
948 // by default. The inline deferral is unnecessary there. So we disable the
949 // inline deferral logic in module inliner.
950 IP.EnableDeferral = false;
951
952 MPM.addPass(ModuleInlinerPass(IP, Opts.enable_ml_inliner, Phase));
954 MPM.addPass(GlobalOptPass());
955 MPM.addPass(GlobalDCEPass());
956 MPM.addPass(AssignGUIDPass());
957 MPM.addPass(PGOCtxProfFlatteningPass(/*IsPreThinlink=*/false));
958 }
959
962 PTO.EagerlyInvalidateAnalyses));
963
964 if (!isThinLTOPreLink(Phase)) {
967 MPM.addPass(
969 }
970
971 return MPM;
972}
973
977 assert(Level != OptimizationLevel::O0 &&
978 "Should not be used for O0 pipeline");
979
981 "FullLTOPostLink shouldn't call buildModuleSimplificationPipeline!");
982
984
985 // Place pseudo probe instrumentation as the first pass of the pipeline to
986 // minimize the impact of optimization changes.
987 if (PGOOpt && PGOOpt->PseudoProbeForProfiling && !isThinLTOPostLink(Phase))
989
990 bool HasSampleProfile = PGOOpt && (PGOOpt->Action == PGOOptions::SampleUse);
991
992 // In ThinLTO mode, when flattened profile is used, all the available
993 // profile information will be annotated in PreLink phase so there is
994 // no need to load the profile again in PostLink.
995 bool LoadSampleProfile = HasSampleProfile && !(Opts.flattened_profile_used &&
997
998 // During the ThinLTO backend phase we perform early indirect call promotion
999 // here, before globalopt. Otherwise imported available_externally functions
1000 // look unreferenced and are removed. If we are going to load the sample
1001 // profile then defer until later.
1002 // TODO: See if we can move later and consolidate with the location where
1003 // we perform ICP when we are loading a sample profile.
1004 // TODO: We pass HasSampleProfile (whether there was a sample profile file
1005 // passed to the compile) to the SamplePGO flag of ICP. This is used to
1006 // determine whether the new direct calls are annotated with prof metadata.
1007 // Ideally this should be determined from whether the IR is annotated with
1008 // sample profile, and not whether the a sample profile was provided on the
1009 // command line. E.g. for flattened profiles where we will not be reloading
1010 // the sample profile in the ThinLTO backend, we ideally shouldn't have to
1011 // provide the sample profile file.
1012 if (isThinLTOPostLink(Phase) && !LoadSampleProfile)
1013 MPM.addPass(PGOIndirectCallPromotion(true /* InLTO */, HasSampleProfile));
1014
1015 // Create an early function pass manager to cleanup the output of the
1016 // frontend. Not necessary with LTO post link pipelines since the pre link
1017 // pipeline already cleaned up the frontend output.
1018 if (!isThinLTOPostLink(Phase)) {
1019 // Do basic inference of function attributes from known properties of system
1020 // libraries and other oracles.
1022 MPM.addPass(CoroEarlyPass());
1023
1024 FunctionPassManager EarlyFPM;
1025 EarlyFPM.addPass(EntryExitInstrumenterPass(/*PostInlining=*/false));
1026 // Lower llvm.expect to metadata before attempting transforms.
1027 // Compare/branch metadata may alter the behavior of passes like
1028 // SimplifyCFG.
1030 EarlyFPM.addPass(SimplifyCFGPass());
1032 EarlyFPM.addPass(EarlyCSEPass());
1033 if (Level == OptimizationLevel::O3)
1034 EarlyFPM.addPass(CallSiteSplittingPass());
1036 std::move(EarlyFPM), PTO.EagerlyInvalidateAnalyses));
1037 }
1038
1039 if (LoadSampleProfile) {
1040 // Annotate sample profile right after early FPM to ensure freshness of
1041 // the debug info.
1043 PGOOpt->ProfileFile, PGOOpt->ProfileRemappingFile, Phase, FS));
1044 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
1045 // RequireAnalysisPass for PSI before subsequent non-module passes.
1047 // Do not invoke ICP in the LTOPrelink phase as it makes it hard
1048 // for the profile annotation to be accurate in the LTO backend.
1049 if (!isLTOPreLink(Phase))
1050 // We perform early indirect call promotion here, before globalopt.
1051 // This is important for the ThinLTO backend phase because otherwise
1052 // imported available_externally functions look unreferenced and are
1053 // removed.
1054 MPM.addPass(
1055 PGOIndirectCallPromotion(true /* IsInLTO */, true /* SamplePGO */));
1056 }
1057
1058 // Try to perform OpenMP specific optimizations on the module. This is a
1059 // (quick!) no-op if there are no OpenMP runtime calls present in the module.
1061
1062 if (Opts.attributor_enable & AttributorRunOption::MODULE)
1063 MPM.addPass(AttributorPass());
1064 else if (Opts.attributor_enable & AttributorRunOption::MODULE_LIGHT)
1066
1067 // Lower type metadata and the type.test intrinsic in the ThinLTO
1068 // post link pipeline after ICP. This is to enable usage of the type
1069 // tests in ICP sequences.
1072
1074
1075 // Interprocedural constant propagation now that basic cleanup has occurred
1076 // and prior to optimizing globals.
1077 // FIXME: This position in the pipeline hasn't been carefully considered in
1078 // years, it should be re-analyzed.
1079 MPM.addPass(
1080 IPSCCPPass(IPSCCPOptions(/*AllowFuncSpec=*/!isLTOPreLink(Phase))));
1081
1082 // Attach metadata to indirect call sites indicating the set of functions
1083 // they may target at run-time. This should follow IPSCCP.
1085
1086 // Optimize globals to try and fold them into constants.
1087 MPM.addPass(GlobalOptPass());
1088
1089 // Create a small function pass pipeline to cleanup after all the global
1090 // optimizations.
1091 FunctionPassManager GlobalCleanupPM;
1092 // FIXME: Should this instead by a run of SROA?
1093 GlobalCleanupPM.addPass(PromotePass());
1094 GlobalCleanupPM.addPass(InstCombinePass());
1095 invokePeepholeEPCallbacks(GlobalCleanupPM, Level);
1096 GlobalCleanupPM.addPass(
1097 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
1098 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(GlobalCleanupPM),
1099 PTO.EagerlyInvalidateAnalyses));
1100
1101 // We already asserted this happens in non-FullLTOPostLink earlier.
1102 const bool IsPreLink = !isThinLTOPostLink(Phase);
1103 // Enable contextual profiling instrumentation.
1104 const bool IsCtxProfGen =
1106 const bool IsPGOPreLink = !IsCtxProfGen && PGOOpt && IsPreLink;
1107 const bool IsPGOInstrGen =
1108 IsPGOPreLink && PGOOpt->Action == PGOOptions::IRInstr;
1109 const bool IsPGOInstrUse =
1110 IsPGOPreLink && PGOOpt->Action == PGOOptions::IRUse;
1111 const bool IsMemprofUse = IsPGOPreLink && !PGOOpt->MemoryProfile.empty();
1112 // We don't want to mix pgo ctx gen and pgo gen; we also don't currently
1113 // enable ctx profiling from the frontend.
1115 "Enabling both instrumented PGO and contextual instrumentation is not "
1116 "supported.");
1117 const bool IsCtxProfUse = !UseCtxProfile.empty() && isThinLTOPreLink(Phase);
1118
1119 assert((Opts.instrument_cold_function_only_path.empty() ||
1121 "--instrument-cold-function-only-path is provided but "
1122 "--pgo-instrument-cold-function-only is not enabled");
1123 const bool IsColdFuncOnlyInstrGen =
1124 isPGOInstrumentColdFunctionOnly() && IsPGOPreLink &&
1125 !Opts.instrument_cold_function_only_path.empty();
1126
1127 if (IsPGOInstrGen || IsPGOInstrUse || IsMemprofUse || IsCtxProfGen ||
1128 IsCtxProfUse || IsColdFuncOnlyInstrGen)
1129 addPreInlinerPasses(MPM, Level, Phase);
1130
1131 // Add all the requested passes for instrumentation PGO, if requested.
1132 if (IsPGOInstrGen || IsPGOInstrUse) {
1133 addPGOInstrPasses(MPM, Level,
1134 /*RunProfileGen=*/IsPGOInstrGen,
1135 /*IsCS=*/false, PGOOpt->AtomicCounterUpdate,
1136 PGOOpt->ProfileFile, PGOOpt->ProfileRemappingFile);
1137 } else if (IsCtxProfGen || IsCtxProfUse) {
1139 // In pre-link, we just want the instrumented IR. We use the contextual
1140 // profile in the post-thinlink phase.
1141 // The instrumentation will be removed in post-thinlink after IPO.
1142 if (IsCtxProfUse) {
1143 MPM.addPass(AssignGUIDPass());
1144 MPM.addPass(PGOCtxProfFlatteningPass(/*IsPreThinlink=*/true));
1145 return MPM;
1146 }
1147 // Block further inlining in the instrumented ctxprof case. This avoids
1148 // confusingly collecting profiles for the same GUID corresponding to
1149 // different variants of the function. We could do like PGO and identify
1150 // functions by a (GUID, Hash) tuple, but since the ctxprof "use" waits for
1151 // thinlto to happen before performing any further optimizations, it's
1152 // unnecessary to collect profiles for non-prevailing copies.
1154 addPostPGOLoopRotation(MPM, Level);
1155 MPM.addPass(AssignGUIDPass());
1157 } else if (IsColdFuncOnlyInstrGen) {
1158 addPGOInstrPasses(MPM, Level, /* RunProfileGen */ true, /* IsCS */ false,
1159 /* AtomicCounterUpdate */ false,
1160 Opts.instrument_cold_function_only_path.str(),
1161 /* ProfileRemappingFile */ "");
1162 }
1163
1164 if (IsPGOInstrGen || IsPGOInstrUse || IsCtxProfGen)
1165 MPM.addPass(PGOIndirectCallPromotion(false, false));
1166
1167 if (IsPGOPreLink && PGOOpt->CSAction == PGOOptions::CSIRInstr)
1169 PGOOpt->CSProfileGenFile, Opts.enable_sampled_instrumentation));
1170
1171 if (IsMemprofUse)
1172 MPM.addPass(MemProfUsePass(PGOOpt->MemoryProfile, FS));
1173
1174 if (PGOOpt && (PGOOpt->Action == PGOOptions::IRUse ||
1175 PGOOpt->Action == PGOOptions::SampleUse))
1176 MPM.addPass(PGOForceFunctionAttrsPass(PGOOpt->ColdOptType));
1177
1178 MPM.addPass(AlwaysInlinerPass(/*InsertLifetimeIntrinsics=*/true));
1179
1180 if (Opts.enable_module_inliner)
1182 else
1183 MPM.addPass(buildInlinerPipeline(Level, Phase));
1184
1185 // Remove any dead arguments exposed by cleanups, constant folding globals,
1186 // and argument promotion.
1188
1191
1192 if (!isThinLTOPreLink(Phase))
1193 MPM.addPass(CoroCleanupPass());
1194
1195 // Optimize globals now that functions are fully simplified.
1196 MPM.addPass(GlobalOptPass());
1197 MPM.addPass(GlobalDCEPass());
1198
1199 return MPM;
1200}
1201
1202/// TODO: Should LTO cause any differences to this set of passes?
1203void PassBuilder::addVectorPasses(OptimizationLevel Level,
1205 ThinOrFullLTOPhase LTOPhase) {
1208
1209 // Drop dereferenceable assumes after vectorization, as they are no longer
1210 // needed and can inhibit further optimization.
1211 if (!isLTOPreLink(LTOPhase))
1212 FPM.addPass(DropUnnecessaryAssumesPass(/*DropDereferenceable=*/true));
1213
1215 if (isFullLTOPostLink(LTOPhase)) {
1216 // The vectorizer may have significantly shortened a loop body; unroll
1217 // again. Unroll small loops to hide loop backedge latency and saturate any
1218 // parallel execution resources of an out-of-order processor. We also then
1219 // need to clean up redundancies and loop invariant code.
1220 // FIXME: It would be really good to use a loop-integrated instruction
1221 // combiner for cleanup here so that the unrolling and LICM can be pipelined
1222 // across the loop nests.
1223 // We do UnrollAndJam in a separate LPM to ensure it happens before unroll
1224 if (Opts.enable_unroll_and_jam && PTO.LoopUnrolling)
1226 LoopUnrollAndJamPass(static_cast<int>(Level))));
1228 static_cast<int>(Level), /*OnlyWhenForced=*/!PTO.LoopUnrolling,
1231 // Now that we are done with loop unrolling, be it either by LoopVectorizer,
1232 // or LoopUnroll passes, some variable-offset GEP's into alloca's could have
1233 // become constant-offset, thus enabling SROA and alloca promotion. Do so.
1234 // NOTE: we are very late in the pipeline, and we don't have any LICM
1235 // or SimplifyCFG passes scheduled after us, that would cleanup
1236 // the CFG mess this may created if allowed to modify CFG, so forbid that.
1237
1238 // We also turn on struct to vector canonicalization here, which allows
1239 // converting allocas of homogeneous structs into vector allocas when the
1240 // allocas' users are all memory intrinsics. This allows promotion in some
1241 // cases because structs cannot promote to SSA values, but vectors can. We
1242 // only turn this on after memcpyopt runs because this might hinder
1243 // memcpyopt's optimizations if done before. Look at the documentation for
1244 // `tryCanonicalizeStructToVector` in SROA.cpp to see why.
1246 /*AggregateToVector=*/true)));
1247 }
1248
1249 if (!isFullLTOPostLink(LTOPhase)) {
1250 // Eliminate loads by forwarding stores from the previous iteration to loads
1251 // of the current iteration.
1253 }
1254 // Cleanup after the loop optimization passes.
1255 FPM.addPass(InstCombinePass());
1256
1257 if (Level > OptimizationLevel::O1 && Opts.extra_vectorizer_passes) {
1258 ExtraFunctionPassManager<ShouldRunExtraVectorPasses> ExtraPasses;
1259 // At higher optimization levels, try to clean up any runtime overlap and
1260 // alignment checks inserted by the vectorizer. We want to track correlated
1261 // runtime checks for two inner loops in the same outer loop, fold any
1262 // common computations, hoist loop-invariant aspects out of any outer loop,
1263 // and unswitch the runtime checks if possible. Once hoisted, we may have
1264 // dead (or speculatable) control flows or more combining opportunities.
1265 ExtraPasses.addPass(EarlyCSEPass());
1266 ExtraPasses.addPass(CorrelatedValuePropagationPass());
1267 ExtraPasses.addPass(InstCombinePass());
1268 LoopPassManager LPM;
1269 LPM.addPass(LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
1270 /*AllowSpeculation=*/true));
1271 LPM.addPass(SimpleLoopUnswitchPass(/* NonTrivial */ Level ==
1273 ExtraPasses.addPass(
1274 createFunctionToLoopPassAdaptor(std::move(LPM), /*UseMemorySSA=*/true));
1275 ExtraPasses.addPass(
1276 SimplifyCFGPass(SimplifyCFGOptions().convertSwitchRangeToICmp(true)));
1277 ExtraPasses.addPass(InstCombinePass());
1278 FPM.addPass(std::move(ExtraPasses));
1279 }
1280
1281 // Now that we've formed fast to execute loop structures, we do further
1282 // optimizations. These are run afterward as they might block doing complex
1283 // analyses and transforms such as what are needed for loop vectorization.
1284
1285 // Cleanup after loop vectorization, etc. Simplification passes like CVP and
1286 // GVN, loop transforms, and others have already run, so it's now better to
1287 // convert to more optimized IR using more aggressive simplify CFG options.
1288 // The extra sinking transform can create larger basic blocks, so do this
1289 // before SLP vectorization.
1290 FPM.addPass(SimplifyCFGPass(SimplifyCFGOptions()
1291 .forwardSwitchCondToPhi(true)
1292 .convertSwitchRangeToICmp(true)
1293 .convertSwitchToArithmetic(true)
1294 .convertSwitchToLookupTable(true)
1295 .needCanonicalLoops(false)
1296 .hoistCommonInsts(true)
1297 .sinkCommonInsts(true)));
1298
1299 if (isFullLTOPostLink(LTOPhase)) {
1300 FPM.addPass(SCCPPass());
1301 FPM.addPass(InstCombinePass());
1302 FPM.addPass(BDCEPass());
1303 }
1304
1305 // Optimize parallel scalar instruction chains into SIMD instructions.
1306 if (PTO.SLPVectorization) {
1307 FPM.addPass(SLPVectorizerPass());
1308 if (Level >= OptimizationLevel::O2 && Opts.extra_vectorizer_passes) {
1309 FPM.addPass(EarlyCSEPass());
1310 }
1311 }
1312 // Enhance/cleanup vector code.
1313 FPM.addPass(VectorCombinePass());
1314
1315 if (!isFullLTOPostLink(LTOPhase)) {
1316 FPM.addPass(InstCombinePass());
1317 // Unroll small loops to hide loop backedge latency and saturate any
1318 // parallel execution resources of an out-of-order processor. We also then
1319 // need to clean up redundancies and loop invariant code.
1320 // FIXME: It would be really good to use a loop-integrated instruction
1321 // combiner for cleanup here so that the unrolling and LICM can be pipelined
1322 // across the loop nests.
1323 // We do UnrollAndJam in a separate LPM to ensure it happens before unroll
1324 if (Opts.enable_unroll_and_jam && PTO.LoopUnrolling) {
1326 LoopUnrollAndJamPass(static_cast<int>(Level))));
1327 }
1328 FPM.addPass(LoopUnrollPass(LoopUnrollOptions(
1329 static_cast<int>(Level), /*OnlyWhenForced=*/!PTO.LoopUnrolling,
1330 PTO.ForgetAllSCEVInLoopUnroll)));
1331 FPM.addPass(WarnMissedTransformationsPass());
1332 // Now that we are done with loop unrolling, be it either by LoopVectorizer,
1333 // or LoopUnroll passes, some variable-offset GEP's into alloca's could have
1334 // become constant-offset, thus enabling SROA and alloca promotion. Do so.
1335 // NOTE: we are very late in the pipeline, and we don't have any LICM
1336 // or SimplifyCFG passes scheduled after us, that would cleanup
1337 // the CFG mess this may created if allowed to modify CFG, so forbid that.
1338
1339 // We also turn on struct to vector canonicalization here, which allows
1340 // converting allocas of homogeneous structs into vector allocas when the
1341 // allocas' users are all memory intrinsics. This allows promotion in some
1342 // cases because structs cannot promote to SSA values, but vectors can. We
1343 // only turn this on after memcpyopt runs because this might hinder
1344 // memcpyopt's optimizations if done before. Look at the documentation for
1345 // `tryCanonicalizeStructToVector` in SROA.cpp to see why.
1346 FPM.addPass(SROAPass(SROAOptions(SROAOptions::PreserveCFG,
1347 /*AggregateToVector=*/true)));
1348 }
1349
1350 FPM.addPass(InferAlignmentPass());
1351 FPM.addPass(InstCombinePass());
1352
1353 // This is needed for two reasons:
1354 // 1. It works around problems that instcombine introduces, such as sinking
1355 // expensive FP divides into loops containing multiplications using the
1356 // divide result.
1357 // 2. It helps to clean up some loop-invariant code created by the loop
1358 // unroll pass when IsFullLTO=false.
1360 LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
1361 /*AllowSpeculation=*/true),
1362 /*UseMemorySSA=*/true));
1363
1364 // Now that we've vectorized and unrolled loops, we may have more refined
1365 // alignment information, try to re-derive it here.
1366 FPM.addPass(AlignmentFromAssumptionsPass());
1367}
1368
1371 ThinOrFullLTOPhase LTOPhase) {
1373
1374 // Run partial inlining pass to partially inline functions that have
1375 // large bodies.
1376 if (Opts.enable_partial_inlining)
1378
1379 // Remove avail extern fns and globals definitions since we aren't compiling
1380 // an object file for later LTO. For LTO we want to preserve these so they
1381 // are eligible for inlining at link-time. Note if they are unreferenced they
1382 // will be removed by GlobalDCE later, so this only impacts referenced
1383 // available externally globals. Eventually they will be suppressed during
1384 // codegen, but eliminating here enables more opportunity for GlobalDCE as it
1385 // may make globals referenced by available external functions dead and saves
1386 // running remaining passes on the eliminated functions. These should be
1387 // preserved during prelinking for link-time inlining decisions.
1388 if (!isLTOPreLink(LTOPhase))
1390
1391 // Do RPO function attribute inference across the module to forward-propagate
1392 // attributes where applicable.
1393 // FIXME: Is this really an optimization rather than a canonicalization?
1395
1396 // Do a post inline PGO instrumentation and use pass. This is a context
1397 // sensitive PGO pass. We don't want to do this in LTOPreLink phrase as
1398 // cross-module inline has not been done yet. The context sensitive
1399 // instrumentation is after all the inlines are done.
1400 if (!isLTOPreLink(LTOPhase) && PGOOpt) {
1401 if (PGOOpt->CSAction == PGOOptions::CSIRInstr)
1402 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/true,
1403 /*IsCS=*/true, PGOOpt->AtomicCounterUpdate,
1404 PGOOpt->CSProfileGenFile, PGOOpt->ProfileRemappingFile);
1405 else if (PGOOpt->CSAction == PGOOptions::CSIRUse)
1406 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/false,
1407 /*IsCS=*/true, PGOOpt->AtomicCounterUpdate,
1408 PGOOpt->ProfileFile, PGOOpt->ProfileRemappingFile);
1409 }
1410
1411 // Re-compute GlobalsAA here prior to function passes. This is particularly
1412 // useful as the above will have inlined, DCE'ed, and function-attr
1413 // propagated everything. We should at this point have a reasonably minimal
1414 // and richly annotated call graph. By computing aliasing and mod/ref
1415 // information for all local globals here, the late loop passes and notably
1416 // the vectorizer will be able to use them to help recognize vectorizable
1417 // memory operations.
1418 if (Opts.enable_global_analyses)
1420
1421 invokeOptimizerEarlyEPCallbacks(MPM, Level, LTOPhase);
1422
1423 FunctionPassManager OptimizePM;
1424
1425 // Only drop unnecessary assumes post-inline and post-link, as otherwise
1426 // additional uses of the affected value may be introduced through inlining
1427 // and CSE.
1428 if (!isLTOPreLink(LTOPhase))
1429 OptimizePM.addPass(DropUnnecessaryAssumesPass());
1430
1431 // Scheduling LoopVersioningLICM when inlining is over, because after that
1432 // we may see more accurate aliasing. Reason to run this late is that too
1433 // early versioning may prevent further inlining due to increase of code
1434 // size. Other optimizations which runs later might get benefit of no-alias
1435 // assumption in clone loop.
1436 if (Opts.enable_loop_versioning_licm) {
1437 OptimizePM.addPass(
1439 // LoopVersioningLICM pass might increase new LICM opportunities.
1441 LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
1442 /*AllowSpeculation=*/true),
1443 /*USeMemorySSA=*/true));
1444 }
1445
1446 OptimizePM.addPass(Float2IntPass());
1447 // Defer until LTO post-link where some constants may become known.
1448 if (!isLTOPreLink(LTOPhase))
1450
1451 if (Opts.enable_matrix) {
1452 OptimizePM.addPass(LowerMatrixIntrinsicsPass());
1453 OptimizePM.addPass(EarlyCSEPass());
1454 }
1455
1456 // CHR pass should only be applied with the profile information.
1457 // The check is to check the profile summary information in CHR.
1458 if (Opts.enable_chr && Level == OptimizationLevel::O3)
1459 OptimizePM.addPass(ControlHeightReductionPass());
1460
1461 // FIXME: We need to run some loop optimizations to re-rotate loops after
1462 // simplifycfg and others undo their rotation.
1463
1464 // Optimize the loop execution. These passes operate on entire loop nests
1465 // rather than on each loop in an inside-out manner, and so they are actually
1466 // function passes.
1467
1468 invokeVectorizerStartEPCallbacks(OptimizePM, Level);
1469
1470 LoopPassManager LPM;
1471 // First rotate loops that may have been un-rotated by prior passes.
1472 // Disable header duplication at -Oz.
1473 LPM.addPass(LoopRotatePass(/*EnableLoopHeaderDuplication=*/true,
1474 isLTOPreLink(LTOPhase),
1475 /*CheckExitCount=*/true));
1476 // Some loops may have become dead by now. Try to delete them.
1477 // FIXME: see discussion in https://reviews.llvm.org/D112851,
1478 // this may need to be revisited once we run GVN before loop deletion
1479 // in the simplification pipeline.
1480 LPM.addPass(LoopDeletionPass());
1481
1482 if (PTO.LoopInterchange)
1483 LPM.addPass(LoopInterchangePass());
1484
1485 OptimizePM.addPass(
1486 createFunctionToLoopPassAdaptor(std::move(LPM), /*UseMemorySSA=*/false));
1487
1488 // FIXME: This may not be the right place in the pipeline.
1489 // We need to have the data to support the right place.
1490 if (PTO.LoopFusion)
1491 OptimizePM.addPass(LoopFusePass());
1492
1493 // Distribute loops to allow partial vectorization. I.e. isolate dependences
1494 // into separate loop that would otherwise inhibit vectorization. This is
1495 // currently only performed for loops marked with the metadata
1496 // llvm.loop.distribute=true or when -enable-loop-distribute is specified.
1497 OptimizePM.addPass(LoopDistributePass());
1498
1499 // Populates the VFABI attribute with the scalar-to-vector mappings
1500 // from the TargetLibraryInfo.
1501 OptimizePM.addPass(InjectTLIMappings());
1502
1503 addVectorPasses(Level, OptimizePM, LTOPhase);
1504
1505 invokeVectorizerEndEPCallbacks(OptimizePM, Level);
1506
1507 // LoopSink pass sinks instructions hoisted by LICM, which serves as a
1508 // canonicalization pass that enables other optimizations. As a result,
1509 // LoopSink pass needs to be a very late IR pass to avoid undoing LICM
1510 // result too early.
1511 OptimizePM.addPass(LoopSinkPass());
1512
1513 // And finally clean up LCSSA form before generating code.
1514 OptimizePM.addPass(InstSimplifyPass());
1515
1516 // This hoists/decomposes div/rem ops. It should run after other sink/hoist
1517 // passes to avoid re-sinking, but before SimplifyCFG because it can allow
1518 // flattening of blocks.
1519 OptimizePM.addPass(DivRemPairsPass());
1520
1521 // Merge adjacent icmps into memcmp, then expand memcmp to loads/compares.
1522 // TODO: move this furter up so that it can be optimized by GVN, etc.
1523 if (Opts.enable_mergeicmps)
1524 OptimizePM.addPass(MergeICmpsPass());
1525 OptimizePM.addPass(ExpandMemCmpPass());
1526
1527 // Try to annotate calls that were created during optimization.
1528 OptimizePM.addPass(
1529 TailCallElimPass(/*UpdateFunctionEntryCount=*/isInstrumentedPGOUse()));
1530
1531 // LoopSink (and other loop passes since the last simplifyCFG) might have
1532 // resulted in single-entry-single-exit or empty blocks. Clean up the CFG.
1533 OptimizePM.addPass(
1535 .convertSwitchRangeToICmp(true)
1536 .convertSwitchToArithmetic(true)
1537 .speculateUnpredictables(true)
1538 .hoistLoadsStoresWithCondFaulting(true)));
1539
1540 // Add the core optimizing pipeline.
1541 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(OptimizePM),
1542 PTO.EagerlyInvalidateAnalyses));
1543
1544 // AllocToken transforms heap allocation calls; this needs to run late after
1545 // other allocation call transformations (such as those in InstCombine).
1546 if (!isLTOPreLink(LTOPhase))
1547 MPM.addPass(AllocTokenPass());
1548
1549 invokeOptimizerLastEPCallbacks(MPM, Level, LTOPhase);
1550
1551 // Run the Instrumentor pass late.
1552 if (Opts.enable_instrumentor)
1553 MPM.addPass(InstrumentorPass(FS));
1554
1555 // Split out cold code. Splitting is done late to avoid hiding context from
1556 // other optimizations and inadvertently regressing performance. The tradeoff
1557 // is that this has a higher code size cost than splitting early.
1558 if (Opts.hot_cold_split && !isLTOPreLink(LTOPhase))
1560
1561 // Now we need to do some global optimization transforms.
1562 // FIXME: It would seem like these should come first in the optimization
1563 // pipeline and maybe be the bottom of the canonicalization pipeline? Weird
1564 // ordering here.
1565 MPM.addPass(GlobalDCEPass());
1567
1568 // Merge functions if requested. It has a better chance to merge functions
1569 // after ConstantMerge folded jump tables.
1570 if (PTO.MergeFunctions)
1572
1573 if (PTO.CallGraphProfile && !isLTOPreLink(LTOPhase))
1574 MPM.addPass(CGProfilePass(isLTOPostLink(LTOPhase)));
1575
1576 // RelLookupTableConverterPass runs later in LTO post-link pipeline.
1577 if (!isLTOPreLink(LTOPhase))
1579
1580 // Add devirtualization pass only when LTO is not enabled, as otherwise
1581 // the pass is already enabled in the LTO pipeline.
1582 if (PTO.DevirtualizeSpeculatively && LTOPhase == ThinOrFullLTOPhase::None) {
1583 // TODO: explore a better pipeline configuration that can improve
1584 // compilation time overhead.
1585 // FIXME: move this earlier (lots of pass ordering tests will need fixing)
1586 MPM.addPass(AssignGUIDPass());
1588 /*ExportSummary*/ nullptr,
1589 /*ImportSummary*/ nullptr,
1590 /*DevirtSpeculatively*/ PTO.DevirtualizeSpeculatively));
1592 // Given that the devirtualization creates more opportunities for inlining,
1593 // we run the Inliner again here to maximize the optimization gain we
1594 // get from devirtualization.
1595 // Also, we can't run devirtualization before inlining because the
1596 // devirtualization depends on the passes optimizing/eliminating vtable GVs
1597 // and those passes are only effective after inlining.
1599 }
1600
1601 // Attach !implicit.ref metadata from all functions to copyright strings.
1603
1604 return MPM;
1605}
1606
1610 if (Level == OptimizationLevel::O0)
1611 return buildO0DefaultPipeline(Level, Phase);
1612
1614 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1615 // Currently this pipeline is only invoked in an LTO pre link pass or when we
1616 // are not running LTO. If that changes the below checks may need updating.
1618
1619 // If we are invoking this in non-LTO mode, remove any MemProf related
1620 // attributes and metadata, as we don't know whether we are linking with
1621 // a library containing the necessary interfaces.
1624
1625 // Convert @llvm.global.annotations to !annotation metadata.
1627
1628 // Force any function attributes we want the rest of the pipeline to observe.
1630
1631 if (Opts.opt_pipeline_trigger_crash)
1633
1634 if (PGOOpt && PGOOpt->DebugInfoForProfiling)
1636
1637 // Apply module pipeline start EP callback.
1639
1640 // Add the core simplification pipeline.
1642
1643 // Now add the optimization pipeline.
1645
1646 if (PGOOpt && PGOOpt->PseudoProbeForProfiling &&
1647 PGOOpt->Action == PGOOptions::SampleUse)
1649
1650 // Emit annotation remarks.
1652
1653 if (isLTOPreLink(Phase))
1654 addRequiredLTOPreLinkPasses(MPM);
1655
1656 instructionCountersPass(MPM, /* IsPreOptimization */ false);
1657 return MPM;
1658}
1659
1662 bool EmitSummary, bool Verify) {
1664
1665 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1666
1667 if (ThinLTO)
1669 else
1671 // AssignGUIDPass attaches !guid metadata (MD_unique_id) to global objects,
1672 // triggering the bitcode writer to emit a METADATA_KIND_BLOCK. Standard LTO
1673 // bitcode emission runs VerifierPass by default, which registers metadata
1674 // kind IDs in LLVMContext. Running VerifierPass here before EmbedBitcodePass
1675 // to get the same behavior.
1676 if (Verify)
1677 MPM.addPass(VerifierPass());
1678 MPM.addPass(EmbedBitcodePass(ThinLTO, EmitSummary));
1679
1680 // Perform any cleanups to the IR that aren't suitable for per TU compilation,
1681 // like removing CFI/WPD related instructions. Note, we reuse
1682 // DropTypeTestsPass to clean up type tests rather than duplicate that logic
1683 // in FatLtoCleanup.
1684 MPM.addPass(FatLtoCleanup());
1685
1686 // If we're doing FatLTO w/ CFI enabled, we don't want the type tests in the
1687 // object code, only in the bitcode section, so drop it before we run
1688 // module optimization and generate machine code. If llvm.type.test() isn't in
1689 // the IR, this won't do anything.
1691
1692 // Use the ThinLTO post-link pipeline with sample profiling
1693 if (ThinLTO && PGOOpt && PGOOpt->Action == PGOOptions::SampleUse)
1694 MPM.addPass(buildThinLTODefaultPipeline(Level, /*ImportSummary=*/nullptr));
1695 else {
1696 // ModuleSimplification does not run the coroutine passes for
1697 // ThinLTOPreLink, so we need the coroutine passes to run for ThinLTO
1698 // builds, otherwise they will miscompile.
1699 if (ThinLTO) {
1700 // TODO: replace w/ buildCoroWrapper() when it takes phase and level into
1701 // consideration.
1702 CGSCCPassManager CGPM;
1706 MPM.addPass(CoroCleanupPass());
1707 }
1708
1709 // otherwise, just use module optimization
1710 MPM.addPass(
1712 // Emit annotation remarks.
1714 }
1715
1716 instructionCountersPass(MPM, /* IsPreOptimization */ false);
1717
1718 return MPM;
1719}
1720
1723 if (Level == OptimizationLevel::O0)
1725
1727
1728 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1729
1730 // Convert @llvm.global.annotations to !annotation metadata.
1732
1733 // Force any function attributes we want the rest of the pipeline to observe.
1735
1736 if (PGOOpt && PGOOpt->DebugInfoForProfiling)
1738
1739 // Apply module pipeline start EP callback.
1741
1742 // If we are planning to perform ThinLTO later, we don't bloat the code with
1743 // unrolling/vectorization/... now. Just simplify the module as much as we
1744 // can.
1747 // In pre-link, for ctx prof use, we stop here with an instrumented IR. We let
1748 // thinlto use the contextual info to perform imports; then use the contextual
1749 // profile in the post-thinlink phase.
1750 if (!UseCtxProfile.empty()) {
1751 addRequiredLTOPreLinkPasses(MPM);
1752 return MPM;
1753 }
1754
1755 // Run partial inlining pass to partially inline functions that have
1756 // large bodies.
1757 // FIXME: It isn't clear whether this is really the right place to run this
1758 // in ThinLTO. Because there is another canonicalization and simplification
1759 // phase that will run after the thin link, running this here ends up with
1760 // less information than will be available later and it may grow functions in
1761 // ways that aren't beneficial.
1762 if (Opts.enable_partial_inlining)
1764
1765 if (PGOOpt && PGOOpt->PseudoProbeForProfiling &&
1766 PGOOpt->Action == PGOOptions::SampleUse)
1768
1769 // Handle Optimizer{Early,Last}EPCallbacks added by clang on PreLink. Actual
1770 // optimization is going to be done in PostLink stage, but clang can't add
1771 // callbacks there in case of in-process ThinLTO called by linker.
1776
1777 // Emit annotation remarks.
1779
1780 // Attach !implicit.ref metadata from all functions to copyright strings.
1782
1783 addRequiredLTOPreLinkPasses(MPM);
1784
1785 instructionCountersPass(MPM, /* IsPreOptimization */ false);
1786
1787 return MPM;
1788}
1789
1791 OptimizationLevel Level, const ModuleSummaryIndex *ImportSummary) {
1793
1794 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1795
1797
1798 // If we are invoking this without a summary index noting that we are linking
1799 // with a library containing the necessary APIs, remove any MemProf related
1800 // attributes and metadata.
1801 if (!ImportSummary || !ImportSummary->withSupportsHotColdNew())
1803
1804 if (ImportSummary) {
1805 // For ThinLTO we must apply the context disambiguation decisions early, to
1806 // ensure we can correctly match the callsites to summary data.
1809 ImportSummary, PGOOpt && PGOOpt->Action == PGOOptions::SampleUse));
1810
1811 // These passes import type identifier resolutions for whole-program
1812 // devirtualization and CFI. They must run early because other passes may
1813 // disturb the specific instruction patterns that these passes look for,
1814 // creating dependencies on resolutions that may not appear in the summary.
1815 //
1816 // For example, GVN may transform the pattern assume(type.test) appearing in
1817 // two basic blocks into assume(phi(type.test, type.test)), which would
1818 // transform a dependency on a WPD resolution into a dependency on a type
1819 // identifier resolution for CFI.
1820 //
1821 // Also, WPD has access to more precise information than ICP and can
1822 // devirtualize more effectively, so it should operate on the IR first.
1823 //
1824 // The WPD and LowerTypeTest passes need to run at -O0 to lower type
1825 // metadata and intrinsics.
1826 MPM.addPass(WholeProgramDevirtPass(nullptr, ImportSummary));
1827 MPM.addPass(LowerTypeTestsPass(nullptr, ImportSummary));
1828 }
1829
1830 if (Level == OptimizationLevel::O0) {
1831 // Run a second time to clean up any type tests left behind by WPD for use
1832 // in ICP.
1835
1836 // AllocToken transforms heap allocation calls; this needs to run late after
1837 // other allocation call transformations (such as those in InstCombine).
1838 MPM.addPass(AllocTokenPass());
1839
1840 // Drop available_externally and unreferenced globals. This is necessary
1841 // with ThinLTO in order to avoid leaving undefined references to dead
1842 // globals in the object file.
1844 MPM.addPass(GlobalDCEPass());
1845
1847
1848 return MPM;
1849 }
1850 if (!UseCtxProfile.empty()) {
1851 MPM.addPass(
1853 } else {
1854 // Add the core simplification pipeline.
1857 }
1858 // Now add the optimization pipeline.
1861
1863
1864 // Emit annotation remarks.
1866
1867 instructionCountersPass(MPM, /* IsPreOptimization */ false);
1868
1869 return MPM;
1870}
1871
1874 // FIXME: We should use a customized pre-link pipeline!
1875 return buildPerModuleDefaultPipeline(Level,
1877}
1878
1881 ModuleSummaryIndex *ExportSummary) {
1883
1884 instructionCountersPass(MPM, /* IsPreOptimization */ true);
1885
1887
1888 // If we are invoking this without a summary index noting that we are linking
1889 // with a library containing the necessary APIs, remove any MemProf related
1890 // attributes and metadata.
1891 if (!ExportSummary || !ExportSummary->withSupportsHotColdNew())
1893
1894 // Create a function that performs CFI checks for cross-DSO calls with targets
1895 // in the current module.
1896 MPM.addPass(CrossDSOCFIPass());
1897
1898 if (Level == OptimizationLevel::O0) {
1899 // The WPD and LowerTypeTest passes need to run at -O0 to lower type
1900 // metadata and intrinsics.
1901 MPM.addPass(WholeProgramDevirtPass(ExportSummary, nullptr));
1902 MPM.addPass(LowerTypeTestsPass(ExportSummary, nullptr));
1903 // Run a second time to clean up any type tests left behind by WPD for use
1904 // in ICP.
1906
1908
1909 // AllocToken transforms heap allocation calls; this needs to run late after
1910 // other allocation call transformations (such as those in InstCombine).
1911 MPM.addPass(AllocTokenPass());
1912
1914
1915 // Emit annotation remarks.
1917
1918 return MPM;
1919 }
1920
1921 if (PGOOpt && PGOOpt->Action == PGOOptions::SampleUse) {
1922 // Load sample profile before running the LTO optimization pipeline.
1923 MPM.addPass(SampleProfileLoaderPass(PGOOpt->ProfileFile,
1924 PGOOpt->ProfileRemappingFile,
1926 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
1927 // RequireAnalysisPass for PSI before subsequent non-module passes.
1929 }
1930
1931 // Try to run OpenMP optimizations, quick no-op if no OpenMP metadata present.
1933
1934 // Remove unused virtual tables to improve the quality of code generated by
1935 // whole-program devirtualization and bitset lowering.
1936 MPM.addPass(GlobalDCEPass(/*InLTOPostLink=*/true));
1937
1938 // Do basic inference of function attributes from known properties of system
1939 // libraries and other oracles.
1941
1942 if (Level >= OptimizationLevel::O2) {
1944 CallSiteSplittingPass(), PTO.EagerlyInvalidateAnalyses));
1945
1946 // Indirect call promotion. This should promote all the targets that are
1947 // left by the earlier promotion pass that promotes intra-module targets.
1948 // This two-step promotion is to save the compile time. For LTO, it should
1949 // produce the same result as if we only do promotion here.
1951 true /* InLTO */, PGOOpt && PGOOpt->Action == PGOOptions::SampleUse));
1952
1953 // Promoting by-reference arguments to by-value exposes more constants to
1954 // IPSCCP.
1955 CGSCCPassManager CGPM;
1958 CGPM.addPass(
1961
1962 // Propagate constants at call sites into the functions they call. This
1963 // opens opportunities for globalopt (and inlining) by substituting function
1964 // pointers passed as arguments to direct uses of functions.
1965 MPM.addPass(IPSCCPPass(IPSCCPOptions(/*AllowFuncSpec=*/true)));
1966
1967 // Attach metadata to indirect call sites indicating the set of functions
1968 // they may target at run-time. This should follow IPSCCP.
1970 }
1971
1972 // Do RPO function attribute inference across the module to forward-propagate
1973 // attributes where applicable.
1974 // FIXME: Is this really an optimization rather than a canonicalization?
1976
1977 // Use in-range annotations on GEP indices to split globals where beneficial.
1978 MPM.addPass(GlobalSplitPass());
1979
1980 // Run whole program optimization of virtual call when the list of callees
1981 // is fixed.
1982 MPM.addPass(WholeProgramDevirtPass(ExportSummary, nullptr));
1983
1985 // Stop here at -O1.
1986 if (Level == OptimizationLevel::O1) {
1988 LowerConstantIntrinsicsPass(), PTO.EagerlyInvalidateAnalyses));
1989
1990 // The LowerTypeTestsPass needs to run to lower type metadata and the
1991 // type.test intrinsics. The pass does nothing if CFI is disabled.
1992 MPM.addPass(LowerTypeTestsPass(ExportSummary, nullptr));
1993 // Run a second time to clean up any type tests left behind by WPD for use
1994 // in ICP (which is performed earlier than this in the regular LTO
1995 // pipeline).
1997
1999
2000 // AllocToken transforms heap allocation calls; this needs to run late after
2001 // other allocation call transformations (such as those in InstCombine).
2002 MPM.addPass(AllocTokenPass());
2003
2005
2006 // Emit annotation remarks.
2008
2009 instructionCountersPass(MPM, /* IsPreOptimization */ false);
2010
2011 return MPM;
2012 }
2013
2014 // TODO: Skip to match buildCoroWrapper.
2015 MPM.addPass(CoroEarlyPass());
2016
2017 // Optimize globals to try and fold them into constants.
2018 MPM.addPass(GlobalOptPass());
2019
2020 // Promote any localized globals to SSA registers.
2022
2023 // Linking modules together can lead to duplicate global constant, only
2024 // keep one copy of each constant.
2026
2027 // Remove unused arguments from functions.
2029
2030 // Reduce the code after globalopt and ipsccp. Both can open up significant
2031 // simplification opportunities, and both can propagate functions through
2032 // function pointers. When this happens, we often have to resolve varargs
2033 // calls, etc, so let instcombine do this.
2034 FunctionPassManager PeepholeFPM;
2035 PeepholeFPM.addPass(InstCombinePass());
2036 if (Level >= OptimizationLevel::O2)
2037 PeepholeFPM.addPass(AggressiveInstCombinePass());
2038 invokePeepholeEPCallbacks(PeepholeFPM, Level);
2039
2040 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(PeepholeFPM),
2041 PTO.EagerlyInvalidateAnalyses));
2042
2043 // Lower variadic functions for supported targets prior to inlining.
2045
2046 // Note: historically, the PruneEH pass was run first to deduce nounwind and
2047 // generally clean up exception handling overhead. It isn't clear this is
2048 // valuable as the inliner doesn't currently care whether it is inlining an
2049 // invoke or a call.
2050 // Run the inliner now.
2052
2053 // Perform context disambiguation after inlining, since that would reduce the
2054 // amount of additional cloning required to distinguish the allocation
2055 // contexts.
2058 /*Summary=*/nullptr,
2059 PGOOpt && PGOOpt->Action == PGOOptions::SampleUse));
2060
2061 // Optimize globals again after we ran the inliner.
2062 MPM.addPass(GlobalOptPass());
2063
2064 // Run the OpenMPOpt pass again after global optimizations.
2066
2067 // Garbage collect dead functions.
2068 MPM.addPass(GlobalDCEPass(/*InLTOPostLink=*/true));
2069
2070 // If we didn't decide to inline a function, check to see if we can
2071 // transform it to pass arguments by value instead of by reference.
2072 CGSCCPassManager CGPM;
2078
2080 // The IPO Passes may leave cruft around. Clean up after them.
2081 FPM.addPass(InstCombinePass());
2082 invokePeepholeEPCallbacks(FPM, Level);
2083
2084 if (Opts.enable_constraint_elimination)
2086
2088
2089 // Do a post inline PGO instrumentation and use pass. This is a context
2090 // sensitive PGO pass.
2091 if (PGOOpt) {
2092 if (PGOOpt->CSAction == PGOOptions::CSIRInstr)
2093 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/true,
2094 /*IsCS=*/true, PGOOpt->AtomicCounterUpdate,
2095 PGOOpt->CSProfileGenFile, PGOOpt->ProfileRemappingFile);
2096 else if (PGOOpt->CSAction == PGOOptions::CSIRUse)
2097 addPGOInstrPasses(MPM, Level, /*RunProfileGen=*/false,
2098 /*IsCS=*/true, PGOOpt->AtomicCounterUpdate,
2099 PGOOpt->ProfileFile, PGOOpt->ProfileRemappingFile);
2100 }
2101
2102 // Break up allocas
2104
2105 // LTO provides additional opportunities for tailcall elimination due to
2106 // link-time inlining, and visibility of nocapture attribute.
2107 FPM.addPass(
2108 TailCallElimPass(/*UpdateFunctionEntryCount=*/isInstrumentedPGOUse()));
2109
2110 // Run a few AA driver optimizations here and now to cleanup the code.
2111 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(FPM),
2112 PTO.EagerlyInvalidateAnalyses));
2113
2114 MPM.addPass(
2116
2117 // Require the GlobalsAA analysis for the module so we can query it within
2118 // MainFPM.
2119 if (Opts.enable_global_analyses) {
2121 // Invalidate AAManager so it can be recreated and pick up the newly
2122 // available GlobalsAA.
2123 MPM.addPass(
2125 }
2126
2127 FunctionPassManager MainFPM;
2129 LICMPass(PTO.LicmMssaOptCap, PTO.LicmMssaNoAccForPromotionCap,
2130 /*AllowSpeculation=*/true),
2131 /*USeMemorySSA=*/true));
2132
2133 if (Opts.enable_newgvn)
2134 MainFPM.addPass(NewGVNPass());
2135 else
2136 MainFPM.addPass(GVNPass());
2137
2138 // Remove dead memcpy()'s.
2139 MainFPM.addPass(MemCpyOptPass());
2140
2141 // Nuke dead stores.
2142 MainFPM.addPass(DSEPass());
2143 MainFPM.addPass(MoveAutoInitPass());
2145
2147
2148 invokeVectorizerStartEPCallbacks(MainFPM, Level);
2149
2150 LoopPassManager LPM;
2151 if (Opts.enable_loop_flatten && Level >= OptimizationLevel::O2)
2152 LPM.addPass(LoopFlattenPass());
2153 LPM.addPass(IndVarSimplifyPass());
2154 LPM.addPass(LoopDeletionPass());
2155 // FIXME: Add loop interchange.
2156
2157 // Unroll small loops and perform peeling.
2158 LPM.addPass(LoopFullUnrollPass(static_cast<int>(Level),
2159 /* OnlyWhenForced= */ !PTO.LoopUnrolling,
2160 PTO.ForgetAllSCEVInLoopUnroll));
2161 // The loop passes in LPM (LoopFullUnrollPass) do not preserve MemorySSA.
2162 // *All* loop passes must preserve it, in order to be able to use it.
2163 MainFPM.addPass(
2164 createFunctionToLoopPassAdaptor(std::move(LPM), /*UseMemorySSA=*/false));
2165
2166 MainFPM.addPass(LoopDistributePass());
2167
2168 addVectorPasses(Level, MainFPM, ThinOrFullLTOPhase::FullLTOPostLink);
2169
2170 invokeVectorizerEndEPCallbacks(MainFPM, Level);
2171
2172 // Run the OpenMPOpt CGSCC pass again late.
2175
2176 invokePeepholeEPCallbacks(MainFPM, Level);
2177 MainFPM.addPass(JumpThreadingPass());
2178 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(MainFPM),
2179 PTO.EagerlyInvalidateAnalyses));
2180
2181 // Lower type metadata and the type.test intrinsic. This pass supports
2182 // clang's control flow integrity mechanisms (-fsanitize=cfi*) and needs
2183 // to be run at link time if CFI is enabled. This pass does nothing if
2184 // CFI is disabled.
2185 MPM.addPass(LowerTypeTestsPass(ExportSummary, nullptr));
2186 // Run a second time to clean up any type tests left behind by WPD for use
2187 // in ICP (which is performed earlier than this in the regular LTO pipeline).
2189
2190 // Enable splitting late in the FullLTO post-link pipeline.
2191 if (Opts.hot_cold_split)
2193
2194 // Add late LTO optimization passes.
2195 FunctionPassManager LateFPM;
2196
2197 // LoopSink pass sinks instructions hoisted by LICM, which serves as a
2198 // canonicalization pass that enables other optimizations. As a result,
2199 // LoopSink pass needs to be a very late IR pass to avoid undoing LICM
2200 // result too early.
2201 LateFPM.addPass(LoopSinkPass());
2202
2203 // This hoists/decomposes div/rem ops. It should run after other sink/hoist
2204 // passes to avoid re-sinking, but before SimplifyCFG because it can allow
2205 // flattening of blocks.
2206 LateFPM.addPass(DivRemPairsPass());
2207
2208 // Delete basic blocks, which optimization passes may have killed.
2210 .convertSwitchRangeToICmp(true)
2211 .convertSwitchToArithmetic(true)
2212 .hoistCommonInsts(true)
2213 .speculateUnpredictables(true)));
2214 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(LateFPM)));
2215
2216 // Drop bodies of available eternally objects to improve GlobalDCE.
2218
2219 // Now that we have optimized the program, discard unreachable functions.
2220 MPM.addPass(GlobalDCEPass(/*InLTOPostLink=*/true));
2221
2222 if (PTO.MergeFunctions)
2224
2226
2227 if (PTO.CallGraphProfile)
2228 MPM.addPass(CGProfilePass(/*InLTOPostLink=*/true));
2229
2230 MPM.addPass(CoroCleanupPass());
2231
2232 // AllocToken transforms heap allocation calls; this needs to run late after
2233 // other allocation call transformations (such as those in InstCombine).
2234 MPM.addPass(AllocTokenPass());
2235
2237
2238 // Emit annotation remarks.
2240
2241 instructionCountersPass(MPM, /* IsPreOptimization */ false);
2242
2243 return MPM;
2244}
2245
2249 assert(Level == OptimizationLevel::O0 &&
2250 "buildO0DefaultPipeline should only be used with O0");
2251
2253
2254 instructionCountersPass(MPM, /* IsPreOptimization */ true);
2255
2256 // Perform pseudo probe instrumentation in O0 mode. This is for the
2257 // consistency between different build modes. For example, a LTO build can be
2258 // mixed with an O0 prelink and an O2 postlink. Loading a sample profile in
2259 // the postlink will require pseudo probe instrumentation in the prelink.
2260 if (PGOOpt && PGOOpt->PseudoProbeForProfiling)
2262
2263 if (PGOOpt && (PGOOpt->Action == PGOOptions::IRInstr ||
2264 PGOOpt->Action == PGOOptions::IRUse))
2266 MPM,
2267 /*RunProfileGen=*/(PGOOpt->Action == PGOOptions::IRInstr),
2268 /*IsCS=*/false, PGOOpt->AtomicCounterUpdate, PGOOpt->ProfileFile,
2269 PGOOpt->ProfileRemappingFile);
2270
2271 // Instrument function entry and exit before all inlining.
2273 EntryExitInstrumenterPass(/*PostInlining=*/false)));
2274
2276
2277 if (PGOOpt && PGOOpt->DebugInfoForProfiling)
2279
2280 if (PGOOpt && PGOOpt->Action == PGOOptions::SampleUse) {
2281 // Explicitly disable sample loader inlining and use flattened profile in O0
2282 // pipeline.
2283 MPM.addPass(SampleProfileLoaderPass(PGOOpt->ProfileFile,
2284 PGOOpt->ProfileRemappingFile,
2286 /*DisableSampleProfileInlining=*/true,
2287 /*UseFlattenedProfile=*/true));
2288 // Cache ProfileSummaryAnalysis once to avoid the potential need to insert
2289 // RequireAnalysisPass for PSI before subsequent non-module passes.
2291 }
2292
2294
2295 // Build a minimal pipeline based on the semantics required by LLVM,
2296 // which is just that always inlining occurs. Further, disable generating
2297 // lifetime intrinsics to avoid enabling further optimizations during
2298 // code generation.
2300 /*InsertLifetimeIntrinsics=*/false));
2301
2302 if (PTO.MergeFunctions)
2304
2305 if (Opts.enable_matrix)
2306 MPM.addPass(
2308
2309 if (!CGSCCOptimizerLateEPCallbacks.empty()) {
2310 CGSCCPassManager CGPM;
2312 if (!CGPM.isEmpty())
2314 }
2315 if (!LateLoopOptimizationsEPCallbacks.empty()) {
2316 LoopPassManager LPM;
2318 if (!LPM.isEmpty()) {
2320 createFunctionToLoopPassAdaptor(std::move(LPM))));
2321 }
2322 }
2323 if (!LoopOptimizerEndEPCallbacks.empty()) {
2324 LoopPassManager LPM;
2326 if (!LPM.isEmpty()) {
2328 createFunctionToLoopPassAdaptor(std::move(LPM))));
2329 }
2330 }
2331 if (!ScalarOptimizerLateEPCallbacks.empty()) {
2334 if (!FPM.isEmpty())
2335 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(FPM)));
2336 }
2337
2339
2340 if (!VectorizerStartEPCallbacks.empty()) {
2343 if (!FPM.isEmpty())
2344 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(FPM)));
2345 }
2346
2347 if (!VectorizerEndEPCallbacks.empty()) {
2350 if (!FPM.isEmpty())
2351 MPM.addPass(createModuleToFunctionPassAdaptor(std::move(FPM)));
2352 }
2353
2355
2356 // AllocToken transforms heap allocation calls; this needs to run late after
2357 // other allocation call transformations (such as those in InstCombine).
2358 if (!isLTOPreLink(Phase))
2359 MPM.addPass(AllocTokenPass());
2360
2362
2363 if (Opts.enable_instrumentor)
2364 MPM.addPass(InstrumentorPass(FS));
2365
2366 // Attach !implicit.ref metadata from all functions to copyright strings.
2368
2369 if (isLTOPreLink(Phase))
2370 addRequiredLTOPreLinkPasses(MPM);
2371
2372 // Emit annotation remarks.
2374
2375 instructionCountersPass(MPM, /* IsPreOptimization */ false);
2376
2377 return MPM;
2378}
2379
2381 AAManager AA;
2382
2383 // The order in which these are registered determines their priority when
2384 // being queried.
2385
2386 // Add any target-specific alias analyses that should be run early.
2387 if (TM)
2388 TM->registerEarlyDefaultAliasAnalyses(AA);
2389
2390 // First we register the basic alias analysis that provides the majority of
2391 // per-function local AA logic. This is a stateless, on-demand local set of
2392 // AA techniques.
2393 AA.registerFunctionAnalysis<BasicAA>();
2394
2395 // Next we query fast, specialized alias analyses that wrap IR-embedded
2396 // information about aliasing.
2397 AA.registerFunctionAnalysis<ScopedNoAliasAA>();
2398 AA.registerFunctionAnalysis<TypeBasedAA>();
2399
2400 // Add support for querying global aliasing information when available.
2401 // Because the `AAManager` is a function analysis and `GlobalsAA` is a module
2402 // analysis, all that the `AAManager` can do is query for any *cached*
2403 // results from `GlobalsAA` through a readonly proxy.
2404 if (Opts.enable_global_analyses)
2405 AA.registerModuleAnalysis<GlobalsAA>();
2406
2407 // Add target-specific alias analyses.
2408 if (TM)
2409 TM->registerDefaultAliasAnalyses(AA);
2410
2411 return AA;
2412}
2413
2414bool PassBuilder::isInstrumentedPGOUse() const {
2415 return (PGOOpt && PGOOpt->Action == PGOOptions::IRUse) ||
2416 !UseCtxProfile.empty();
2417}
aarch64 falkor hwpf fix Falkor HW Prefetch Fix Late Phase
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AggressiveInstCombiner - Combine expression patterns to form expressions with fewer,...
Provides passes to inlining "always_inline" functions.
This is the interface for LLVM's primary stateless and local alias analysis.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
This file provides the interface for LLVM's Call Graph Profile pass.
This header provides classes for managing passes over SCCs of the call graph.
This file provides the interface for a simple, fast CSE pass.
This file provides a pass which clones the current module and runs the provided pass pipeline on the ...
This file provides a pass manager that only runs its passes if the provided marker analysis has been ...
Super simple passes to force specific function attrs from the commandline into the IR for debugging p...
Provides passes for computing function attributes based on interprocedural analyses.
This file provides the interface for the GVNHoist pass.
This file provides the interface for the GVNSink pass.
This file provides the interface for LLVM's Global Value Numbering pass which eliminates fully redund...
This is the interface for a simple mod/ref and alias analysis over globals.
AcceleratorCodeSelection - Identify all functions reachable from a kernel, removing those that are un...
This header defines various interfaces for pass management in LLVM.
Interfaces for passes which infer implicit function attributes from the name and signature of functio...
This file provides the primary interface to the instcombine pass.
Defines passes for running instruction simplification across chunks of IR.
This file provides the interface for LLVM's PGO Instrumentation lowering pass.
See the comments on JumpThreadingPass.
static LVOptions Options
Definition LVOptions.cpp:25
This file implements the Loop Fusion pass.
This header defines the LoopLoadEliminationPass object.
This header provides classes for managing a pipeline of passes over loops in LLVM IR.
The header file for the LowerConstantIntrinsics pass as used by the new pass manager.
The header file for the LowerExpectIntrinsic pass as used by the new pass manager.
This pass performs merges of loads and stores on both sides of a.
This file provides the interface for LLVM's Global Value Numbering pass.
This header enumerates the LLVM-provided high-level optimization levels.
This file provides the interface for IR based instrumentation passes ( (profile-gen,...
Define option tunables for PGO.
ppc ctr loops PowerPC CTR Loops Verify
static bool isThinLTOPostLink(ThinOrFullLTOPhase Phase)
static void addAnnotationRemarksPass(ModulePassManager &MPM)
static CoroConditionalWrapper buildCoroWrapper(ThinOrFullLTOPhase Phase)
static bool isFullLTOPostLink(ThinOrFullLTOPhase Phase)
static void addModuleInlinerPass(ModulePassManager &MPM, const PassesOptions &Opts, OptimizationLevel Level, ThinOrFullLTOPhase Phase)
static bool isThinLTOPreLink(ThinOrFullLTOPhase Phase)
static bool isLTOPreLink(ThinOrFullLTOPhase Phase)
static void instructionCountersPass(ModulePassManager &MPM, bool IsPreOptimization)
static bool isFullLTOPreLink(ThinOrFullLTOPhase Phase)
static bool isLTOPostLink(ThinOrFullLTOPhase Phase)
This file implements relative lookup table converter that converts lookup tables to relative lookup t...
This file provides the interface for LLVM's Scalar Replacement of Aggregates pass.
This file provides the interface for the pseudo probe implementation for AutoFDO.
This file provides the interface for the sampled PGO loader pass.
This is the interface for a metadata-based scoped no-alias analysis.
This file provides the interface for the pass responsible for both simplifying and canonicalizing the...
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
This is the interface for a metadata-based TBAA.
A manager for alias analyses.
A module pass that rewrites heap allocations to use token-enabled allocation functions based on vario...
Definition AllocToken.h:36
Inlines functions marked as "always_inline".
Argument promotion pass.
Analysis pass providing a never-invalidated alias analysis result.
Simple pass that canonicalizes aliases.
A pass that merges duplicate global constants into a single constant.
This class implements a trivial dead store elimination.
Eliminate dead arguments (and return values) from functions.
A pass that transforms external global definitions into declarations.
Pass embeds a copy of the module optimized with the provided pass pipeline into a global variable.
A pass manager to run a set of extra loop passes if the MarkerTy analysis is present.
Statistics pass for the FunctionPropertiesAnalysis results.
Pass to remove unused function declarations.
Definition GlobalDCE.h:38
Optimize globals that never have their address taken.
Definition GlobalOpt.h:25
Pass to perform split of global variables.
Definition GlobalSplit.h:26
Analysis pass providing a never-invalidated alias analysis result.
Pass to outline cold regions.
Pass to perform interprocedural constant propagation.
Definition SCCP.h:48
Run instruction simplification across each instruction in the function.
Instrumentation based profiling lowering pass.
The Instrumentor pass.
This pass performs 'jump threading', which looks at blocks that have multiple predecessors and multip...
Performs Loop Invariant Code Motion Pass.
Definition LICM.h:67
Loop unroll pass that only does full loop unrolling and peeling.
Performs Loop Idiom Recognize Pass.
Performs Loop Inst Simplify Pass.
A simple loop rotation transformation.
Performs basic CFG simplifications to assist other loop passes.
A pass that does profile-guided sinking of instructions into loops.
Definition LoopSink.h:33
A simple loop rotation transformation.
Loop unroll pass that will support both full and partial unrolling.
Strips MemProf attributes and metadata.
Merge identical functions.
The module inliner pass for the new pass manager.
Module pass, wrapping the inliner pass.
Definition Inliner.h:65
void addModulePass(T Pass)
Add a module pass that runs before the CGSCC passes.
Definition Inliner.h:81
CGSCCPassManager & getPM()
Allow adding more CGSCC passes, besides inlining.
Definition Inliner.h:78
void addLateModulePass(T Pass)
Add a module pass that runs after the CGSCC passes.
Definition Inliner.h:86
Class to hold module path string table and global value map, and encapsulate methods for operating on...
Simple pass that provides a name to every anonymous globals.
Additional 'norecurse' attribute deduction during postlink LTO phase.
OpenMP optimizations pass.
Definition OpenMPOpt.h:42
static LLVM_ABI bool isCtxIRPGOInstrEnabled()
The indirect function call promotion pass.
The instrumentation (profile-instr-gen) pass for IR based PGO.
The instrumentation (profile-instr-gen) pass for IR based PGO.
The profile annotation (profile-instr-use) pass for IR based PGO.
The profile size based optimization pass for memory intrinsics.
Pass to remove unused function declarations.
LLVM_ABI void invokeFullLinkTimeOptimizationLastEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level)
LLVM_ABI ModuleInlinerWrapperPass buildInlinerPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase)
Construct the module pipeline that performs inlining as well as the inlining-driven cleanups.
LLVM_ABI void invokeOptimizerEarlyEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level, ThinOrFullLTOPhase Phase)
LLVM_ABI ModulePassManager buildFatLTODefaultPipeline(OptimizationLevel Level, bool ThinLTO, bool EmitSummary, bool Verify=true)
Build a fat object default optimization pipeline.
LLVM_ABI void invokeVectorizerStartEPCallbacks(FunctionPassManager &FPM, OptimizationLevel Level)
LLVM_ABI AAManager buildDefaultAAPipeline()
Build the default AAManager with the default alias analysis pipeline registered.
LLVM_ABI void invokeCGSCCOptimizerLateEPCallbacks(CGSCCPassManager &CGPM, OptimizationLevel Level)
LLVM_ABI ModulePassManager buildThinLTOPreLinkDefaultPipeline(OptimizationLevel Level)
Build a pre-link, ThinLTO-targeting default optimization pipeline to a pass manager.
LLVM_ABI void addPGOInstrPassesForO0(ModulePassManager &MPM, bool RunProfileGen, bool IsCS, bool AtomicCounterUpdate, std::string ProfileFile, std::string ProfileRemappingFile)
Add PGOInstrumenation passes for O0 only.
LLVM_ABI void invokeScalarOptimizerLateEPCallbacks(FunctionPassManager &FPM, OptimizationLevel Level)
LLVM_ABI ModulePassManager buildPerModuleDefaultPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase=ThinOrFullLTOPhase::None)
Build a per-module default optimization pipeline.
LLVM_ABI void invokePipelineStartEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level)
LLVM_ABI void invokeVectorizerEndEPCallbacks(FunctionPassManager &FPM, OptimizationLevel Level)
LLVM_ABI void invokeThinLinkTimeOptimizationLastEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level)
LLVM_ABI ModulePassManager buildO0DefaultPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase=ThinOrFullLTOPhase::None)
Build an O0 pipeline with the minimal semantically required passes.
LLVM_ABI FunctionPassManager buildFunctionSimplificationPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase)
Construct the core LLVM function canonicalization and simplification pipeline.
LLVM_ABI void invokePeepholeEPCallbacks(FunctionPassManager &FPM, OptimizationLevel Level)
LLVM_ABI void invokePipelineEarlySimplificationEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level, ThinOrFullLTOPhase Phase)
LLVM_ABI void invokeLoopOptimizerEndEPCallbacks(LoopPassManager &LPM, OptimizationLevel Level)
LLVM_ABI ModulePassManager buildLTODefaultPipeline(OptimizationLevel Level, ModuleSummaryIndex *ExportSummary)
Build an LTO default optimization pipeline to a pass manager.
LLVM_ABI ModulePassManager buildModuleInlinerPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase)
Construct the module pipeline that performs inlining with module inliner pass.
LLVM_ABI ModulePassManager buildThinLTODefaultPipeline(OptimizationLevel Level, const ModuleSummaryIndex *ImportSummary)
Build a ThinLTO default optimization pipeline to a pass manager.
LLVM_ABI void invokeLateLoopOptimizationsEPCallbacks(LoopPassManager &LPM, OptimizationLevel Level)
LLVM_ABI void invokeFullLinkTimeOptimizationEarlyEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level)
LLVM_ABI ModulePassManager buildModuleSimplificationPipeline(OptimizationLevel Level, ThinOrFullLTOPhase Phase)
Construct the core LLVM module canonicalization and simplification pipeline.
LLVM_ABI ModulePassManager buildModuleOptimizationPipeline(OptimizationLevel Level, ThinOrFullLTOPhase LTOPhase)
Construct the core LLVM module optimization pipeline.
LLVM_ABI void invokeThinLinkTimeOptimizationEarlyEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level)
LLVM_ABI void invokeOptimizerLastEPCallbacks(ModulePassManager &MPM, OptimizationLevel Level, ThinOrFullLTOPhase Phase)
LLVM_ABI ModulePassManager buildLTOPreLinkDefaultPipeline(OptimizationLevel Level)
Build a pre-link, LTO-targeting default optimization pipeline to a pass manager.
LLVM_ATTRIBUTE_MINSIZE std::enable_if_t<!std::is_same_v< PassT, PassManager > > addPass(PassT &&Pass)
bool isEmpty() const
Returns if the pass manager contains any passes.
unsigned LicmMssaNoAccForPromotionCap
Tuning option to disable promotion to scalars in LICM with MemorySSA, if the number of access is too ...
Definition PassBuilder.h:79
bool SLPVectorization
Tuning option to enable/disable slp loop vectorization, set based on opt level.
Definition PassBuilder.h:57
int InlinerThreshold
Tuning option to override the default inliner threshold.
Definition PassBuilder.h:93
bool LoopFusion
Tuning option to enable/disable loop fusion. Its default value is false.
Definition PassBuilder.h:67
bool CallGraphProfile
Tuning option to enable/disable call graph profile.
Definition PassBuilder.h:83
bool MergeFunctions
Tuning option to enable/disable function merging.
Definition PassBuilder.h:90
bool ForgetAllSCEVInLoopUnroll
Tuning option to forget all SCEV loops in LoopUnroll.
Definition PassBuilder.h:71
unsigned LicmMssaOptCap
Tuning option to cap the number of calls to retrive clobbering accesses in MemorySSA,...
Definition PassBuilder.h:75
bool LoopInterleaving
Tuning option to set loop interleaving on/off, set based on opt level.
Definition PassBuilder.h:49
LLVM_ABI PipelineTuningOptions()
Constructor sets pipeline tuning defaults based on cl::opts.
bool LoopUnrolling
Tuning option to enable/disable loop unrolling. Its default value is true.
Definition PassBuilder.h:60
bool LoopInterchange
Tuning option to enable/disable loop interchange.
Definition PassBuilder.h:64
bool LoopVectorization
Tuning option to enable/disable loop vectorization, set based on opt level.
Definition PassBuilder.h:53
Reassociate commutative expressions.
Definition Reassociate.h:75
A pass to do RPO deduction and propagation of function attributes.
This pass performs function-level constant propagation and merging.
Definition SCCP.h:30
The sample profiler data loader pass.
Analysis pass providing a never-invalidated alias analysis result.
This pass transforms loops that contain branches or switches on loop- invariant conditions to have mu...
A pass to simplify and canonicalize the CFG of a function.
Definition SimplifyCFG.h:30
Analysis pass providing a never-invalidated alias analysis result.
Optimize scalar/vector interactions in IR using target cost models.
Create a verifier pass.
Definition Verifier.h:134
Interfaces for registering analysis passes, producing common pass manager configurations,...
Abstract Attribute helper functions.
Definition Attributor.h:165
@ All
Drop only llvm.assumes using type test value.
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI cl::opt< bool > EnableKnowledgeRetention
ModuleToFunctionPassAdaptor createModuleToFunctionPassAdaptor(FunctionPassT &&Pass, bool EagerlyInvalidate=false)
A function to deduce a function pass type and wrap it in the templated adaptor.
cl::opt< std::string > UseCtxProfile("use-ctx-profile", cl::init(""), cl::Hidden, cl::desc("Use the specified contextual profile file"))
LLVM_ABI bool getForgetSCEVInLoopUnroll()
Returns -forget-scev-loop-unroll.
@ O1
Optimize quickly without destroying debuggability.
@ O0
Disable as many optimizations as possible.
@ O3
Optimize for fast execution as much as possible.
@ O2
Optimize for fast execution as much as possible without triggering significant incremental compile ti...
PassManager< LazyCallGraph::SCC, CGSCCAnalysisManager, LazyCallGraph &, CGSCCUpdateResult & > CGSCCPassManager
The CGSCC pass manager.
@ CGSCC_LIGHT
@ MODULE_LIGHT
ThinOrFullLTOPhase
This enumerates the LLVM full LTO or ThinLTO optimization phases.
Definition Pass.h:77
@ FullLTOPreLink
Full LTO prelink phase.
Definition Pass.h:85
@ ThinLTOPostLink
ThinLTO postlink (backend compile) phase.
Definition Pass.h:83
@ None
No LTO/ThinLTO behavior needed.
Definition Pass.h:79
@ FullLTOPostLink
Full LTO postlink (backend compile) phase.
Definition Pass.h:87
@ ThinLTOPreLink
ThinLTO prelink (summary) phase.
Definition Pass.h:81
PassManager< Loop, LoopAnalysisManager, LoopStandardAnalysisResults &, LPMUpdater & > LoopPassManager
The Loop pass manager.
ModuleToPostOrderCGSCCPassAdaptor createModuleToPostOrderCGSCCPassAdaptor(CGSCCPassT &&Pass)
A function to deduce a function pass type and wrap it in the templated adaptor.
FunctionToLoopPassAdaptor createFunctionToLoopPassAdaptor(LoopPassT &&Pass, bool UseMemorySSA=false)
A function to deduce a loop pass type and wrap it in the templated adaptor.
CGSCCToFunctionPassAdaptor createCGSCCToFunctionPassAdaptor(FunctionPassT &&Pass, bool EagerlyInvalidate=false, bool NoRerun=false)
A function to deduce a function pass type and wrap it in the templated adaptor.
PassManager< Module > ModulePassManager
Convenience typedef for a pass manager over modules.
LLVM_ABI bool AreStatisticsEnabled()
Check if statistics are enabled.
LLVM_ABI unsigned getLicmMssaNoAccForPromotionCap()
Returns -licm-mssa-max-acc-promotion.
Definition LICM.cpp:123
cl::opt< bool > EnableMemProfContextDisambiguation
Enable MemProf context disambiguation for thin link.
PassManager< Function > FunctionPassManager
Convenience typedef for a pass manager over functions.
LLVM_ABI InlineParams getInlineParams()
Generate the parameters to tune the inline cost analysis based only on the commandline options.
LLVM_ABI unsigned getLicmMssaOptCap()
Returns -licm-mssa-optimization-cap.
Definition LICM.cpp:119
LLVM_ABI InlineParams getInlineParamsFromOptLevel(unsigned OptLevel)
Generate the parameters to tune the inline cost analysis based on command line options.
cl::opt< unsigned > MaxDevirtIterations("max-devirt-iterations", cl::ReallyHidden, cl::init(4))
LLVM_ABI bool isPGOInstrumentColdFunctionOnly()
Return the value of -pgo-instrument-cold-function-only.
A DCE pass that assumes instructions are dead until proven otherwise.
Definition ADCE.h:31
Pass to convert @llvm.global.annotations to !annotation metadata.
This pass attempts to minimize the number of assume without loosing any information.
A more lightweight version of the Attributor which only runs attribute inference but no simplificatio...
A more lightweight version of the Attributor which only runs attribute inference but no simplificatio...
Hoist/decompose integer division and remainder instructions to enable CFG improvements and better cod...
Definition DivRemPairs.h:23
A simple and fast domtree-based CSE pass.
Definition EarlyCSE.h:31
Pass which forces specific function attributes into the IR, primarily as a debugging tool.
A simple and fast domtree-based GVN pass to hoist common expressions from sibling branches.
Definition GVNHoist.h:23
Uses an "inverted" value numbering to decide the similarity of expressions and sinks similar expressi...
Definition GVNSink.h:23
A set of parameters to control various transforms performed by IPSCCP pass.
Definition SCCP.h:35
A pass which infers function attributes from the names and signatures of function declarations in a m...
Provides context on when an inline advisor is constructed in the pipeline (e.g., link phase,...
Thresholds to tune inline cost analysis.
Definition InlineCost.h:207
std::optional< int > OptSizeHintThreshold
Threshold to use for callees with inline hint, when the caller is optimized for size.
Definition InlineCost.h:216
std::optional< int > HotCallSiteThreshold
Threshold to use when the callsite is considered hot.
Definition InlineCost.h:228
int DefaultThreshold
The default threshold to start with for a callee.
Definition InlineCost.h:209
std::optional< bool > EnableDeferral
Indicate whether we should allow inline deferral.
Definition InlineCost.h:241
std::optional< int > HintThreshold
Threshold to use for callees with inline hint.
Definition InlineCost.h:212
Options for the frontend instrumentation based profiling pass.
A no-op pass template which simply forces a specific analysis result to be invalidated.
Pass to forward loads in a loop around the backedge to subsequent iterations.
A set of parameters used to control various transforms performed by the LoopUnroll pass.
The LoopVectorize Pass.
Computes function attributes in post-order over the call graph.
A utility pass template to force an analysis result to be available.