LLVM 24.0.0git
AArch64TargetMachine.cpp
Go to the documentation of this file.
1//===-- AArch64TargetMachine.cpp - Define TargetMachine for AArch64 -------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9//
10//===----------------------------------------------------------------------===//
11
13#include "AArch64.h"
14#include "AArch64AsmPrinter.h"
17#include "AArch64MacroFusion.h"
18#include "AArch64Subtarget.h"
36#include "llvm/CodeGen/Passes.h"
39#include "llvm/IR/Attributes.h"
40#include "llvm/IR/Function.h"
42#include "llvm/MC/MCAsmInfo.h"
45#include "llvm/Pass.h"
57#include <memory>
58
59using namespace llvm;
60
61static cl::opt<bool> EnableCCMP("aarch64-enable-ccmp",
62 cl::desc("Enable the CCMP formation pass"),
63 cl::init(true), cl::Hidden);
64
65static cl::opt<bool>
66 EnableCondBrTuning("aarch64-enable-cond-br-tune",
67 cl::desc("Enable the conditional branch tuning pass"),
68 cl::init(true), cl::Hidden);
69
71 "aarch64-enable-copy-propagation",
72 cl::desc("Enable the copy propagation with AArch64 copy instr"),
73 cl::init(true), cl::Hidden);
74
75static cl::opt<bool> EnableMCR("aarch64-enable-mcr",
76 cl::desc("Enable the machine combiner pass"),
77 cl::init(true), cl::Hidden);
78
79static cl::opt<bool> EnableStPairSuppress("aarch64-enable-stp-suppress",
80 cl::desc("Suppress STP for AArch64"),
81 cl::init(true), cl::Hidden);
82
84 "aarch64-enable-simd-scalar",
85 cl::desc("Enable use of AdvSIMD scalar integer instructions"),
86 cl::init(false), cl::Hidden);
87
88static cl::opt<bool>
89 EnablePromoteConstant("aarch64-enable-promote-const",
90 cl::desc("Enable the promote constant pass"),
91 cl::init(true), cl::Hidden);
92
94 "aarch64-enable-collect-loh",
95 cl::desc("Enable the pass that emits the linker optimization hints (LOH)"),
96 cl::init(true), cl::Hidden);
97
98static cl::opt<bool>
99 EnableDeadRegisterElimination("aarch64-enable-dead-defs", cl::Hidden,
100 cl::desc("Enable the pass that removes dead"
101 " definitions and replaces stores to"
102 " them with stores to the zero"
103 " register"),
104 cl::init(true));
105
107 "aarch64-enable-copyelim",
108 cl::desc("Enable the redundant copy elimination pass"), cl::init(true),
109 cl::Hidden);
110
111static cl::opt<bool> EnableLoadStoreOpt("aarch64-enable-ldst-opt",
112 cl::desc("Enable the load/store pair"
113 " optimization pass"),
114 cl::init(true), cl::Hidden);
115
117 "aarch64-enable-atomic-cfg-tidy", cl::Hidden,
118 cl::desc("Run SimplifyCFG after expanding atomic operations"
119 " to make use of cmpxchg flow-based information"),
120 cl::init(true));
121
122static cl::opt<bool>
123EnableEarlyIfConversion("aarch64-enable-early-ifcvt", cl::Hidden,
124 cl::desc("Run early if-conversion"),
125 cl::init(true));
126
127static cl::opt<bool>
128 EnableCondOpt("aarch64-enable-condopt",
129 cl::desc("Enable the condition optimizer pass"),
130 cl::init(true), cl::Hidden);
131
132static cl::opt<bool>
133 EnableGEPOpt("aarch64-enable-gep-opt", cl::Hidden,
134 cl::desc("Enable optimizations on complex GEPs"),
135 cl::init(false));
136
137static cl::opt<bool>
138 EnableSelectOpt("aarch64-select-opt", cl::Hidden,
139 cl::desc("Enable select to branch optimizations"),
140 cl::init(true));
141
142static cl::opt<bool>
143 BranchRelaxation("aarch64-enable-branch-relax", cl::Hidden, cl::init(true),
144 cl::desc("Relax out of range conditional branches"));
145
147 "aarch64-enable-compress-jump-tables", cl::Hidden, cl::init(true),
148 cl::desc("Use smallest entry possible for jump tables"));
149
150// FIXME: Unify control over GlobalMerge.
152 EnableGlobalMerge("aarch64-enable-global-merge", cl::Hidden,
153 cl::desc("Enable the global merge pass"));
154
155static cl::opt<bool>
156 EnableLoopDataPrefetch("aarch64-enable-loop-data-prefetch", cl::Hidden,
157 cl::desc("Enable the loop data prefetch pass"),
158 cl::init(true));
159
161 "aarch64-enable-global-isel-at-O", cl::Hidden,
162 cl::desc("Enable GlobalISel at or below an opt level (-1 to disable)"),
163 cl::init(0));
164
165static cl::opt<bool>
166 EnableSMEPeepholeOpt("enable-aarch64-sme-peephole-opt", cl::init(true),
168 cl::desc("Perform SME peephole optimization"));
169
170static cl::opt<bool> EnableFalkorHWPFFix("aarch64-enable-falkor-hwpf-fix",
171 cl::init(true), cl::Hidden);
172
173static cl::opt<bool>
174 EnableBranchTargets("aarch64-enable-branch-targets", cl::Hidden,
175 cl::desc("Enable the AArch64 branch target pass"),
176 cl::init(true));
177
179 "aarch64-sve-vector-bits-max",
180 cl::desc("Assume SVE vector registers are at most this big, "
181 "with zero meaning no maximum size is assumed."),
182 cl::init(0), cl::Hidden);
183
185 "aarch64-sve-vector-bits-min",
186 cl::desc("Assume SVE vector registers are at least this big, "
187 "with zero meaning no minimum size is assumed."),
188 cl::init(0), cl::Hidden);
189
191 "force-streaming",
192 cl::desc("Force the use of streaming code for all functions"),
193 cl::init(false), cl::Hidden);
194
196 "force-streaming-compatible",
197 cl::desc("Force the use of streaming-compatible code for all functions"),
198 cl::init(false), cl::Hidden);
199
201
203 "aarch64-enable-gisel-ldst-prelegal",
204 cl::desc("Enable GlobalISel's pre-legalizer load/store optimization pass"),
205 cl::init(true), cl::Hidden);
206
208 "aarch64-enable-gisel-ldst-postlegal",
209 cl::desc("Enable GlobalISel's post-legalizer load/store optimization pass"),
210 cl::init(false), cl::Hidden);
211
212static cl::opt<bool>
213 EnableSinkFold("aarch64-enable-sink-fold",
214 cl::desc("Enable sinking and folding of instruction copies"),
215 cl::init(true), cl::Hidden);
216
217static cl::opt<bool>
218 EnableMachinePipeliner("aarch64-enable-pipeliner",
219 cl::desc("Enable Machine Pipeliner for AArch64"),
220 cl::init(false), cl::Hidden);
221
223 "aarch64-srlt-mitigate-sr2r",
224 cl::desc("Enable SUBREG_TO_REG mitigation by adding 'implicit-def' for "
225 "super-regs when using Subreg Liveness Tracking"),
226 cl::init(true), cl::Hidden);
227
229 "aarch64-enable-sve-shuffle-opts",
230 cl::desc("Enable pattern matching of shuffles that could make use of SVE "
231 "instructions like tbl or the bottom/top variants"),
232 cl::init(true), cl::Hidden);
233
235 "aarch64-enable-predicate-as-counter-loop-rewrites",
236 cl::desc("Enable rewriting loops with wide loop-carried masks to use "
237 "predicate-as-counter"),
238 cl::init(false), cl::Hidden);
239
242 // Register the target.
248 auto &PR = *PassRegistry::getPassRegistry();
294}
295
297 const bool GlobalISelFlag = getCGPassBuilderOption().EnableGlobalISelOption ==
299
301 (static_cast<unsigned>(getOptLevel()) >
302 static_cast<unsigned>(EnableGlobalISelAtO) &&
303 !GlobalISelFlag);
304}
305
307 SubtargetMap.clear();
308 LastSubtarget = nullptr;
309}
310
311//===----------------------------------------------------------------------===//
312// AArch64 Lowering public interface.
313//===----------------------------------------------------------------------===//
314static std::unique_ptr<TargetLoweringObjectFile> createTLOF(const Triple &TT) {
315 if (TT.isOSBinFormatMachO())
316 return std::make_unique<AArch64_MachoTargetObjectFile>();
317 if (TT.isOSBinFormatCOFF())
318 return std::make_unique<AArch64_COFFTargetObjectFile>();
319
320 return std::make_unique<AArch64_ELFTargetObjectFile>();
321}
322
324 if (CPU.empty() && TT.isArm64e())
325 return "apple-a12";
326 return CPU;
327}
328
330 std::optional<Reloc::Model> RM) {
331 // AArch64 Darwin and Windows are always PIC.
332 if (TT.isOSDarwin() || TT.isOSWindows())
333 return Reloc::PIC_;
334 // On ELF platforms the default static relocation model has a smart enough
335 // linker to cope with referencing external symbols defined in a shared
336 // library. Hence DynamicNoPIC doesn't need to be promoted to PIC.
337 if (!RM || *RM == Reloc::DynamicNoPIC)
338 return Reloc::Static;
339 return *RM;
340}
341
342static CodeModel::Model
344 std::optional<CodeModel::Model> CM, bool JIT) {
345 if (CM) {
346 if (*CM != CodeModel::Small && *CM != CodeModel::Tiny &&
347 *CM != CodeModel::Large) {
349 "Only small, tiny and large code models are allowed on AArch64");
350 } else if (*CM == CodeModel::Tiny && !TT.isOSBinFormatELF()) {
351 report_fatal_error("tiny code model is only supported on ELF");
352 }
353 return *CM;
354 }
355 // The default MCJIT memory managers make no guarantees about where they can
356 // find an executable page; JITed code needs to be able to refer to globals
357 // no matter how far away they are.
358 // We should set the CodeModel::Small for Windows ARM64 in JIT mode,
359 // since with large code model LLVM generating 4 MOV instructions, and
360 // Windows doesn't support relocating these long branch (4 MOVs).
361 if (JIT && !TT.isOSWindows())
362 return CodeModel::Large;
363 return CodeModel::Small;
364}
365
366/// Create an AArch64 architecture model.
367///
369 StringRef CPU, StringRef FS,
370 const TargetOptions &Options,
371 std::optional<Reloc::Model> RM,
372 std::optional<CodeModel::Model> CM,
373 CodeGenOptLevel OL, bool JIT,
374 bool LittleEndian)
377 getEffectiveAArch64CodeModel(TT, CM, JIT), OL),
378 TLOF(createTLOF(getTargetTriple())), isLittle(LittleEndian) {
379 initAsmInfo();
380
381 if (TT.isOSBinFormatMachO()) {
382 this->Options.TrapUnreachable = true;
383 this->Options.NoTrapAfterNoreturn = true;
384 }
385
386 if (getMCAsmInfo().usesWindowsCFI()) {
387 // Unwinding can get confused if the last instruction in an
388 // exception-handling region (function, funclet, try block, etc.)
389 // is a call.
390 //
391 // FIXME: We could elide the trap if the next instruction would be in
392 // the same region anyway.
393 this->Options.TrapUnreachable = true;
394 }
395
396 if (this->Options.TLSSize == 0) // default
397 this->Options.TLSSize = 24;
398 if ((getCodeModel() == CodeModel::Small ||
400 this->Options.TLSSize > 32)
401 // for the small (and kernel) code model, the maximum TLS size is 4GiB
402 this->Options.TLSSize = 32;
403 else if (getCodeModel() == CodeModel::Tiny && this->Options.TLSSize > 24)
404 // for the tiny code model, the maximum TLS size is 1MiB (< 16MiB)
405 this->Options.TLSSize = 24;
406
407 const bool TargetSupportsGISel =
408 TT.getArch() != Triple::aarch64_32 &&
409 TT.getEnvironment() != Triple::GNUILP32 &&
410 !(getCodeModel() == CodeModel::Large && TT.isOSBinFormatMachO());
411
412 const bool GlobalISelFlag = getCGPassBuilderOption().EnableGlobalISelOption ==
414
415 // Enable GlobalISel at or below EnableGlobalISelAt0, unless this is
416 // MachO/CodeModel::Large, which GlobalISel does not support.
417 if (TargetSupportsGISel && EnableGlobalISelAtO != -1 &&
418 (static_cast<int>(getOptLevel()) <= EnableGlobalISelAtO ||
419 (!GlobalISelFlag && !Options.EnableGlobalISel))) {
420 setGlobalISel(true);
422 }
423
425
426 // AArch64 supports the MachineOutliner.
427 setMachineOutliner(true);
428
429 // AArch64 supports default outlining behaviour.
431
432 // AArch64 supports the debug entry values.
434
435 // AArch64 supports fixing up the DWARF unwind information.
436 if (!getMCAsmInfo().usesWindowsCFI())
437 setCFIFixup(true);
438}
439
443
445
446const AArch64Subtarget *
448 // Constructing the subtarget key is not cheap, avoid rebuilding it for
449 // repeated queries with the same function attributes.
450 AttributeSet FnAttrs = F.getAttributes().getFnAttrs();
451 if (LastSubtarget && LastSubtargetAttrs == FnAttrs)
452 return LastSubtarget;
453
454 Attribute CPUAttr = F.getFnAttribute("target-cpu");
455 Attribute TuneAttr = F.getFnAttribute("tune-cpu");
456 Attribute FSAttr = F.getFnAttribute("target-features");
457
458 StringRef CPU = CPUAttr.isValid() ? CPUAttr.getValueAsString() : TargetCPU;
459 StringRef TuneCPU = TuneAttr.isValid() ? TuneAttr.getValueAsString() : CPU;
460 StringRef FS = FSAttr.isValid() ? FSAttr.getValueAsString() : TargetFS;
461 bool HasMinSize = F.hasMinSize();
462
463 bool IsStreaming = ForceStreaming ||
464 F.hasFnAttribute("aarch64_pstate_sm_enabled") ||
465 F.hasFnAttribute("aarch64_pstate_sm_body");
466 bool IsStreamingCompatible = ForceStreamingCompatible ||
467 F.hasFnAttribute("aarch64_pstate_sm_compatible");
468
469 unsigned MinSVEVectorSize = 0;
470 unsigned MaxSVEVectorSize = 0;
471 if (F.hasFnAttribute(Attribute::VScaleRange)) {
472 ConstantRange CR = getVScaleRange(&F, 64);
473 MinSVEVectorSize = CR.getUnsignedMin().getZExtValue() * 128;
474 MaxSVEVectorSize = CR.getUnsignedMax().getZExtValue() * 128;
475 } else {
476 MinSVEVectorSize = SVEVectorBitsMinOpt;
477 MaxSVEVectorSize = SVEVectorBitsMaxOpt;
478 }
479
480 assert(MinSVEVectorSize % 128 == 0 &&
481 "SVE requires vector length in multiples of 128!");
482 assert(MaxSVEVectorSize % 128 == 0 &&
483 "SVE requires vector length in multiples of 128!");
484 assert((MaxSVEVectorSize >= MinSVEVectorSize || MaxSVEVectorSize == 0) &&
485 "Minimum SVE vector size should not be larger than its maximum!");
486
487 // Sanitize user input in case of no asserts
488 if (MaxSVEVectorSize != 0) {
489 MinSVEVectorSize = std::min(MinSVEVectorSize, MaxSVEVectorSize);
490 MaxSVEVectorSize = std::max(MinSVEVectorSize, MaxSVEVectorSize);
491 }
492
494 // This lookup is hot during repeated TTI queries, so build the key directly
495 // instead of formatting through raw_svector_ostream.
496 Key += "SVEMin";
497 Key += utostr(MinSVEVectorSize);
498 Key += "SVEMax";
499 Key += utostr(MaxSVEVectorSize);
500 Key += "IsStreaming=";
501 Key += utostr(IsStreaming);
502 Key += "IsStreamingCompatible=";
503 Key += utostr(IsStreamingCompatible);
504 Key += CPU;
505 Key += TuneCPU;
506 Key += FS;
507 Key += "HasMinSize=";
508 Key += utostr(HasMinSize);
509
510 auto &I = SubtargetMap[Key];
511 if (!I) {
512 I = std::make_unique<AArch64Subtarget>(
513 TargetTriple, CPU, TuneCPU, FS, *this, isLittle, MinSVEVectorSize,
514 MaxSVEVectorSize, IsStreaming, IsStreamingCompatible, HasMinSize,
516 }
517
518 if (IsStreaming && !I->hasSME())
519 reportFatalUsageError("streaming SVE functions require SME");
520
521 LastSubtargetAttrs = FnAttrs;
522 LastSubtarget = I.get();
523 return LastSubtarget;
524}
525
526// Encourage placing FORM_TRANSPOSED_REG immediately before the instruction that
527// uses/consumes it. This ensures its def has a short live range, which means
528// we're more likely to allocate registers its operands first (which works best
529// for the hints in AArch64RegisterInfo::getRegAllocationHints).
531 const TargetInstrInfo &TII, const TargetSubtargetInfo &TSI,
532 const MachineInstr *FirstMI, const MachineInstr &SecondMI,
533 const SDep *Dep) {
534 if (isNonDataDep(Dep))
535 return false;
536 return !FirstMI ||
537 FirstMI->getOpcode() == AArch64::FORM_TRANSPOSED_REG_TUPLE_X2_PSEUDO ||
538 FirstMI->getOpcode() == AArch64::FORM_TRANSPOSED_REG_TUPLE_X4_PSEUDO;
539}
540
543 const AArch64Subtarget &ST = C->MF->getSubtarget<AArch64Subtarget>();
545 DAG->addMutation(createLoadClusterDAGMutation(DAG->TII));
546 DAG->addMutation(createStoreClusterDAGMutation(DAG->TII));
547 if (ST.hasFusion())
548 DAG->addMutation(createAArch64MacroFusionDAGMutation());
549 if (ST.hasSME() && ST.isStreaming())
550 DAG->addMutation(createMacroFusionDAGMutation(
552 return DAG;
553}
554
557 const AArch64Subtarget &ST = C->MF->getSubtarget<AArch64Subtarget>();
559 if (ST.hasFusion()) {
560 // Run the Macro Fusion after RA again since literals are expanded from
561 // pseudos then (v. addPreSched2()).
562 DAG->addMutation(createAArch64MacroFusionDAGMutation());
563 return DAG;
564 }
565
566 return DAG;
567}
568
570 const SmallPtrSetImpl<MachineInstr *> &MIs) const {
571 if (MIs.empty())
572 return 0;
573 auto *MI = *MIs.begin();
574 auto *FuncInfo = MI->getMF()->getInfo<AArch64FunctionInfo>();
575 return FuncInfo->clearLinkerOptimizationHints(MIs);
576}
577
578void AArch64leTargetMachine::anchor() { }
579
581 const Target &T, const Triple &TT, StringRef CPU, StringRef FS,
582 const TargetOptions &Options, std::optional<Reloc::Model> RM,
583 std::optional<CodeModel::Model> CM, CodeGenOptLevel OL, bool JIT)
584 : AArch64TargetMachine(T, TT, CPU, FS, Options, RM, CM, OL, JIT, true) {}
585
586void AArch64beTargetMachine::anchor() { }
587
589 const Target &T, const Triple &TT, StringRef CPU, StringRef FS,
590 const TargetOptions &Options, std::optional<Reloc::Model> RM,
591 std::optional<CodeModel::Model> CM, CodeGenOptLevel OL, bool JIT)
592 : AArch64TargetMachine(T, TT, CPU, FS, Options, RM, CM, OL, JIT, false) {}
593
594namespace {
595
596/// AArch64 Code Generator Pass Configuration Options.
597class AArch64PassConfig : public TargetPassConfig {
598public:
599 AArch64PassConfig(AArch64TargetMachine &TM, PassManagerBase &PM)
600 : TargetPassConfig(TM, PM) {
602 substitutePass(&PostRASchedulerID, &PostMachineSchedulerID);
603 setEnableSinkAndFold(EnableSinkFold);
604 }
605
606 AArch64TargetMachine &getAArch64TargetMachine() const {
608 }
609
610 void addIRPasses() override;
611 bool addPreISel() override;
612 void addCodeGenPrepare() override;
613 bool addInstSelector() override;
614 bool addIRTranslator() override;
615 void addPreLegalizeMachineIR() override;
616 bool addLegalizeMachineIR() override;
617 void addPreRegBankSelect() override;
618 bool addRegBankSelect() override;
619 bool addGlobalInstructionSelect() override;
620 void addMachineSSAOptimization() override;
621 bool addILPOpts() override;
622 void addPreRegAlloc() override;
623 void addPostRewrite() override;
624 void addPostRegAlloc() override;
625 void addPreSched2() override;
626 void addPreEmitPass() override;
627 void addPostBBSections() override;
628 void addPreEmitPass2() override;
629 bool addRegAssignAndRewriteOptimized() override;
630
631 std::unique_ptr<CSEConfigBase> getCSEConfig() const override;
632};
633
634} // end anonymous namespace
635
637#define GET_PASS_REGISTRY "AArch64PassRegistry.def"
639
640 PB.registerLateLoopOptimizationsEPCallback(
641 [=](LoopPassManager &LPM, OptimizationLevel Level) {
642 if (Level != OptimizationLevel::O0)
643 LPM.addPass(LoopIdiomVectorizePass());
644 });
645 if (getTargetTriple().isOSWindows())
646 PB.registerPipelineEarlySimplificationEPCallback(
649 });
650}
651
654 return TargetTransformInfo(std::make_unique<AArch64TTIImpl>(this, F));
655}
656
658 return new AArch64PassConfig(*this, PM);
659}
660
661std::unique_ptr<CSEConfigBase> AArch64PassConfig::getCSEConfig() const {
662 return getStandardCSEConfigForOpt(TM->getOptLevel());
663}
664
665void AArch64PassConfig::addIRPasses() {
666 // Always expand atomic operations, we don't deal with atomicrmw or cmpxchg
667 // ourselves.
669
670 if (getOptLevel() >= CodeGenOptLevel::Default &&
673
674 // Cmpxchg instructions are often used with a subsequent comparison to
675 // determine whether it succeeded. We can exploit existing control-flow in
676 // ldrex/strex loops to simplify this, but it needs tidying up.
677 if (TM->getOptLevel() != CodeGenOptLevel::None && EnableAtomicTidy)
679 .forwardSwitchCondToPhi(true)
680 .convertSwitchRangeToICmp(true)
681 .convertSwitchToLookupTable(true)
682 .needCanonicalLoops(false)
683 .hoistCommonInsts(true)
684 .sinkCommonInsts(true)));
685
686 // Run LoopDataPrefetch
687 //
688 // Run this before LSR to remove the multiplies involved in computing the
689 // pointer values N iterations ahead.
690 if (TM->getOptLevel() != CodeGenOptLevel::None) {
695 }
696
697 if (EnableGEPOpt) {
698 // Call SeparateConstOffsetFromGEP pass to extract constants within indices
699 // and lower a GEP with multiple indices to either arithmetic operations or
700 // multiple GEPs with single index.
702 // Call EarlyCSE pass to find and remove subexpressions in the lowered
703 // result.
704 addPass(createEarlyCSEPass());
705 // Do loop invariant code motion in case part of the lowered result is
706 // invariant.
707 addPass(createLICMPass());
708 }
709
711
712 if (getOptLevel() == CodeGenOptLevel::Aggressive && EnableSelectOpt)
713 addPass(createSelectOptimizePass());
714
716 /*IsOptNone=*/TM->getOptLevel() == CodeGenOptLevel::None));
717
718 // Try to use tbl in place of other shuffling operations if doing so would
719 // reduce the total number of instructions. Shuffle masks for big endian may
720 // be different, so require a little endian target.
721 if (getOptLevel() >= CodeGenOptLevel::Default && EnableSVEShuffleOpt &&
722 TM->getTargetTriple().isLittleEndian())
723 addPass(createSVEShuffleOptsPass());
724
725 // Match complex arithmetic patterns
726 if (TM->getOptLevel() >= CodeGenOptLevel::Default)
728
729 // Match interleaved memory accesses to ldN/stN intrinsics.
730 if (TM->getOptLevel() != CodeGenOptLevel::None) {
733 }
734
735 // Add Control Flow Guard checks.
736 if (TM->getTargetTriple().isOSWindows()) {
737 if (TM->getTargetTriple().isWindowsArm64EC())
739 else
740 addPass(createCFGuardPass());
741 }
742
743 if (TM->Options.JMCInstrument)
744 addPass(createJMCInstrumenterPass());
745}
746
747// Pass Pipeline Configuration
748bool AArch64PassConfig::addPreISel() {
749 // Run promote constant before global merge, so that the promoted constants
750 // get a chance to be merged
751 if (TM->getOptLevel() != CodeGenOptLevel::None && EnablePromoteConstant)
753 // FIXME: On AArch64, this depends on the type.
754 // Basically, the addressable offsets are up to 4095 * Ty.getSizeInBytes().
755 // and the offset has to be a multiple of the related size in bytes.
756 if ((TM->getOptLevel() != CodeGenOptLevel::None &&
759 bool OnlyOptimizeForSize =
760 (TM->getOptLevel() < CodeGenOptLevel::Aggressive) &&
762
763 // Merging of extern globals is enabled by default on non-Mach-O as we
764 // expect it to be generally either beneficial or harmless. On Mach-O it
765 // is disabled as we emit the .subsections_via_symbols directive which
766 // means that merging extern globals is not safe.
767 bool MergeExternalByDefault = !TM->getTargetTriple().isOSBinFormatMachO();
768 addPass(createGlobalMergePass(TM, 4095, OnlyOptimizeForSize,
769 MergeExternalByDefault));
770 }
771
772 return false;
773}
774
775void AArch64PassConfig::addCodeGenPrepare() {
776 if (getOptLevel() != CodeGenOptLevel::None)
779}
780
781bool AArch64PassConfig::addInstSelector() {
782 addPass(createAArch64ISelDag(getAArch64TargetMachine(), getOptLevel()));
783 return false;
784}
785
786bool AArch64PassConfig::addIRTranslator() {
787 addPass(new IRTranslatorLegacy(getOptLevel()));
788 return false;
789}
790
791void AArch64PassConfig::addPreLegalizeMachineIR() {
792 if (getAArch64TargetMachine().isGlobalISelOptNone()) {
794 addPass(new LocalizerLegacy());
795 } else {
797 addPass(new LocalizerLegacy());
799 addPass(new LoadStoreOptLegacy());
800 }
801}
802
803bool AArch64PassConfig::addLegalizeMachineIR() {
804 addPass(new LegalizerLegacy());
805 return false;
806}
807
808void AArch64PassConfig::addPreRegBankSelect() {
809 const bool IsGlobalISelOptNone =
810 getAArch64TargetMachine().isGlobalISelOptNone();
811 if (!IsGlobalISelOptNone) {
812 addPass(createAArch64PostLegalizerCombinerLegacy(IsGlobalISelOptNone));
814 addPass(new LoadStoreOptLegacy());
815 }
817}
818
819bool AArch64PassConfig::addRegBankSelect() {
820 addPass(new RegBankSelectLegacy());
821 return false;
822}
823
824bool AArch64PassConfig::addGlobalInstructionSelect() {
825 addPass(new InstructionSelectLegacy(getOptLevel()));
826 if (!getAArch64TargetMachine().isGlobalISelOptNone())
828
829 return false;
830}
831
832void AArch64PassConfig::addMachineSSAOptimization() {
833 // For ELF, cleanup any local-dynamic TLS accesses
834 // (i.e. combine as many references to _TLS_MODULE_BASE_ as possible.
835 if (TM->getTargetTriple().isOSBinFormatELF() &&
836 getOptLevel() != CodeGenOptLevel::None)
838
839 if (TM->getOptLevel() != CodeGenOptLevel::None)
840 addPass(createMachineSMEABIPass(TM->getOptLevel()));
841
842 if (TM->getOptLevel() != CodeGenOptLevel::None && EnableSMEPeepholeOpt)
843 addPass(createSMEPeepholeOptPass());
844
845 // Run default MachineSSAOptimization first.
847
848 if (TM->getOptLevel() != CodeGenOptLevel::None) {
851 }
852}
853
854bool AArch64PassConfig::addILPOpts() {
855 if (EnableCondOpt)
857 if (EnableCCMP)
859 if (EnableMCR)
860 addPass(&MachineCombinerID);
862 addPass(createAArch64CondBrTuning());
864 addPass(&EarlyIfConverterLegacyID);
868 if (TM->getOptLevel() != CodeGenOptLevel::None)
870 return true;
871}
872
873void AArch64PassConfig::addPreRegAlloc() {
874 if (TM->getOptLevel() == CodeGenOptLevel::None)
876
877 // Change dead register definitions to refer to the zero register.
878 if (TM->getOptLevel() != CodeGenOptLevel::None &&
881
882 // Use AdvSIMD scalar instructions whenever profitable.
883 if (TM->getOptLevel() != CodeGenOptLevel::None && EnableAdvSIMDScalar) {
885 // The AdvSIMD pass may produce copies that can be rewritten to
886 // be register coalescer friendly.
888 }
889 if (TM->getOptLevel() != CodeGenOptLevel::None && EnableMachinePipeliner)
890 addPass(&MachinePipelinerID);
891}
892
893void AArch64PassConfig::addPostRewrite() {
896}
897
898void AArch64PassConfig::addPostRegAlloc() {
899 // Remove redundant copy instructions.
900 if (TM->getOptLevel() != CodeGenOptLevel::None &&
903
904 if (TM->getOptLevel() != CodeGenOptLevel::None && usingDefaultRegAlloc())
905 // Improve performance for some FP/SIMD code for A57.
907}
908
909void AArch64PassConfig::addPreSched2() {
910 // Lower homogeneous frame instructions
913 // Expand some pseudo instructions to allow proper scheduling.
915 // Use load/store pair instructions when possible.
916 if (TM->getOptLevel() != CodeGenOptLevel::None) {
919 }
920 // Emit KCFI checks for indirect calls.
921 addPass(createKCFIPass());
922
923 // The AArch64SpeculationHardeningPass destroys dominator tree and natural
924 // loop info, which is needed for the FalkorHWPFFixPass and also later on.
925 // Therefore, run the AArch64SpeculationHardeningPass before the
926 // FalkorHWPFFixPass to avoid recomputing dominator tree and natural loop
927 // info.
929
930 if (TM->getOptLevel() != CodeGenOptLevel::None) {
932 addPass(createFalkorHWPFFixPass());
933 }
934}
935
936void AArch64PassConfig::addPreEmitPass() {
937 // Machine Block Placement might have created new opportunities when run
938 // at O3, where the Tail Duplication Threshold is set to 4 instructions.
939 // Run the load/store optimizer once more.
940 if (TM->getOptLevel() >= CodeGenOptLevel::Aggressive && EnableLoadStoreOpt)
942
943 if (TM->getOptLevel() >= CodeGenOptLevel::Aggressive &&
946 if (TM->getOptLevel() != CodeGenOptLevel::None)
948
950
951 if (TM->getTargetTriple().isOSWindows()) {
952 // Identify valid longjmp targets for Windows Control Flow Guard.
953 addPass(createCFGuardLongjmpPass());
954 // Identify valid eh continuation targets for Windows EHCont Guard.
956 }
957
958 if (TM->getOptLevel() != CodeGenOptLevel::None && EnableCollectLOH &&
959 TM->getTargetTriple().isOSBinFormatMachO())
961
962 // Apply code layout optimizations. Run late so detection reflects the
963 // final MI stream.
964 if (getOptLevel() != CodeGenOptLevel::None)
966}
967
968void AArch64PassConfig::addPostBBSections() {
973 // Relax conditional branch instructions if they're otherwise out of
974 // range of their destination.
976 addPass(&BranchRelaxationPassID);
977
978 if (TM->getOptLevel() != CodeGenOptLevel::None && EnableCompressJumpTables)
980}
981
982void AArch64PassConfig::addPreEmitPass2() {
983 // Insert pseudo probe annotation for callsite profiling
984 addPass(createPseudoProbeInserter());
985
986 // SVE bundles move prefixes with destructive operations. BLR_RVMARKER pseudo
987 // instructions are lowered to bundles as well.
988 addPass(createUnpackMachineBundlesLegacy(nullptr));
989}
990
991bool AArch64PassConfig::addRegAssignAndRewriteOptimized() {
994}
995
1002
1007
1010 const auto *MFI = MF.getInfo<AArch64FunctionInfo>();
1011 return new yaml::AArch64FunctionInfo(*MFI);
1012}
1013
1016 SMDiagnostic &Error, SMRange &SourceRange) const {
1017 const auto &YamlMFI = static_cast<const yaml::AArch64FunctionInfo &>(MFI);
1018 MachineFunction &MF = PFS.MF;
1019 MF.getInfo<AArch64FunctionInfo>()->initializeBaseYamlFields(YamlMFI);
1020 return false;
1021}
cl::opt< bool > EnableHomogeneousPrologEpilog("homogeneous-prolog-epilog", cl::Hidden, cl::desc("Emit homogeneous prologue and epilogue for the size " "optimization (default = off)"))
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static cl::opt< bool > EnableBranchTargets("aarch64-enable-branch-targets", cl::Hidden, cl::desc("Enable the AArch64 branch target pass"), cl::init(true))
static cl::opt< bool > EnableAArch64CopyPropagation("aarch64-enable-copy-propagation", cl::desc("Enable the copy propagation with AArch64 copy instr"), cl::init(true), cl::Hidden)
static cl::opt< bool > BranchRelaxation("aarch64-enable-branch-relax", cl::Hidden, cl::init(true), cl::desc("Relax out of range conditional branches"))
static cl::opt< bool > EnablePromoteConstant("aarch64-enable-promote-const", cl::desc("Enable the promote constant pass"), cl::init(true), cl::Hidden)
static cl::opt< bool > EnableCondBrTuning("aarch64-enable-cond-br-tune", cl::desc("Enable the conditional branch tuning pass"), cl::init(true), cl::Hidden)
static cl::opt< bool > EnablePredicateAsCounterLoopRewrites("aarch64-enable-predicate-as-counter-loop-rewrites", cl::desc("Enable rewriting loops with wide loop-carried masks to use " "predicate-as-counter"), cl::init(false), cl::Hidden)
static cl::opt< bool > EnableSinkFold("aarch64-enable-sink-fold", cl::desc("Enable sinking and folding of instruction copies"), cl::init(true), cl::Hidden)
static cl::opt< bool > EnableDeadRegisterElimination("aarch64-enable-dead-defs", cl::Hidden, cl::desc("Enable the pass that removes dead" " definitions and replaces stores to" " them with stores to the zero" " register"), cl::init(true))
static cl::opt< bool > EnableGEPOpt("aarch64-enable-gep-opt", cl::Hidden, cl::desc("Enable optimizations on complex GEPs"), cl::init(false))
static bool scheduleFormTransposedTupleAdjacentToUsers(const TargetInstrInfo &TII, const TargetSubtargetInfo &TSI, const MachineInstr *FirstMI, const MachineInstr &SecondMI, const SDep *Dep)
static cl::opt< bool > EnableSelectOpt("aarch64-select-opt", cl::Hidden, cl::desc("Enable select to branch optimizations"), cl::init(true))
static cl::opt< bool > EnableLoadStoreOpt("aarch64-enable-ldst-opt", cl::desc("Enable the load/store pair" " optimization pass"), cl::init(true), cl::Hidden)
static cl::opt< bool > EnableGISelLoadStoreOptPostLegal("aarch64-enable-gisel-ldst-postlegal", cl::desc("Enable GlobalISel's post-legalizer load/store optimization pass"), cl::init(false), cl::Hidden)
static StringRef computeDefaultCPU(const Triple &TT, StringRef CPU)
static cl::opt< unsigned > SVEVectorBitsMinOpt("aarch64-sve-vector-bits-min", cl::desc("Assume SVE vector registers are at least this big, " "with zero meaning no minimum size is assumed."), cl::init(0), cl::Hidden)
static cl::opt< bool > EnableMCR("aarch64-enable-mcr", cl::desc("Enable the machine combiner pass"), cl::init(true), cl::Hidden)
static cl::opt< cl::boolOrDefault > EnableGlobalMerge("aarch64-enable-global-merge", cl::Hidden, cl::desc("Enable the global merge pass"))
static cl::opt< bool > EnableStPairSuppress("aarch64-enable-stp-suppress", cl::desc("Suppress STP for AArch64"), cl::init(true), cl::Hidden)
static CodeModel::Model getEffectiveAArch64CodeModel(const Triple &TT, std::optional< CodeModel::Model > CM, bool JIT)
static cl::opt< bool > EnableCondOpt("aarch64-enable-condopt", cl::desc("Enable the condition optimizer pass"), cl::init(true), cl::Hidden)
static cl::opt< bool > ForceStreaming("force-streaming", cl::desc("Force the use of streaming code for all functions"), cl::init(false), cl::Hidden)
static cl::opt< bool > EnableCollectLOH("aarch64-enable-collect-loh", cl::desc("Enable the pass that emits the linker optimization hints (LOH)"), cl::init(true), cl::Hidden)
static cl::opt< bool > EnableGISelLoadStoreOptPreLegal("aarch64-enable-gisel-ldst-prelegal", cl::desc("Enable GlobalISel's pre-legalizer load/store optimization pass"), cl::init(true), cl::Hidden)
static cl::opt< bool > EnableRedundantCopyElimination("aarch64-enable-copyelim", cl::desc("Enable the redundant copy elimination pass"), cl::init(true), cl::Hidden)
static cl::opt< bool > EnableAtomicTidy("aarch64-enable-atomic-cfg-tidy", cl::Hidden, cl::desc("Run SimplifyCFG after expanding atomic operations" " to make use of cmpxchg flow-based information"), cl::init(true))
static cl::opt< bool > EnableAdvSIMDScalar("aarch64-enable-simd-scalar", cl::desc("Enable use of AdvSIMD scalar integer instructions"), cl::init(false), cl::Hidden)
static cl::opt< int > EnableGlobalISelAtO("aarch64-enable-global-isel-at-O", cl::Hidden, cl::desc("Enable GlobalISel at or below an opt level (-1 to disable)"), cl::init(0))
static cl::opt< bool > EnableLoopDataPrefetch("aarch64-enable-loop-data-prefetch", cl::Hidden, cl::desc("Enable the loop data prefetch pass"), cl::init(true))
static cl::opt< bool > EnableSMEPeepholeOpt("enable-aarch64-sme-peephole-opt", cl::init(true), cl::Hidden, cl::desc("Perform SME peephole optimization"))
static cl::opt< bool > EnableEarlyIfConversion("aarch64-enable-early-ifcvt", cl::Hidden, cl::desc("Run early if-conversion"), cl::init(true))
static cl::opt< bool > EnableSRLTSubregToRegMitigation("aarch64-srlt-mitigate-sr2r", cl::desc("Enable SUBREG_TO_REG mitigation by adding 'implicit-def' for " "super-regs when using Subreg Liveness Tracking"), cl::init(true), cl::Hidden)
static cl::opt< bool > EnableMachinePipeliner("aarch64-enable-pipeliner", cl::desc("Enable Machine Pipeliner for AArch64"), cl::init(false), cl::Hidden)
static cl::opt< bool > EnableFalkorHWPFFix("aarch64-enable-falkor-hwpf-fix", cl::init(true), cl::Hidden)
static cl::opt< bool > EnableSVEShuffleOpt("aarch64-enable-sve-shuffle-opts", cl::desc("Enable pattern matching of shuffles that could make use of SVE " "instructions like tbl or the bottom/top variants"), cl::init(true), cl::Hidden)
static cl::opt< unsigned > SVEVectorBitsMaxOpt("aarch64-sve-vector-bits-max", cl::desc("Assume SVE vector registers are at most this big, " "with zero meaning no maximum size is assumed."), cl::init(0), cl::Hidden)
static cl::opt< bool > ForceStreamingCompatible("force-streaming-compatible", cl::desc("Force the use of streaming-compatible code for all functions"), cl::init(false), cl::Hidden)
static std::unique_ptr< TargetLoweringObjectFile > createTLOF(const Triple &TT)
static cl::opt< bool > EnableCompressJumpTables("aarch64-enable-compress-jump-tables", cl::Hidden, cl::init(true), cl::desc("Use smallest entry possible for jump tables"))
LLVM_ABI LLVM_EXTERNAL_VISIBILITY void LLVMInitializeAArch64Target()
static cl::opt< bool > EnableCCMP("aarch64-enable-ccmp", cl::desc("Enable the CCMP formation pass"), cl::init(true), cl::Hidden)
This file a TargetTransformInfoImplBase conforming object specific to the AArch64 target machine.
static Reloc::Model getEffectiveRelocModel()
This file contains the simple types necessary to represent the attributes associated with functions a...
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
Provides analysis for continuously CSEing during GISel passes.
#define LLVM_ABI
Definition Compiler.h:215
#define LLVM_EXTERNAL_VISIBILITY
Definition Compiler.h:132
static cl::opt< bool > EnableGlobalMerge("enable-global-merge", cl::Hidden, cl::desc("Enable the global merge pass"), cl::init(true))
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
This file declares the IRTranslator pass.
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
#define T
PassBuilder PB(Machine, PassOpts->PTO, std::nullopt, &PIC)
This file describes the interface of the MachineFunctionPass responsible for assigning the generic vi...
const GCNTargetMachine & getTM(const GCNSubtarget *STI)
This file contains some functions that are useful when dealing with strings.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
Target-Independent Code Generator Pass Configuration Options pass.
This pass exposes codegen information to IR-level passes.
static std::unique_ptr< TargetLoweringObjectFile > createTLOF()
AArch64FunctionInfo - This class is derived from MachineFunctionInfo and contains private AArch64-spe...
size_t clearLinkerOptimizationHints(const SmallPtrSetImpl< MachineInstr * > &MIs)
size_t clearLinkerOptimizationHints(const SmallPtrSetImpl< MachineInstr * > &MIs) const override
Remove all Linker Optimization Hints (LOH) associated with instructions in MIs and.
StringMap< std::unique_ptr< AArch64Subtarget > > SubtargetMap
MachineFunctionInfo * createMachineFunctionInfo(BumpPtrAllocator &Allocator, const Function &F, const TargetSubtargetInfo *STI) const override
Create the target's instance of MachineFunctionInfo.
void registerPassBuilderCallbacks(PassBuilder &PB) override
Allow the target to modify the pass pipeline.
const AArch64Subtarget * getSubtargetImpl() const =delete
yaml::MachineFunctionInfo * createDefaultFuncInfoYAML() const override
Allocate and return a default initialized instance of the YAML representation for the MachineFunction...
ScheduleDAGInstrs * createPostMachineScheduler(MachineSchedContext *C) const override
Similar to createMachineScheduler but used when postRA machine scheduling is enabled.
unsigned getEnableGlobalISelAtO() const
Returns the optimisation level that enables GlobalISel.
const AArch64Subtarget * LastSubtarget
std::unique_ptr< TargetLoweringObjectFile > TLOF
yaml::MachineFunctionInfo * convertFuncInfoToYAML(const MachineFunction &MF) const override
Allocate and initialize an instance of the YAML representation of the MachineFunctionInfo.
bool parseMachineFunctionInfo(const yaml::MachineFunctionInfo &, PerFunctionMIParsingState &PFS, SMDiagnostic &Error, SMRange &SourceRange) const override
Parse out the target's MachineFunctionInfo from the YAML reprsentation.
TargetPassConfig * createPassConfig(PassManagerBase &PM) override
Create a pass configuration object to be used by addPassToEmitX methods for generating a pipeline of ...
bool isGlobalISelOptNone() const
This function checks whether the opt level is explicitly set to none, or whether GlobalISel was enabl...
void reset() override
Reset internal state.
AArch64TargetMachine(const Target &T, const Triple &TT, StringRef CPU, StringRef FS, const TargetOptions &Options, std::optional< Reloc::Model > RM, std::optional< CodeModel::Model > CM, CodeGenOptLevel OL, bool JIT, bool IsLittleEndian)
Create an AArch64 architecture model.
ScheduleDAGInstrs * createMachineScheduler(MachineSchedContext *C) const override
Create an instance of ScheduleDAGInstrs to be run within the standard MachineScheduler pass for this ...
TargetTransformInfo getTargetTransformInfo(const Function &F) const override
Return a TargetTransformInfo for a given function.
AArch64beTargetMachine(const Target &T, const Triple &TT, StringRef CPU, StringRef FS, const TargetOptions &Options, std::optional< Reloc::Model > RM, std::optional< CodeModel::Model > CM, CodeGenOptLevel OL, bool JIT)
AArch64leTargetMachine(const Target &T, const Triple &TT, StringRef CPU, StringRef FS, const TargetOptions &Options, std::optional< Reloc::Model > RM, std::optional< CodeModel::Model > CM, CodeGenOptLevel OL, bool JIT)
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1560
This class holds the attributes for a particular argument, parameter, function, or return value.
Definition Attributes.h:410
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:106
LLVM_ABI StringRef getValueAsString() const
Return the attribute's value as a string.
bool isValid() const
Return true if the attribute is any kind of attribute.
Definition Attributes.h:266
CodeGenTargetMachineImpl(const Target &T, const Triple &TT, StringRef CPU, StringRef FS, const TargetOptions &Options, Reloc::Model RM, CodeModel::Model CM, CodeGenOptLevel OL)
This class represents a range of values.
LLVM_ABI APInt getUnsignedMin() const
Return the smallest unsigned value contained in the ConstantRange.
LLVM_ABI APInt getUnsignedMax() const
Return the largest unsigned value contained in the ConstantRange.
Lightweight error class with error context and mandatory checking.
Definition Error.h:159
This pass is responsible for selecting generic machine instructions to target-specific instructions.
static void setUseExtended(bool Enable)
This pass implements the localization mechanism described at the top of this file.
Definition Localizer.h:40
Pass to replace calls to ifuncs with indirect calls.
Definition LowerIFunc.h:19
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
This class provides access to building LLVM's passes.
LLVM_ATTRIBUTE_MINSIZE std::enable_if_t<!std::is_same_v< PassT, PassManager > > addPass(PassT &&Pass)
static LLVM_ABI PassRegistry * getPassRegistry()
getPassRegistry - Access the global registry object, which is automatically initialized at applicatio...
This pass implements the reg bank selector pass used in the GlobalISel pipeline.
Scheduling dependency.
Definition ScheduleDAG.h:53
Instances of this class encapsulate one diagnostic report, allowing printing to a raw_ostream as a ca...
Definition SourceMgr.h:305
Represents a range in source code.
Definition SMLoc.h:47
A ScheduleDAG for scheduling lists of MachineInstr.
ScheduleDAGMILive is an implementation of ScheduleDAGInstrs that schedules machine instructions while...
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
iterator begin() const
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
TargetInstrInfo - Interface to description of machine instruction set.
CodeGenOptLevel getOptLevel() const
Returns the optimization level: None, Less, Default, or Aggressive.
void setSupportsDebugEntryValues(bool Enable)
Triple TargetTriple
Triple string, CPU name, and target feature strings the TargetMachine instance is created with.
const Triple & getTargetTriple() const
void setMachineOutliner(bool Enable)
void setCFIFixup(bool Enable)
void setSupportsDefaultOutlining(bool Enable)
void setGlobalISelAbort(GlobalISelAbortMode Mode)
const MCAsmInfo & getMCAsmInfo() const
Return target specific asm information.
std::unique_ptr< const MCSubtargetInfo > STI
void setGlobalISel(bool Enable)
TargetOptions Options
CodeModel::Model getCodeModel() const
Returns the code model.
unsigned TLSSize
Bit size of immediate TLS offsets (0 == use the default).
unsigned NoTrapAfterNoreturn
Do not emit a trap instruction for 'unreachable' IR instructions behind noreturn calls,...
unsigned TrapUnreachable
Emit target-specific trap instruction for 'unreachable' IR instructions.
Target-Independent Code Generator Pass Configuration Options.
virtual void addCodeGenPrepare()
Add pass to prepare the LLVM IR for code generation.
virtual void addIRPasses()
Add common target configurable passes that perform LLVM IR to IR transforms following machine indepen...
virtual void addMachineSSAOptimization()
addMachineSSAOptimization - Add standard passes that optimize machine instructions in SSA form.
virtual bool addRegAssignAndRewriteOptimized()
TargetSubtargetInfo - Generic base class for all target subtargets.
This pass provides access to the codegen interfaces that are needed for IR-level transformations.
Target - Wrapper for Target specific information.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
PassManagerBase - An abstract interface to allow code to add passes to a pass manager without having ...
Interfaces for registering analysis passes, producing common pass manager configurations,...
@ DynamicNoPIC
Definition CodeGen.h:26
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
ScheduleDAGMILive * createSchedLive(MachineSchedContext *C)
Create the standard converging machine scheduler.
void initializeAArch64PredicateAsCounterLoopRewritesPass(PassRegistry &)
FunctionPass * createAArch64PreLegalizerCombiner()
void initializeLDTLSCleanupPass(PassRegistry &)
LLVM_ABI FunctionPass * createCFGSimplificationPass(SimplifyCFGOptions Options=SimplifyCFGOptions(), std::function< bool(const Function &)> Ftor=nullptr)
void initializeMachineSMEABIPass(PassRegistry &)
FunctionPass * createAArch64PostSelectOptimize()
void initializeAArch64PTrueCoalescingLegacyPass(PassRegistry &)
FunctionPass * createAArch64ConditionOptimizerLegacyPass()
void initializeAArch64A53Fix835769LegacyPass(PassRegistry &)
LLVM_ABI ModulePass * createJMCInstrumenterPass()
JMC instrument pass.
void initializeAArch64SpeculationHardeningPass(PassRegistry &)
FunctionPass * createAArch64RedundantCopyEliminationPass()
LLVM_ABI std::unique_ptr< ScheduleDAGMutation > createMacroFusionDAGMutation(ArrayRef< MacroFusionPredTy > Predicates, bool BranchOnly=false)
Create a DAG scheduling mutation to pair instructions back to back for instructions that benefit acco...
LLVM_ABI FunctionPass * createTypePromotionLegacyPass()
Create IR Type Promotion pass.
void initializeAArch64StackTaggingPreRALegacyPass(PassRegistry &)
FunctionPass * createMachineSMEABIPass(CodeGenOptLevel)
LLVM_ABI FunctionPass * createEHContGuardTargetsLegacy()
Creates Windows EH Continuation Guard target identification pass.
LLVM_ABI FunctionPass * createSelectOptimizePass()
This pass converts conditional moves to conditional jumps when profitable.
FunctionPass * createAArch64A53Fix835769LegacyPass()
LLVM_ABI Pass * createGlobalMergePass(const TargetMachine *TM, unsigned MaximalOffset, bool OnlyOptimizeForSize=false, bool MergeExternalByDefault=false, bool MergeConstantByDefault=false, bool MergeConstAggressiveByDefault=false)
GlobalMerge - This pass merges internal (by default) globals into structs to enable reuse of a base p...
void initializeAArch64BranchTargetsLegacyPass(PassRegistry &)
LLVM_ABI std::unique_ptr< ScheduleDAGMutation > createLoadClusterDAGMutation(const TargetInstrInfo *TII, bool ReorderWhileClustering=false)
If ReorderWhileClustering is set to true, no attempt will be made to reduce reordering due to store c...
LLVM_ABI FunctionPass * createPseudoProbeInserter()
This pass inserts pseudo probe annotation for callsite profiling.
FunctionPass * createAArch64PostCoalescerPass()
void initializeAArch64PromoteConstantPass(PassRegistry &)
FunctionPass * createFalkorMarkStridedAccessesPass()
Target & getTheAArch64beTarget()
FunctionPass * createAArch64PointerAuthPass()
FunctionPass * createFalkorHWPFFixPass()
LLVM_ABI char & PostRASchedulerID
PostRAScheduler - This pass performs post register allocation scheduling.
std::string utostr(uint64_t X, bool isNeg=false)
FunctionPass * createAArch64O0PreLegalizerCombiner()
FunctionPass * createAArch64SLSHardeningLegacyPass()
@ O0
Disable as many optimizations as possible.
void initializeAArch64CollectLOHLegacyPass(PassRegistry &)
Pass * createAArch64PredicateAsCounterLoopRewritesPass()
void initializeAArch64PostLegalizerCombinerLegacyPass(PassRegistry &)
FunctionPass * createAArch64PostLegalizerCombinerLegacy(bool IsOptNone)
FunctionPass * createAArch64LoadStoreOptLegacyPass()
createAArch64LoadStoreOptimizationPass - returns an instance of the load / store optimization pass.
FunctionPass * createAArch64CondBrTuning()
LLVM_ABI std::unique_ptr< CSEConfigBase > getStandardCSEConfigForOpt(CodeGenOptLevel Level)
Definition CSEInfo.cpp:85
void initializeAArch64Arm64ECCallLoweringPass(PassRegistry &)
void initializeAArch64SIMDInstrOptLegacyPass(PassRegistry &)
LLVM_ABI char & PostMachineSchedulerID
PostMachineScheduler - This pass schedules machine instructions postRA.
LLVM_ABI char & PeepholeOptimizerLegacyID
PeepholeOptimizer - This pass performs peephole optimizations - like extension and comparison elimina...
LLVM_ABI std::unique_ptr< ScheduleDAGMutation > createStoreClusterDAGMutation(const TargetInstrInfo *TII, bool ReorderWhileClustering=false)
If ReorderWhileClustering is set to true, no attempt will be made to reduce reordering due to store c...
LLVM_ABI Pass * createLICMPass()
Definition LICM.cpp:381
FunctionPass * createAArch64A57FPLoadBalancingLegacyPass()
Target & getTheAArch64leTarget()
FunctionPass * createAArch64DeadRegisterDefinitions()
LLVM_ABI char & EarlyIfConverterLegacyID
EarlyIfConverter - This pass performs if-conversion on SSA form by inserting cmov instructions.
void initializeAArch64RedundantCondBranchLegacyPass(PassRegistry &)
void initializeAArch64PostSelectOptimizeLegacyPass(PassRegistry &)
FunctionPass * createSMEPeepholeOptPass()
FunctionPass * createAArch64PostLegalizerLowering()
ThinOrFullLTOPhase
This enumerates the LLVM full LTO or ThinLTO optimization phases.
Definition Pass.h:77
FunctionPass * createAArch64StackTaggingPreRALegacyPass()
void initializeAArch64CodeLayoutOptPass(PassRegistry &)
FunctionPass * createAArch64PTrueCoalescingLegacyPass()
LLVM_ABI void initializeMachineKCFILegacyPass(PassRegistry &)
PassManager< Loop, LoopAnalysisManager, LoopStandardAnalysisResults &, LPMUpdater & > LoopPassManager
The Loop pass manager.
LLVM_ABI char & MachineCombinerID
This pass performs instruction combining using trace metrics to estimate critical-path and resource d...
void initializeAArch64AsmPrinterPass(PassRegistry &)
FunctionPass * createAArch64MIPeepholeOptLegacyPass()
LLVM_ABI FunctionPass * createUnpackMachineBundlesLegacy(std::function< bool(const MachineFunction &)> Ftor)
static Reloc::Model getEffectiveRelocModel(std::optional< Reloc::Model > RM)
void initializeAArch64AdvSIMDScalarLegacyPass(PassRegistry &)
FunctionPass * createAArch64CompressJumpTablesPass()
Target & getTheAArch64_32Target()
FunctionPass * createAArch64ConditionalCompares()
FunctionPass * createAArch64ExpandPseudoLegacyPass()
Returns an instance of the pseudo instruction expansion pass.
void initializeAArch64PointerAuthLegacyPass(PassRegistry &)
ScheduleDAGMI * createSchedPostRA(MachineSchedContext *C)
Create a generic scheduler with no vreg liveness or DAG mutation passes.
LLVM_ABI char & BranchRelaxationPassID
BranchRelaxation - This pass replaces branches that need to jump further than is supported by a branc...
void initializeFalkorMarkStridedAccessesLegacyPass(PassRegistry &)
void initializeAArch64StackTaggingPass(PassRegistry &)
void initializeAArch64PostLegalizerLoweringLegacyPass(PassRegistry &)
LLVM_ABI FunctionPass * createKCFIPass()
Lowers KCFI operand bundles for indirect calls.
Definition KCFI.cpp:75
std::unique_ptr< ScheduleDAGMutation > createAArch64MacroFusionDAGMutation()
Note that you have to add: DAG.addMutation(createAArch64MacroFusionDAGMutation()); to AArch64TargetMa...
LLVM_ABI FunctionPass * createComplexDeinterleavingPass(const TargetMachine *TM)
This pass implements generation of target-specific intrinsics to support handling of complex number a...
PassManager< Module > ModulePassManager
Convenience typedef for a pass manager over modules.
ModulePass * createAArch64Arm64ECCallLoweringPass()
void initializeAArch64ConditionOptimizerLegacyPass(PassRegistry &)
LLVM_ABI FunctionPass * createLoopDataPrefetchPass()
FunctionPass * createAArch64SIMDInstrOptPass()
Returns an instance of the high cost ASIMD instruction replacement optimization pass.
void initializeSMEPeepholeOptPass(PassRegistry &)
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
FunctionPass * createAArch64StorePairSuppressPass()
void initializeAArch64PostCoalescerLegacyPass(PassRegistry &)
FunctionPass * createAArch64CollectLOHPass()
LLVM_ABI ConstantRange getVScaleRange(const Function *F, unsigned BitWidth)
Determine the possible constant range of vscale with the given bit width, based on the vscale_range f...
CodeGenOptLevel
Code generation optimization level.
Definition CodeGen.h:227
@ Default
-O2, -Os, -Oz
Definition CodeGen.h:230
LLVM_ABI FunctionPass * createCFGuardLongjmpPass()
Creates CFGuard longjmp target identification pass.
void initializeAArch64SLSHardeningLegacyPass(PassRegistry &)
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
Target & getTheARM64_32Target()
void initializeAArch64StorePairSuppressPass(PassRegistry &)
LLVM_ABI FunctionPass * createSeparateConstOffsetFromGEPPass(bool LowerGEP=false)
LLVM_ABI FunctionPass * createInterleavedAccessPass()
InterleavedAccess Pass - This pass identifies and matches interleaved memory accesses to target speci...
LLVM_ABI void initializeGlobalISel(PassRegistry &)
Initialize all passes linked into the GlobalISel library.
void initializeAArch64PreLegalizerCombinerLegacyPass(PassRegistry &)
FunctionPass * createAArch64ISelDag(AArch64TargetMachine &TM, CodeGenOptLevel OptLevel)
createAArch64ISelDag - This pass converts a legalized DAG into a AArch64-specific DAG,...
void initializeSVEShuffleOptsPass(PassRegistry &)
void initializeAArch64LowerHomogeneousPrologEpilogLegacyPass(PassRegistry &)
LLVM_ABI FunctionPass * createCFGuardPass()
Insert Control Flow Guard checks on indirect function calls.
Definition CFGuard.cpp:315
void initializeAArch64CondBrTuningPass(PassRegistry &)
LLVM_ABI char & MachinePipelinerID
This pass performs software pipelining on machine instructions.
void initializeAArch64A57FPLoadBalancingLegacyPass(PassRegistry &)
FunctionPass * createAArch64BranchTargetsPass()
Target & getTheARM64Target()
void initializeFalkorHWPFFixPass(PassRegistry &)
void initializeAArch64ExpandPseudoLegacyPass(PassRegistry &)
ModulePass * createAArch64LowerHomogeneousPrologEpilogPass()
LLVM_ABI bool isNonDataDep(const SDep *Dep)
Returns true if Dep is a non-null non-data dependency.
void initializeAArch64SRLTDefineSuperRegsLegacyPass(PassRegistry &)
FunctionPass * createAArch64SRLTDefineSuperRegsLegacyPass()
FunctionPass * createAArch64StackTaggingPass(bool IsOptNone)
LLVM_ABI FunctionPass * createAtomicExpandLegacyPass()
AtomicExpandPass - At IR level this pass replace atomic instructions with __atomic_* library calls,...
FunctionPass * createAArch64CleanupLocalDynamicTLSPass()
BumpPtrAllocatorImpl<> BumpPtrAllocator
The standard BumpPtrAllocator which just uses the default template parameters.
Definition Allocator.h:391
ModulePass * createAArch64PromoteConstantPass()
void initializeAArch64CompressJumpTablesLegacyPass(PassRegistry &)
LLVM_ABI FunctionPass * createEarlyCSEPass(bool UseMemorySSA=false)
LLVM_ABI MachineFunctionPass * createMachineCopyPropagationPass(bool UseCopyInstr)
void initializeAArch64RedundantCopyEliminationLegacyPass(PassRegistry &)
Pass * createSVEShuffleOptsPass()
FunctionPass * createAArch64CodeLayoutOptPass()
FunctionPass * createAArch64AdvSIMDScalar()
FunctionPass * createAArch64RedundantCondBranchPass()
void initializeAArch64DAGToDAGISelLegacyPass(PassRegistry &)
FunctionPass * createAArch64SpeculationHardeningPass()
Returns an instance of the pseudo instruction expansion pass.
void initializeAArch64MIPeepholeOptLegacyPass(PassRegistry &)
void initializeAArch64DeadRegisterDefinitionsLegacyPass(PassRegistry &)
void initializeAArch64ConditionalComparesLegacyPass(PassRegistry &)
void initializeAArch64O0PreLegalizerCombinerLegacyPass(PassRegistry &)
LLVM_ABI FunctionPass * createInterleavedLoadCombinePass()
InterleavedLoadCombines Pass - This pass identifies interleaved loads and combines them into wide loa...
void initializeAArch64LoadStoreOptLegacyPass(PassRegistry &)
LLVM_ABI CGPassBuilderOption getCGPassBuilderOption()
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
cl::boolOrDefault EnableGlobalISelOption
MachineFunctionInfo - This class can be derived from and used by targets to hold private target-speci...
static FuncInfoTy * create(BumpPtrAllocator &Allocator, const Function &F, const SubtargetTy *STI)
Factory function: default behavior is to call new using the supplied allocator.
MachineSchedContext provides enough context from the MachineScheduler pass for the target to instanti...
RegisterTargetMachine - Helper template for registering a target machine implementation,...
Targets should override this in a way that mirrors the implementation of llvm::MachineFunctionInfo.