52#include "llvm/Config/llvm-config.h"
76#define DEBUG_TYPE "machine-scheduler"
79 "Number of instructions in source order after pre-RA scheduling");
81 "Number of instructions in source order after post-RA scheduling");
83 "Number of instructions scheduled by pre-RA scheduler");
85 "Number of instructions scheduled by post-RA scheduler");
86STATISTIC(NumClustered,
"Number of load/store pairs clustered");
89 "Number of scheduling units chosen from top queue pre-RA");
91 "Number of scheduling units chosen from bottom queue pre-RA");
93 "Number of scheduling units chosen for NoCand heuristic pre-RA");
95 "Number of scheduling units chosen for Only1 heuristic pre-RA");
97 "Number of scheduling units chosen for PhysReg heuristic pre-RA");
99 "Number of scheduling units chosen for RegExcess heuristic pre-RA");
101 "Number of scheduling units chosen for RegCritical heuristic pre-RA");
103 "Number of scheduling units chosen for Stall heuristic pre-RA");
105 "Number of scheduling units chosen for Cluster heuristic pre-RA");
107 "Number of scheduling units chosen for Weak heuristic pre-RA");
109 "Number of scheduling units chosen for RegMax heuristic pre-RA");
111 NumResourceReducePreRA,
112 "Number of scheduling units chosen for ResourceReduce heuristic pre-RA");
114 NumResourceDemandPreRA,
115 "Number of scheduling units chosen for ResourceDemand heuristic pre-RA");
117 NumTopDepthReducePreRA,
118 "Number of scheduling units chosen for TopDepthReduce heuristic pre-RA");
120 NumTopPathReducePreRA,
121 "Number of scheduling units chosen for TopPathReduce heuristic pre-RA");
123 NumBotHeightReducePreRA,
124 "Number of scheduling units chosen for BotHeightReduce heuristic pre-RA");
126 NumBotPathReducePreRA,
127 "Number of scheduling units chosen for BotPathReduce heuristic pre-RA");
129 "Number of scheduling units chosen for NodeOrder heuristic pre-RA");
131 "Number of scheduling units chosen for FirstValid heuristic pre-RA");
134 "Number of scheduling units chosen from top queue post-RA");
136 "Number of scheduling units chosen from bottom queue post-RA");
138 "Number of scheduling units chosen for NoCand heuristic post-RA");
140 "Number of scheduling units chosen for Only1 heuristic post-RA");
142 "Number of scheduling units chosen for PhysReg heuristic post-RA");
144 "Number of scheduling units chosen for RegExcess heuristic post-RA");
146 NumRegCriticalPostRA,
147 "Number of scheduling units chosen for RegCritical heuristic post-RA");
149 "Number of scheduling units chosen for Stall heuristic post-RA");
151 "Number of scheduling units chosen for Cluster heuristic post-RA");
153 "Number of scheduling units chosen for Weak heuristic post-RA");
155 "Number of scheduling units chosen for RegMax heuristic post-RA");
157 NumResourceReducePostRA,
158 "Number of scheduling units chosen for ResourceReduce heuristic post-RA");
160 NumResourceDemandPostRA,
161 "Number of scheduling units chosen for ResourceDemand heuristic post-RA");
163 NumTopDepthReducePostRA,
164 "Number of scheduling units chosen for TopDepthReduce heuristic post-RA");
166 NumTopPathReducePostRA,
167 "Number of scheduling units chosen for TopPathReduce heuristic post-RA");
169 NumBotHeightReducePostRA,
170 "Number of scheduling units chosen for BotHeightReduce heuristic post-RA");
172 NumBotPathReducePostRA,
173 "Number of scheduling units chosen for BotPathReduce heuristic post-RA");
175 "Number of scheduling units chosen for NodeOrder heuristic post-RA");
177 "Number of scheduling units chosen for FirstValid heuristic post-RA");
181 cl::desc(
"Pre reg-alloc list scheduling direction"),
185 "Force top-down pre reg-alloc list scheduling"),
187 "Force bottom-up pre reg-alloc list scheduling"),
189 "Force bidirectional pre reg-alloc list scheduling")));
193 cl::desc(
"Post reg-alloc list scheduling direction"),
197 "Force top-down post reg-alloc list scheduling"),
199 "Force bottom-up post reg-alloc list scheduling"),
201 "Force bidirectional post reg-alloc list scheduling")));
205 cl::desc(
"Print critical path length to stdout"));
209 cl::desc(
"Verify machine instrs before and after machine scheduling"));
217 cl::desc(
"Pop up a window to show MISched dags after they are processed"));
222 cl::desc(
"Dump resource usage at schedule boundary."));
225 cl::desc(
"Show details of invoking getNextResoufceCycle."));
230#ifdef LLVM_ENABLE_DUMP
239 cl::desc(
"Hide nodes with more predecessor/successor than cutoff"));
245 cl::desc(
"Only schedule this function"));
247 cl::desc(
"Only schedule this MBB#"));
262 cl::desc(
"Enable memop clustering."),
266 cl::desc(
"Switch to fast cluster algorithm with the lost "
267 "of some fusion opportunities"),
271 cl::desc(
"The threshold for fast cluster"),
274#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
277 cl::desc(
"Dump resource usage at schedule boundary."));
280 cl::desc(
"Set width of the columns with "
281 "the resources and schedule units"),
285 cl::desc(
"Set width of the columns showing resource booking."),
289 cl::desc(
"Sort the resources printed in the dump trace"));
300void MachineSchedStrategy::anchor() {}
302void ScheduleDAGMutation::anchor() {}
342 const RequiredAnalyses &Analyses);
370 const RequiredAnalyses &Analyses);
394 const RequiredAnalyses &Analyses);
411 MachineSchedulerImpl Impl;
414 MachineSchedulerLegacy();
415 void getAnalysisUsage(AnalysisUsage &AU)
const override;
423 SSAMachineSchedulerImpl Impl;
426 SSAMachineSchedulerLegacy();
427 void getAnalysisUsage(AnalysisUsage &AU)
const override;
435 PostMachineSchedulerImpl Impl;
438 PostMachineSchedulerLegacy();
439 void getAnalysisUsage(AnalysisUsage &AU)
const override;
447char MachineSchedulerLegacy::ID = 0;
452 "Machine Instruction Scheduler",
false,
false)
459 "Machine Instruction Scheduler",
false,
false)
463void MachineSchedulerLegacy::getAnalysisUsage(
AnalysisUsage &AU)
const {
476char SSAMachineSchedulerLegacy::ID = 0;
481 "SSA Machine Instruction Scheduler",
false,
false)
488 "SSA Machine Instruction Scheduler",
false,
false)
490SSAMachineSchedulerLegacy::SSAMachineSchedulerLegacy()
495void SSAMachineSchedulerLegacy::getAnalysisUsage(
AnalysisUsage &AU)
const {
508char PostMachineSchedulerLegacy::ID = 0;
513 "PostRA Machine Instruction Scheduler",
false,
false)
520PostMachineSchedulerLegacy::PostMachineSchedulerLegacy()
523void PostMachineSchedulerLegacy::getAnalysisUsage(
AnalysisUsage &AU)
const {
545 cl::desc(
"Machine instruction scheduler to use"));
553 cl::desc(
"Enable the machine instruction scheduling pass."),
cl::init(
true),
557 "enable-ssa-misched",
558 cl::desc(
"Enable the machine instruction scheduling pass in SSA."),
562 "enable-post-misched",
563 cl::desc(
"Enable the post-ra machine instruction scheduling pass."),
570 assert(
I != Beg &&
"reached the top of the region, cannot decrement");
572 if (!
I->isDebugOrPseudoInstr())
591 for(;
I != End; ++
I) {
592 if (!
I->isDebugOrPseudoInstr())
634 const char *MSchedBanner =
"Before machine scheduling.";
636 MF->verify(P, MSchedBanner, &
errs());
638 MF->verify(*MFAM, MSchedBanner, &
errs());
648 const char *MSchedBanner =
"After machine scheduling.";
650 MF->verify(P, MSchedBanner, &
errs());
652 MF->verify(*MFAM, MSchedBanner, &
errs());
681 const char *MSchedBanner =
"Before machine scheduling.";
683 MF->verify(P, MSchedBanner, &
errs());
685 MF->verify(*MFAM, MSchedBanner, &
errs());
696 const char *MSchedBanner =
"After machine scheduling.";
698 MF->verify(P, MSchedBanner, &
errs());
700 MF->verify(*MFAM, MSchedBanner, &
errs());
727 const char *PostMSchedBanner =
"Before post machine scheduling.";
729 MF->verify(P, PostMSchedBanner, &
errs());
731 MF->verify(*MFAM, PostMSchedBanner, &
errs());
740 const char *PostMSchedBanner =
"After post machine scheduling.";
742 MF->verify(P, PostMSchedBanner, &
errs());
744 MF->verify(*MFAM, PostMSchedBanner, &
errs());
765bool MachineSchedulerLegacy::runOnMachineFunction(
MachineFunction &MF) {
778 auto &MLI = getAnalysis<MachineLoopInfoWrapperPass>().getLI();
779 auto &TM = getAnalysis<TargetPassConfig>().getTM<
TargetMachine>();
780 auto &
AA = getAnalysis<AAResultsWrapperPass>().getAAResults();
781 auto &LIS = getAnalysis<LiveIntervalsWrapperPass>().getLIS();
783 getAnalysis<MachineRegisterClassInfoWrapperPass>().getRCI();
784 auto &MBFI = getAnalysis<MachineBlockFrequencyInfoWrapperPass>().getMBFI();
786 Impl.setLegacyPass(
this);
787 return Impl.run(MF, TM, {MLI,
AA, LIS, RegClassInfo, MBFI});
790bool SSAMachineSchedulerLegacy::runOnMachineFunction(
MachineFunction &MF) {
801 auto &MLI = getAnalysis<MachineLoopInfoWrapperPass>().getLI();
802 auto &TM = getAnalysis<TargetPassConfig>().getTM<
TargetMachine>();
803 auto &
AA = getAnalysis<AAResultsWrapperPass>().getAAResults();
804 auto &LIS = getAnalysis<LiveIntervalsWrapperPass>().getLIS();
806 getAnalysis<MachineRegisterClassInfoWrapperPass>().getRCI();
807 auto &MBFI = getAnalysis<MachineBlockFrequencyInfoWrapperPass>().getMBFI();
809 Impl.setLegacyPass(
this);
810 return Impl.run(MF, TM, {MLI,
AA, LIS, RegClassInfo, MBFI});
814 : Impl(
std::make_unique<MachineSchedulerImpl>()), TM(TM) {}
820 : Impl(
std::make_unique<SSAMachineSchedulerImpl>()), TM(TM) {}
826 : Impl(
std::make_unique<PostMachineSchedulerImpl>()), TM(TM) {}
850 Impl->setMFAM(&MFAM);
851 bool Changed = Impl->run(MF, *TM, {MLI,
AA, LIS, RegClassInfo, MBFI});
857 .preserve<SlotIndexesAnalysis>()
880 Impl->setMFAM(&MFAM);
881 bool Changed = Impl->run(MF, *TM, {MLI,
AA, LIS, RegClassInfo, MBFI});
890bool PostMachineSchedulerLegacy::runOnMachineFunction(
MachineFunction &MF) {
902 auto &MLI = getAnalysis<MachineLoopInfoWrapperPass>().getLI();
903 auto &TM = getAnalysis<TargetPassConfig>().getTM<
TargetMachine>();
904 auto &
AA = getAnalysis<AAResultsWrapperPass>().getAAResults();
905 Impl.setLegacyPass(
this);
906 return Impl.run(MF, TM, {MLI,
AA});
925 Impl->setMFAM(&MFAM);
926 bool Changed = Impl->run(MF, *TM, {MLI,
AA});
948 return MI->isCall() ||
TII->isSchedulingBoundary(*
MI,
MBB, *MF) ||
949 MI->isFakeUse() ||
MI->isPHI();
957 bool RegionsTopDown) {
963 RegionEnd !=
MBB->begin(); RegionEnd =
I) {
966 if (RegionEnd !=
MBB->end() ||
973 unsigned NumRegionInstrs = 0;
975 for (;
I !=
MBB->begin(); --
I) {
979 if (!
MI.isDebugOrPseudoInstr()) {
988 if (NumRegionInstrs != 0)
993 std::reverse(Regions.
begin(), Regions.
end());
1031 bool ScheduleSingleMI =
Scheduler.shouldScheduleSingleMIRegions();
1035 unsigned NumRegionInstrs = R.NumRegionInstrs;
1039 Scheduler.enterRegion(&*
MBB,
I, RegionEnd, NumRegionInstrs);
1043 if (
I == RegionEnd || (!ScheduleSingleMI &&
I == std::prev(RegionEnd))) {
1049 auto DumpRegionHeader = [&] {
1050 dbgs() <<
"Current Schedule Region\n";
1052 <<
MBB->getName() <<
"\n From: " << *
I <<
" To: ";
1053 if (RegionEnd !=
MBB->end())
1054 dbgs() << *RegionEnd;
1057 dbgs() <<
" RegionInstrs: " << NumRegionInstrs <<
'\n';
1065 errs() <<
":%bb. " <<
MBB->getNumber();
1066 errs() <<
" " <<
MBB->getName() <<
" \n";
1086#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1088 dbgs() <<
"Queue " << Name <<
": ";
1089 for (
const SUnit *SU : Queue)
1090 dbgs() << SU->NodeNum <<
" ";
1111 if (SuccEdge->
isWeak()) {
1117 dbgs() <<
"*** Scheduling failed! ***\n";
1119 dbgs() <<
" has been released too many times!\n";
1146 if (PredEdge->
isWeak()) {
1152 dbgs() <<
"*** Scheduling failed! ***\n";
1154 dbgs() <<
" has been released too many times!\n";
1191 unsigned regioninstrs)
1201 else if (
SchedImpl->getPolicy().OnlyBottomUp)
1217 BB->splice(InsertPos,
BB,
MI);
1221 LIS->handleMove(*
MI,
true);
1229#if LLVM_ENABLE_ABI_BREAKING_CHECKS && !defined(NDEBUG)
1234 ++NumInstrsScheduled;
1266 bool IsTopNode =
false;
1271 LLVM_DEBUG(
dbgs() <<
"** ScheduleDAGMI::schedule picking next node\n");
1288 if (&*priorII ==
MI)
1310 dbgs() <<
"*** Final schedule for "
1327 assert(!SU.isBoundaryNode() &&
"Boundary node should not be in SUnits");
1330 SU.biasCriticalPath();
1333 if (!SU.NumPredsLeft)
1336 if (!SU.NumSuccsLeft)
1339 ExitSU.biasCriticalPath();
1349 for (
SUnit *SU : TopRoots)
1388 for (std::vector<std::pair<MachineInstr *, MachineInstr *>>
::iterator
1390 std::pair<MachineInstr *, MachineInstr *>
P = *std::prev(DI);
1401#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1413 dbgs() <<
" * Schedule table (TopDown):\n";
1431 for (
unsigned C = FirstCycle;
C <= LastCycle; ++
C)
1438 dbgs() <<
"Missing SUnit\n";
1441 std::string NodeName(
"SU(");
1442 NodeName += std::to_string(SU->
NodeNum) +
")";
1444 unsigned C = FirstCycle;
1445 for (;
C <= LastCycle; ++
C) {
1463 return std::tie(LHS.AcquireAtCycle, LHS.ReleaseAtCycle) <
1464 std::tie(RHS.AcquireAtCycle, RHS.ReleaseAtCycle);
1468 const std::string ResName =
1469 SchedModel.getResourceName(PI.ProcResourceIdx);
1474 for (
unsigned I = 0, E = PI.ReleaseAtCycle - PI.AcquireAtCycle;
I != E;
1477 while (
C++ <= LastCycle)
1494 dbgs() <<
" * Schedule table (BottomUp):\n";
1507 if ((
int)SU->
BotReadyCycle - PI->ReleaseAtCycle + 1 < LastCycle)
1508 LastCycle = (int)SU->
BotReadyCycle - PI->ReleaseAtCycle + 1;
1513 for (
int C = FirstCycle;
C >= LastCycle; --
C)
1520 dbgs() <<
"Missing SUnit\n";
1523 std::string NodeName(
"SU(");
1524 NodeName += std::to_string(SU->
NodeNum) +
")";
1527 for (;
C >= LastCycle; --
C) {
1544 return std::tie(LHS.AcquireAtCycle, LHS.ReleaseAtCycle) <
1545 std::tie(RHS.AcquireAtCycle, RHS.ReleaseAtCycle);
1549 const std::string ResName =
1550 SchedModel.getResourceName(PI.ProcResourceIdx);
1555 for (
unsigned I = 0, E = PI.ReleaseAtCycle - PI.AcquireAtCycle;
I != E;
1558 while (
C-- >= LastCycle)
1567#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1575 dbgs() <<
"* Schedule table (Bidirectional): not implemented\n";
1577 dbgs() <<
"* Schedule table: DumpDirection not set.\n";
1585 dbgs() <<
"Missing SUnit\n";
1610 if (!Reg.isVirtual())
1615 bool FoundDef =
false;
1617 if (MO2.getReg() == Reg && !MO2.isDead()) {
1628 for (; UI !=
VRegUses.end(); ++UI) {
1644 unsigned regioninstrs)
1658 "ShouldTrackLaneMasks requires ShouldTrackPressure");
1709 dbgs() <<
"Bottom Pressure: ";
1715 "Can't find the region bottom");
1723 unsigned Limit =
RegClassInfo->getRegPressureSetLimit(i);
1732 dbgs() <<
"Excess PSets: ";
1734 dbgs() <<
TRI->getRegPressureSetName(RCPS.getPSet()) <<
" ";
1742 const std::vector<unsigned> &NewMaxPressure) {
1748 unsigned ID = PC.getPSet();
1753 && NewMaxPressure[ID] <= (
unsigned)std::numeric_limits<int16_t>::max())
1756 unsigned Limit =
RegClassInfo->getRegPressureSetLimit(ID);
1757 if (NewMaxPressure[ID] >= Limit - 2) {
1759 << NewMaxPressure[ID]
1760 << ((NewMaxPressure[ID] > Limit) ?
" > " :
" <= ")
1772 if (!
P.VRegOrUnit.isVirtualReg())
1774 Register Reg =
P.VRegOrUnit.asVirtualReg();
1781 bool Decrement =
P.LaneMask.any();
1785 SUnit &SU = *V2SU.SU;
1795 <<
" UpdateRegPressure: " << SU <<
" "
1818 assert(VNI &&
"No live value at use.");
1821 SUnit *SU = V2SU.SU;
1845#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1846 if (
EntrySU.getInstr() !=
nullptr)
1851 dbgs() <<
" Pressure Diff : ";
1854 dbgs() <<
" Single Issue : ";
1855 if (
SchedModel.mustBeginGroup(SU.getInstr()) &&
1862 if (
ExitSU.getInstr() !=
nullptr)
1898 bool IsTopNode =
false;
1903 LLVM_DEBUG(
dbgs() <<
"** ScheduleDAGMILive::schedule picking next node\n");
1912 unsigned SubtreeID =
DFSResult->getSubtreeID(SU);
1930 dbgs() <<
"*** Final schedule for "
1999 if (!
BB->isSuccessor(
BB))
2002 unsigned MaxCyclicLatency = 0;
2005 if (!
P.VRegOrUnit.isVirtualReg())
2007 Register Reg =
P.VRegOrUnit.asVirtualReg();
2018 unsigned LiveOutHeight = DefSU->
getHeight();
2023 SUnit *SU = V2SU.SU;
2035 unsigned CyclicLatency = 0;
2037 CyclicLatency = LiveOutDepth - SU->
getDepth();
2040 if (LiveInHeight > LiveOutHeight) {
2041 if (LiveInHeight - LiveOutHeight < CyclicLatency)
2042 CyclicLatency = LiveInHeight - LiveOutHeight;
2046 LLVM_DEBUG(
dbgs() <<
"Cyclic Path: " << *DefSU <<
" -> " << *SU <<
" = "
2047 << CyclicLatency <<
"c\n");
2048 if (CyclicLatency > MaxCyclicLatency)
2049 MaxCyclicLatency = CyclicLatency;
2052 LLVM_DEBUG(
dbgs() <<
"Cyclic Critical Path: " << MaxCyclicLatency <<
"c\n");
2053 return MaxCyclicLatency;
2105 if (&*priorII ==
MI)
2156 bool OffsetIsScalable;
2160 : SU(SU), BaseOps(BaseOps),
Offset(
Offset), Width(Width),
2161 OffsetIsScalable(OffsetIsScalable) {}
2165 if (
A->getType() !=
B->getType())
2166 return A->getType() <
B->getType();
2168 return A->getReg() <
B->getReg();
2180 if (AIsFixed != BIsFixed)
2181 return StackGrowsDown ? !AIsFixed : AIsFixed;
2189 if (AOffset != BOffset)
2190 return AOffset < BOffset;
2192 return StackGrowsDown ?
A->getIndex() >
B->getIndex()
2193 :
A->getIndex() <
B->getIndex();
2203 if (std::lexicographical_compare(BaseOps.
begin(), BaseOps.
end(),
2204 RHS.BaseOps.begin(),
RHS.BaseOps.end(),
2207 if (std::lexicographical_compare(
RHS.BaseOps.begin(),
RHS.BaseOps.end(),
2208 BaseOps.
begin(), BaseOps.
end(), Compare))
2216 const TargetInstrInfo *
TII;
2218 bool ReorderWhileClustering;
2221 BaseMemOpClusterMutation(
const TargetInstrInfo *tii,
bool IsLoad,
2222 bool ReorderWhileClustering)
2223 :
TII(tii), IsLoad(IsLoad),
2224 ReorderWhileClustering(ReorderWhileClustering) {}
2226 void apply(ScheduleDAGInstrs *DAGInstrs)
override;
2230 ScheduleDAGInstrs *DAG);
2231 void collectMemOpRecords(std::vector<SUnit> &SUnits,
2232 SmallVectorImpl<MemOpInfo> &MemOpRecords);
2237class StoreClusterMutation :
public BaseMemOpClusterMutation {
2239 StoreClusterMutation(
const TargetInstrInfo *tii,
bool ReorderWhileClustering)
2240 : BaseMemOpClusterMutation(tii,
false, ReorderWhileClustering) {}
2243class LoadClusterMutation :
public BaseMemOpClusterMutation {
2245 LoadClusterMutation(
const TargetInstrInfo *tii,
bool ReorderWhileClustering)
2246 : BaseMemOpClusterMutation(tii,
true, ReorderWhileClustering) {}
2251std::unique_ptr<ScheduleDAGMutation>
2253 bool ReorderWhileClustering) {
2255 TII, ReorderWhileClustering)
2259std::unique_ptr<ScheduleDAGMutation>
2261 bool ReorderWhileClustering) {
2263 TII, ReorderWhileClustering)
2272void BaseMemOpClusterMutation::clusterNeighboringMemOps(
2281 for (
unsigned Idx = 0, End = MemOpRecords.
size(); Idx < (End - 1); ++Idx) {
2283 auto MemOpa = MemOpRecords[Idx];
2286 unsigned NextIdx = Idx + 1;
2287 for (; NextIdx < End; ++NextIdx)
2290 if (!SUnit2ClusterInfo.
count(MemOpRecords[NextIdx].SU->NodeNum) &&
2292 (!DAG->
IsReachable(MemOpRecords[NextIdx].SU, MemOpa.SU) &&
2293 !DAG->
IsReachable(MemOpa.SU, MemOpRecords[NextIdx].SU))))
2298 auto MemOpb = MemOpRecords[NextIdx];
2299 unsigned ClusterLength = 2;
2300 unsigned CurrentClusterBytes = MemOpa.Width.getValue().getKnownMinValue() +
2301 MemOpb.Width.getValue().getKnownMinValue();
2302 auto It = SUnit2ClusterInfo.
find(MemOpa.SU->NodeNum);
2303 if (It != SUnit2ClusterInfo.
end()) {
2304 const auto &[Len, Bytes] = It->second;
2305 ClusterLength = Len + 1;
2306 CurrentClusterBytes = Bytes + MemOpb.Width.getValue().getKnownMinValue();
2309 if (!
TII->shouldClusterMemOps(MemOpa.BaseOps, MemOpa.Offset,
2310 MemOpa.OffsetIsScalable, MemOpb.BaseOps,
2311 MemOpb.Offset, MemOpb.OffsetIsScalable,
2312 ClusterLength, CurrentClusterBytes))
2315 SUnit *SUa = MemOpa.SU;
2316 SUnit *SUb = MemOpb.SU;
2326 LLVM_DEBUG(
dbgs() <<
"Cluster ld/st " << *SUa <<
" - " << *SUb <<
"\n");
2356 SUnit2ClusterInfo[MemOpb.SU->NodeNum] = {ClusterLength,
2357 CurrentClusterBytes};
2360 <<
", Curr cluster bytes: " << CurrentClusterBytes
2371 unsigned ClusterIdx = AllClusters.size();
2373 MemberI->ParentClusterIdx = ClusterIdx;
2376 AllClusters.push_back(Group);
2380void BaseMemOpClusterMutation::collectMemOpRecords(
2382 for (
auto &SU : SUnits) {
2390 bool OffsetIsScalable;
2393 OffsetIsScalable, Width)) {
2398 MemOpInfo(&SU, BaseOps,
Offset, OffsetIsScalable, Width));
2401 <<
Offset <<
", OffsetIsScalable: " << OffsetIsScalable
2402 <<
", Width: " << Width <<
"\n");
2405 for (
const auto *
Op : BaseOps)
2411bool BaseMemOpClusterMutation::groupMemOps(
2418 for (
const auto &
MemOp : MemOps) {
2419 unsigned ChainPredID = DAG->
SUnits.size();
2421 for (
const SDep &Pred :
MemOp.SU->Preds) {
2445 collectMemOpRecords(DAG->
SUnits, MemOpRecords);
2447 if (MemOpRecords.
size() < 2)
2454 bool FastCluster = groupMemOps(MemOpRecords, DAG,
Groups);
2456 for (
auto &Group :
Groups) {
2462 clusterNeighboringMemOps(Group.second, FastCluster, DAG);
2477 SlotIndex RegionBeginIdx;
2481 SlotIndex RegionEndIdx;
2484 CopyConstrain(
const TargetInstrInfo *) {}
2486 void apply(ScheduleDAGInstrs *DAGInstrs)
override;
2489 void constrainLocalCopy(SUnit *CopySU, ScheduleDAGMILive *DAG);
2494std::unique_ptr<ScheduleDAGMutation>
2496 return std::make_unique<CopyConstrain>(
TII);
2539 unsigned LocalReg = SrcReg;
2540 unsigned GlobalReg = DstReg;
2542 if (!LocalLI->
isLocal(RegionBeginIdx, RegionEndIdx)) {
2546 if (!LocalLI->
isLocal(RegionBeginIdx, RegionEndIdx))
2557 if (GlobalSegment == GlobalLI->
end())
2564 if (GlobalSegment->contains(LocalLI->
beginIndex()))
2567 if (GlobalSegment == GlobalLI->
end())
2571 if (GlobalSegment != GlobalLI->
begin()) {
2574 GlobalSegment->start)) {
2585 assert(std::prev(GlobalSegment)->start < LocalLI->beginIndex() &&
2586 "Disconnected LRG within the scheduling region.");
2602 for (
const SDep &Succ : LastLocalSU->
Succs) {
2617 for (
const SDep &Pred : GlobalSU->
Preds) {
2620 if (Pred.
getSUnit() == FirstLocalSU)
2628 for (
SUnit *LU : LocalUses) {
2629 LLVM_DEBUG(
dbgs() <<
" Local use SU(" << LU->NodeNum <<
") -> SU("
2630 << GlobalSU->
NodeNum <<
")\n");
2633 for (
SUnit *GU : GlobalUses) {
2634 LLVM_DEBUG(
dbgs() <<
" Global use " << *GU <<
" -> " << *FirstLocalSU
2647 if (FirstPos == DAG->
end())
2675 unsigned Latency,
bool AfterSchedNode) {
2678 return ResCntFactor >= (int)LFactor;
2680 return ResCntFactor > (int)LFactor;
2691 CheckPending =
false;
2694 MinReadyCycle = std::numeric_limits<unsigned>::max();
2695 ExpectedLatency = 0;
2696 DependentLatency = 0;
2698 MaxExecutedResCount = 0;
2700 IsResourceLimited =
false;
2701 ReservedCycles.clear();
2702 ReservedResourceSegments.clear();
2703 ReservedCyclesIndex.clear();
2704 ResourceGroupSubUnitMasks.clear();
2705#if LLVM_ENABLE_ABI_BREAKING_CHECKS
2709 MaxObservedStall = 0;
2712 ExecutedResCounts.resize(1);
2713 assert(!ExecutedResCounts[0] &&
"nonzero count for bad resource");
2729 unsigned PIdx = PI->ProcResourceIdx;
2731 assert(PI->ReleaseAtCycle >= PI->AcquireAtCycle);
2733 (Factor * (PI->ReleaseAtCycle - PI->AcquireAtCycle));
2745 unsigned ResourceCount =
SchedModel->getNumProcResourceKinds();
2746 ReservedCyclesIndex.resize(ResourceCount);
2747 ExecutedResCounts.resize(ResourceCount);
2748 ResourceGroupSubUnitMasks.resize(ResourceCount,
APInt(ResourceCount, 0));
2749 unsigned NumUnits = 0;
2751 for (
unsigned i = 0; i < ResourceCount; ++i) {
2752 ReservedCyclesIndex[i] = NumUnits;
2753 NumUnits +=
SchedModel->getProcResource(i)->NumUnits;
2755 auto SubUnits =
SchedModel->getProcResource(i)->SubUnitsIdxBegin;
2756 for (
unsigned U = 0, UE =
SchedModel->getProcResource(i)->NumUnits;
2758 ResourceGroupSubUnitMasks[i].setBit(SubUnits[U]);
2778 if (ReadyCycle > CurrCycle)
2779 return ReadyCycle - CurrCycle;
2786 unsigned ReleaseAtCycle,
2787 unsigned AcquireAtCycle) {
2790 return ReservedResourceSegments[InstanceIdx].getFirstAvailableAtFromTop(
2791 CurrCycle, AcquireAtCycle, ReleaseAtCycle);
2793 return ReservedResourceSegments[InstanceIdx].getFirstAvailableAtFromBottom(
2794 CurrCycle, AcquireAtCycle, ReleaseAtCycle);
2797 unsigned NextUnreserved = ReservedCycles[InstanceIdx];
2803 NextUnreserved = std::max(CurrCycle, NextUnreserved + ReleaseAtCycle);
2804 return NextUnreserved;
2810std::pair<unsigned, unsigned>
2812 unsigned ReleaseAtCycle,
2813 unsigned AcquireAtCycle) {
2815 LLVM_DEBUG(
dbgs() <<
" Resource booking (@" << CurrCycle <<
"c): \n");
2817 LLVM_DEBUG(
dbgs() <<
" getNextResourceCycle (@" << CurrCycle <<
"c): \n");
2820 unsigned InstanceIdx = 0;
2821 unsigned StartIndex = ReservedCyclesIndex[PIdx];
2822 unsigned NumberOfInstances =
SchedModel->getProcResource(PIdx)->NumUnits;
2823 assert(NumberOfInstances > 0 &&
2824 "Cannot have zero instances of a ProcResource");
2841 if (ResourceGroupSubUnitMasks[PIdx][PE.ProcResourceIdx])
2843 StartIndex, ReleaseAtCycle, AcquireAtCycle),
2846 auto SubUnits =
SchedModel->getProcResource(PIdx)->SubUnitsIdxBegin;
2847 for (
unsigned I = 0, End = NumberOfInstances;
I < End; ++
I) {
2848 unsigned NextUnreserved, NextInstanceIdx;
2849 std::tie(NextUnreserved, NextInstanceIdx) =
2851 if (MinNextUnreserved > NextUnreserved) {
2852 InstanceIdx = NextInstanceIdx;
2853 MinNextUnreserved = NextUnreserved;
2856 return std::make_pair(MinNextUnreserved, InstanceIdx);
2859 for (
unsigned I = StartIndex, End = StartIndex + NumberOfInstances;
I < End;
2861 unsigned NextUnreserved =
2865 << NextUnreserved <<
"c\n");
2866 if (MinNextUnreserved > NextUnreserved) {
2868 MinNextUnreserved = NextUnreserved;
2873 <<
"[" << InstanceIdx - StartIndex <<
"]"
2874 <<
" available @" << MinNextUnreserved <<
"c"
2876 return std::make_pair(MinNextUnreserved, InstanceIdx);
2896 <<
"hazard: " << *SU <<
" reported by HazardRec\n");
2901 if ((CurrMOps > 0) && (CurrMOps + uops >
SchedModel->getIssueWidth())) {
2903 <<
", CurrMOps = " << CurrMOps <<
", "
2904 <<
"CurrMOps + uops > issue width of "
2913 << (
isTop() ?
"begin" :
"end") <<
" group\n");
2922 unsigned ResIdx = PE.ProcResourceIdx;
2923 unsigned ReleaseAtCycle = PE.ReleaseAtCycle;
2924 unsigned AcquireAtCycle = PE.AcquireAtCycle;
2925 unsigned NRCycle, InstanceIdx;
2926 std::tie(NRCycle, InstanceIdx) =
2928 if (NRCycle > CurrCycle) {
2929#if LLVM_ENABLE_ABI_BREAKING_CHECKS
2930 MaxObservedStall = std::max(ReleaseAtCycle, MaxObservedStall);
2933 <<
"hazard: " << *SU <<
" "
2934 <<
SchedModel->getResourceName(ResIdx) <<
'['
2935 << InstanceIdx - ReservedCyclesIndex[ResIdx] <<
']' <<
"="
2936 << NRCycle <<
"c, is later than "
2937 <<
"CurrCycle = " << CurrCycle <<
"c\n");
2948 SUnit *LateSU =
nullptr;
2949 unsigned RemLatency = 0;
2950 for (
SUnit *SU : ReadySUs) {
2952 if (L > RemLatency) {
2959 << RemLatency <<
"c\n");
2973 unsigned OtherCritCount =
Rem->RemIssueCount
2974 + (RetiredMOps *
SchedModel->getMicroOpFactor());
2976 << OtherCritCount /
SchedModel->getMicroOpFactor() <<
'\n');
2977 for (
unsigned PIdx = 1, PEnd =
SchedModel->getNumProcResourceKinds();
2978 PIdx != PEnd; ++PIdx) {
2980 if (OtherCount > OtherCritCount) {
2981 OtherCritCount = OtherCount;
2982 OtherCritIdx = PIdx;
2987 dbgs() <<
" " <<
Available.getName() <<
" + Remain CritRes: "
2988 << OtherCritCount /
SchedModel->getResourceFactor(OtherCritIdx)
2989 <<
" " <<
SchedModel->getResourceName(OtherCritIdx) <<
"\n");
2991 return OtherCritCount;
2998#if LLVM_ENABLE_ABI_BREAKING_CHECKS
3002 if (ReadyCycle > CurrCycle)
3003 MaxObservedStall = std::max(ReadyCycle - CurrCycle, MaxObservedStall);
3006 if (ReadyCycle < MinReadyCycle)
3007 MinReadyCycle = ReadyCycle;
3011 bool IsBuffered =
SchedModel->getMicroOpBufferSize() != 0;
3012 bool HazardDetected = !IsBuffered && ReadyCycle > CurrCycle;
3015 <<
"hazard: " << *SU <<
" ReadyCycle = " << ReadyCycle
3016 <<
" is later than CurrCycle = " << CurrCycle
3017 <<
" on an unbuffered resource" <<
"\n");
3022 HazardDetected =
true;
3027 if (!HazardDetected) {
3042 if (
SchedModel->getMicroOpBufferSize() == 0) {
3043 assert(MinReadyCycle < std::numeric_limits<unsigned>::max() &&
3044 "MinReadyCycle uninitialized");
3045 if (MinReadyCycle > NextCycle)
3046 NextCycle = MinReadyCycle;
3049 unsigned DecMOps =
SchedModel->getIssueWidth() * (NextCycle - CurrCycle);
3050 CurrMOps = (CurrMOps <= DecMOps) ? 0 : CurrMOps - DecMOps;
3053 if ((NextCycle - CurrCycle) > DependentLatency)
3054 DependentLatency = 0;
3056 DependentLatency -= (NextCycle - CurrCycle);
3060 CurrCycle = NextCycle;
3063 for (; CurrCycle != NextCycle; ++CurrCycle) {
3070 CheckPending =
true;
3080 ExecutedResCounts[PIdx] +=
Count;
3081 if (ExecutedResCounts[PIdx] > MaxExecutedResCount)
3082 MaxExecutedResCount = ExecutedResCounts[PIdx];
3096 unsigned ReleaseAtCycle,
3098 unsigned AcquireAtCycle) {
3099 unsigned Factor =
SchedModel->getResourceFactor(PIdx);
3100 unsigned Count = Factor * (ReleaseAtCycle- AcquireAtCycle);
3102 << ReleaseAtCycle <<
"x" << Factor <<
"u\n");
3106 assert(
Rem->RemainingCounts[PIdx] >=
Count &&
"resource double counted");
3107 Rem->RemainingCounts[PIdx] -=
Count;
3112 ZoneCritResIdx = PIdx;
3119 unsigned NextAvailable, InstanceIdx;
3120 std::tie(NextAvailable, InstanceIdx) =
3122 if (NextAvailable > CurrCycle) {
3125 <<
'[' << InstanceIdx - ReservedCyclesIndex[PIdx] <<
']'
3126 <<
" reserved until @" << NextAvailable <<
"\n");
3128 return NextAvailable;
3138 (CurrMOps == 0 || (CurrMOps + IncMOps) <=
SchedModel->getIssueWidth()) &&
3139 "Cannot schedule this instruction's MicroOps in the current cycle.");
3144 unsigned NextCycle = CurrCycle;
3145 switch (
SchedModel->getMicroOpBufferSize()) {
3147 assert(ReadyCycle <= CurrCycle &&
"Broken PendingQueue");
3150 if (ReadyCycle > NextCycle) {
3151 NextCycle = ReadyCycle;
3152 LLVM_DEBUG(
dbgs() <<
" *** Stall until: " << ReadyCycle <<
"\n");
3161 NextCycle = ReadyCycle;
3164 RetiredMOps += IncMOps;
3168 unsigned DecRemIssue = IncMOps *
SchedModel->getMicroOpFactor();
3169 assert(
Rem->RemIssueCount >= DecRemIssue &&
"MOps double counted");
3170 Rem->RemIssueCount -= DecRemIssue;
3171 if (ZoneCritResIdx) {
3173 unsigned ScaledMOps =
3174 RetiredMOps *
SchedModel->getMicroOpFactor();
3182 << ScaledMOps /
SchedModel->getLatencyFactor()
3188 PE =
SchedModel->getWriteProcResEnd(SC); PI != PE; ++PI) {
3190 countResource(SC, PI->ProcResourceIdx, PI->ReleaseAtCycle, NextCycle,
3191 PI->AcquireAtCycle);
3192 if (RCycle > NextCycle)
3202 PE =
SchedModel->getWriteProcResEnd(SC); PI != PE; ++PI) {
3203 unsigned PIdx = PI->ProcResourceIdx;
3204 if (
SchedModel->getResourceBufferSize(PIdx) == 0) {
3207 unsigned ReservedUntil, InstanceIdx;
3209 SC, PIdx, PI->ReleaseAtCycle, PI->AcquireAtCycle);
3211 ReservedResourceSegments[InstanceIdx].add(
3213 NextCycle, PI->AcquireAtCycle, PI->ReleaseAtCycle),
3216 ReservedResourceSegments[InstanceIdx].add(
3218 NextCycle, PI->AcquireAtCycle, PI->ReleaseAtCycle),
3223 unsigned ReservedUntil, InstanceIdx;
3225 SC, PIdx, PI->ReleaseAtCycle, PI->AcquireAtCycle);
3227 ReservedCycles[InstanceIdx] =
3228 std::max(ReservedUntil, NextCycle + PI->ReleaseAtCycle);
3230 ReservedCycles[InstanceIdx] = NextCycle;
3237 unsigned &TopLatency =
isTop() ? ExpectedLatency : DependentLatency;
3238 unsigned &BotLatency =
isTop() ? DependentLatency : ExpectedLatency;
3242 <<
" " << TopLatency <<
"c\n");
3247 <<
" " << BotLatency <<
"c\n");
3250 if (NextCycle > CurrCycle)
3268 CheckPending =
true;
3275 CurrMOps += IncMOps;
3288 while (CurrMOps >=
SchedModel->getIssueWidth()) {
3289 LLVM_DEBUG(
dbgs() <<
" *** Max MOps " << CurrMOps <<
" at cycle "
3290 << CurrCycle <<
'\n');
3301 MinReadyCycle = std::numeric_limits<unsigned>::max();
3305 for (
unsigned I = 0, E =
Pending.size();
I < E; ++
I) {
3311 if (ReadyCycle < MinReadyCycle)
3312 MinReadyCycle = ReadyCycle;
3323 CheckPending =
false;
3352 for (
unsigned i = 0;
Available.empty(); ++i) {
3369#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3378 unsigned ResourceCount =
SchedModel->getNumProcResourceKinds();
3379 unsigned StartIdx = 0;
3381 for (
unsigned ResIdx = 0; ResIdx < ResourceCount; ++ResIdx) {
3382 const unsigned NumUnits =
SchedModel->getProcResource(ResIdx)->NumUnits;
3383 std::string ResName =
SchedModel->getResourceName(ResIdx);
3384 for (
unsigned UnitIdx = 0; UnitIdx < NumUnits; ++UnitIdx) {
3385 dbgs() << ResName <<
"(" << UnitIdx <<
") = ";
3387 if (ReservedResourceSegments.count(StartIdx + UnitIdx))
3388 dbgs() << ReservedResourceSegments.at(StartIdx + UnitIdx);
3392 dbgs() << ReservedCycles[StartIdx + UnitIdx] <<
"\n";
3394 StartIdx += NumUnits;
3403 if (ZoneCritResIdx) {
3404 ResFactor =
SchedModel->getResourceFactor(ZoneCritResIdx);
3408 ResCount = RetiredMOps * ResFactor;
3410 unsigned LFactor =
SchedModel->getLatencyFactor();
3412 <<
" Retired: " << RetiredMOps;
3414 dbgs() <<
"\n Critical: " << ResCount / LFactor <<
"c, "
3415 << ResCount / ResFactor <<
" "
3416 <<
SchedModel->getResourceName(ZoneCritResIdx)
3417 <<
"\n ExpectedLatency: " << ExpectedLatency <<
"c\n"
3418 << (IsResourceLimited ?
" - Resource" :
" - Latency")
3438 PE =
SchedModel->getWriteProcResEnd(SC); PI != PE; ++PI) {
3439 if (PI->ProcResourceIdx ==
Policy.ReduceResIdx)
3440 ResDelta.CritResources += PI->ReleaseAtCycle;
3441 if (PI->ProcResourceIdx ==
Policy.DemandResIdx)
3442 ResDelta.DemandedResources += PI->ReleaseAtCycle;
3448bool GenericSchedulerBase::shouldReduceLatency(
const CandPolicy &Policy,
3450 bool ComputeRemLatency,
3451 unsigned &RemLatency)
const {
3461 if (ComputeRemLatency)
3477 unsigned OtherCritIdx = 0;
3478 unsigned OtherCount =
3481 bool OtherResLimited =
false;
3482 unsigned RemLatency = 0;
3483 bool RemLatencyComputed =
false;
3484 if (
SchedModel->hasInstrSchedModel() && OtherCount != 0) {
3486 RemLatencyComputed =
true;
3488 OtherCount, RemLatency,
false);
3494 if (!OtherResLimited &&
3495 (IsPostRA || shouldReduceLatency(Policy, CurrZone, !RemLatencyComputed,
3499 <<
" RemainingLatency " << RemLatency <<
" + "
3501 <<
Rem.CriticalPath <<
"\n");
3508 dbgs() <<
" " << CurrZone.Available.getName() <<
" ResourceLimited: "
3509 << SchedModel->getResourceName(CurrZone.getZoneCritResIdx()) <<
"\n";
3510 }
if (OtherResLimited)
dbgs()
3511 <<
" RemainingLimit: "
3512 <<
SchedModel->getResourceName(OtherCritIdx) <<
"\n";
3514 <<
" Latency limited both directions.\n");
3519 if (OtherResLimited)
3528 case NoCand:
return "NOCAND ";
3529 case Only1:
return "ONLY1 ";
3530 case PhysReg:
return "PHYS-REG ";
3533 case Stall:
return "STALL ";
3534 case Cluster:
return "CLUSTER ";
3535 case Weak:
return "WEAK ";
3536 case RegMax:
return "REG-MAX ";
3552 unsigned ResIdx = 0;
3587 dbgs() <<
" " <<
TRI->getRegPressureSetName(
P.getPSet())
3588 <<
":" <<
P.getUnitInc() <<
" ";
3592 dbgs() <<
" " <<
SchedModel->getProcResource(ResIdx)->Name <<
" ";
3621 RemLatency = std::max(RemLatency,
3623 RemLatency = std::max(RemLatency,
3635 if (TryVal < CandVal) {
3639 if (TryVal > CandVal) {
3640 if (Cand.
Reason > Reason)
3651 if (TryVal > CandVal) {
3655 if (TryVal < CandVal) {
3656 if (Cand.
Reason > Reason)
3698 const bool IsTop,
const bool IsPostRA =
false) {
3699 assert(SU &&
"SU must not be null for tracing");
3700 LLVM_DEBUG(
dbgs() <<
"Pick " << (IsTop ?
"Top " :
"Bot ") <<
"Cand " << *SU
3702 << (IsPostRA ?
"post-RA" :
"pre-RA") <<
"]\n");
3721 NumRegExcessPostRA++;
3724 NumRegCriticalPostRA++;
3739 NumResourceReducePostRA++;
3742 NumResourceDemandPostRA++;
3745 NumTopDepthReducePostRA++;
3748 NumTopPathReducePostRA++;
3751 NumBotHeightReducePostRA++;
3754 NumBotPathReducePostRA++;
3757 NumNodeOrderPostRA++;
3760 NumFirstValidPostRA++;
3780 NumRegExcessPreRA++;
3783 NumRegCriticalPreRA++;
3798 NumResourceReducePreRA++;
3801 NumResourceDemandPreRA++;
3804 NumTopDepthReducePreRA++;
3807 NumTopPathReducePreRA++;
3810 NumBotHeightReducePreRA++;
3813 NumBotPathReducePreRA++;
3816 NumNodeOrderPreRA++;
3819 NumFirstValidPreRA++;
3827 const bool IsPostRA =
false) {
3833 "(PreRA)GenericScheduler needs vreg liveness");
3839 DAG->computeDFSResult();
3851 Top.HazardRec.reset(
DAG->TII->CreateTargetMIHazardRecognizer(Itin,
DAG));
3853 Bot.HazardRec.reset(
DAG->TII->CreateTargetMIHazardRecognizer(Itin,
DAG));
3873 for (
unsigned VT = MVT::i64; VT > (
unsigned)MVT::i1; --VT) {
3876 unsigned NIntRegs =
Context->RegClassInfo->getNumAllocatableRegs(
3914#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3915 dbgs() <<
"GenericScheduler RegionPolicy: "
3916 <<
" ShouldTrackPressure=" <<
RegionPolicy.ShouldTrackPressure
3933 if (
Rem.CyclicCritPath == 0 ||
Rem.CyclicCritPath >=
Rem.CriticalPath)
3937 unsigned IterCount =
3938 std::max(
Rem.CyclicCritPath *
SchedModel->getLatencyFactor(),
3941 unsigned AcyclicCount =
Rem.CriticalPath *
SchedModel->getLatencyFactor();
3943 unsigned InFlightCount =
3944 (AcyclicCount *
Rem.RemIssueCount + IterCount-1) / IterCount;
3945 unsigned BufferLimit =
3948 Rem.IsAcyclicLatencyLimited = InFlightCount > BufferLimit;
3951 dbgs() <<
"IssueCycles="
3952 <<
Rem.RemIssueCount /
SchedModel->getLatencyFactor() <<
"c "
3953 <<
"IterCycles=" << IterCount /
SchedModel->getLatencyFactor()
3954 <<
"c NumIters=" << (AcyclicCount + IterCount - 1) / IterCount
3955 <<
" InFlight=" << InFlightCount /
SchedModel->getMicroOpFactor()
3956 <<
"m BufferLim=" <<
SchedModel->getMicroOpBufferSize() <<
"m\n";
3957 if (
Rem.IsAcyclicLatencyLimited)
dbgs() <<
" ACYCLIC LATENCY LIMIT\n");
3961 Rem.CriticalPath =
DAG->ExitSU.getDepth();
3964 for (
const SUnit *SU :
Bot.Available) {
3970 errs() <<
"Critical Path(GS-RR ): " <<
Rem.CriticalPath <<
" \n";
3974 Rem.CyclicCritPath =
DAG->computeCyclicCriticalPath();
4000 if (TryPSet == CandPSet) {
4005 int TryRank = TryP.
isValid() ?
TRI->getRegPressureSetScore(MF, TryPSet) :
4006 std::numeric_limits<int>::max();
4008 int CandRank = CandP.
isValid() ?
TRI->getRegPressureSetScore(MF, CandPSet) :
4009 std::numeric_limits<int>::max();
4014 return tryGreater(TryRank, CandRank, TryCand, Cand, Reason);
4032 unsigned ScheduledOper = isTop ? 1 : 0;
4033 unsigned UnscheduledOper = isTop ? 0 : 1;
4036 if (
MI->getOperand(ScheduledOper).getReg().isPhysical())
4041 if (
MI->getOperand(UnscheduledOper).getReg().isPhysical())
4042 return AtBoundary ? -1 : 1;
4045 if (
MI->isMoveImmediate()) {
4051 if (
Op.isReg() && !
Op.getReg().isPhysical()) {
4058 return isTop ? -1 : 1;
4061 if (BiasPRegsExtra && !isTop &&
MI->getNumExplicitDefs() == 1)
4064 return MI->getOperand(0).getReg().isPhysical();
4074 if (
tryGreater(TryCandPRegBias, CandPRegBias, TryCand, Cand,
4077 if (BiasPRegsExtra && Zone !=
nullptr && TryCandPRegBias &&
4078 TryCandPRegBias == CandPRegBias) {
4097 if (
DAG->isTrackingPressure()) {
4102 DAG->getRegionCriticalPSets(),
4103 DAG->getRegPressure().MaxSetPressure);
4108 &
DAG->getPressureDiff(Cand.
SU),
4110 DAG->getRegionCriticalPSets(),
4111 DAG->getRegPressure().MaxSetPressure);
4115 DAG->getPressureDiff(Cand.
SU),
4117 DAG->getRegionCriticalPSets(),
4118 DAG->getRegPressure().MaxSetPressure);
4123 <<
" Try " << *Cand.
SU <<
" "
4171 bool SameBoundary = Zone !=
nullptr;
4194 bool CandIsClusterSucc =
4196 bool TryCandIsClusterSucc =
4199 if (
tryGreater(TryCandIsClusterSucc, CandIsClusterSucc, TryCand, Cand,
4207 TryCand, Cand,
Weak))
4232 !
Rem.IsAcyclicLatencyLimited &&
tryLatency(TryCand, Cand, *Zone))
4259 for (
SUnit *SU : Q) {
4279 if (
SUnit *SU =
Bot.pickOnlyChoice()) {
4284 if (
SUnit *SU =
Top.pickOnlyChoice()) {
4301 BotCand.Policy != BotPolicy) {
4313 "Last pick result should correspond to re-picking right now");
4321 TopCand.Policy != TopPolicy) {
4333 "Last pick result should correspond to re-picking right now");
4348 IsTopNode = Cand.
AtTop;
4355 if (
DAG->top() ==
DAG->bottom()) {
4357 Bot.Available.empty() &&
Bot.Pending.empty() &&
"ReadyQ garbage");
4362 SU =
Top.pickOnlyChoice();
4375 SU =
Bot.pickOnlyChoice();
4408 Top.removeReady(SU);
4410 Bot.removeReady(SU);
4416 ++NumInstrsInSourceOrderPreRA;
4420 ++NumInstrsInSourceOrderPreRA;
4423 NumInstrsScheduledPreRA += 1;
4436 for (
SDep &Dep : Deps) {
4437 if (Dep.getKind() !=
SDep::Data || !Dep.getReg().isPhysical())
4439 SUnit *DepSU = Dep.getSUnit();
4440 if (isTop ? DepSU->
Succs.size() > 1 : DepSU->
Preds.size() > 1)
4443 if (!Copy->isCopy() && !Copy->isMoveImmediate())
4446 DAG->dumpNode(*Dep.getSUnit()));
4447 DAG->moveInstruction(Copy, InsertPos);
4465 dbgs() <<
" Top Cluster: ";
4466 for (
auto *
N : *TopCluster)
4467 dbgs() <<
N->NodeNum <<
'\t';
4480 dbgs() <<
" Bot Cluster: ";
4481 for (
auto *
N : *BotCluster)
4482 dbgs() <<
N->NodeNum <<
'\t';
4496static MachineSchedRegistry
4517 Top.HazardRec.reset(
DAG->TII->CreateTargetMIHazardRecognizer(Itin,
DAG));
4519 Bot.HazardRec.reset(
DAG->TII->CreateTargetMIHazardRecognizer(Itin,
DAG));
4555 Rem.CriticalPath =
DAG->ExitSU.getDepth();
4558 for (
const SUnit *SU :
Bot.Available) {
4564 errs() <<
"Critical Path(PGS-RR ): " <<
Rem.CriticalPath <<
" \n";
4583 Top.getLatencyStallCycles(Cand.
SU), TryCand, Cand,
Stall))
4589 bool CandIsClusterSucc =
4591 bool TryCandIsClusterSucc =
4594 if (
tryGreater(TryCandIsClusterSucc, CandIsClusterSucc, TryCand, Cand,
4627 for (
SUnit *SU : Q) {
4646 if (
SUnit *SU =
Bot.pickOnlyChoice()) {
4651 if (
SUnit *SU =
Top.pickOnlyChoice()) {
4668 BotCand.Policy != BotPolicy) {
4680 "Last pick result should correspond to re-picking right now");
4688 TopCand.Policy != TopPolicy) {
4700 "Last pick result should correspond to re-picking right now");
4715 IsTopNode = Cand.
AtTop;
4722 if (
DAG->top() ==
DAG->bottom()) {
4724 Bot.Available.empty() &&
Bot.Pending.empty() &&
"ReadyQ garbage");
4729 SU =
Bot.pickOnlyChoice();
4745 SU =
Top.pickOnlyChoice();
4766 Top.removeReady(SU);
4768 Bot.removeReady(SU);
4774 ++NumInstrsInSourceOrderPostRA;
4778 ++NumInstrsInSourceOrderPostRA;
4781 NumInstrsScheduledPostRA += 1;
4809 const BitVector *ScheduledTrees =
nullptr;
4812 ILPOrder(
bool MaxILP) : MaximizeILP(MaxILP) {}
4817 bool operator()(
const SUnit *
A,
const SUnit *
B)
const {
4820 if (SchedTreeA != SchedTreeB) {
4822 if (ScheduledTrees->
test(SchedTreeA) != ScheduledTrees->
test(SchedTreeB))
4823 return ScheduledTrees->
test(SchedTreeB);
4840class ILPScheduler :
public MachineSchedStrategy {
4841 ScheduleDAGMILive *DAG =
nullptr;
4844 std::vector<SUnit*> ReadyQ;
4847 ILPScheduler(
bool MaximizeILP) :
Cmp(MaximizeILP) {}
4849 void initialize(ScheduleDAGMI *dag)
override {
4851 DAG =
static_cast<ScheduleDAGMILive*
>(dag);
4858 void registerRoots()
override {
4860 std::make_heap(ReadyQ.begin(), ReadyQ.end(), Cmp);
4867 SUnit *pickNode(
bool &IsTopNode)
override {
4868 if (ReadyQ.empty())
return nullptr;
4869 std::pop_heap(ReadyQ.begin(), ReadyQ.end(), Cmp);
4870 SUnit *SU = ReadyQ.back();
4880 <<
"Scheduling " << *SU->
getInstr());
4885 void scheduleTree(
unsigned SubtreeID)
override {
4886 std::make_heap(ReadyQ.begin(), ReadyQ.end(), Cmp);
4891 void schedNode(SUnit *SU,
bool IsTopNode)
override {
4892 assert(!IsTopNode &&
"SchedDFSResult needs bottom-up");
4895 void releaseTopNode(SUnit *)
override { }
4897 void releaseBottomNode(SUnit *SU)
override {
4898 ReadyQ.push_back(SU);
4899 std::push_heap(ReadyQ.begin(), ReadyQ.end(), Cmp);
4926template<
bool IsReverse>
4930 return A->NodeNum >
B->NodeNum;
4932 return A->NodeNum <
B->NodeNum;
4937class InstructionShuffler :
public MachineSchedStrategy {
4944 PriorityQueue<SUnit*, std::vector<SUnit*>, SUnitOrder<false>>
4948 PriorityQueue<SUnit*, std::vector<SUnit*>, SUnitOrder<true>>
4952 InstructionShuffler(
bool alternate,
bool topdown)
4953 : IsAlternating(alternate), IsTopDown(topdown) {}
4963 SUnit *pickNode(
bool &IsTopNode)
override {
4967 if (TopQ.empty())
return nullptr;
4974 if (BottomQ.empty())
return nullptr;
4981 IsTopDown = !IsTopDown;
4985 void schedNode(SUnit *SU,
bool IsTopNode)
override {}
4987 void releaseTopNode(SUnit *SU)
override {
4990 void releaseBottomNode(SUnit *SU)
override {
5002 C, std::make_unique<InstructionShuffler>(Alternate, TopDown));
5006 "shuffle",
"Shuffle machine instructions alternating directions",
5014#if !defined(NDEBUG) && LLVM_ENABLE_ABI_BREAKING_CHECKS
5021struct llvm::DOTGraphTraits<ScheduleDAGMI *> :
public DefaultDOTGraphTraits {
5022 DOTGraphTraits(
bool isSimple =
false) : DefaultDOTGraphTraits(
isSimple) {}
5024 static std::string getGraphName(
const ScheduleDAG *
G) {
5025 return std::string(
G->MF.getName());
5028 static bool renderGraphFromBottomUp() {
5032 static bool isNodeHidden(
const SUnit *Node,
const ScheduleDAG *
G) {
5041 static std::string getEdgeAttributes(
const SUnit *Node,
5043 const ScheduleDAG *Graph) {
5045 return "color=cyan,style=dashed";
5047 return "color=blue,style=dashed";
5051 static std::string
getNodeLabel(
const SUnit *SU,
const ScheduleDAG *
G) {
5053 raw_string_ostream
SS(Str);
5054 const ScheduleDAGMI *DAG =
static_cast<const ScheduleDAGMI*
>(
G);
5056 static_cast<const ScheduleDAGMILive*
>(
G)->getDFSResult() :
nullptr;
5063 static std::string getNodeDescription(
const SUnit *SU,
const ScheduleDAG *
G) {
5064 return G->getGraphNodeLabel(SU);
5067 static std::string getNodeAttributes(
const SUnit *
N,
const ScheduleDAG *
G) {
5068 std::string Str(
"shape=Mrecord");
5069 const ScheduleDAGMI *DAG =
static_cast<const ScheduleDAGMI*
>(
G);
5071 static_cast<const ScheduleDAGMILive*
>(
G)->getDFSResult() :
nullptr;
5073 Str +=
",style=filled,fillcolor=\"#";
5086#if !defined(NDEBUG) && LLVM_ENABLE_ABI_BREAKING_CHECKS
5089 errs() <<
"ScheduleDAGMI::viewGraph is only available in debug builds on "
5090 <<
"systems with Graphviz or gv!\n";
5105 return A.first <
B.first;
5108unsigned ResourceSegments::getFirstAvailableAt(
5109 unsigned CurrCycle,
unsigned AcquireAtCycle,
unsigned ReleaseAtCycle,
5111 IntervalBuilder)
const {
5113 "Cannot execute on an un-sorted set of intervals.");
5117 if (AcquireAtCycle == ReleaseAtCycle)
5120 unsigned RetCycle = CurrCycle;
5122 IntervalBuilder(RetCycle, AcquireAtCycle, ReleaseAtCycle);
5123 for (
auto &
Interval : _Intervals) {
5130 "Invalid intervals configuration.");
5131 RetCycle += (unsigned)
Interval.second - (
unsigned)NewInterval.first;
5132 NewInterval = IntervalBuilder(RetCycle, AcquireAtCycle, ReleaseAtCycle);
5138 const unsigned CutOff) {
5139 assert(
A.first <=
A.second &&
"Cannot add negative resource usage");
5140 assert(CutOff > 0 &&
"0-size interval history has no use.");
5146 if (
A.first ==
A.second)
5153 "A resource is being overwritten");
5154 _Intervals.push_back(
A);
5160 while (_Intervals.size() > CutOff)
5161 _Intervals.pop_front();
5166 assert(
A.first <=
A.second &&
"Invalid interval");
5167 assert(
B.first <=
B.second &&
"Invalid interval");
5170 if ((
A.first ==
B.first) || (
A.second ==
B.second))
5175 if ((
A.first >
B.first) && (
A.second <
B.second))
5180 if ((
A.first >
B.first) && (
A.first <
B.second) && (
A.second >
B.second))
5185 if ((
A.first <
B.first) && (
B.first <
A.second) && (
B.second >
B.first))
5191void ResourceSegments::sortAndMerge() {
5192 if (_Intervals.size() <= 1)
5199 auto next = std::next(std::begin(_Intervals));
5200 auto E = std::end(_Intervals);
5201 for (; next != E; ++next) {
5202 if (std::prev(next)->second >= next->first) {
5203 next->first = std::prev(next)->first;
5204 _Intervals.erase(std::prev(next));
MachineInstrBuilder MachineInstrBuilder & DefMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
Function Alias Analysis false
static const Function * getParent(const Value *V)
This file implements the BitVector class.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
#define clEnumValN(ENUMVAL, FLAGNAME, DESC)
#define LLVM_DUMP_METHOD
Mark debug helper function definitions like dump() that should not be stripped from debug builds.
static std::optional< ArrayRef< InsnRange >::iterator > intersects(const MachineInstr *StartMI, const MachineInstr *EndMI, ArrayRef< InsnRange > Ranges, const InstructionOrdering &Ordering)
Check if the instruction range [StartMI, EndMI] intersects any instruction range in Ranges.
This file defines the DenseMap class.
Generic implementation of equivalence classes through the use Tarjan's efficient union-find algorithm...
const HexagonInstrInfo * TII
A common definition of LaneBitmask for use in TableGen and CodeGen.
static cl::opt< MISched::Direction > PostRADirection("misched-postra-direction", cl::Hidden, cl::desc("Post reg-alloc list scheduling direction"), cl::init(MISched::Unspecified), cl::values(clEnumValN(MISched::TopDown, "topdown", "Force top-down post reg-alloc list scheduling"), clEnumValN(MISched::BottomUp, "bottomup", "Force bottom-up post reg-alloc list scheduling"), clEnumValN(MISched::Bidirectional, "bidirectional", "Force bidirectional post reg-alloc list scheduling")))
static bool isSchedBoundary(MachineBasicBlock::iterator MI, MachineBasicBlock *MBB, MachineFunction *MF, const TargetInstrInfo *TII)
Return true of the given instruction should not be included in a scheduling region.
static MachineSchedRegistry ILPMaxRegistry("ilpmax", "Schedule bottom-up for max ILP", createILPMaxScheduler)
static cl::opt< bool > EnableMemOpCluster("misched-cluster", cl::Hidden, cl::desc("Enable memop clustering."), cl::init(true))
PostRA Machine Instruction Scheduler
static MachineBasicBlock::const_iterator nextIfDebug(MachineBasicBlock::const_iterator I, MachineBasicBlock::const_iterator End)
If this iterator is a debug value, increment until reaching the End or a non-debug instruction.
static const unsigned MinSubtreeSize
static cl::opt< bool > EnableSSAMachineSched("enable-ssa-misched", cl::desc("Enable the machine instruction scheduling pass in SSA."), cl::init(false), cl::Hidden)
static cl::opt< bool > VerifyScheduling("verify-misched", cl::Hidden, cl::desc("Verify machine instrs before and after machine scheduling"))
static const unsigned InvalidCycle
static cl::opt< bool > MISchedSortResourcesInTrace("misched-sort-resources-in-trace", cl::Hidden, cl::init(true), cl::desc("Sort the resources printed in the dump trace"))
static cl::opt< bool > EnableCyclicPath("misched-cyclicpath", cl::Hidden, cl::desc("Enable cyclic critical path analysis."), cl::init(true))
static MachineBasicBlock::const_iterator priorNonDebug(MachineBasicBlock::const_iterator I, MachineBasicBlock::const_iterator Beg)
Decrement this iterator until reaching the top or a non-debug instr.
static cl::opt< MachineSchedRegistry::ScheduleDAGCtor, false, RegisterPassParser< MachineSchedRegistry > > MachineSchedOpt("misched", cl::init(&useDefaultMachineSched), cl::Hidden, cl::desc("Machine instruction scheduler to use"))
MachineSchedOpt allows command line selection of the scheduler.
static cl::opt< bool > EnableMachineSched("enable-misched", cl::desc("Enable the machine instruction scheduling pass."), cl::init(true), cl::Hidden)
static cl::opt< unsigned > MISchedCutoff("misched-cutoff", cl::Hidden, cl::desc("Stop scheduling after N instructions"), cl::init(~0U))
static cl::opt< unsigned > SchedOnlyBlock("misched-only-block", cl::Hidden, cl::desc("Only schedule this MBB#"))
static cl::opt< bool > EnableRegPressure("misched-regpressure", cl::Hidden, cl::desc("Enable register pressure scheduling."), cl::init(true))
static MachineSchedRegistry GenericSchedRegistry("converge", "Standard converging scheduler.", createConvergingSched)
static cl::opt< unsigned > HeaderColWidth("misched-dump-schedule-trace-col-header-width", cl::Hidden, cl::desc("Set width of the columns with " "the resources and schedule units"), cl::init(19))
static cl::opt< bool > ForceFastCluster("force-fast-cluster", cl::Hidden, cl::desc("Switch to fast cluster algorithm with the lost " "of some fusion opportunities"), cl::init(false))
static cl::opt< unsigned > FastClusterThreshold("fast-cluster-threshold", cl::Hidden, cl::desc("The threshold for fast cluster"), cl::init(1000))
static bool checkResourceLimit(unsigned LFactor, unsigned Count, unsigned Latency, bool AfterSchedNode)
Given a Count of resource usage and a Latency value, return true if a SchedBoundary becomes resource ...
static ScheduleDAGInstrs * createInstructionShuffler(MachineSchedContext *C)
static ScheduleDAGInstrs * useDefaultMachineSched(MachineSchedContext *C)
A dummy default scheduler factory indicates whether the scheduler is overridden on the command line.
static bool sortIntervals(const ResourceSegments::IntervalTy &A, const ResourceSegments::IntervalTy &B)
Sort predicate for the intervals stored in an instance of ResourceSegments.
static cl::opt< unsigned > ColWidth("misched-dump-schedule-trace-col-width", cl::Hidden, cl::desc("Set width of the columns showing resource booking."), cl::init(5))
static cl::opt< MISched::Direction > PreRADirection("misched-prera-direction", cl::Hidden, cl::desc("Pre reg-alloc list scheduling direction"), cl::init(MISched::Unspecified), cl::values(clEnumValN(MISched::TopDown, "topdown", "Force top-down pre reg-alloc list scheduling"), clEnumValN(MISched::BottomUp, "bottomup", "Force bottom-up pre reg-alloc list scheduling"), clEnumValN(MISched::Bidirectional, "bidirectional", "Force bidirectional pre reg-alloc list scheduling")))
static MachineSchedRegistry DefaultSchedRegistry("default", "Use the target's default scheduler choice.", useDefaultMachineSched)
static cl::opt< std::string > SchedOnlyFunc("misched-only-func", cl::Hidden, cl::desc("Only schedule this function"))
static const char * scheduleTableLegend
static ScheduleDAGInstrs * createConvergingSched(MachineSchedContext *C)
static cl::opt< bool > MischedDetailResourceBooking("misched-detail-resource-booking", cl::Hidden, cl::init(false), cl::desc("Show details of invoking getNextResoufceCycle."))
static cl::opt< unsigned > ViewMISchedCutoff("view-misched-cutoff", cl::Hidden, cl::desc("Hide nodes with more predecessor/successor than cutoff"))
In some situations a few uninteresting nodes depend on nearly all other nodes in the graph,...
static MachineSchedRegistry ShufflerRegistry("shuffle", "Shuffle machine instructions alternating directions", createInstructionShuffler)
static void tracePick(const SUnit *SU, const GenericSchedulerBase::CandReason Reason, const bool IsTop, const bool IsPostRA=false)
static cl::opt< bool > EnablePostRAMachineSched("enable-post-misched", cl::desc("Enable the post-ra machine instruction scheduling pass."), cl::init(true), cl::Hidden)
static void getSchedRegions(MachineBasicBlock *MBB, MBBRegionsVector &Regions, bool RegionsTopDown)
static cl::opt< unsigned > MIResourceCutOff("misched-resource-cutoff", cl::Hidden, cl::desc("Number of intervals to track"), cl::init(10))
static ScheduleDAGInstrs * createILPMaxScheduler(MachineSchedContext *C)
SmallVector< SchedRegion, 16 > MBBRegionsVector
static cl::opt< bool > MISchedDumpReservedCycles("misched-dump-reserved-cycles", cl::Hidden, cl::init(false), cl::desc("Dump resource usage at schedule boundary."))
static cl::opt< unsigned > ReadyListLimit("misched-limit", cl::Hidden, cl::desc("Limit ready list to N instructions"), cl::init(256))
Avoid quadratic complexity in unusually large basic blocks by limiting the size of the ready lists.
static cl::opt< bool > DumpCriticalPathLength("misched-dcpl", cl::Hidden, cl::desc("Print critical path length to stdout"))
static ScheduleDAGInstrs * createILPMinScheduler(MachineSchedContext *C)
static cl::opt< bool > MISchedDumpScheduleTrace("misched-dump-schedule-trace", cl::Hidden, cl::init(false), cl::desc("Dump resource usage at schedule boundary."))
static MachineSchedRegistry ILPMinRegistry("ilpmin", "Schedule bottom-up for min ILP", createILPMinScheduler)
Register const TargetRegisterInfo * TRI
std::pair< uint64_t, uint64_t > Interval
static std::string getNodeLabel(const ValueInfo &VI, GlobalValueSummary *GVS)
FunctionAnalysisManager FAM
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
This file defines the PriorityQueue class.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
static void initialize(TargetLibraryInfoImpl &TLI, const Triple &T, const llvm::StringTable &StandardNames, VectorLibrary VecLib)
Initialize the set of available library functions based on the specified target triple.
This file describes how to lower LLVM code to machine code.
Target-Independent Code Generator Pass Configuration Options pass.
static const X86InstrFMA3Group Groups[]
Class recording the (high level) value of a variable.
A manager for alias analyses.
A wrapper pass to provide the legacy pass manager access to a suitably prepared AAResults object.
Class for arbitrary precision integers.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Represent a constant reference to an array (0 or more elements consecutively in memory),...
reverse_iterator rend() const
size_t size() const
Get the array size.
reverse_iterator rbegin() const
bool test(unsigned Idx) const
Returns true if bit Idx is set.
Represents analyses that only rely on functions' control flow.
size_type count(const_arg_type_t< KeyT > Val) const
Return 1 if the specified key is in the map, 0 otherwise.
iterator find(const_arg_type_t< KeyT > Val)
The EquivalenceClasses data structure is just a set of these.
This represents a collection of equivalence classes and supports three efficient operations: insert a...
iterator_range< member_iterator > members(const ECValue &ECV) const
member_iterator unionSets(const ElemTy &V1, const ElemTy &V2)
Merge the two equivalence sets for the specified values, inserting them if they do not already exist ...
void traceCandidate(const SchedCandidate &Cand)
LLVM_ABI void setPolicy(CandPolicy &Policy, bool IsPostRA, SchedBoundary &CurrZone, SchedBoundary *OtherZone)
Set the CandPolicy given a scheduling zone given the current resources and latencies inside and outsi...
MachineSchedPolicy RegionPolicy
const TargetSchedModel * SchedModel
static const char * getReasonStr(GenericSchedulerBase::CandReason Reason)
const MachineSchedContext * Context
CandReason
Represent the type of SchedCandidate found within a single queue.
const TargetRegisterInfo * TRI
void checkAcyclicLatency()
Set IsAcyclicLatencyLimited if the acyclic path is longer than the cyclic critical path by more cycle...
SchedCandidate BotCand
Candidate last picked from Bot boundary.
SchedCandidate TopCand
Candidate last picked from Top boundary.
virtual bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const
Apply a set of heuristics to a new candidate.
void dumpPolicy() const override
void initialize(ScheduleDAGMI *dag) override
Initialize the strategy after building the DAG for a new region.
void initCandidate(SchedCandidate &Cand, SUnit *SU, bool AtTop, const RegPressureTracker &RPTracker, RegPressureTracker &TempTracker)
void registerRoots() override
Notify this strategy that all roots have been released (including those that depend on EntrySU or Exi...
void initPolicy(MachineBasicBlock::iterator Begin, MachineBasicBlock::iterator End, unsigned NumRegionInstrs) override
Initialize the per-region scheduling policy.
void reschedulePhysReg(SUnit *SU, bool isTop)
SUnit * pickNode(bool &IsTopNode) override
Pick the best node to balance the schedule. Implements MachineSchedStrategy.
void pickNodeFromQueue(SchedBoundary &Zone, const CandPolicy &ZonePolicy, const RegPressureTracker &RPTracker, SchedCandidate &Candidate)
Pick the best candidate from the queue.
void schedNode(SUnit *SU, bool IsTopNode) override
Update the scheduler's state after scheduling a node.
SUnit * pickNodeBidirectional(bool &IsTopNode)
Pick the best candidate node from either the top or bottom queue.
bool getMemOperandsWithOffsetWidth(const MachineInstr &LdSt, SmallVectorImpl< const MachineOperand * > &BaseOps, int64_t &Offset, bool &OffsetIsScalable, LocationSize &Width) const override
Get the base register and byte offset of a load/store instr.
Itinerary data supplied by a subtarget to be used by a target.
LiveInterval - This class represents the liveness of a register, or stack slot.
MachineInstr * getInstructionFromIndex(SlotIndex index) const
Returns the instruction associated with the given index.
SlotIndex getInstructionIndex(const MachineInstr &Instr) const
Returns the base index of the given instruction.
LiveInterval & getInterval(Register Reg)
Result of a LiveRange query.
VNInfo * valueIn() const
Return the value that is live-in to the instruction.
Segments::iterator iterator
LiveQueryResult Query(SlotIndex Idx) const
Query Liveness at Idx.
VNInfo * getVNInfoBefore(SlotIndex Idx) const
getVNInfoBefore - Return the VNInfo that is live up to but not necessarily including Idx,...
SlotIndex beginIndex() const
beginIndex - Return the lowest numbered slot covered.
SlotIndex endIndex() const
endNumber - return the maximum point of the range of the whole, exclusive.
bool isLocal(SlotIndex Start, SlotIndex End) const
True iff this segment is a single segment that lies between the specified boundaries,...
LLVM_ABI iterator find(SlotIndex Pos)
find - Return an iterator pointing to the first segment that ends after Pos, or end().
static LocationSize precise(uint64_t Value)
MachineInstrBundleIterator< const MachineInstr > const_iterator
MachineInstrBundleIterator< MachineInstr > iterator
MachineBlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate machine basic b...
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
int64_t getObjectOffset(int ObjectIdx) const
Return the assigned stack offset of the specified object from the incoming stack pointer.
bool isFixedObjectIndex(int ObjectIdx) const
Returns true if the specified index corresponds to a fixed stack object.
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
void print(raw_ostream &OS, const SlotIndexes *=nullptr) const
print - Print out the MachineFunction in a format suitable for debugging to the specified stream.
nonconst_iterator getNonConstIterator() const
Representation of each machine instruction.
bool mayLoad(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read memory.
bool mayStore(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly modify memory.
Analysis pass that exposes the MachineLoopInfo for a machine function.
MachineOperand class - Representation of each machine instruction operand.
MachinePassRegistry - Track the registration of machine passes.
MachineSchedRegistry provides a selection of available machine instruction schedulers.
static LLVM_ABI MachinePassRegistry< ScheduleDAGCtor > Registry
ScheduleDAGInstrs *(*)(MachineSchedContext *) ScheduleDAGCtor
LLVM_ABI PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
LLVM_ABI MachineSchedulerPass(const TargetMachine *TM)
LLVM_ABI ~MachineSchedulerPass()
static LLVM_ABI PassRegistry * getPassRegistry()
getPassRegistry - Access the global registry object, which is automatically initialized at applicatio...
void initPolicy(MachineBasicBlock::iterator Begin, MachineBasicBlock::iterator End, unsigned NumRegionInstrs) override
Optionally override the per-region scheduling policy.
virtual bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand)
Apply a set of heuristics to a new candidate for PostRA scheduling.
void schedNode(SUnit *SU, bool IsTopNode) override
Called after ScheduleDAGMI has scheduled an instruction and updated scheduled/remaining flags in the ...
SchedCandidate BotCand
Candidate last picked from Bot boundary.
void pickNodeFromQueue(SchedBoundary &Zone, SchedCandidate &Cand)
void initialize(ScheduleDAGMI *Dag) override
Initialize the strategy after building the DAG for a new region.
SchedCandidate TopCand
Candidate last picked from Top boundary.
SUnit * pickNodeBidirectional(bool &IsTopNode)
Pick the best candidate node from either the top or bottom queue.
void registerRoots() override
Notify this strategy that all roots have been released (including those that depend on EntrySU or Exi...
SUnit * pickNode(bool &IsTopNode) override
Pick the next node to schedule.
LLVM_ABI PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
LLVM_ABI PostMachineSchedulerPass(const TargetMachine *TM)
LLVM_ABI ~PostMachineSchedulerPass()
A set of analyses that are preserved following a run of a transformation pass.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
PreservedAnalyses & preserveSet()
Mark an analysis set as preserved.
Capture a change in pressure for a single pressure set.
unsigned getPSetOrMax() const
List of PressureChanges in order of increasing, unique PSetID.
LLVM_ABI void dump(const TargetRegisterInfo &TRI) const
LLVM_ABI void addPressureChange(VirtRegOrUnit VRegOrUnit, bool IsDec, const MachineRegisterInfo *MRI)
Add a change in pressure to the pressure diff of a given instruction.
void clear()
clear - Erase all elements from the queue.
Helpers for implementing custom MachineSchedStrategy classes.
ArrayRef< SUnit * > elements()
LLVM_ABI void dump() const
std::vector< SUnit * >::iterator iterator
StringRef getName() const
Track the current register pressure at some position in the instruction stream, and remember the high...
LLVM_ABI void getMaxUpwardPressureDelta(const MachineInstr *MI, PressureDiff *PDiff, RegPressureDelta &Delta, ArrayRef< PressureChange > CriticalPSets, ArrayRef< unsigned > MaxPressureLimit)
Consider the pressure increase caused by traversing this instruction bottom-up.
LLVM_ABI void getMaxDownwardPressureDelta(const MachineInstr *MI, RegPressureDelta &Delta, ArrayRef< PressureChange > CriticalPSets, ArrayRef< unsigned > MaxPressureLimit)
Consider the pressure increase caused by traversing this instruction top-down.
LLVM_ABI void getUpwardPressureDelta(const MachineInstr *MI, PressureDiff &PDiff, RegPressureDelta &Delta, ArrayRef< PressureChange > CriticalPSets, ArrayRef< unsigned > MaxPressureLimit) const
This is the fast version of querying register pressure that does not directly depend on current liven...
List of registers defined and used by a machine instruction.
LLVM_ABI void detectDeadDefs(const MachineInstr &MI, LiveIntervals &LIS, const MachineRegisterInfo &MRI)
Use liveness information to find dead defs at MI's dead slot not marked with a dead flag and move the...
LLVM_ABI void adjustLaneLiveness(LiveIntervals &LIS, const MachineRegisterInfo &MRI, SlotIndex Pos)
Use liveness information to find out which uses/defs are partially undefined/dead at Pos and adjust t...
LLVM_ABI void collect(const MachineInstr &MI, const TargetRegisterInfo &TRI, const MachineRegisterInfo &MRI, bool TrackLaneMasks, bool IgnoreDead)
Analyze the given instruction MI and fill in the Uses, Defs and DeadDefs list based on the MachineOpe...
RegisterPassParser class - Handle the addition of new machine passes.
Wrapper class representing virtual and physical registers.
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
LLVM_ABI void add(IntervalTy A, const unsigned CutOff=10)
Adds an interval [a, b) to the collection of the instance.
static IntervalTy getResourceIntervalBottom(unsigned C, unsigned AcquireAtCycle, unsigned ReleaseAtCycle)
These function return the interval used by a resource in bottom and top scheduling.
static LLVM_ABI bool intersects(IntervalTy A, IntervalTy B)
Checks whether intervals intersect.
std::pair< int64_t, int64_t > IntervalTy
Represents an interval of discrete integer values closed on the left and open on the right: [a,...
static IntervalTy getResourceIntervalTop(unsigned C, unsigned AcquireAtCycle, unsigned ReleaseAtCycle)
Kind getKind() const
Returns an enum value representing the kind of the dependence.
@ Anti
A register anti-dependence (aka WAR).
@ Data
Regular data dependence (aka true-dependence).
bool isWeak() const
Tests if this a weak dependence.
@ Cluster
Weak DAG edge linking a chain of clustered instrs.
@ Artificial
Arbitrary strong DAG edge (no real dependence).
@ Weak
Arbitrary weak DAG edge.
unsigned getLatency() const
Returns the latency value for this edge, which roughly means the minimum number of cycles that must e...
bool isArtificial() const
Tests if this is an Order dependence that is marked as "artificial", meaning it isn't necessary for c...
bool isCtrl() const
Shorthand for getKind() != SDep::Data.
Register getReg() const
Returns the register associated with this edge.
LLVM_ABI ~SSAMachineSchedulerPass()
LLVM_ABI PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
LLVM_ABI SSAMachineSchedulerPass(const TargetMachine *TM)
bool isArtificialDep() const
bool isCtrlDep() const
Tests if this is not an SDep::Data dependence.
Scheduling unit. This is a node in the scheduling DAG.
bool isCall
Is a function call.
unsigned TopReadyCycle
Cycle relative to start when node is ready.
unsigned NodeNum
Entry # of node in the node vector.
bool isUnbuffered
Uses an unbuffered resource.
unsigned getHeight() const
Returns the height of this node, which is the length of the maximum path down to any node which has n...
unsigned short Latency
Node latency.
unsigned getDepth() const
Returns the depth of this node, which is the length of the maximum path up to any node which has no p...
bool isScheduled
True once scheduled.
unsigned ParentClusterIdx
The parent cluster id.
bool hasPhysRegDefs
Has physreg defs that are being used.
unsigned BotReadyCycle
Cycle relative to end when node is ready.
SmallVector< SDep, 4 > Succs
All sunit successors.
bool hasReservedResource
Uses a reserved resource.
bool isBottomReady() const
bool hasPhysRegUses
Has physreg uses.
SmallVector< SDep, 4 > Preds
All sunit predecessors.
MachineInstr * getInstr() const
Returns the representative MachineInstr for this SUnit.
Each Scheduling boundary is associated with ready queues.
LLVM_ABI unsigned getNextResourceCycleByInstance(unsigned InstanceIndex, unsigned ReleaseAtCycle, unsigned AcquireAtCycle)
Compute the next cycle at which the given processor resource unit can be scheduled.
LLVM_ABI void releasePending()
Release pending ready nodes in to the available queue.
unsigned getDependentLatency() const
bool isReservedGroup(unsigned PIdx) const
unsigned getScheduledLatency() const
Get the number of latency cycles "covered" by the scheduled instructions.
LLVM_ABI void incExecutedResources(unsigned PIdx, unsigned Count)
bool isResourceLimited() const
const TargetSchedModel * SchedModel
unsigned getExecutedCount() const
Get a scaled count for the minimum execution time of the scheduled micro-ops that are ready to execut...
LLVM_ABI unsigned getLatencyStallCycles(SUnit *SU)
Get the difference between the given SUnit's ready time and the current cycle.
LLVM_ABI unsigned findMaxLatency(ArrayRef< SUnit * > ReadySUs)
LLVM_ABI void dumpReservedCycles() const
Dump the state of the information that tracks resource usage.
LLVM_ABI unsigned getOtherResourceCount(unsigned &OtherCritIdx)
LLVM_ABI void bumpNode(SUnit *SU)
Move the boundary of scheduled code by one SUnit.
unsigned getCriticalCount() const
Get the scaled count of scheduled micro-ops and resources, including executed resources.
LLVM_ABI SUnit * pickOnlyChoice()
Call this before applying any other heuristics to the Available queue.
LLVM_ABI void releaseNode(SUnit *SU, unsigned ReadyCycle, bool InPQueue, unsigned Idx=0)
Release SU to make it ready.
LLVM_ABI unsigned countResource(const MCSchedClassDesc *SC, unsigned PIdx, unsigned Cycles, unsigned ReadyCycle, unsigned StartAtCycle)
Add the given processor resource to this scheduled zone.
LLVM_ABI ~SchedBoundary()
LLVM_ABI void init(ScheduleDAGMI *dag, const TargetSchedModel *smodel, SchedRemainder *rem)
unsigned getResourceCount(unsigned ResIdx) const
LLVM_ABI void bumpCycle(unsigned NextCycle)
Move the boundary of scheduled code by one cycle.
unsigned getCurrMOps() const
Micro-ops issued in the current cycle.
unsigned getCurrCycle() const
Number of cycles to issue the instructions scheduled in this zone.
std::unique_ptr< ScheduleHazardRecognizer > HazardRec
LLVM_ABI bool checkHazard(SUnit *SU)
Does this SU have a hazard within the current instruction group.
LLVM_ABI std::pair< unsigned, unsigned > getNextResourceCycle(const MCSchedClassDesc *SC, unsigned PIdx, unsigned ReleaseAtCycle, unsigned AcquireAtCycle)
Compute the next cycle at which the given processor resource can be scheduled.
LLVM_ABI void dumpScheduledState() const
LLVM_ABI void removeReady(SUnit *SU)
Remove SU from the ready set for this boundary.
unsigned getZoneCritResIdx() const
unsigned getUnscheduledLatency(SUnit *SU) const
Compute the values of each DAG node for various metrics during DFS.
unsigned getNumInstrs(const SUnit *SU) const
Get the number of instructions in the given subtree and its children.
unsigned getSubtreeID(const SUnit *SU) const
Get the ID of the subtree the given DAG node belongs to.
ILPValue getILP(const SUnit *SU) const
Get the ILP value for a DAG node.
unsigned getSubtreeLevel(unsigned SubtreeID) const
Get the connection level of a subtree.
A ScheduleDAG for scheduling lists of MachineInstr.
SmallVector< ClusterInfo > & getClusters()
Returns the array of the clusters.
virtual void finishBlock()
Cleans up after scheduling in the given block.
MachineBasicBlock::iterator end() const
Returns an iterator to the bottom of the current scheduling region.
std::string getDAGName() const override
Returns a label for the region of code covered by the DAG.
MachineBasicBlock * BB
The block in which to insert instructions.
MachineInstr * FirstDbgValue
virtual void startBlock(MachineBasicBlock *BB)
Prepares to perform scheduling in the given block.
MachineBasicBlock::iterator RegionEnd
The end of the range to be scheduled.
const MCSchedClassDesc * getSchedClass(SUnit *SU) const
Resolves and cache a resolved scheduling class for an SUnit.
DbgValueVector DbgValues
Remember instruction that precedes DBG_VALUE.
bool addEdge(SUnit *SuccSU, const SDep &PredDep)
Add a DAG edge to the given SU with the given predecessor dependence data.
DumpDirection
The direction that should be used to dump the scheduled Sequence.
bool TrackLaneMasks
Whether lane masks should get tracked.
void dumpNode(const SUnit &SU) const override
bool IsReachable(SUnit *SU, SUnit *TargetSU)
IsReachable - Checks if SU is reachable from TargetSU.
MachineBasicBlock::iterator begin() const
Returns an iterator to the top of the current scheduling region.
void buildSchedGraph(AAResults *AA, RegPressureTracker *RPTracker=nullptr, PressureDiffs *PDiffs=nullptr, LiveIntervals *LIS=nullptr, bool TrackLaneMasks=false)
Builds SUnits for the current region.
SUnit * getSUnit(MachineInstr *MI) const
Returns an existing SUnit for this MI, or nullptr.
TargetSchedModel SchedModel
TargetSchedModel provides an interface to the machine model.
bool canAddEdge(SUnit *SuccSU, SUnit *PredSU)
True if an edge can be added from PredSU to SuccSU without creating a cycle.
MachineBasicBlock::iterator RegionBegin
The beginning of the range to be scheduled.
virtual void enterRegion(MachineBasicBlock *bb, MachineBasicBlock::iterator begin, MachineBasicBlock::iterator end, unsigned regioninstrs)
Initialize the DAG and common scheduler state for a new scheduling region.
void dump() const override
void setDumpDirection(DumpDirection D)
ScheduleDAGMILive is an implementation of ScheduleDAGInstrs that schedules machine instructions while...
void scheduleMI(SUnit *SU, bool IsTopNode)
Move an instruction and update register pressure.
void schedule() override
Implement ScheduleDAGInstrs interface for scheduling a sequence of reorderable instructions.
VReg2SUnitMultiMap VRegUses
Maps vregs to the SUnits of their uses in the current scheduling region.
void computeDFSResult()
Compute a DFSResult after DAG building is complete, and before any queue comparisons.
PressureDiff & getPressureDiff(const SUnit *SU)
SchedDFSResult * DFSResult
Information about DAG subtrees.
void enterRegion(MachineBasicBlock *bb, MachineBasicBlock::iterator begin, MachineBasicBlock::iterator end, unsigned regioninstrs) override
Implement the ScheduleDAGInstrs interface for handling the next scheduling region.
void initQueues(ArrayRef< SUnit * > TopRoots, ArrayRef< SUnit * > BotRoots)
Release ExitSU predecessors and setup scheduler queues.
bool ShouldTrackLaneMasks
RegPressureTracker BotRPTracker
void buildDAGWithRegPressure()
Call ScheduleDAGInstrs::buildSchedGraph with register pressure tracking enabled.
std::vector< PressureChange > RegionCriticalPSets
List of pressure sets that exceed the target's pressure limit before scheduling, listed in increasing...
void updateScheduledPressure(const SUnit *SU, const std::vector< unsigned > &NewMaxPressure)
PressureDiffs SUPressureDiffs
unsigned computeCyclicCriticalPath()
Compute the cyclic critical path through the DAG.
void updatePressureDiffs(ArrayRef< VRegMaskOrUnit > LiveUses)
Update the PressureDiff array for liveness after scheduling this instruction.
void collectVRegUses(SUnit &SU)
RegisterClassInfo * RegClassInfo
const SchedDFSResult * getDFSResult() const
Return a non-null DFS result if the scheduling strategy initialized it.
RegPressureTracker RPTracker
bool ShouldTrackPressure
Register pressure in this region computed by initRegPressure.
~ScheduleDAGMILive() override
void dump() const override
BitVector & getScheduledTrees()
MachineBasicBlock::iterator LiveRegionEnd
RegPressureTracker TopRPTracker
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
void dumpSchedule() const
dump the scheduled Sequence.
std::unique_ptr< MachineSchedStrategy > SchedImpl
void startBlock(MachineBasicBlock *bb) override
Prepares to perform scheduling in the given block.
void releasePred(SUnit *SU, SDep *PredEdge)
ReleasePred - Decrement the NumSuccsLeft count of a predecessor.
void initQueues(ArrayRef< SUnit * > TopRoots, ArrayRef< SUnit * > BotRoots)
Release ExitSU predecessors and setup scheduler queues.
void moveInstruction(MachineInstr *MI, MachineBasicBlock::iterator InsertPos)
Change the position of an instruction within the basic block and update live ranges and region bounda...
void releasePredecessors(SUnit *SU)
releasePredecessors - Call releasePred on each of SU's predecessors.
void postProcessDAG()
Apply each ScheduleDAGMutation step in order.
void dumpScheduleTraceTopDown() const
Print execution trace of the schedule top-down or bottom-up.
void schedule() override
Implement ScheduleDAGInstrs interface for scheduling a sequence of reorderable instructions.
void findRootsAndBiasEdges(SmallVectorImpl< SUnit * > &TopRoots, SmallVectorImpl< SUnit * > &BotRoots)
MachineBasicBlock::iterator CurrentBottom
The bottom of the unscheduled zone.
virtual bool hasVRegLiveness() const
Return true if this DAG supports VReg liveness and RegPressure.
void enterRegion(MachineBasicBlock *bb, MachineBasicBlock::iterator begin, MachineBasicBlock::iterator end, unsigned regioninstrs) override
Implement the ScheduleDAGInstrs interface for handling the next scheduling region.
LiveIntervals * getLIS() const
void viewGraph(const Twine &Name, const Twine &Title) override
viewGraph - Pop up a ghostview window with the reachable parts of the DAG rendered using 'dot'.
void viewGraph() override
Out-of-line implementation with no arguments is handy for gdb.
void releaseSucc(SUnit *SU, SDep *SuccEdge)
ReleaseSucc - Decrement the NumPredsLeft count of a successor.
void dumpScheduleTraceBottomUp() const
~ScheduleDAGMI() override
void finishBlock() override
Cleans up after scheduling in the given block.
void updateQueues(SUnit *SU, bool IsTopNode)
Update scheduler DAG and queues after scheduling an instruction.
void placeDebugValues()
Reinsert debug_values recorded in ScheduleDAGInstrs::DbgValues.
MachineBasicBlock::iterator CurrentTop
The top of the unscheduled zone.
void releaseSuccessors(SUnit *SU)
releaseSuccessors - Call releaseSucc on each of SU's successors.
std::vector< std::unique_ptr< ScheduleDAGMutation > > Mutations
Ordered list of DAG postprocessing steps.
Mutate the DAG as a postpass after normal DAG building.
MachineRegisterInfo & MRI
Virtual/real register map.
std::vector< SUnit > SUnits
The scheduling units.
const TargetRegisterInfo * TRI
Target processor register info.
SUnit EntrySU
Special node for the region entry.
MachineFunction & MF
Machine function.
void dumpNodeAll(const SUnit &SU) const
SUnit ExitSU
Special node for the region exit.
static bool isSameInstr(SlotIndex A, SlotIndex B)
isSameInstr - Return true if A and B refer to the same instruction.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
std::reverse_iterator< const_iterator > const_reverse_iterator
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
iterator_base< SparseMultiSet * > iterator
Information about stack frame layout on the target.
StackDirection getStackGrowthDirection() const
getStackGrowthDirection - Return the direction the stack grows
TargetInstrInfo - Interface to description of machine instruction set.
virtual const TargetRegisterClass * getRegClassFor(MVT VT, bool isDivergent=false) const
Return the register class that should be used for the specified value type.
bool isTypeLegal(EVT VT) const
Return true if the target has native support for the specified value type.
This class defines information used to lower LLVM code to legal SelectionDAG operators that the targe...
Primary interface to the complete machine description for the target machine.
Target-Independent Code Generator Pass Configuration Options.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
Provide an instruction scheduling machine model to CodeGen passes.
unsigned getMicroOpFactor() const
Multiply number of micro-ops by this factor to normalize it relative to other resources.
ProcResIter getWriteProcResEnd(const MCSchedClassDesc *SC) const
LLVM_ABI bool hasInstrSchedModel() const
Return true if this machine model includes an instruction-level scheduling model.
const MCWriteProcResEntry * ProcResIter
unsigned getResourceFactor(unsigned ResIdx) const
Multiply the number of units consumed for a resource by this factor to normalize it relative to other...
LLVM_ABI unsigned getNumMicroOps(const MachineInstr *MI, const MCSchedClassDesc *SC=nullptr) const
Return the number of issue slots required for this MI.
unsigned getNumProcResourceKinds() const
Get the number of kinds of resources for this target.
ProcResIter getWriteProcResBegin(const MCSchedClassDesc *SC) const
virtual void overridePostRASchedPolicy(MachineSchedPolicy &Policy, const SchedRegion &Region) const
Override generic post-ra scheduling policy within a region.
virtual void overrideSchedPolicy(MachineSchedPolicy &Policy, const SchedRegion &Region) const
Override generic scheduling policy within a region.
virtual bool enableMachineScheduler() const
True if the subtarget should run MachineScheduler after aggressive coalescing.
virtual bool enableSSAMachineScheduler() const
True if the subtarget should run a machine scheduler before PHI elimination.
virtual bool enablePostRAMachineScheduler() const
True if the subtarget should run a machine scheduler after register allocation.
virtual const TargetFrameLowering * getFrameLowering() const
virtual const TargetInstrInfo * getInstrInfo() const
virtual const TargetLowering * getTargetLowering() const
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
VNInfo - Value Number Information.
SlotIndex def
The index of the defining instruction.
bool isPHIDef() const
Returns true if this value is defined by a PHI instruction (or was, PHI instructions may have been el...
Wrapper class representing a virtual register or register unit.
int getNumOccurrences() const
Base class for the machine scheduler classes.
void scheduleRegions(ScheduleDAGInstrs &Scheduler, bool FixKillFlags)
Main driver for both MachineScheduler and PostMachineScheduler.
Impl class for MachineScheduler.
void setMFAM(MachineFunctionAnalysisManager *MFAM)
void setLegacyPass(MachineFunctionPass *P)
bool run(MachineFunction &MF, const TargetMachine &TM, const RequiredAnalyses &Analyses)
MachineSchedulerImpl()=default
ScheduleDAGInstrs * createMachineScheduler()
Instantiate a ScheduleDAGInstrs that will be owned by the caller.
Impl class for PostMachineScheduler.
bool run(MachineFunction &Func, const TargetMachine &TM, const RequiredAnalyses &Analyses)
void setMFAM(MachineFunctionAnalysisManager *MFAM)
ScheduleDAGInstrs * createPostMachineScheduler()
Instantiate a ScheduleDAGInstrs for PostRA scheduling that will be owned by the caller.
void setLegacyPass(MachineFunctionPass *P)
PostMachineSchedulerImpl()=default
Impl class for SSAMachineScheduler.
void setLegacyPass(MachineFunctionPass *P)
bool run(MachineFunction &MF, const TargetMachine &TM, const RequiredAnalyses &Analyses)
ScheduleDAGInstrs * createMachineScheduler()
Instantiate a ScheduleDAGInstrs that will be owned by the caller.
void setMFAM(MachineFunctionAnalysisManager *MFAM)
SSAMachineSchedulerImpl()
This provides a very simple, boring adaptor for a begin and end iterator into a range type.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
Abstract Attribute helper functions.
LLVM_ABI StringRef getColorString(unsigned NodeNumber)
Get a color string for this node number.
void apply(Opt *O, const Mod &M, const Mods &... Ms)
ValuesClass values(OptsTy... Options)
Helper to build a ValuesClass by forwarding a variable number of arguments as an initializer list to ...
initializer< Ty > init(const Ty &Val)
NodeAddr< NodeBase * > Node
bool isSimple(Instruction *I)
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI int biasPhysReg(const SUnit *SU, bool isTop, bool BiasPRegsExtra=false)
Minimize physical register live ranges.
ScheduleDAGMILive * createSchedLive(MachineSchedContext *C)
Create the standard converging machine scheduler.
bool operator<(int64_t V1, const APSInt &V2)
void stable_sort(R &&Range)
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI unsigned getWeakLeft(const SUnit *SU, bool isTop)
FormattedString right_justify(StringRef Str, unsigned Width)
right_justify - add spaces before string so total output is Width characters.
LLVM_ABI std::unique_ptr< ScheduleDAGMutation > createLoadClusterDAGMutation(const TargetInstrInfo *TII, bool ReorderWhileClustering=false)
If ReorderWhileClustering is set to true, no attempt will be made to reduce reordering due to store c...
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
Printable PrintLaneMask(LaneBitmask LaneMask)
Create Printable object to print LaneBitmasks on a raw_ostream.
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI char & MachineSchedulerID
MachineScheduler - This pass schedules machine instructions.
LLVM_ABI char & PostMachineSchedulerID
PostMachineScheduler - This pass schedules machine instructions postRA.
LLVM_ABI std::unique_ptr< ScheduleDAGMutation > createStoreClusterDAGMutation(const TargetInstrInfo *TII, bool ReorderWhileClustering=false)
If ReorderWhileClustering is set to true, no attempt will be made to reduce reordering due to store c...
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI bool tryPressure(const PressureChange &TryP, const PressureChange &CandP, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason, const TargetRegisterInfo *TRI, const MachineFunction &MF)
LLVM_ABI void initializeSSAMachineSchedulerLegacyPass(PassRegistry &)
ScheduleDAGMI * createSchedPostRA(MachineSchedContext *C)
Create a generic scheduler with no vreg liveness or DAG mutation passes.
void sort(IteratorTy Start, IteratorTy End)
cl::opt< bool > ViewMISchedDAGs
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool is_sorted(R &&Range, Compare C)
Wrapper function around std::is_sorted to check if elements in a range R are sorted with respect to a...
LLVM_ABI bool shouldVerifyScheduling()
Returns whether -verify-misched is set.
LLVM_ABI bool tryLatency(GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, SchedBoundary &Zone)
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
constexpr unsigned InvalidClusterId
FormattedString left_justify(StringRef Str, unsigned Width)
left_justify - append spaces after string so total output is Width characters.
bool isTheSameCluster(unsigned A, unsigned B)
Return whether the input cluster ID's are the same and valid.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
LLVM_ABI bool tryBiasPhysRegs(GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, SchedBoundary *Zone, bool BiasPRegsExtra)
LLVM_ABI char & SSAMachineSchedulerID
SSAMachineScheduler - This pass schedules machine instructions in SSA.
LLVM_ABI std::unique_ptr< ScheduleDAGMutation > createCopyConstrainDAGMutation(const TargetInstrInfo *TII)
DWARFExpression::Operation Op
LLVM_ABI bool tryGreater(int TryVal, int CandVal, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason)
SmallPtrSet< SUnit *, 8 > ClusterInfo
Keep record of which SUnit are in the same cluster group.
void ViewGraph(const GraphType &G, const Twine &Name, bool ShortNames=false, const Twine &Title="", GraphProgram::Name Program=GraphProgram::DOT)
ViewGraph - Emit a dot graph, run 'dot', run gv on the postscript file, then cleanup.
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI unsigned computeRemLatency(SchedBoundary &CurrZone)
Compute remaining latency.
LLVM_ABI void dumpRegSetPressure(ArrayRef< unsigned > SetPressure, const TargetRegisterInfo *TRI)
LLVM_ABI MISched::Direction getPreRADirection()
Returns -misched-prera-direction.
LLVM_ABI bool tryLess(int TryVal, int CandVal, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason)
Return true if this heuristic determines order.
LLVM_ABI Printable printReg(Register Reg, const TargetRegisterInfo *TRI=nullptr, unsigned SubIdx=0, const MachineRegisterInfo *MRI=nullptr)
Prints virtual and physical registers with or without a TRI instance.
LLVM_ABI Printable printMBBReference(const MachineBasicBlock &MBB)
Prints a machine basic block reference.
cl::opt< bool > PrintDAGs
Implement std::hash so that hash_code can be used in STL containers.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Policy for scheduling the next instruction in the candidate's zone.
Store the state used by GenericScheduler heuristics, required for the lifetime of one invocation of p...
void setBest(SchedCandidate &Best)
void reset(const CandPolicy &NewPolicy)
LLVM_ABI void initResourceDelta(const ScheduleDAGMI *DAG, const TargetSchedModel *SchedModel)
SchedResourceDelta ResDelta
Status of an instruction's critical resource consumption.
unsigned DemandedResources
static constexpr LaneBitmask getNone()
Summarize the scheduling resources required for an instruction of a particular scheduling class.
Identify one of the processor resource kinds consumed by a particular scheduling class for the specif...
MachineSchedContext provides enough context from the MachineScheduler pass for the target to instanti...
RegisterClassInfo * RegClassInfo
MachineBlockFrequencyInfo * MBFI
const MachineLoopInfo * MLI
virtual ~MachineSchedContext()
PressureChange CriticalMax
PressureChange CurrentMax
RegisterPressure computed within a region of instructions delimited by TopPos and BottomPos.
A region of an MBB for scheduling.
Summarize the unscheduled region.
LLVM_ABI void init(ScheduleDAGMI *DAG, const TargetSchedModel *SchedModel)
SmallVector< unsigned, 16 > RemainingCounts
An individual mapping from virtual register number to SUnit.
RegisterClassInfo & RegClassInfo
MachineBlockFrequencyInfo & MBFI
MachineBlockFrequencyInfo & MBFI
RegisterClassInfo & RegClassInfo