46#define DEBUG_TYPE "machine-scheduler"
51 "amdgpu-disable-unclustered-high-rp-reschedule",
cl::Hidden,
52 cl::desc(
"Disable unclustered high register pressure "
53 "reduction scheduling stage."),
57 "amdgpu-disable-clustered-low-occupancy-reschedule",
cl::Hidden,
58 cl::desc(
"Disable clustered low occupancy "
59 "rescheduling for ILP scheduling stage."),
65 "Sets the bias which adds weight to occupancy vs latency. Set it to "
66 "100 to chase the occupancy only."),
71 cl::desc(
"Relax occupancy targets for kernels which are memory "
72 "bound (amdgpu-membound-threshold), or "
73 "Wave Limited (amdgpu-limit-wave-threshold)."),
78 cl::desc(
"Use the AMDGPU specific RPTrackers during scheduling"),
82 "amdgpu-scheduler-pending-queue-limit",
cl::Hidden,
84 "Max (Available+Pending) size to inspect pending queue (0 disables)"),
87#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
88#define DUMP_MAX_REG_PRESSURE
90 "amdgpu-print-max-reg-pressure-regusage-before-scheduler",
cl::Hidden,
91 cl::desc(
"Print a list of live registers along with their def/uses at the "
92 "point of maximum register pressure before scheduling."),
96 "amdgpu-print-max-reg-pressure-regusage-after-scheduler",
cl::Hidden,
97 cl::desc(
"Print a list of live registers along with their def/uses at the "
98 "point of maximum register pressure after scheduling."),
103 "amdgpu-disable-rewrite-mfma-form-sched-stage",
cl::Hidden,
108struct VGPRThresholdParser :
public cl::parser<unsigned> {
111 bool parse(cl::Option &O, StringRef ArgName, StringRef Arg,
unsigned &
Value) {
113 return O.error(
"'" + Arg +
"' value invalid for uint argument!");
116 return O.error(
"'" + Arg +
"' value must be in the range [0, 100]!");
126 cl::desc(
"Percent of VGPR limits that we should use as RP threshold "
127 "during scheduling. We have two limits relevant to scheduling: "
128 "Critical (avoid decreasing occupancy), Excess (avoid spilling). "
129 "This flag scales both limits back by an equal percent: (0 = use "
130 " default calculation, 1-100 = use percentage), default: 0"),
150 Context->RegClassInfo->getNumAllocatableRegs(&AMDGPU::SGPR_32RegClass);
152 Context->RegClassInfo->getNumAllocatableRegs(&AMDGPU::VGPR_32RegClass);
174 "VGPRCriticalLimit calculation method.\n");
178 unsigned Addressable =
181 VGPRBudget = std::max(VGPRBudget, Granule);
194 <<
". VGPRCriticalLimit: " << OriginalVGPRCriticalLimit
234 if (!
Op.isReg() ||
Op.isImplicit())
236 if (
Op.getReg().isPhysical() ||
237 (
Op.isDef() &&
Op.getSubReg() != AMDGPU::NoSubRegister))
272 Pressure[AMDGPU::RegisterPressureSets::VGPR_32] =
280 if (!Zone.
isTop() || !SU)
297 if (NextAvail > CurrCycle)
298 Stall = std::max(
Stall, NextAvail - CurrCycle);
317 unsigned SGPRPressure,
318 unsigned VGPRPressure,
bool IsBottomUp) {
322 if (!
DAG->isTrackingPressure())
345 Pressure[AMDGPU::RegisterPressureSets::SReg_32] = SGPRPressure;
346 Pressure[AMDGPU::RegisterPressureSets::VGPR_32] = VGPRPressure;
348 for (
const auto &Diff :
DAG->getPressureDiff(SU)) {
354 (IsBottomUp ? Diff.getUnitInc() : -Diff.getUnitInc());
357#ifdef EXPENSIVE_CHECKS
358 std::vector<unsigned> CheckPressure, CheckMaxPressure;
361 if (
Pressure[AMDGPU::RegisterPressureSets::SReg_32] !=
362 CheckPressure[AMDGPU::RegisterPressureSets::SReg_32] ||
363 Pressure[AMDGPU::RegisterPressureSets::VGPR_32] !=
364 CheckPressure[AMDGPU::RegisterPressureSets::VGPR_32]) {
365 errs() <<
"Register Pressure is inaccurate when calculated through "
367 <<
"SGPR got " <<
Pressure[AMDGPU::RegisterPressureSets::SReg_32]
369 << CheckPressure[AMDGPU::RegisterPressureSets::SReg_32] <<
"\n"
370 <<
"VGPR got " <<
Pressure[AMDGPU::RegisterPressureSets::VGPR_32]
372 << CheckPressure[AMDGPU::RegisterPressureSets::VGPR_32] <<
"\n";
378 unsigned NewSGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::SReg_32];
379 unsigned NewVGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::VGPR_32];
389 const unsigned MaxVGPRPressureInc = 16;
390 bool ShouldTrackVGPRs = VGPRPressure + MaxVGPRPressureInc >=
VGPRExcessLimit;
391 bool ShouldTrackSGPRs = !ShouldTrackVGPRs && SGPRPressure >=
SGPRExcessLimit;
422 if (SGPRDelta >= 0 || VGPRDelta >= 0) {
424 if (SGPRDelta > VGPRDelta) {
438 bool HasBufferedModel =
457 dbgs() <<
"Prefer:\t\t";
458 DAG->dumpNode(*Preferred.
SU);
462 DAG->dumpNode(*Current.
SU);
465 dbgs() <<
"Reason:\t\t";
479 unsigned SGPRPressure = 0;
480 unsigned VGPRPressure = 0;
482 if (
DAG->isTrackingPressure()) {
484 SGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::SReg_32];
485 VGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::VGPR_32];
490 SGPRPressure =
T->getPressure().getSGPRNum();
491 VGPRPressure =
T->getPressure().getArchVGPRNum();
496 for (
SUnit *SU : AQ) {
500 VGPRPressure, IsBottomUp);
520 for (
SUnit *SU : PQ) {
524 VGPRPressure, IsBottomUp);
544 bool &PickedPending) {
564 bool BotPending =
false;
584 "Last pick result should correspond to re-picking right now");
589 bool TopPending =
false;
609 "Last pick result should correspond to re-picking right now");
619 PickedPending = BotPending && TopPending;
622 if (BotPending || TopPending) {
629 Cand.setBest(TryCand);
634 IsTopNode = Cand.AtTop;
641 if (
DAG->top() ==
DAG->bottom()) {
643 Bot.Available.empty() &&
Bot.Pending.empty() &&
"ReadyQ garbage");
649 PickedPending =
false;
683 if (ReadyCycle > CurrentCycle)
755 if (
DAG->isTrackingPressure() &&
761 if (
DAG->isTrackingPressure() &&
766 bool SameBoundary = Zone !=
nullptr;
790 if (IsLegacyScheduler)
809 if (
DAG->isTrackingPressure() &&
819 bool SameBoundary = Zone !=
nullptr;
854 bool CandIsClusterSucc =
856 bool TryCandIsClusterSucc =
858 if (
tryGreater(TryCandIsClusterSucc, CandIsClusterSucc, TryCand, Cand,
863 if (
DAG->isTrackingPressure() &&
869 if (
DAG->isTrackingPressure() &&
915 if (
DAG->isTrackingPressure()) {
931 bool CandIsClusterSucc =
933 bool TryCandIsClusterSucc =
935 if (
tryGreater(TryCandIsClusterSucc, CandIsClusterSucc, TryCand, Cand,
944 bool SameBoundary = Zone !=
nullptr;
961 if (TryMayLoad || CandMayLoad) {
962 bool TryLongLatency =
964 bool CandLongLatency =
968 Zone->
isTop() ? CandLongLatency : TryLongLatency, TryCand,
986 if (
DAG->isTrackingPressure() &&
1005 !
Rem.IsAcyclicLatencyLimited &&
tryLatency(TryCand, Cand, *Zone))
1023 StartingOccupancy(MFI.getOccupancy()), MinOccupancy(StartingOccupancy),
1024 RegionLiveOuts(this,
true) {
1030 LLVM_DEBUG(
dbgs() <<
"Starting occupancy is " << StartingOccupancy <<
".\n");
1032 MinOccupancy = std::min(MFI.getMinAllowedOccupancy(), StartingOccupancy);
1033 if (MinOccupancy != StartingOccupancy)
1034 LLVM_DEBUG(
dbgs() <<
"Allowing Occupancy drops to " << MinOccupancy
1039std::unique_ptr<GCNSchedStage>
1041 switch (SchedStageID) {
1043 return std::make_unique<OccInitialScheduleStage>(SchedStageID, *
this);
1045 return std::make_unique<RewriteMFMAFormStage>(SchedStageID, *
this);
1047 return std::make_unique<UnclusteredHighRPStage>(SchedStageID, *
this);
1049 return std::make_unique<ClusteredLowOccStage>(SchedStageID, *
this);
1051 return std::make_unique<PreRARematStage>(SchedStageID, *
this);
1053 return std::make_unique<ILPInitialScheduleStage>(SchedStageID, *
this);
1055 return std::make_unique<MemoryClauseInitialScheduleStage>(SchedStageID,
1069GCNScheduleDAGMILive::getRealRegPressure(
unsigned RegionIdx)
const {
1070 if (Regions[RegionIdx].first == Regions[RegionIdx].second)
1074 &LiveIns[RegionIdx]);
1080 assert(RegionBegin != RegionEnd &&
"Region must not be empty");
1084void GCNScheduleDAGMILive::computeBlockPressure(
unsigned RegionIdx,
1096 const MachineBasicBlock *OnlySucc =
nullptr;
1099 if (!Candidate->empty() && Candidate->pred_size() == 1) {
1100 SlotIndexes *Ind =
LIS->getSlotIndexes();
1102 OnlySucc = Candidate;
1107 size_t CurRegion = RegionIdx;
1108 for (
size_t E = Regions.size(); CurRegion !=
E; ++CurRegion)
1109 if (Regions[CurRegion].first->getParent() !=
MBB)
1114 auto LiveInIt = MBBLiveIns.find(
MBB);
1115 auto &Rgn = Regions[CurRegion];
1117 if (LiveInIt != MBBLiveIns.end()) {
1118 auto LiveIn = std::move(LiveInIt->second);
1120 MBBLiveIns.erase(LiveInIt);
1123 auto LRS = BBLiveInMap.lookup(NonDbgMI);
1124#ifdef EXPENSIVE_CHECKS
1133 if (Regions[CurRegion].first ==
I || NonDbgMI ==
I) {
1134 LiveIns[CurRegion] =
RPTracker.getLiveRegs();
1138 if (Regions[CurRegion].second ==
I) {
1139 Pressure[CurRegion] =
RPTracker.moveMaxPressure();
1140 if (CurRegion-- == RegionIdx)
1142 auto &Rgn = Regions[CurRegion];
1155 MBBLiveIns[OnlySucc] =
RPTracker.moveLiveRegs();
1160GCNScheduleDAGMILive::getRegionLiveInMap()
const {
1161 assert(!Regions.empty());
1162 std::vector<MachineInstr *> RegionFirstMIs;
1163 RegionFirstMIs.reserve(Regions.size());
1165 RegionFirstMIs.push_back(
1172GCNScheduleDAGMILive::getRegionLiveOutMap()
const {
1173 assert(!Regions.empty());
1174 std::vector<MachineInstr *> RegionLastMIs;
1175 RegionLastMIs.reserve(Regions.size());
1186 IdxToInstruction.clear();
1189 IsLiveOut ? DAG->getRegionLiveOutMap() : DAG->getRegionLiveInMap();
1190 for (
unsigned I = 0;
I < DAG->Regions.size();
I++) {
1191 auto &[RegionBegin, RegionEnd] = DAG->Regions[
I];
1193 if (RegionBegin == RegionEnd)
1197 IdxToInstruction[
I] = RegionKey;
1205 LiveIns.resize(Regions.size());
1206 Pressure.resize(Regions.size());
1207 RegionsWithHighRP.resize(Regions.size());
1208 RegionsWithExcessRP.resize(Regions.size());
1209 RegionsWithIGLPInstrs.resize(Regions.size());
1210 RegionsWithHighRP.reset();
1211 RegionsWithExcessRP.reset();
1212 RegionsWithIGLPInstrs.reset();
1217void GCNScheduleDAGMILive::runSchedStages() {
1218 LLVM_DEBUG(
dbgs() <<
"All regions recorded, starting actual scheduling.\n");
1221 if (!Regions.
empty()) {
1222 BBLiveInMap = getRegionLiveInMap();
1227#ifdef DUMP_MAX_REG_PRESSURE
1237 if (!Stage->initGCNSchedStage())
1240 for (
auto Region : Regions) {
1244 if (!Stage->initGCNRegion()) {
1245 Stage->advanceRegion();
1251 const unsigned RegionIdx = Stage->getRegionIdx();
1254 MRI, RegionLiveOuts.getLiveRegsForRegionIdx(RegionIdx));
1258 Stage->finalizeGCNRegion();
1259 Stage->advanceRegion();
1263 Stage->finalizeGCNSchedStage();
1266#ifdef DUMP_MAX_REG_PRESSURE
1279 OS <<
"Max Occupancy Initial Schedule";
1282 OS <<
"Instruction Rewriting Reschedule";
1285 OS <<
"Unclustered High Register Pressure Reschedule";
1288 OS <<
"Clustered Low Occupancy Reschedule";
1291 OS <<
"Pre-RA Rematerialize";
1294 OS <<
"Max ILP Initial Schedule";
1297 OS <<
"Max memory clause Initial Schedule";
1317void RewriteMFMAFormStage::findReachingDefs(
1339 while (!Worklist.
empty()) {
1354 for (MachineBasicBlock *PredMBB : DefMBB->
predecessors()) {
1355 if (Visited.
insert(PredMBB).second)
1361void RewriteMFMAFormStage::findReachingUses(
1365 for (MachineOperand &UseMO :
1368 findReachingDefs(UseMO, LIS, ReachingDefIndexes);
1372 if (
any_of(ReachingDefIndexes, [DefIdx](SlotIndex RDIdx) {
1384 if (!
ST.hasGFX90AInsts() ||
MFI.getMinWavesPerEU() > 1)
1387 RegionsWithExcessArchVGPR.resize(
DAG.Regions.size());
1388 RegionsWithExcessArchVGPR.reset();
1392 RegionsWithExcessArchVGPR[
Region] =
true;
1395 if (RegionsWithExcessArchVGPR.none())
1398 TII =
ST.getInstrInfo();
1399 SRI =
ST.getRegisterInfo();
1401 std::vector<std::pair<MachineInstr *, unsigned>> RewriteCands;
1405 if (!initHeuristics(RewriteCands, CopyForUse, CopyForDef))
1408 int64_t
Cost = getRewriteCost(RewriteCands, CopyForUse, CopyForDef);
1415 return rewrite(RewriteCands);
1425 if (
DAG.RegionsWithHighRP.none() &&
DAG.RegionsWithExcessRP.none())
1432 InitialOccupancy =
DAG.MinOccupancy;
1435 TempTargetOccupancy =
MFI.getMaxWavesPerEU() >
DAG.MinOccupancy
1436 ? InitialOccupancy + 1
1438 IsAnyRegionScheduled =
false;
1439 S.SGPRLimitBias =
S.HighRPSGPRBias;
1440 S.VGPRLimitBias =
S.HighRPVGPRBias;
1444 <<
"Retrying function scheduling without clustering. "
1445 "Aggressively try to reduce register pressure to achieve occupancy "
1446 << TempTargetOccupancy <<
".\n");
1461 if (
DAG.StartingOccupancy <=
DAG.MinOccupancy)
1465 dbgs() <<
"Retrying function scheduling with lowest recorded occupancy "
1466 <<
DAG.MinOccupancy <<
".\n");
1471#define REMAT_PREFIX "[PreRARemat] "
1472#define REMAT_DEBUG(X) LLVM_DEBUG(dbgs() << REMAT_PREFIX; X;)
1474#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1475Printable PreRARematStage::ScoredRemat::print()
const {
1477 OS <<
'(' << MaxFreq <<
", " << FreqDiff <<
", " << RegionImpact <<
')';
1492 auto PrintTargetRegions = [&]() ->
void {
1493 if (TargetRegions.none()) {
1498 for (
unsigned I : TargetRegions.set_bits())
1505 dbgs() <<
"Analyzing ";
1506 MF.getFunction().printAsOperand(
dbgs(),
false);
1509 if (!setObjective()) {
1510 LLVM_DEBUG(
dbgs() <<
"no objective to achieve, occupancy is maximal at "
1511 <<
MFI.getMaxWavesPerEU() <<
'\n');
1516 dbgs() <<
"increase occupancy from " << *TargetOcc - 1 <<
'\n';
1518 dbgs() <<
"reduce spilling (minimum target occupancy is "
1519 <<
MFI.getMinWavesPerEU() <<
")\n";
1521 PrintTargetRegions();
1526 DAG.RegionLiveOuts.buildLiveRegMap();
1528 if (!Remater.analyze()) {
1542 for (
unsigned RegIdx = 0, E = Remater.getNumRegs(); RegIdx < E; ++RegIdx) {
1546 if (CandReg.
Uses.size() != 1)
1548 const auto [UseRegion,
Users] = *CandReg.
Uses.begin();
1567 "user must have at least one operand");
1574 assert(FirstUseMI &&
"there must be a user in the region");
1576 DAG.LIS->getInstructionIndex(*FirstUseMI).getRegSlot(
true);
1578 DAG.LIS->getInstructionIndex(*CandReg.
getLastDef()).getRegSlot(
true);
1580 const Rematerializer::Reg &DepReg = Remater.getReg(DepRegIdx);
1581 Register DepDefReg = DepReg.getDefReg();
1582 return MarkedRegs.contains(DepDefReg) ||
1583 !Remater.isRegIdenticalAtUses(DepDefReg, DepReg.Mask, RefIdx,
1588 [&](
const std::pair<Register, LaneBitmask> &RegAndMask) {
1589 const auto &[Reg, Mask] = RegAndMask;
1590 return !Remater.isRegIdenticalAtUses(Reg, Mask, RefIdx,
1595 MarkedRegs.
insert(CandReg.getDefReg());
1597 Cand.init(RegIdx, FreqInfo, Remater,
DAG);
1598 Cand.update(TargetRegions, RPTargets, FreqInfo, !TargetOcc);
1599 if (!Cand.hasNullScore())
1610 Rollback = std::make_unique<RollbackSupport>(Remater);
1617 RecomputeRP.reset();
1620 sort(CandidateOrder, [&](
unsigned LHSIndex,
unsigned RHSIndex) {
1621 return Candidates[LHSIndex] < Candidates[RHSIndex];
1625 dbgs() <<
"==== NEW REMAT ROUND ====\n"
1627 <<
"Candidates with non-null score, in rematerialization order:\n";
1628 for (
const ScoredRemat &Cand :
reverse(Candidates)) {
1630 << Remater.printRematReg(Cand.RegIdx) <<
'\n';
1632 PrintTargetRegions();
1638 while (!CandidateOrder.empty()) {
1639 const ScoredRemat &Cand = Candidates[CandidateOrder.back()];
1640 const Rematerializer::Reg &
Reg = Remater.getReg(Cand.RegIdx);
1648 if (!Cand.maybeBeneficial(TargetRegions, RPTargets)) {
1650 << Cand.print() <<
" | "
1651 << Remater.printRematReg(Cand.RegIdx));
1654 CandidateOrder.pop_back();
1656#ifdef EXPENSIVE_CHECKS
1659 for (
const MachineInstr *
DefMI :
Reg.Defs) {
1664 if (!MO.isReg() || !MO.getReg() || !MO.readsReg() || MO.isDef())
1671 LiveInterval &LI =
DAG.LIS->getInterval(
UseReg);
1672 LaneBitmask LM =
DAG.MRI.getMaxLaneMaskForVReg(MO.getReg());
1674 LM =
DAG.TRI->getSubRegIndexLaneMask(MO.getSubReg());
1676 const unsigned UseRegion =
Reg.Uses.begin()->first;
1677 LaneBitmask LiveInMask =
DAG.LiveIns[UseRegion].at(
UseReg);
1678 LaneBitmask UncoveredLanes = LM & ~(LiveInMask & LM);
1682 if (UncoveredLanes.
any()) {
1684 for (LiveInterval::SubRange &SR : LI.
subranges())
1685 assert((SR.LaneMask & UncoveredLanes).none());
1693 REMAT_DEBUG(
dbgs() <<
"** REMAT " << Remater.printRematReg(Cand.RegIdx)
1695 removeFromLiveMaps(
Reg.getDefReg(), Cand.LiveIn, Cand.LiveOut);
1697 Rollback->LiveMapUpdates.emplace_back(Cand.RegIdx, Cand.LiveIn,
1700 Cand.rematerialize(Remater);
1705 updateRPTargets(Cand.Live, Cand.RPSave);
1706 RecomputeRP |= Cand.UnpredictableRPSave;
1707 RescheduleRegions |= Cand.Live;
1708 if (!TargetRegions.any()) {
1714 if (!updateAndVerifyRPTargets(RecomputeRP) && !TargetRegions.any()) {
1723 unsigned NumUsefulCandidates = 0;
1724 for (
unsigned CandIdx : CandidateOrder) {
1725 ScoredRemat &Candidate = Candidates[CandIdx];
1726 Candidate.update(TargetRegions, RPTargets, FreqInfo, !TargetOcc);
1727 if (!Candidate.hasNullScore())
1728 CandidateOrder[NumUsefulCandidates++] = CandIdx;
1730 if (NumUsefulCandidates == 0) {
1731 REMAT_DEBUG(
dbgs() <<
"Stop on exhausted rematerialization candidates\n");
1734 CandidateOrder.truncate(NumUsefulCandidates);
1737 if (RescheduleRegions.none())
1743 unsigned DynamicVGPRBlockSize =
MFI.getDynamicVGPRBlockSize();
1744 for (
unsigned I : RescheduleRegions.set_bits()) {
1745 DAG.Pressure[
I] = RPTargets[
I].getCurrentRP();
1747 <<
DAG.Pressure[
I].getOccupancy(
ST, DynamicVGPRBlockSize)
1748 <<
" (" << RPTargets[
I] <<
")\n");
1750 AchievedOcc =
MFI.getMaxWavesPerEU();
1751 for (
const GCNRegPressure &RP :
DAG.Pressure) {
1753 std::min(AchievedOcc,
RP.getOccupancy(
ST, DynamicVGPRBlockSize));
1757 dbgs() <<
"Retrying function scheduling with new min. occupancy of "
1758 << AchievedOcc <<
" from rematerializing (original was "
1759 <<
DAG.MinOccupancy;
1761 dbgs() <<
", target was " << *TargetOcc;
1765 DAG.setTargetOccupancy(getStageTargetOccupancy());
1776 S.SGPRLimitBias =
S.VGPRLimitBias = 0;
1777 if (
DAG.MinOccupancy > InitialOccupancy) {
1778 assert(IsAnyRegionScheduled);
1780 <<
" stage successfully increased occupancy to "
1781 <<
DAG.MinOccupancy <<
'\n');
1782 }
else if (!IsAnyRegionScheduled) {
1783 assert(
DAG.MinOccupancy == InitialOccupancy);
1785 <<
": No regions scheduled, min occupancy stays at "
1786 <<
DAG.MinOccupancy <<
", MFI occupancy stays at "
1787 <<
MFI.getOccupancy() <<
".\n");
1795 if (
DAG.begin() ==
DAG.end())
1802 unsigned NumRegionInstrs = std::distance(
DAG.begin(),
DAG.end());
1806 if (
DAG.begin() == std::prev(
DAG.end()))
1812 <<
"\n From: " << *
DAG.begin() <<
" To: ";
1814 else dbgs() <<
"End";
1815 dbgs() <<
" RegionInstrs: " << NumRegionInstrs <<
'\n');
1823 for (
auto &
I :
DAG) {
1836 dbgs() <<
"Pressure before scheduling:\nRegion live-ins:"
1838 <<
"Region live-in pressure: "
1842 S.HasHighPressure =
false;
1864 unsigned DynamicVGPRBlockSize =
DAG.MFI.getDynamicVGPRBlockSize();
1867 unsigned CurrentTargetOccupancy =
1868 IsAnyRegionScheduled ?
DAG.MinOccupancy : TempTargetOccupancy;
1870 (CurrentTargetOccupancy <= InitialOccupancy ||
1871 DAG.Pressure[
RegionIdx].getOccupancy(
ST, DynamicVGPRBlockSize) !=
1878 if (!IsAnyRegionScheduled && IsSchedulingThisRegion) {
1879 IsAnyRegionScheduled =
true;
1880 if (
MFI.getMaxWavesPerEU() >
DAG.MinOccupancy)
1881 DAG.setTargetOccupancy(TempTargetOccupancy);
1883 return IsSchedulingThisRegion;
1899 return !RevertAllRegions && RescheduleRegions[
RegionIdx] &&
1919 if (
S.HasHighPressure)
1940 if (
DAG.MinOccupancy < *TargetOcc) {
1942 <<
" cannot meet occupancy target, interrupting "
1943 "re-scheduling in all regions\n");
1944 RevertAllRegions =
true;
1955 unsigned DynamicVGPRBlockSize =
DAG.MFI.getDynamicVGPRBlockSize();
1966 unsigned TargetOccupancy = std::min(
1967 S.getTargetOccupancy(),
ST.getOccupancyWithWorkGroupSizes(
MF).second);
1968 unsigned WavesAfter = std::min(
1969 TargetOccupancy,
PressureAfter.getOccupancy(
ST, DynamicVGPRBlockSize));
1970 unsigned WavesBefore = std::min(
1972 LLVM_DEBUG(
dbgs() <<
"Occupancy before scheduling: " << WavesBefore
1973 <<
", after " << WavesAfter <<
".\n");
1979 unsigned NewOccupancy = std::max(WavesAfter, WavesBefore);
1983 if (WavesAfter < WavesBefore && WavesAfter <
DAG.MinOccupancy &&
1984 WavesAfter >=
MFI.getMinAllowedOccupancy()) {
1985 LLVM_DEBUG(
dbgs() <<
"Function is memory bound, allow occupancy drop up to "
1986 <<
MFI.getMinAllowedOccupancy() <<
" waves\n");
1987 NewOccupancy = WavesAfter;
1990 if (NewOccupancy <
DAG.MinOccupancy) {
1991 DAG.MinOccupancy = NewOccupancy;
1992 MFI.limitOccupancy(
DAG.MinOccupancy);
1994 <<
DAG.MinOccupancy <<
".\n");
1998 unsigned MaxVGPRs =
ST.getMaxNumVGPRs(
MF);
2001 unsigned MaxArchVGPRs = std::min(MaxVGPRs,
ST.getAddressableNumArchVGPRs());
2002 unsigned MaxSGPRs =
ST.getMaxNumSGPRs(
MF);
2026 unsigned ReadyCycle = CurrCycle;
2027 for (
auto &
D : SU.
Preds) {
2028 if (
D.isAssignedRegDep()) {
2031 unsigned DefReady = ReadyCycles[
DAG.getSUnit(
DefMI)->NodeNum];
2032 ReadyCycle = std::max(ReadyCycle, DefReady +
Latency);
2035 ReadyCycles[SU.
NodeNum] = ReadyCycle;
2042 std::pair<MachineInstr *, unsigned>
B)
const {
2043 return A.second <
B.second;
2049 if (ReadyCycles.empty())
2051 unsigned BBNum = ReadyCycles.begin()->first->getParent()->getNumber();
2052 dbgs() <<
"\n################## Schedule time ReadyCycles for MBB : " << BBNum
2053 <<
" ##################\n# Cycle #\t\t\tInstruction "
2057 for (
auto &
I : ReadyCycles) {
2058 if (
I.second > IPrev + 1)
2059 dbgs() <<
"****************************** BUBBLE OF " <<
I.second - IPrev
2060 <<
" CYCLES DETECTED ******************************\n\n";
2061 dbgs() <<
"[ " <<
I.second <<
" ] : " << *
I.first <<
"\n";
2074 unsigned SumBubbles = 0;
2076 unsigned CurrCycle = 0;
2077 for (
auto &SU : InputSchedule) {
2078 unsigned ReadyCycle =
2080 SumBubbles += ReadyCycle - CurrCycle;
2082 ReadyCyclesSorted.insert(std::make_pair(SU.getInstr(), ReadyCycle));
2084 CurrCycle = ++ReadyCycle;
2107 unsigned SumBubbles = 0;
2109 unsigned CurrCycle = 0;
2110 for (
auto &
MI :
DAG) {
2114 unsigned ReadyCycle =
2116 SumBubbles += ReadyCycle - CurrCycle;
2118 ReadyCyclesSorted.insert(std::make_pair(SU->
getInstr(), ReadyCycle));
2120 CurrCycle = ++ReadyCycle;
2137 if (WavesAfter <
DAG.MinOccupancy)
2141 if (
DAG.MFI.isDynamicVGPREnabled()) {
2144 DAG.MFI.getDynamicVGPRBlockSize());
2147 if (BlocksAfter > BlocksBefore)
2184 <<
"\n\t *** In shouldRevertScheduling ***\n"
2185 <<
" *********** BEFORE UnclusteredHighRPStage ***********\n");
2189 <<
"\n *********** AFTER UnclusteredHighRPStage ***********\n");
2191 unsigned OldMetric = MBefore.
getMetric();
2192 unsigned NewMetric = MAfter.
getMetric();
2193 unsigned WavesBefore = std::min(
2194 S.getTargetOccupancy(),
2201 LLVM_DEBUG(
dbgs() <<
"\tMetric before " << MBefore <<
"\tMetric after "
2202 << MAfter <<
"Profit: " << Profit <<
"\n");
2233 unsigned WavesAfter) {
2240 LLVM_DEBUG(
dbgs() <<
"New pressure will result in more spilling.\n");
2252 "instruction number mismatch");
2253 if (MIOrder.
empty())
2266 if (MII != RegionEnd) {
2268 bool NonDebugReordered =
2269 !
MI->isDebugInstr() &&
2275 if (NonDebugReordered)
2276 DAG.LIS->handleMove(*
MI,
true);
2283 if (!
MI->isDebugInstr()) {
2285 SlotIndex PrevIdx =
DAG.LIS->getSlotIndexes()->getIndexBefore(*
MI);
2286 if (PrevIdx >= MIIdx)
2287 DAG.LIS->handleMove(*
MI,
true);
2291 if (
MI->isDebugInstr()) {
2298 Op.setIsUndef(
false);
2301 if (
DAG.ShouldTrackLaneMasks) {
2326 if (RD->
getOpcode() == AMDGPU::AV_MOV_B32_IMM_PSEUDO ||
2327 RD->
getOpcode() == AMDGPU::AV_MOV_B64_IMM_PSEUDO)
2334bool RewriteMFMAFormStage::hasUseRequiringVGPR(
2336 const SmallPtrSetImpl<MachineInstr *> &RewriteSet) {
2337 for (SlotIndex RDIdx : Src2ReachingDefs) {
2338 const MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIdx);
2340 findReachingUses(RD,
DAG.LIS, ReachingUses);
2341 for (
const MachineOperand *UseMO : ReachingUses) {
2353void RewriteMFMAFormStage::resetRewriteCandsToVGPR(
2354 ArrayRef<std::pair<MachineInstr *, unsigned>> RewriteCands) {
2355 for (
auto [
MI, OriginalOpcode] : RewriteCands) {
2358 DAG.MRI.getRegClass(
MI->getOperand(0).getReg());
2360 DAG.MRI.setRegClass(
MI->getOperand(0).getReg(), VDefRC);
2361 MI->setDesc(
TII->get(OriginalOpcode));
2363 MachineOperand *Src2 =
TII->getNamedOperand(*
MI, AMDGPU::OpName::src2);
2372 DAG.MRI.setRegClass(Src2->
getReg(), VUseRC);
2376bool RewriteMFMAFormStage::isRewriteCandidate(MachineInstr *
MI)
const {
2377 if (!
static_cast<const SIInstrInfo *
>(
DAG.TII)->isMAI(*
MI))
2382 Register DstReg =
MI->getOperand(0).getReg();
2383 for (
const MachineInstr &
UseMI :
DAG.MRI.use_nodbg_instructions(DstReg)) {
2390bool RewriteMFMAFormStage::initHeuristics(
2391 std::vector<std::pair<MachineInstr *, unsigned>> &RewriteCands,
2392 DenseMap<MachineBasicBlock *, std::set<Register>> &CopyForUse,
2393 SmallPtrSetImpl<MachineInstr *> &CopyForDef) {
2398 SmallPtrSet<MachineInstr *, 16> RewriteSet;
2399 DenseSet<Register> CandSrc2Regs;
2400 for (MachineBasicBlock &
MBB :
MF) {
2401 for (MachineInstr &
MI :
MBB) {
2402 if (!isRewriteCandidate(&
MI))
2405 MachineOperand *Src2 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src2);
2406 if (Src2 && Src2->
isReg())
2412 for (MachineBasicBlock &
MBB :
MF) {
2413 for (MachineInstr &
MI :
MBB) {
2414 if (!isRewriteCandidate(&
MI))
2418 assert(ReplacementOp != -1);
2420 RewriteCands.push_back({&
MI,
MI.getOpcode()});
2421 MI.setDesc(
TII->get(ReplacementOp));
2423 MachineOperand *Src2 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src2);
2424 if (Src2->
isReg()) {
2426 findReachingDefs(*Src2,
DAG.LIS, Src2ReachingDefs);
2430 bool Src2NeedsVGPR = hasUseRequiringVGPR(Src2ReachingDefs, RewriteSet);
2431 Src2NeedsVGPRCache[&
MI] = Src2NeedsVGPR;
2433 for (SlotIndex RDIdx : Src2ReachingDefs) {
2434 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIdx);
2435 if (!Src2NeedsVGPR &&
2442 MachineOperand &Dst =
MI.getOperand(0);
2445 findReachingUses(&
MI,
DAG.LIS, DstReachingUses);
2447 for (MachineOperand *RUOp : DstReachingUses) {
2448 MachineInstr *UserMI = RUOp->getParent();
2450 if (
TII->isMAI(*UserMI) && RewriteSet.
contains(UserMI))
2456 CopyForUse[UserMI->
getParent()].insert(RUOp->getReg());
2458 if (
TII->isMAI(*UserMI))
2462 findReachingDefs(*RUOp,
DAG.LIS, DstUsesReachingDefs);
2464 for (SlotIndex RDIndex : DstUsesReachingDefs) {
2465 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIndex);
2466 if (
TII->isMAI(*RD))
2480 DAG.MRI.setRegClass(Dst.getReg(), ADefRC);
2481 if (Src2->
isReg()) {
2487 DAG.MRI.setRegClass(Src2->
getReg(), AUseRC);
2496int64_t RewriteMFMAFormStage::getRewriteCost(
2497 ArrayRef<std::pair<MachineInstr *, unsigned>> RewriteCands,
2498 const DenseMap<MachineBasicBlock *, std::set<Register>> &CopyForUse,
2499 const SmallPtrSetImpl<MachineInstr *> &CopyForDef) {
2500 MachineBlockFrequencyInfo *MBFI =
DAG.MBFI;
2502 int64_t BestSpillCost = 0;
2506 std::pair<unsigned, unsigned> MaxVectorRegs =
2507 ST.getMaxNumVectorRegs(
MF.getFunction());
2508 unsigned ArchVGPRThreshold = MaxVectorRegs.first;
2509 unsigned AGPRThreshold = MaxVectorRegs.second;
2510 unsigned CombinedThreshold =
ST.getMaxNumVGPRs(
MF);
2513 if (!RegionsWithExcessArchVGPR[Region])
2518 MF, ArchVGPRThreshold, AGPRThreshold, CombinedThreshold);
2526 MF, ArchVGPRThreshold, AGPRThreshold, CombinedThreshold);
2532 bool RelativeFreqIsDenom = EntryFreq > BlockFreq;
2533 uint64_t RelativeFreq = EntryFreq && BlockFreq
2534 ? (RelativeFreqIsDenom ? EntryFreq / BlockFreq
2535 : BlockFreq / EntryFreq)
2540 int64_t SpillCost = ((int)SpillCostAfter - (int)SpillCostBefore) * 2;
2543 if (RelativeFreqIsDenom)
2544 SpillCost /= (int64_t)RelativeFreq;
2546 SpillCost *= (int64_t)RelativeFreq;
2549 if (SpillCost > 0) {
2550 resetRewriteCandsToVGPR(RewriteCands);
2554 if (SpillCost < BestSpillCost)
2555 BestSpillCost = SpillCost;
2560 Cost = BestSpillCost;
2563 unsigned CopyCost = 0;
2567 for (MachineInstr *
DefMI : CopyForDef) {
2579 for (
auto &[UseBlock, UseRegs] : CopyForUse) {
2593 resetRewriteCandsToVGPR(RewriteCands);
2595 return Cost + CopyCost;
2598bool RewriteMFMAFormStage::rewrite(
2599 ArrayRef<std::pair<MachineInstr *, unsigned>> RewriteCands) {
2600 DenseMap<MachineInstr *, unsigned> FirstMIToRegion;
2601 DenseMap<MachineInstr *, unsigned> LastMIToRegion;
2609 if (
Entry.second !=
Entry.first->getParent()->end())
2652 DenseSet<Register> RewriteRegs;
2655 DenseMap<Register, Register> RedefMap;
2657 DenseMap<Register, DenseSet<MachineOperand *>>
ReplaceMap;
2659 DenseMap<Register, SmallPtrSet<MachineInstr *, 8>> ReachingDefCopyMap;
2662 DenseMap<unsigned, DenseMap<Register, SmallPtrSet<MachineOperand *, 8>>>
2667 SmallPtrSet<MachineInstr *, 16> RewriteCandsSet;
2668 DenseSet<Register> RewriteSrc2Regs;
2669 for (
auto &[
MI, OriginalOpcode] : RewriteCands) {
2671 MachineOperand *Src2 =
TII->getNamedOperand(*
MI, AMDGPU::OpName::src2);
2672 if (Src2 && Src2->
isReg())
2676 for (
auto &[
MI, OriginalOpcode] : RewriteCands) {
2678 if (ReplacementOp == -1)
2680 MI->setDesc(
TII->get(ReplacementOp));
2683 MachineOperand *Src2 =
TII->getNamedOperand(*
MI, AMDGPU::OpName::src2);
2684 if (Src2->
isReg()) {
2691 findReachingDefs(*Src2,
DAG.LIS, Src2ReachingDefs);
2692 SmallSetVector<MachineInstr *, 8> Src2DefsReplace;
2696 bool Src2NeedsVGPR = Src2NeedsVGPRCache.lookup(
MI);
2698 for (SlotIndex RDIndex : Src2ReachingDefs) {
2699 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIndex);
2700 if (!Src2NeedsVGPR &&
2704 Src2DefsReplace.
insert(RD);
2707 if (!Src2DefsReplace.
empty()) {
2708 auto RI = RedefMap.
find(Src2Reg);
2709 if (RI != RedefMap.
end()) {
2710 MappedReg = RI->second;
2715 SRI->getEquivalentVGPRClass(Src2RC);
2718 MappedReg =
DAG.MRI.createVirtualRegister(VGPRRC);
2719 RedefMap[Src2Reg] = MappedReg;
2724 for (MachineInstr *RD : Src2DefsReplace) {
2726 if (ReachingDefCopyMap[Src2Reg].insert(RD).second) {
2727 MachineInstrBuilder VGPRCopy =
2730 .
addDef(MappedReg, {}, 0)
2731 .addUse(Src2Reg, {}, 0);
2732 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
2737 unsigned UpdateRegion = LastMIToRegion[RD];
2738 DAG.Regions[UpdateRegion].second = VGPRCopy;
2739 LastMIToRegion.
erase(RD);
2746 RewriteRegs.
insert(Src2Reg);
2756 MachineOperand *Dst = &
MI->getOperand(0);
2765 SmallVector<MachineInstr *, 8> DstUseDefsReplace;
2767 findReachingUses(
MI,
DAG.LIS, DstReachingUses);
2769 for (MachineOperand *RUOp : DstReachingUses) {
2770 MachineInstr *UserMI = RUOp->
getParent();
2772 if (
TII->isMAI(*UserMI) && RewriteCandsSet.
contains(UserMI))
2776 if (
find(DstReachingUseCopies, RUOp) == DstReachingUseCopies.
end())
2780 if (
TII->isMAI(*UserMI))
2784 findReachingDefs(*RUOp,
DAG.LIS, DstUsesReachingDefs);
2786 for (SlotIndex RDIndex : DstUsesReachingDefs) {
2787 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIndex);
2788 if (
TII->isMAI(*RD))
2793 if (
find(DstUseDefsReplace, RD) == DstUseDefsReplace.
end())
2798 if (!DstUseDefsReplace.
empty()) {
2799 auto RI = RedefMap.
find(DstReg);
2800 if (RI != RedefMap.
end()) {
2801 MappedReg = RI->second;
2808 MappedReg =
DAG.MRI.createVirtualRegister(VGPRRC);
2809 RedefMap[DstReg] = MappedReg;
2814 for (MachineInstr *RD : DstUseDefsReplace) {
2816 if (ReachingDefCopyMap[DstReg].insert(RD).second) {
2817 MachineInstrBuilder VGPRCopy =
2820 .
addDef(MappedReg, {}, 0)
2821 .addUse(DstReg, {}, 0);
2822 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
2826 auto LMI = LastMIToRegion.
find(RD);
2827 if (LMI != LastMIToRegion.
end()) {
2828 unsigned UpdateRegion = LMI->second;
2829 DAG.Regions[UpdateRegion].second = VGPRCopy;
2830 LastMIToRegion.
erase(RD);
2836 DenseSet<MachineOperand *> &DstRegSet =
ReplaceMap[DstReg];
2839 MachineInstr *EarliestSameBlockUse =
nullptr;
2840 for (MachineOperand *RU : DstReachingUseCopies) {
2841 MachineBasicBlock *RUBlock = RU->getParent()->getParent();
2844 if (RUBlock !=
MI->getParent()) {
2850 if (!SameBlockCopyReg.
isValid()) {
2853 SameBlockCopyReg =
DAG.MRI.createVirtualRegister(VGPRRC);
2857 MachineInstr *UseInst = RU->getParent();
2858 if (!EarliestSameBlockUse ||
2860 DAG.LIS->getInstructionIndex(*UseInst),
2861 DAG.LIS->getInstructionIndex(*EarliestSameBlockUse)))
2862 EarliestSameBlockUse = UseInst;
2863 RU->setReg(SameBlockCopyReg);
2867 if (SameBlockCopyReg.
isValid()) {
2868 MachineInstrBuilder VGPRCopy =
2871 TII->get(TargetOpcode::COPY), SameBlockCopyReg)
2873 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
2878 RewriteRegs.
insert(DstReg);
2888 std::pair<unsigned, DenseMap<Register, SmallPtrSet<MachineOperand *, 8>>>;
2889 for (RUBType RUBlockEntry : ReachingUseTracker) {
2890 using RUDType = std::pair<Register, SmallPtrSet<MachineOperand *, 8>>;
2891 for (RUDType RUDst : RUBlockEntry.second) {
2892 MachineOperand *OpBegin = *RUDst.second.begin();
2893 SlotIndex InstPt =
DAG.LIS->getInstructionIndex(*OpBegin->
getParent());
2896 for (MachineOperand *User : RUDst.second) {
2897 SlotIndex NewInstPt =
DAG.LIS->getInstructionIndex(*
User->getParent());
2904 Register NewUseReg =
DAG.MRI.createVirtualRegister(VGPRRC);
2905 MachineInstr *UseInst =
DAG.LIS->getInstructionFromIndex(InstPt);
2907 MachineInstrBuilder VGPRCopy =
2910 .
addDef(NewUseReg, {}, 0)
2911 .addUse(RUDst.first, {}, 0);
2912 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
2916 auto FI = FirstMIToRegion.
find(UseInst);
2917 if (FI != FirstMIToRegion.
end()) {
2918 unsigned UpdateRegion = FI->second;
2919 DAG.Regions[UpdateRegion].first = VGPRCopy;
2920 FirstMIToRegion.
erase(UseInst);
2924 for (MachineOperand *User : RUDst.second) {
2925 User->setReg(NewUseReg);
2936 for (std::pair<Register, Register> NewDef : RedefMap) {
2941 for (MachineOperand *ReplaceOp :
ReplaceMap[OldReg])
2942 ReplaceOp->setReg(NewReg);
2946 for (
Register RewriteReg : RewriteRegs) {
2947 Register RegToRewrite = RewriteReg;
2950 auto RI = RedefMap.find(RewriteReg);
2951 if (RI != RedefMap.end())
2952 RegToRewrite = RI->second;
2957 DAG.MRI.setRegClass(RegToRewrite, AGPRRC);
2961 DAG.LIS->reanalyze(
DAG.MF);
2963 RegionPressureMap LiveInUpdater(&
DAG,
false);
2964 LiveInUpdater.buildLiveRegMap();
2967 DAG.LiveIns[Region] = LiveInUpdater.getLiveRegsForRegionIdx(Region);
2974unsigned PreRARematStage::getStageTargetOccupancy()
const {
2975 return TargetOcc ? *TargetOcc :
MFI.getMinWavesPerEU();
2978bool PreRARematStage::setObjective() {
2982 unsigned MaxSGPRs =
ST.getMaxNumSGPRs(
F);
2983 unsigned MaxVGPRs =
ST.getMaxNumVGPRs(
F);
2984 bool HasVectorRegisterExcess =
false;
2985 for (
unsigned I = 0,
E =
DAG.Regions.size();
I !=
E; ++
I) {
2986 const GCNRegPressure &
RP =
DAG.Pressure[
I];
2987 GCNRPTarget &
Target = RPTargets.emplace_back(MaxSGPRs, MaxVGPRs,
MF, RP);
2989 TargetRegions.set(
I);
2990 HasVectorRegisterExcess |=
Target.hasVectorRegisterExcess();
2993 if (HasVectorRegisterExcess ||
DAG.MinOccupancy >=
MFI.getMaxWavesPerEU()) {
2996 TargetOcc = std::nullopt;
3000 TargetOcc =
DAG.MinOccupancy + 1;
3001 const unsigned VGPRBlockSize =
MFI.getDynamicVGPRBlockSize();
3002 MaxSGPRs =
ST.getMaxNumSGPRs(*TargetOcc,
false);
3003 MaxVGPRs =
ST.getMaxNumVGPRs(*TargetOcc, VGPRBlockSize);
3004 for (
auto [
I, Target] :
enumerate(RPTargets)) {
3005 Target.setTarget(MaxSGPRs, MaxVGPRs);
3007 TargetRegions.set(
I);
3011 return TargetRegions.any();
3014bool PreRARematStage::ScoredRemat::maybeBeneficial(
3016 for (
unsigned I : TargetRegions.set_bits()) {
3017 if (Live[
I] && RPTargets[
I].isSaveBeneficial(RPSave))
3030 const unsigned NumRegions =
DAG.Regions.size();
3034 for (
unsigned I = 0;
I < NumRegions; ++
I) {
3038 if (BlockFreq && BlockFreq <
MinFreq)
3047 if (
MinFreq >= ScaleFactor * ScaleFactor) {
3048 for (uint64_t &Freq :
Regions)
3049 Freq /= ScaleFactor;
3055void PreRARematStage::ScoredRemat::init(RegisterIdx RegIdx,
3059 this->RegIdx = RegIdx;
3060 const unsigned NumRegions =
DAG.Regions.size();
3061 LiveIn.resize(NumRegions);
3062 LiveOut.resize(NumRegions);
3063 Live.resize(NumRegions);
3064 UnpredictableRPSave.resize(NumRegions);
3068 assert(Reg.Uses.size() == 1 &&
"expected users in single region");
3069 const unsigned UseRegion = Reg.Uses.begin()->first;
3072 for (
unsigned I = 0, E = NumRegions;
I != E; ++
I) {
3073 if (
DAG.LiveIns[
I].contains(DefReg))
3075 if (
DAG.RegionLiveOuts.getLiveRegsForRegionIdx(
I).contains(DefReg))
3080 if (!LiveIn[
I] || !LiveOut[
I] ||
I == UseRegion)
3081 UnpredictableRPSave.set(
I);
3090 int64_t DefOrMin = std::max(Freq.
Regions[Reg.DefRegion], Freq.
MinFreq);
3091 int64_t UseOrMax = Freq.
Regions[UseRegion];
3094 FreqDiff = DefOrMin - UseOrMax;
3097void PreRARematStage::ScoredRemat::update(
const BitVector &TargetRegions,
3099 const FreqInfo &FreqInfo,
3103 for (
unsigned I : TargetRegions.
set_bits()) {
3112 if (!NumRegsBenefit)
3116 RegionImpact += (UnpredictableRPSave[
I] ? 1 : 2) * NumRegsBenefit;
3119 uint64_t Freq = FreqInfo.
Regions[
I];
3120 if (UnpredictableRPSave[
I]) {
3125 MaxFreq = std::max(MaxFreq, Freq);
3130void PreRARematStage::ScoredRemat::rematerialize(
3131 Rematerializer &Remater)
const {
3132 const Rematerializer::Reg &
Reg = Remater.getReg(RegIdx);
3133 Rematerializer::DependencyReuseInfo DRI;
3134 for (RegisterIdx DepRegIdx :
Reg.Dependencies)
3135 DRI.
reuse(DepRegIdx);
3136 unsigned UseRegion =
Reg.Uses.begin()->first;
3137 Remater.rematerializeToRegion(RegIdx, UseRegion, DRI);
3140void PreRARematStage::updateRPTargets(
const BitVector &Regions,
3141 const GCNRegPressure &RPSave) {
3143 RPTargets[
I].saveRP(RPSave);
3144 if (TargetRegions[
I] && RPTargets[
I].satisfied()) {
3146 TargetRegions.reset(
I);
3151bool PreRARematStage::updateAndVerifyRPTargets(
const BitVector &Regions) {
3152 bool TooOptimistic =
false;
3154 GCNRPTarget &
Target = RPTargets[
I];
3160 if (!TargetRegions[
I] && !
Target.satisfied()) {
3162 TooOptimistic =
true;
3163 TargetRegions.set(
I);
3166 return TooOptimistic;
3169void PreRARematStage::removeFromLiveMaps(
Register Reg,
const BitVector &LiveIn,
3170 const BitVector &LiveOut) {
3172 LiveOut.
size() ==
DAG.Regions.size() &&
"region num mismatch");
3176 DAG.RegionLiveOuts.getLiveRegsForRegionIdx(
I).erase(
Reg);
3179void PreRARematStage::addToLiveMaps(
Register Reg, LaneBitmask Mask,
3180 const BitVector &LiveIn,
3181 const BitVector &LiveOut) {
3183 LiveOut.
size() ==
DAG.Regions.size() &&
"region num mismatch");
3184 std::pair<Register, LaneBitmask> LiveReg(
Reg, Mask);
3186 DAG.LiveIns[
I].insert(LiveReg);
3188 DAG.RegionLiveOuts.getLiveRegsForRegionIdx(
I).insert(LiveReg);
3200 if (
DAG.MinOccupancy >= *TargetOcc)
3204 for (
const auto &[
RegionIdx, OrigMIOrder, MaxPressure] : RegionReverts) {
3214 if (AchievedOcc >= *TargetOcc) {
3215 DAG.setTargetOccupancy(AchievedOcc);
3220 DAG.setTargetOccupancy(*TargetOcc - 1);
3225 assert(Rollback &&
"rollbacker should be defined");
3226 Rollback->Listener.rollback(Remater);
3227 for (
const auto &[RegIdx, LiveIn, LiveOut] : Rollback->LiveMapUpdates) {
3228 const Rematerializer::Reg &
Reg = Remater.getReg(RegIdx);
3229 addToLiveMaps(
Reg.getDefReg(),
Reg.Mask, LiveIn, LiveOut);
3232#ifdef EXPENSIVE_CHECKS
3237 for (
unsigned I : RescheduleRegions.set_bits())
3238 DAG.Pressure[
I] =
DAG.getRealRegPressure(
I);
3243void GCNScheduleDAGMILive::setTargetOccupancy(
unsigned TargetOccupancy) {
3244 MinOccupancy = TargetOccupancy;
3245 if (
MFI.getOccupancy() < TargetOccupancy)
3246 MFI.increaseOccupancy(
MF, MinOccupancy);
3248 MFI.limitOccupancy(MinOccupancy);
3265 if (HasIGLPInstrs) {
3266 SavedMutations.clear();
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static SUnit * pickOnlyChoice(SchedBoundary &Zone)
This file implements the BitVector class.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file defines the GCNRegPressure class, which tracks registry pressure by bookkeeping number of S...
static cl::opt< bool > GCNTrackers("amdgpu-use-amdgpu-trackers", cl::Hidden, cl::desc("Use the AMDGPU specific RPTrackers during scheduling"), cl::init(false))
static cl::opt< bool > DisableClusteredLowOccupancy("amdgpu-disable-clustered-low-occupancy-reschedule", cl::Hidden, cl::desc("Disable clustered low occupancy " "rescheduling for ILP scheduling stage."), cl::init(false))
#define REMAT_PREFIX
Allows to easily filter for this stage's debug output.
static cl::opt< unsigned, false, VGPRThresholdParser > VGPRThresholdPercentOpt("amdgpu-vgpr-threshold-percent", cl::Hidden, cl::desc("Percent of VGPR limits that we should use as RP threshold " "during scheduling. We have two limits relevant to scheduling: " "Critical (avoid decreasing occupancy), Excess (avoid spilling). " "This flag scales both limits back by an equal percent: (0 = use " " default calculation, 1-100 = use percentage), default: 0"), cl::init(0))
static MachineInstr * getLastMIForRegion(MachineBasicBlock::iterator RegionBegin, MachineBasicBlock::iterator RegionEnd)
static bool shouldCheckPending(SchedBoundary &Zone, const TargetSchedModel *SchedModel)
static cl::opt< bool > RelaxedOcc("amdgpu-schedule-relaxed-occupancy", cl::Hidden, cl::desc("Relax occupancy targets for kernels which are memory " "bound (amdgpu-membound-threshold), or " "Wave Limited (amdgpu-limit-wave-threshold)."), cl::init(false))
static cl::opt< bool > DisableUnclusterHighRP("amdgpu-disable-unclustered-high-rp-reschedule", cl::Hidden, cl::desc("Disable unclustered high register pressure " "reduction scheduling stage."), cl::init(false))
static void printScheduleModel(std::set< std::pair< MachineInstr *, unsigned >, EarlierIssuingCycle > &ReadyCycles)
static bool isReachingDefAGPRForm(MachineInstr *RD, const SmallPtrSetImpl< MachineInstr * > &RewriteSet, const DenseSet< Register > &CandSrc2Regs, const SIInstrInfo &TII)
Returns true if reaching def RD will be in AGPR form after the rewrite and so needs no bridge copy: a...
static cl::opt< bool > PrintMaxRPRegUsageAfterScheduler("amdgpu-print-max-reg-pressure-regusage-after-scheduler", cl::Hidden, cl::desc("Print a list of live registers along with their def/uses at the " "point of maximum register pressure after scheduling."), cl::init(false))
static bool hasIGLPInstrs(ScheduleDAGInstrs *DAG)
static cl::opt< bool > DisableRewriteMFMAFormSchedStage("amdgpu-disable-rewrite-mfma-form-sched-stage", cl::Hidden, cl::desc("Disable rewrite mfma rewrite scheduling stage"), cl::init(true))
static bool canUsePressureDiffs(const SUnit &SU)
Checks whether SU can use the cached DAG pressure diffs to compute the current register pressure.
static cl::opt< unsigned > PendingQueueLimit("amdgpu-scheduler-pending-queue-limit", cl::Hidden, cl::desc("Max (Available+Pending) size to inspect pending queue (0 disables)"), cl::init(256))
static cl::opt< bool > PrintMaxRPRegUsageBeforeScheduler("amdgpu-print-max-reg-pressure-regusage-before-scheduler", cl::Hidden, cl::desc("Print a list of live registers along with their def/uses at the " "point of maximum register pressure before scheduling."), cl::init(false))
static cl::opt< unsigned > ScheduleMetricBias("amdgpu-schedule-metric-bias", cl::Hidden, cl::desc("Sets the bias which adds weight to occupancy vs latency. Set it to " "100 to chase the occupancy only."), cl::init(10))
static Register UseReg(const MachineOperand &MO)
const HexagonInstrInfo * TII
static constexpr std::pair< StringLiteral, StringLiteral > ReplaceMap[]
iv Induction Variable Users
A common definition of LaneBitmask for use in TableGen and CodeGen.
static llvm::Error parse(GsymDataExtractor &Data, uint64_t BaseAddr, LineEntryCallback const &Callback)
Promote Memory to Register
MIR-level target-independent rematerialization helpers.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & front() const
Get the first element.
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
iterator_range< const_set_bits_iterator > set_bits() const
size_type size() const
Returns the number of bits in this bitvector.
uint64_t getFrequency() const
Returns the frequency as a fixpoint number scaled by the entry frequency.
bool initGCNSchedStage() override
bool shouldRevertScheduling(unsigned WavesAfter) override
bool initGCNRegion() override
iterator find(const_arg_type_t< KeyT > Val)
bool erase(const KeyT &Val)
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Implements a dense probed hash-table based set.
bool reset(const MachineInstr &MI, MachineBasicBlock::const_iterator End, const LiveRegSet *LiveRegs=nullptr)
Reset tracker to the point before the MI filling LiveRegs upon this point using LIS.
GCNRegPressure bumpDownwardPressure(const MachineInstr *MI, const SIRegisterInfo *TRI) const
Mostly copy/paste from CodeGen/RegisterPressure.cpp Calculate the impact MI will have on CurPressure ...
GCNMaxILPSchedStrategy(const MachineSchedContext *C)
bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const override
Apply a set of heuristics to a new candidate.
bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const override
GCNMaxMemoryClauseSchedStrategy tries best to clause memory instructions as much as possible.
GCNMaxMemoryClauseSchedStrategy(const MachineSchedContext *C)
GCNMaxOccupancySchedStrategy(const MachineSchedContext *C, bool IsLegacyScheduler=false)
void finalizeSchedule() override
Allow targets to perform final scheduling actions at the level of the whole MachineFunction.
void schedule() override
Orders nodes according to selected style.
GCNPostScheduleDAGMILive(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S, bool RemoveKillFlags)
Models a register pressure target, allowing to evaluate and track register savings against that targe...
unsigned getNumRegsBenefit(const GCNRegPressure &SaveRP) const
Returns the benefit towards achieving the RP target that saving SaveRP represents,...
GCNRegPressure getPressure() const
virtual bool initGCNRegion()
GCNRegPressure PressureBefore
bool isRegionWithExcessRP() const
void modifyRegionSchedule(unsigned RegionIdx, ArrayRef< MachineInstr * > MIOrder)
Sets the schedule of region RegionIdx to MIOrder.
bool mayCauseSpilling(unsigned WavesAfter)
ScheduleMetrics getScheduleMetrics(const std::vector< SUnit > &InputSchedule)
GCNScheduleDAGMILive & DAG
const GCNSchedStageID StageID
std::vector< MachineInstr * > Unsched
GCNRegPressure PressureAfter
virtual void finalizeGCNRegion()
SIMachineFunctionInfo & MFI
unsigned computeSUnitReadyCycle(const SUnit &SU, unsigned CurrCycle, DenseMap< unsigned, unsigned > &ReadyCycles, const TargetSchedModel &SM)
virtual void finalizeGCNSchedStage()
virtual bool initGCNSchedStage()
virtual bool shouldRevertScheduling(unsigned WavesAfter)
std::vector< std::unique_ptr< ScheduleDAGMutation > > SavedMutations
GCNSchedStage(GCNSchedStageID StageID, GCNScheduleDAGMILive &DAG)
MachineBasicBlock * CurrentMBB
This is a minimal scheduler strategy.
GCNDownwardRPTracker DownwardTracker
bool useGCNTrackers() const
void getRegisterPressures(bool AtTop, const RegPressureTracker &RPTracker, SUnit *SU, std::vector< unsigned > &Pressure, std::vector< unsigned > &MaxPressure, GCNDownwardRPTracker &DownwardTracker, GCNUpwardRPTracker &UpwardTracker, ScheduleDAGMI *DAG, const SIRegisterInfo *SRI)
GCNSchedStrategy(const MachineSchedContext *C)
SmallVector< GCNSchedStageID, 4 > SchedStages
unsigned SGPRCriticalLimit
std::vector< unsigned > MaxPressure
bool hasNextStage() const
SUnit * pickNodeBidirectional(bool &IsTopNode, bool &PickedPending)
GCNSchedStageID getCurrentStage()
bool tryPendingCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const
Evaluates instructions in the pending queue using a subset of scheduling heuristics.
SmallVectorImpl< GCNSchedStageID >::iterator CurrentStage
unsigned VGPRCriticalLimit
void schedNode(SUnit *SU, bool IsTopNode) override
Notify MachineSchedStrategy that ScheduleDAGMI has scheduled an instruction and updated scheduled/rem...
std::optional< bool > GCNTrackersOverride
GCNDownwardRPTracker * getDownwardTracker()
std::vector< unsigned > Pressure
void initialize(ScheduleDAGMI *DAG) override
Initialize the strategy after building the DAG for a new region.
GCNUpwardRPTracker UpwardTracker
void printCandidateDecision(const SchedCandidate &Current, const SchedCandidate &Preferred)
void pickNodeFromQueue(SchedBoundary &Zone, const CandPolicy &ZonePolicy, const RegPressureTracker &RPTracker, SchedCandidate &Cand, bool &IsPending, bool IsBottomUp)
unsigned getStructuralStallCycles(SchedBoundary &Zone, SUnit *SU) const
Estimate how many cycles SU must wait due to structural hazards at the current boundary cycle.
void initCandidate(SchedCandidate &Cand, SUnit *SU, bool AtTop, const RegPressureTracker &RPTracker, const SIRegisterInfo *SRI, unsigned SGPRPressure, unsigned VGPRPressure, bool IsBottomUp)
SUnit * pickNode(bool &IsTopNode) override
Pick the next node to schedule, or return NULL.
GCNUpwardRPTracker * getUpwardTracker()
GCNSchedStageID getNextStage() const
void finalizeSchedule() override
Allow targets to perform final scheduling actions at the level of the whole MachineFunction.
void schedule() override
Orders nodes according to selected style.
GCNScheduleDAGMILive(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S)
void recede(const MachineInstr &MI)
Move to the state of RP just before the MI .
void reset(const MachineInstr &MI)
Resets tracker to the point just after MI (in program order), which can be a debug instruction.
void compute(FunctionT &F)
Compute the cycle info for a function.
void traceCandidate(const SchedCandidate &Cand)
LLVM_ABI void setPolicy(CandPolicy &Policy, bool IsPostRA, SchedBoundary &CurrZone, SchedBoundary *OtherZone)
Set the CandPolicy given a scheduling zone given the current resources and latencies inside and outsi...
MachineSchedPolicy RegionPolicy
const TargetSchedModel * SchedModel
const MachineSchedContext * Context
const TargetRegisterInfo * TRI
SchedCandidate BotCand
Candidate last picked from Bot boundary.
SchedCandidate TopCand
Candidate last picked from Top boundary.
virtual bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const
Apply a set of heuristics to a new candidate.
void initialize(ScheduleDAGMI *dag) override
Initialize the strategy after building the DAG for a new region.
void schedNode(SUnit *SU, bool IsTopNode) override
Update the scheduler's state after scheduling a node.
GenericScheduler(const MachineSchedContext *C)
bool shouldRevertScheduling(unsigned WavesAfter) override
LiveInterval - This class represents the liveness of a register, or stack slot.
bool hasSubRanges() const
Returns true if subregister liveness information is available.
iterator_range< subrange_iterator > subranges()
SlotIndex getInstructionIndex(const MachineInstr &Instr) const
Returns the base index of the given instruction.
SlotIndex getMBBEndIdx(const MachineBasicBlock *mbb) const
Return the last index in the given basic block.
LiveInterval & getInterval(Register Reg)
LLVM_ABI void dump() const
MachineBasicBlock * getMBBFromIndex(SlotIndex index) const
VNInfo * getVNInfoAt(SlotIndex Idx) const
getVNInfoAt - Return the VNInfo that is live at Idx, or NULL.
uint8_t getCopyCost() const
getCopyCost - Return the cost of copying a value between two registers in this class.
int getNumber() const
MachineBasicBlocks are uniquely numbered at the function level, unless they're not in a MachineFuncti...
succ_iterator succ_begin()
unsigned succ_size() const
iterator_range< pred_iterator > predecessors()
MachineInstrBundleIterator< MachineInstr > iterator
MachineBlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate machine basic b...
LLVM_ABI BlockFrequency getBlockFreq(const MachineBasicBlock *MBB) const
getblockFreq - Return block frequency.
LLVM_ABI BlockFrequency getEntryFreq() const
Divide a block's BlockFrequency::getFrequency() value by this value to obtain the entry block - relat...
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
unsigned getNumOperands() const
Retuns the total number of operands.
bool mayLoad(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read memory.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
const MachineOperand & getOperand(unsigned i) const
MachineOperand class - Representation of each machine instruction operand.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
Register getReg() const
getReg - Returns the register number.
bool shouldRevertScheduling(unsigned WavesAfter) override
bool shouldRevertScheduling(unsigned WavesAfter) override
bool shouldRevertScheduling(unsigned WavesAfter) override
void finalizeGCNRegion() override
bool initGCNRegion() override
bool initGCNSchedStage() override
Capture a change in pressure for a single pressure set.
Simple wrapper around std::function<void(raw_ostream&)>.
Helpers for implementing custom MachineSchedStrategy classes.
Track the current register pressure at some position in the instruction stream, and remember the high...
LLVM_ABI void advance()
Advance across the current instruction.
LLVM_ABI void getDownwardPressure(const MachineInstr *MI, std::vector< unsigned > &PressureResult, std::vector< unsigned > &MaxPressureResult)
Get the pressure of each PSet after traversing this instruction top-down.
const std::vector< unsigned > & getRegSetPressureAtPos() const
Get the register set pressure at the current position, which may be less than the pressure across the...
LLVM_ABI void getUpwardPressure(const MachineInstr *MI, std::vector< unsigned > &PressureResult, std::vector< unsigned > &MaxPressureResult)
Get the pressure of each PSet after traversing this instruction bottom-up.
List of registers defined and used by a machine instruction.
LLVM_ABI void adjustLaneLiveness(const LiveIntervals &LIS, const MachineRegisterInfo &MRI, SlotIndex Pos)
Use liveness information to find out which uses/defs are partially undefined/dead at Pos and adjust t...
LLVM_ABI void collect(const MachineInstr &MI, const TargetRegisterInfo &TRI, const MachineRegisterInfo &MRI, bool TrackLaneMasks, bool IgnoreDead)
Analyze the given instruction MI and fill in the Uses, Defs and DeadDefs list based on the MachineOpe...
LLVM_ABI void detectDeadDefs(const MachineInstr &MI, const LiveIntervals &LIS)
Use liveness information to find dead defs not marked with a dead flag and move them to the DeadDefs ...
Wrapper class representing virtual and physical registers.
constexpr bool isValid() const
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
MIR-level target-independent rematerializer.
bool isIGLPMutationOnly(unsigned Opcode) const
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
unsigned getOccupancy() const
unsigned getDynamicVGPRBlockSize() const
unsigned getMinAllowedOccupancy() const
Scheduling unit. This is a node in the scheduling DAG.
bool isInstr() const
Returns true if this SUnit refers to a machine instruction as opposed to an SDNode.
unsigned TopReadyCycle
Cycle relative to start when node is ready.
unsigned NodeNum
Entry # of node in the node vector.
unsigned short Latency
Node latency.
bool isScheduled
True once scheduled.
unsigned ParentClusterIdx
The parent cluster id.
unsigned BotReadyCycle
Cycle relative to end when node is ready.
bool hasReservedResource
Uses a reserved resource.
bool isBottomReady() const
SmallVector< SDep, 4 > Preds
All sunit predecessors.
MachineInstr * getInstr() const
Returns the representative MachineInstr for this SUnit.
Each Scheduling boundary is associated with ready queues.
LLVM_ABI void releasePending()
Release pending ready nodes in to the available queue.
LLVM_ABI unsigned getLatencyStallCycles(SUnit *SU)
Get the difference between the given SUnit's ready time and the current cycle.
LLVM_ABI SUnit * pickOnlyChoice()
Call this before applying any other heuristics to the Available queue.
LLVM_ABI void bumpCycle(unsigned NextCycle)
Move the boundary of scheduled code by one cycle.
unsigned getCurrMOps() const
Micro-ops issued in the current cycle.
unsigned getCurrCycle() const
Number of cycles to issue the instructions scheduled in this zone.
std::unique_ptr< ScheduleHazardRecognizer > HazardRec
LLVM_ABI bool checkHazard(SUnit *SU)
Does this SU have a hazard within the current instruction group.
LLVM_ABI std::pair< unsigned, unsigned > getNextResourceCycle(const MCSchedClassDesc *SC, unsigned PIdx, unsigned ReleaseAtCycle, unsigned AcquireAtCycle)
Compute the next cycle at which the given processor resource can be scheduled.
A ScheduleDAG for scheduling lists of MachineInstr.
bool ScheduleSingleMIRegions
True if regions with a single MI should be scheduled.
MachineBasicBlock::iterator RegionEnd
The end of the range to be scheduled.
virtual void finalizeSchedule()
Allow targets to perform final scheduling actions at the level of the whole MachineFunction.
virtual void exitRegion()
Called when the scheduler has finished scheduling the current region.
const MachineLoopInfo * MLI
bool RemoveKillFlags
True if the DAG builder should remove kill flags (in preparation for rescheduling).
MachineBasicBlock::iterator RegionBegin
The beginning of the range to be scheduled.
void schedule() override
Implement ScheduleDAGInstrs interface for scheduling a sequence of reorderable instructions.
ScheduleDAGMILive(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S)
RegPressureTracker RPTracker
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
void addMutation(std::unique_ptr< ScheduleDAGMutation > Mutation)
Add a postprocessing step to the DAG builder.
void schedule() override
Implement ScheduleDAGInstrs interface for scheduling a sequence of reorderable instructions.
ScheduleDAGMI(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S, bool RemoveKillFlags)
std::vector< std::unique_ptr< ScheduleDAGMutation > > Mutations
Ordered list of DAG postprocessing steps.
MachineRegisterInfo & MRI
Virtual/real register map.
const TargetInstrInfo * TII
Target instruction information.
MachineFunction & MF
Machine function.
static const unsigned ScaleFactor
unsigned getMetric() const
bool empty() const
Determine if the SetVector is empty or not.
bool insert(const value_type &X)
Insert a new element into the SetVector.
SlotIndex - An opaque wrapper around machine indexes.
static bool isSameInstr(SlotIndex A, SlotIndex B)
isSameInstr - Return true if A and B refer to the same instruction.
static bool isEarlierInstr(SlotIndex A, SlotIndex B)
isEarlierInstr - Return true if A refers to an instruction earlier than B.
SlotIndex getPrevSlot() const
Returns the previous slot in the index list.
SlotIndex getMBBStartIdx(const MachineBasicBlock *mbb) const
Returns the first index in the given basic block.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
SmallSet - This maintains a set of unique values, optimizing for the case when the set is small (less...
bool contains(const T &V) const
Check if the SmallSet contains the given element.
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Provide an instruction scheduling machine model to CodeGen passes.
LLVM_ABI bool hasInstrSchedModel() const
Return true if this machine model includes an instruction-level scheduling model.
unsigned getMicroOpBufferSize() const
Number of micro-ops that may be buffered for OOO execution.
bool initGCNSchedStage() override
bool initGCNRegion() override
void finalizeGCNSchedStage() override
bool shouldRevertScheduling(unsigned WavesAfter) override
VNInfo - Value Number Information.
SlotIndex def
The index of the defining instruction.
bool isPHIDef() const
Returns true if this value is defined by a PHI instruction (or was, PHI instructions may have been el...
std::pair< iterator, bool > insert(const ValueT &V)
bool contains(const_arg_type_t< ValueT > V) const
Check if the set contains the given element.
self_iterator getIterator()
This class implements an extremely fast bulk output stream that can only output to a stream.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize)
unsigned getAllocatedNumVGPRBlocks(const MCSubtargetInfo &STI, unsigned NumVGPRs, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
LLVM_READONLY int32_t getAGPRFormOp(uint32_t Opcode)
This namespace contains all of the command line option processing machinery.
initializer< Ty > init(const Ty &Val)
@ User
could "use" a pointer
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI int biasPhysReg(const SUnit *SU, bool isTop, bool BiasPRegsExtra=false)
Minimize physical register live ranges.
auto find(R &&Range, const T &Val)
Provide wrappers to std::find which take ranges instead of having to pass begin/end explicitly.
bool isEqual(const GCNRPTracker::LiveRegSet &S1, const GCNRPTracker::LiveRegSet &S2)
Printable print(const GCNRegPressure &RP, const GCNSubtarget *ST=nullptr, unsigned DynamicVGPRBlockSize=0)
LLVM_ABI unsigned getWeakLeft(const SUnit *SU, bool isTop)
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
GCNRegPressure getRegPressure(const MachineRegisterInfo &MRI, Range &&LiveRegs)
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
std::unique_ptr< ScheduleDAGMutation > createIGroupLPDAGMutation(AMDGPU::SchedulingPhase Phase)
Phase specifes whether or not this is a reentry into the IGroupLPDAGMutation.
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
std::pair< MachineBasicBlock::iterator, MachineBasicBlock::iterator > RegionBoundaries
A region's boundaries i.e.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
IterT skipDebugInstructionsForward(IterT It, IterT End, bool SkipPseudoOp=true)
Increment It until it points to a non-debug instruction or to End and return the resulting iterator.
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI bool tryPressure(const PressureChange &TryP, const PressureChange &CandP, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason, const TargetRegisterInfo *TRI, const MachineFunction &MF)
@ UnclusteredHighRPReschedule
@ MemoryClauseInitialSchedule
@ ClusteredLowOccupancyReschedule
auto reverse(ContainerTy &&C)
void sort(IteratorTy Start, IteratorTy End)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
LLVM_ABI cl::opt< bool > VerifyScheduling
LLVM_ABI bool tryLatency(GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, SchedBoundary &Zone)
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
IterT skipDebugInstructionsBackward(IterT It, IterT Begin, bool SkipPseudoOp=true)
Decrement It until it points to a non-debug instruction or to Begin and return the resulting iterator...
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
bool isTheSameCluster(unsigned A, unsigned B)
Return whether the input cluster ID's are the same and valid.
DWARFExpression::Operation Op
LLVM_ABI bool tryGreater(int TryVal, int CandVal, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason)
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
ArrayRef(const T &OneElt) -> ArrayRef< T >
OutputIt move(R &&Range, OutputIt Out)
Provide wrappers to std::move which take ranges instead of having to pass begin/end explicitly.
DenseMap< MachineInstr *, GCNRPTracker::LiveRegSet > getLiveRegMap(Range &&R, bool After, LiveIntervals &LIS)
creates a map MachineInstr -> LiveRegSet R - range of iterators on instructions After - upon entry or...
GCNRPTracker::LiveRegSet getLiveRegsBefore(const MachineInstr &MI, const LiveIntervals &LIS)
LLVM_ABI bool tryLess(int TryVal, int CandVal, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason)
Return true if this heuristic determines order.
LLVM_ABI void dumpMaxRegPressure(MachineFunction &MF, GCNRegPressure::RegKind Kind, LiveIntervals &LIS, const MachineLoopInfo *MLI)
LLVM_ABI Printable printMBBReference(const MachineBasicBlock &MBB)
Prints a machine basic block reference.
MCRegisterClass TargetRegisterClass
Implement std::hash so that hash_code can be used in STL containers.
bool operator()(std::pair< MachineInstr *, unsigned > A, std::pair< MachineInstr *, unsigned > B) const
unsigned getArchVGPRNum() const
unsigned getAGPRNum() const
unsigned getSGPRNum() const
Policy for scheduling the next instruction in the candidate's zone.
Store the state used by GenericScheduler heuristics, required for the lifetime of one invocation of p...
void setBest(SchedCandidate &Best)
void reset(const CandPolicy &NewPolicy)
LLVM_ABI void initResourceDelta(const ScheduleDAGMI *DAG, const TargetSchedModel *SchedModel)
SchedResourceDelta ResDelta
Status of an instruction's critical resource consumption.
unsigned DemandedResources
constexpr bool any() const
static constexpr LaneBitmask getNone()
Summarize the scheduling resources required for an instruction of a particular scheduling class.
Identify one of the processor resource kinds consumed by a particular scheduling class for the specif...
MachineSchedContext provides enough context from the MachineScheduler pass for the target to instanti...
Execution frequency information required by scoring heuristics.
SmallVector< uint64_t > Regions
Per-region execution frequencies. 0 when unknown.
uint64_t MinFreq
Minimum and maximum observed frequencies.
FreqInfo(MachineFunction &MF, const GCNScheduleDAGMILive &DAG)
PressureChange CriticalMax
PressureChange CurrentMax
DependencyReuseInfo & reuse(RegisterIdx DepIdx)
A rematerializable register, potentially defined by multiple instructions.
LLVM_ABI std::pair< MachineInstr *, MachineInstr * > getRegionUseBounds(unsigned UseRegion, const LiveIntervals &LIS) const
Returns the first and last user of the register in region UseRegion.
SmallVector< MachineInstr *, 1 > Defs
All instructions that define the register, in program order.
SmallDenseMap< unsigned, RegionUsers, 2 > Uses
Uses of the register, mapped by region.
MachineInstr * getLastDef() const
SmallVector< RegisterIdx, 2 > Dependencies
This register's rematerializable dependencies, one per unique rematerializable register operand over ...