47#define DEBUG_TYPE "machine-scheduler"
52 "amdgpu-disable-unclustered-high-rp-reschedule",
cl::Hidden,
53 cl::desc(
"Disable unclustered high register pressure "
54 "reduction scheduling stage."),
58 "amdgpu-disable-clustered-low-occupancy-reschedule",
cl::Hidden,
59 cl::desc(
"Disable clustered low occupancy "
60 "rescheduling for ILP scheduling stage."),
66 "Sets the bias which adds weight to occupancy vs latency. Set it to "
67 "100 to chase the occupancy only."),
72 cl::desc(
"Relax occupancy targets for kernels which are memory "
73 "bound (amdgpu-membound-threshold), or "
74 "Wave Limited (amdgpu-limit-wave-threshold)."),
79 cl::desc(
"Use the AMDGPU specific RPTrackers during scheduling"),
83 "amdgpu-scheduler-pending-queue-limit",
cl::Hidden,
85 "Max (Available+Pending) size to inspect pending queue (0 disables)"),
88#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
89#define DUMP_MAX_REG_PRESSURE
91 "amdgpu-print-max-reg-pressure-regusage-before-scheduler",
cl::Hidden,
92 cl::desc(
"Print a list of live registers along with their def/uses at the "
93 "point of maximum register pressure before scheduling."),
97 "amdgpu-print-max-reg-pressure-regusage-after-scheduler",
cl::Hidden,
98 cl::desc(
"Print a list of live registers along with their def/uses at the "
99 "point of maximum register pressure after scheduling."),
104 "amdgpu-disable-rewrite-mfma-form-sched-stage",
cl::Hidden,
110 return O.error(
"'" + Arg +
"' value invalid for uint argument!");
113 return O.error(
"'" + Arg +
"' value must be in the range [0, 100]!");
120 cl::desc(
"Percent of VGPR limits that we should use as RP threshold "
121 "during scheduling. We have two limits relevant to scheduling: "
122 "Critical (avoid decreasing occupancy), Excess (avoid spilling). "
123 "This flag scales both limits back by an equal percent: (0 = use "
124 " default calculation, 1-100 = use percentage), default: 0"),
145 Context->RegClassInfo->getNumAllocatableRegs(&AMDGPU::SGPR_32RegClass);
147 Context->RegClassInfo->getNumAllocatableRegs(&AMDGPU::VGPR_32RegClass);
149 Context->RegClassInfo->getNumAllocatableRegs(&AMDGPU::AGPR_32RegClass);
171 "VGPRCriticalLimit calculation method.\n");
175 unsigned Addressable =
178 VGPRBudget = std::max(VGPRBudget, Granule);
194 <<
". VGPRCriticalLimit: " << OriginalVGPRCriticalLimit
241 if (!
Op.isReg() ||
Op.isImplicit())
243 if (
Op.getReg().isPhysical() ||
244 (
Op.isDef() &&
Op.getSubReg() != AMDGPU::NoSubRegister))
279 Pressure[AMDGPU::RegisterPressureSets::VGPR_32] =
288 unsigned SGPRPressure,
289 unsigned VGPRPressure,
290 unsigned AGPRPressure,
bool IsBottomUp) {
294 if (!
DAG->isTrackingPressure())
317 Pressure[AMDGPU::RegisterPressureSets::SReg_32] = SGPRPressure;
318 Pressure[AMDGPU::RegisterPressureSets::VGPR_32] = VGPRPressure;
319 Pressure[AMDGPU::RegisterPressureSets::AGPR_32] = AGPRPressure;
321 for (
const auto &Diff :
DAG->getPressureDiff(SU)) {
327 (IsBottomUp ? Diff.getUnitInc() : -Diff.getUnitInc());
330#ifdef EXPENSIVE_CHECKS
331 std::vector<unsigned> CheckPressure, CheckMaxPressure;
334 if (
Pressure[AMDGPU::RegisterPressureSets::SReg_32] !=
335 CheckPressure[AMDGPU::RegisterPressureSets::SReg_32] ||
336 Pressure[AMDGPU::RegisterPressureSets::VGPR_32] !=
337 CheckPressure[AMDGPU::RegisterPressureSets::VGPR_32] ||
338 Pressure[AMDGPU::RegisterPressureSets::AGPR_32] !=
339 CheckPressure[AMDGPU::RegisterPressureSets::AGPR_32]) {
340 errs() <<
"Register Pressure is inaccurate when calculated through "
342 <<
"SGPR got " <<
Pressure[AMDGPU::RegisterPressureSets::SReg_32]
344 << CheckPressure[AMDGPU::RegisterPressureSets::SReg_32] <<
"\n"
345 <<
"VGPR got " <<
Pressure[AMDGPU::RegisterPressureSets::VGPR_32]
347 << CheckPressure[AMDGPU::RegisterPressureSets::VGPR_32] <<
"\n"
348 <<
"AGPR got " <<
Pressure[AMDGPU::RegisterPressureSets::AGPR_32]
350 << CheckPressure[AMDGPU::RegisterPressureSets::AGPR_32] <<
"\n";
356 unsigned NewAGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::AGPR_32];
357 unsigned NewSGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::SReg_32];
358 unsigned NewVGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::VGPR_32];
368 const unsigned MaxVGPRPressureInc = 16;
369 bool ShouldTrackVGPRs = VGPRPressure + MaxVGPRPressureInc >=
VGPRExcessLimit;
372 bool ShouldTrackSGPRs =
373 !ShouldTrackVGPRs && !ShouldTrackAGPRs && SGPRPressure >=
SGPRExcessLimit;
408 : std::numeric_limits<int>::min();
410 if (SGPRDelta >= 0 || VGPRDelta >= 0 || AGPRDelta >= 0) {
413 if (VGPRDelta >= SGPRDelta && VGPRDelta >= AGPRDelta) {
417 }
else if (AGPRDelta >= SGPRDelta) {
431 bool HasBufferedModel =
450 dbgs() <<
"Prefer:\t\t";
451 DAG->dumpNode(*Preferred.
SU);
455 DAG->dumpNode(*Current.
SU);
458 dbgs() <<
"Reason:\t\t";
472 unsigned SGPRPressure = 0;
473 unsigned VGPRPressure = 0;
474 unsigned AGPRPressure = 0;
476 if (
DAG->isTrackingPressure()) {
478 SGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::SReg_32];
479 VGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::VGPR_32];
480 AGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::AGPR_32];
485 SGPRPressure =
T->getPressure().getSGPRNum();
486 VGPRPressure =
T->getPressure().getArchVGPRNum();
487 AGPRPressure =
T->getPressure().getAGPRNum();
492 for (
SUnit *SU : AQ) {
496 VGPRPressure, AGPRPressure, IsBottomUp);
516 for (
SUnit *SU : PQ) {
520 VGPRPressure, AGPRPressure, IsBottomUp);
540 bool &PickedPending) {
560 bool BotPending =
false;
580 "Last pick result should correspond to re-picking right now");
585 bool TopPending =
false;
605 "Last pick result should correspond to re-picking right now");
615 PickedPending = BotPending && TopPending;
618 if (BotPending || TopPending) {
625 Cand.setBest(TryCand);
630 IsTopNode = Cand.AtTop;
637 if (
DAG->top() ==
DAG->bottom()) {
639 Bot.Available.empty() &&
Bot.Pending.empty() &&
"ReadyQ garbage");
645 PickedPending =
false;
679 if (ReadyCycle > CurrentCycle)
745 if (
DAG->isTrackingPressure() &&
751 if (
DAG->isTrackingPressure() &&
756 bool SameBoundary = Zone !=
nullptr;
780 if (IsLegacyScheduler)
799 if (
DAG->isTrackingPressure() &&
809 bool SameBoundary = Zone !=
nullptr;
844 bool CandIsClusterSucc =
846 bool TryCandIsClusterSucc =
848 if (
tryGreater(TryCandIsClusterSucc, CandIsClusterSucc, TryCand, Cand,
853 if (
DAG->isTrackingPressure() &&
859 if (
DAG->isTrackingPressure() &&
905 if (
DAG->isTrackingPressure()) {
921 bool CandIsClusterSucc =
923 bool TryCandIsClusterSucc =
925 if (
tryGreater(TryCandIsClusterSucc, CandIsClusterSucc, TryCand, Cand,
934 bool SameBoundary = Zone !=
nullptr;
951 if (TryMayLoad || CandMayLoad) {
952 bool TryLongLatency =
954 bool CandLongLatency =
958 Zone->
isTop() ? CandLongLatency : TryLongLatency, TryCand,
976 if (
DAG->isTrackingPressure() &&
995 !
Rem.IsAcyclicLatencyLimited &&
tryLatency(TryCand, Cand, *Zone))
1013 StartingOccupancy(MFI.getOccupancy()), MinOccupancy(StartingOccupancy),
1014 RegionLiveOuts(this,
true) {
1020 LLVM_DEBUG(
dbgs() <<
"Starting occupancy is " << StartingOccupancy <<
".\n");
1022 MinOccupancy = std::min(MFI.getMinAllowedOccupancy(), StartingOccupancy);
1023 if (MinOccupancy != StartingOccupancy)
1024 LLVM_DEBUG(
dbgs() <<
"Allowing Occupancy drops to " << MinOccupancy
1029std::unique_ptr<GCNSchedStage>
1031 switch (SchedStageID) {
1033 return std::make_unique<OccInitialScheduleStage>(SchedStageID, *
this);
1035 return std::make_unique<RewriteMFMAFormStage>(SchedStageID, *
this);
1037 return std::make_unique<UnclusteredHighRPStage>(SchedStageID, *
this);
1039 return std::make_unique<ClusteredLowOccStage>(SchedStageID, *
this);
1041 return std::make_unique<PreRARematStage>(SchedStageID, *
this);
1043 return std::make_unique<ILPInitialScheduleStage>(SchedStageID, *
this);
1045 return std::make_unique<MemoryClauseInitialScheduleStage>(SchedStageID,
1048 return std::make_unique<LiveIntervalRPStage>(SchedStageID, *
this);
1061GCNScheduleDAGMILive::getRealRegPressure(
unsigned RegionIdx)
const {
1062 if (Regions[RegionIdx].first == Regions[RegionIdx].second)
1066 &LiveIns[RegionIdx]);
1072 assert(RegionBegin != RegionEnd &&
"Region must not be empty");
1076void GCNScheduleDAGMILive::computeBlockPressure(
unsigned RegionIdx,
1088 const MachineBasicBlock *OnlySucc =
nullptr;
1091 if (!Candidate->empty() && Candidate->pred_size() == 1) {
1092 SlotIndexes *Ind =
LIS->getSlotIndexes();
1094 OnlySucc = Candidate;
1099 size_t CurRegion = RegionIdx;
1100 for (
size_t E = Regions.size(); CurRegion !=
E; ++CurRegion)
1101 if (Regions[CurRegion].first->getParent() !=
MBB)
1106 auto LiveInIt = MBBLiveIns.find(
MBB);
1107 auto &Rgn = Regions[CurRegion];
1109 if (LiveInIt != MBBLiveIns.end()) {
1110 auto LiveIn = std::move(LiveInIt->second);
1112 MBBLiveIns.erase(LiveInIt);
1115 auto LRS = BBLiveInMap.lookup(NonDbgMI);
1116#ifdef EXPENSIVE_CHECKS
1125 if (Regions[CurRegion].first ==
I || NonDbgMI ==
I) {
1126 LiveIns[CurRegion] =
RPTracker.getLiveRegs();
1130 if (Regions[CurRegion].second ==
I) {
1131 Pressure[CurRegion] =
RPTracker.moveMaxPressure();
1132 if (CurRegion-- == RegionIdx)
1134 auto &Rgn = Regions[CurRegion];
1147 MBBLiveIns[OnlySucc] =
RPTracker.moveLiveRegs();
1152GCNScheduleDAGMILive::getRegionLiveInMap()
const {
1153 assert(!Regions.empty());
1154 std::vector<MachineInstr *> RegionFirstMIs;
1155 RegionFirstMIs.reserve(Regions.size());
1157 RegionFirstMIs.push_back(
1164GCNScheduleDAGMILive::getRegionLiveOutMap()
const {
1165 assert(!Regions.empty());
1166 std::vector<MachineInstr *> RegionLastMIs;
1167 RegionLastMIs.reserve(Regions.size());
1178 IdxToInstruction.clear();
1181 IsLiveOut ? DAG->getRegionLiveOutMap() : DAG->getRegionLiveInMap();
1182 for (
unsigned I = 0;
I < DAG->Regions.size();
I++) {
1183 auto &[RegionBegin, RegionEnd] = DAG->Regions[
I];
1185 if (RegionBegin == RegionEnd)
1189 IdxToInstruction[
I] = RegionKey;
1197 LiveIns.resize(Regions.size());
1198 Pressure.resize(Regions.size());
1199 RegionsWithHighRP.resize(Regions.size());
1200 RegionsWithExcessRP.resize(Regions.size());
1201 RegionsWithIGLPInstrs.resize(Regions.size());
1202 RegionsWithHighRP.reset();
1203 RegionsWithExcessRP.reset();
1204 RegionsWithIGLPInstrs.reset();
1209void GCNScheduleDAGMILive::runSchedStages() {
1210 LLVM_DEBUG(
dbgs() <<
"All regions recorded, starting actual scheduling.\n");
1213 if (!Regions.
empty()) {
1214 BBLiveInMap = getRegionLiveInMap();
1219#ifdef DUMP_MAX_REG_PRESSURE
1229 if (!Stage->initGCNSchedStage())
1232 for (
auto Region : Regions) {
1236 if (!Stage->initGCNRegion()) {
1237 Stage->advanceRegion();
1243 const unsigned RegionIdx = Stage->getRegionIdx();
1246 MRI, RegionLiveOuts.getLiveRegsForRegionIdx(RegionIdx));
1250 Stage->finalizeGCNRegion();
1251 Stage->advanceRegion();
1255 Stage->finalizeGCNSchedStage();
1258#ifdef DUMP_MAX_REG_PRESSURE
1271 OS <<
"Max Occupancy Initial Schedule";
1274 OS <<
"Instruction Rewriting Reschedule";
1277 OS <<
"Unclustered High Register Pressure Reschedule";
1280 OS <<
"Clustered Low Occupancy Reschedule";
1283 OS <<
"Pre-RA Rematerialize";
1286 OS <<
"Max ILP Initial Schedule";
1289 OS <<
"Max memory clause Initial Schedule";
1292 OS <<
"Live Interval RP Reschedule";
1312void RewriteMFMAFormStage::findReachingDefs(
1334 while (!Worklist.
empty()) {
1349 for (MachineBasicBlock *PredMBB : DefMBB->
predecessors()) {
1350 if (Visited.
insert(PredMBB).second)
1356void RewriteMFMAFormStage::findReachingUses(
1360 for (MachineOperand &UseMO :
1363 findReachingDefs(UseMO, LIS, ReachingDefIndexes);
1367 if (
any_of(ReachingDefIndexes, [DefIdx](SlotIndex RDIdx) {
1379 if (!
ST.hasGFX90AInsts() ||
MFI.getMinWavesPerEU() > 1)
1382 RegionsWithExcessArchVGPR.resize(
DAG.Regions.size());
1383 RegionsWithExcessArchVGPR.reset();
1387 RegionsWithExcessArchVGPR[
Region] =
true;
1390 if (RegionsWithExcessArchVGPR.none())
1393 TII =
ST.getInstrInfo();
1394 SRI =
ST.getRegisterInfo();
1396 std::vector<std::pair<MachineInstr *, unsigned>> RewriteCands;
1400 if (!initHeuristics(RewriteCands, CopyForUse, CopyForDef))
1403 int64_t
Cost = getRewriteCost(RewriteCands, CopyForUse, CopyForDef);
1410 return rewrite(RewriteCands);
1420 if (
DAG.RegionsWithHighRP.none() &&
DAG.RegionsWithExcessRP.none())
1427 InitialOccupancy =
DAG.MinOccupancy;
1430 TempTargetOccupancy =
MFI.getMaxWavesPerEU() >
DAG.MinOccupancy
1431 ? InitialOccupancy + 1
1433 IsAnyRegionScheduled =
false;
1434 S.SGPRLimitBias =
S.HighRPSGPRBias;
1435 S.VGPRLimitBias =
S.HighRPVGPRBias;
1439 <<
"Retrying function scheduling without clustering. "
1440 "Aggressively try to reduce register pressure to achieve occupancy "
1441 << TempTargetOccupancy <<
".\n");
1456 if (
DAG.StartingOccupancy <=
DAG.MinOccupancy)
1460 dbgs() <<
"Retrying function scheduling with lowest recorded occupancy "
1461 <<
DAG.MinOccupancy <<
".\n");
1466#define REMAT_PREFIX "[PreRARemat] "
1467#define REMAT_DEBUG(X) LLVM_DEBUG(dbgs() << REMAT_PREFIX; X;)
1469#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1470Printable PreRARematStage::ScoredRemat::print()
const {
1472 OS <<
'(' << MaxFreq <<
", " << FreqDiff <<
", " << RegionImpact <<
')';
1487 auto PrintTargetRegions = [&]() ->
void {
1488 if (TargetRegions.none()) {
1493 for (
unsigned I : TargetRegions.set_bits())
1500 dbgs() <<
"Analyzing ";
1501 MF.getFunction().printAsOperand(
dbgs(),
false);
1504 if (!setObjective()) {
1505 LLVM_DEBUG(
dbgs() <<
"no objective to achieve, occupancy is maximal at "
1506 <<
MFI.getMaxWavesPerEU() <<
'\n');
1511 dbgs() <<
"increase occupancy from " << *TargetOcc - 1 <<
'\n';
1513 dbgs() <<
"reduce spilling (minimum target occupancy is "
1514 <<
MFI.getMinWavesPerEU() <<
")\n";
1516 PrintTargetRegions();
1521 DAG.RegionLiveOuts.buildLiveRegMap();
1523 if (!Remater.analyze()) {
1539 DefRegToCandIdx.
resize(
DAG.MRI.getNumVirtRegs());
1540 const unsigned NumRegions =
DAG.Regions.size();
1542 for (
unsigned RegIdx = 0, E = Remater.getNumRegs(); RegIdx < E; ++RegIdx) {
1546 if (CandReg.
Uses.size() != 1)
1548 const auto [UseRegion,
Users] = *CandReg.
Uses.begin();
1567 "user must have at least one operand");
1574 assert(FirstUseMI &&
"there must be a user in the region");
1576 DAG.LIS->getInstructionIndex(*FirstUseMI).getRegSlot(
true);
1578 DAG.LIS->getInstructionIndex(*CandReg.
getLastDef()).getRegSlot(
true);
1580 const Rematerializer::Reg &DepReg = Remater.getReg(DepRegIdx);
1581 Register DepDefReg = DepReg.getDefReg();
1582 return MarkedRegs.contains(DepDefReg) ||
1583 !Remater.isRegIdenticalAtUses(DepDefReg, DepReg.Mask, RefIdx,
1588 [&](
const std::pair<Register, LaneBitmask> &RegAndMask) {
1589 const auto &[Reg, Mask] = RegAndMask;
1590 return !Remater.isRegIdenticalAtUses(Reg, Mask, RefIdx,
1595 Register DefReg = CandReg.getDefReg();
1596 MarkedRegs.
insert(DefReg);
1597 DefRegToCandIdx[DefReg] = Candidates.
size();
1605 for (
unsigned I = 0;
I < NumRegions; ++
I) {
1606 for (
const auto &[Reg, Mask] :
DAG.LiveIns[
I]) {
1609 unsigned CandIdx = DefRegToCandIdx[Reg];
1611 Candidates[CandIdx].LiveIn.set(
I);
1613 for (
const auto &[
Reg, Mask] :
1617 unsigned CandIdx = DefRegToCandIdx[
Reg];
1619 Candidates[CandIdx].LiveOut.set(
I);
1624 SmallVector<unsigned> CandidateOrder;
1625 for (
auto [CandIdx, Cand] :
enumerate(Candidates)) {
1626 Cand.init(FreqInfo, Remater,
DAG);
1627 Cand.update(TargetRegions, RPTargets, FreqInfo, !TargetOcc);
1628 if (!Cand.hasNullScore())
1639 Rollback = std::make_unique<RollbackSupport>(Remater);
1644 BitVector RecomputeRP(
DAG.Regions.size());
1646 RecomputeRP.reset();
1649 sort(CandidateOrder, [&](
unsigned LHSIndex,
unsigned RHSIndex) {
1650 return Candidates[LHSIndex] < Candidates[RHSIndex];
1654 dbgs() <<
"==== NEW REMAT ROUND ====\n"
1656 <<
"Candidates with non-null score, in rematerialization order:\n";
1657 for (
const ScoredRemat &Cand :
reverse(Candidates)) {
1659 << Remater.printRematReg(Cand.RegIdx) <<
'\n';
1661 PrintTargetRegions();
1667 while (!CandidateOrder.
empty()) {
1668 const ScoredRemat &Cand = Candidates[CandidateOrder.
back()];
1669 const Rematerializer::Reg &
Reg = Remater.getReg(Cand.RegIdx);
1677 if (!Cand.maybeBeneficial(TargetRegions, RPTargets)) {
1679 << Cand.print() <<
" | "
1680 << Remater.printRematReg(Cand.RegIdx));
1685#ifdef EXPENSIVE_CHECKS
1688 for (
const MachineInstr *
DefMI :
Reg.Defs) {
1693 if (!MO.isReg() || !MO.getReg() || !MO.readsReg() || MO.isDef())
1700 LiveInterval &LI =
DAG.LIS->getInterval(
UseReg);
1701 LaneBitmask LM =
DAG.MRI.getMaxLaneMaskForVReg(MO.getReg());
1703 LM =
DAG.TRI->getSubRegIndexLaneMask(MO.getSubReg());
1705 const unsigned UseRegion =
Reg.Uses.begin()->first;
1706 LaneBitmask LiveInMask =
DAG.LiveIns[UseRegion].at(
UseReg);
1707 LaneBitmask UncoveredLanes = LM & ~(LiveInMask & LM);
1711 if (UncoveredLanes.
any()) {
1713 for (LiveInterval::SubRange &SR : LI.
subranges())
1714 assert((SR.LaneMask & UncoveredLanes).none());
1722 REMAT_DEBUG(
dbgs() <<
"** REMAT " << Remater.printRematReg(Cand.RegIdx)
1724 removeFromLiveMaps(
Reg.getDefReg(), Cand.LiveIn, Cand.LiveOut);
1726 Rollback->LiveMapUpdates.emplace_back(Cand.RegIdx, Cand.LiveIn,
1729 Cand.rematerialize(Remater);
1734 updateRPTargets(Cand.Live, Cand.RPSave);
1735 RecomputeRP |= Cand.UnpredictableRPSave;
1736 RescheduleRegions |= Cand.Live;
1737 if (!TargetRegions.any()) {
1743 if (!updateAndVerifyRPTargets(RecomputeRP) && !TargetRegions.any()) {
1752 unsigned NumUsefulCandidates = 0;
1753 for (
unsigned CandIdx : CandidateOrder) {
1754 ScoredRemat &Candidate = Candidates[CandIdx];
1755 Candidate.update(TargetRegions, RPTargets, FreqInfo, !TargetOcc);
1756 if (!Candidate.hasNullScore())
1757 CandidateOrder[NumUsefulCandidates++] = CandIdx;
1759 if (NumUsefulCandidates == 0) {
1760 REMAT_DEBUG(
dbgs() <<
"Stop on exhausted rematerialization candidates\n");
1763 CandidateOrder.truncate(NumUsefulCandidates);
1766 if (RescheduleRegions.none())
1772 unsigned DynamicVGPRBlockSize =
MFI.getDynamicVGPRBlockSize();
1773 for (
unsigned I : RescheduleRegions.set_bits()) {
1774 DAG.Pressure[
I] = RPTargets[
I].getCurrentRP();
1776 <<
DAG.Pressure[
I].getOccupancy(
ST, DynamicVGPRBlockSize)
1777 <<
" (" << RPTargets[
I] <<
")\n");
1779 AchievedOcc =
MFI.getMaxWavesPerEU();
1780 for (
const GCNRegPressure &RP :
DAG.Pressure) {
1782 std::min(AchievedOcc,
RP.getOccupancy(
ST, DynamicVGPRBlockSize));
1786 dbgs() <<
"Retrying function scheduling with new min. occupancy of "
1787 << AchievedOcc <<
" from rematerializing (original was "
1788 <<
DAG.MinOccupancy;
1790 dbgs() <<
", target was " << *TargetOcc;
1794 DAG.setTargetOccupancy(getStageTargetOccupancy());
1805 S.SGPRLimitBias =
S.VGPRLimitBias = 0;
1806 if (
DAG.MinOccupancy > InitialOccupancy) {
1807 assert(IsAnyRegionScheduled);
1809 <<
" stage successfully increased occupancy to "
1810 <<
DAG.MinOccupancy <<
'\n');
1811 }
else if (!IsAnyRegionScheduled) {
1812 assert(
DAG.MinOccupancy == InitialOccupancy);
1814 <<
": No regions scheduled, min occupancy stays at "
1815 <<
DAG.MinOccupancy <<
", MFI occupancy stays at "
1816 <<
MFI.getOccupancy() <<
".\n");
1824 if (
DAG.begin() ==
DAG.end())
1831 unsigned NumRegionInstrs = std::distance(
DAG.begin(),
DAG.end());
1835 if (
DAG.begin() == std::prev(
DAG.end()))
1841 <<
"\n From: " << *
DAG.begin() <<
" To: ";
1843 else dbgs() <<
"End";
1844 dbgs() <<
" RegionInstrs: " << NumRegionInstrs <<
'\n');
1852 for (
auto &
I :
DAG) {
1865 dbgs() <<
"Pressure before scheduling:\nRegion live-ins:"
1867 <<
"Region live-in pressure: "
1871 S.HasHighPressure =
false;
1893 unsigned DynamicVGPRBlockSize =
DAG.MFI.getDynamicVGPRBlockSize();
1896 unsigned CurrentTargetOccupancy =
1897 IsAnyRegionScheduled ?
DAG.MinOccupancy : TempTargetOccupancy;
1899 (CurrentTargetOccupancy <= InitialOccupancy ||
1900 DAG.Pressure[
RegionIdx].getOccupancy(
ST, DynamicVGPRBlockSize) !=
1907 if (!IsAnyRegionScheduled && IsSchedulingThisRegion) {
1908 IsAnyRegionScheduled =
true;
1909 if (
MFI.getMaxWavesPerEU() >
DAG.MinOccupancy)
1910 DAG.setTargetOccupancy(TempTargetOccupancy);
1912 return IsSchedulingThisRegion;
1928 return !RevertAllRegions && RescheduleRegions[
RegionIdx] &&
1948 if (
S.HasHighPressure)
1969 if (
DAG.MinOccupancy < *TargetOcc) {
1971 <<
" cannot meet occupancy target, interrupting "
1972 "re-scheduling in all regions\n");
1973 RevertAllRegions =
true;
1984 unsigned DynamicVGPRBlockSize =
DAG.MFI.getDynamicVGPRBlockSize();
1995 unsigned TargetOccupancy = std::min(
1996 S.getTargetOccupancy(),
ST.getOccupancyWithWorkGroupSizes(
MF).second);
1997 unsigned WavesAfter = std::min(
1998 TargetOccupancy,
PressureAfter.getOccupancy(
ST, DynamicVGPRBlockSize));
1999 unsigned WavesBefore = std::min(
2001 LLVM_DEBUG(
dbgs() <<
"Occupancy before scheduling: " << WavesBefore
2002 <<
", after " << WavesAfter <<
".\n");
2008 unsigned NewOccupancy = std::max(WavesAfter, WavesBefore);
2012 if (WavesAfter < WavesBefore && WavesAfter <
DAG.MinOccupancy &&
2013 WavesAfter >=
MFI.getMinAllowedOccupancy()) {
2014 LLVM_DEBUG(
dbgs() <<
"Function is memory bound, allow occupancy drop up to "
2015 <<
MFI.getMinAllowedOccupancy() <<
" waves\n");
2016 NewOccupancy = WavesAfter;
2019 if (NewOccupancy <
DAG.MinOccupancy) {
2020 DAG.MinOccupancy = NewOccupancy;
2021 MFI.limitOccupancy(
DAG.MinOccupancy);
2023 <<
DAG.MinOccupancy <<
".\n");
2027 unsigned MaxVGPRs =
ST.getMaxNumVGPRs(
MF);
2030 unsigned MaxArchVGPRs = std::min(MaxVGPRs,
ST.getAddressableNumArchVGPRs());
2031 unsigned MaxSGPRs =
ST.getMaxNumSGPRs(
MF);
2055 unsigned ReadyCycle = CurrCycle;
2056 for (
auto &
D : SU.
Preds) {
2057 if (
D.isAssignedRegDep()) {
2060 unsigned DefReady = ReadyCycles[
DAG.getSUnit(
DefMI)->NodeNum];
2061 ReadyCycle = std::max(ReadyCycle, DefReady +
Latency);
2064 ReadyCycles[SU.
NodeNum] = ReadyCycle;
2071 std::pair<MachineInstr *, unsigned>
B)
const {
2072 return A.second <
B.second;
2078 if (ReadyCycles.empty())
2080 unsigned BBNum = ReadyCycles.begin()->first->getParent()->getNumber();
2081 dbgs() <<
"\n################## Schedule time ReadyCycles for MBB : " << BBNum
2082 <<
" ##################\n# Cycle #\t\t\tInstruction "
2086 for (
auto &
I : ReadyCycles) {
2087 if (
I.second > IPrev + 1)
2088 dbgs() <<
"****************************** BUBBLE OF " <<
I.second - IPrev
2089 <<
" CYCLES DETECTED ******************************\n\n";
2090 dbgs() <<
"[ " <<
I.second <<
" ] : " << *
I.first <<
"\n";
2103 unsigned SumBubbles = 0;
2105 unsigned CurrCycle = 0;
2106 for (
auto &SU : InputSchedule) {
2107 unsigned ReadyCycle =
2109 SumBubbles += ReadyCycle - CurrCycle;
2111 ReadyCyclesSorted.insert(std::make_pair(SU.getInstr(), ReadyCycle));
2113 CurrCycle = ++ReadyCycle;
2136 unsigned SumBubbles = 0;
2138 unsigned CurrCycle = 0;
2139 for (
auto &
MI :
DAG) {
2143 unsigned ReadyCycle =
2145 SumBubbles += ReadyCycle - CurrCycle;
2147 ReadyCyclesSorted.insert(std::make_pair(SU->
getInstr(), ReadyCycle));
2149 CurrCycle = ++ReadyCycle;
2166 if (WavesAfter <
DAG.MinOccupancy)
2170 if (
DAG.MFI.isDynamicVGPREnabled()) {
2173 DAG.MFI.getDynamicVGPRBlockSize());
2176 if (BlocksAfter > BlocksBefore)
2213 <<
"\n\t *** In shouldRevertScheduling ***\n"
2214 <<
" *********** BEFORE UnclusteredHighRPStage ***********\n");
2218 <<
"\n *********** AFTER UnclusteredHighRPStage ***********\n");
2220 unsigned OldMetric = MBefore.
getMetric();
2221 unsigned NewMetric = MAfter.
getMetric();
2222 unsigned WavesBefore = std::min(
2223 S.getTargetOccupancy(),
2230 LLVM_DEBUG(
dbgs() <<
"\tMetric before " << MBefore <<
"\tMetric after "
2231 << MAfter <<
"Profit: " << Profit <<
"\n");
2262 unsigned WavesAfter) {
2272 cl::desc(
"Percent increase of live interval RP over instant pressure to "
2273 "trigger rescheduling"),
2279 "Reduction factor (percent) for VGPR threshold during live interval RP "
2280 "reschedule stage"),
2284 "amdgpu-lirp-instant-lower-bound",
cl::Hidden,
2285 cl::desc(
"Lower bound (percent of the VGPR excess limit) on instant RP, "
2286 "below which a region is skipped"),
2296 if (!
S.VGPRThresholdPercent) {
2297 LLVM_DEBUG(
dbgs() <<
"LIRP: expected VGPRThresholdPercent to be enabled, "
2298 "not using live interval RP reschedule stage\n");
2306 unsigned InstantRP =
DAG.Pressure[
RegionIdx].getArchVGPRNum();
2307 auto [RegionBegin, RegionEnd] =
DAG.Regions[
RegionIdx];
2308 if (RegionBegin == RegionEnd)
2315 unsigned NewVGPRThresholdPercent =
2319 <<
", VGPRThresholdPercent: " <<
S.VGPRThresholdPercent
2320 <<
" -> " << NewVGPRThresholdPercent
2321 <<
", VGPRExcessLimit=" <<
S.VGPRExcessLimit
2322 <<
", VGPRCriticalLimit=" <<
S.VGPRCriticalLimit
2323 <<
", InstantRP=" << InstantRP <<
", LIRP=" << LIRP);
2325 bool DoRescheduling =
false;
2327 unsigned InstantRPLowerBound =
2329 if (LIRP >
S.VGPRExcessLimit) {
2330 LLVM_DEBUG(
dbgs() <<
" [LIRP exceeds the limit (" <<
S.VGPRExcessLimit
2331 <<
"), rescheduling]");
2332 DoRescheduling =
true;
2333 }
else if (LIRP > InstantRP && InstantRP > InstantRPLowerBound) {
2334 unsigned IncreasePercent = ((LIRP - InstantRP) * 100) / InstantRP;
2338 DoRescheduling =
true;
2344 SavedVGPRExcessLimit =
S.VGPRExcessLimit;
2345 SavedVGPRCriticalLimit =
S.VGPRCriticalLimit;
2346 SavedVGPRThresholdPercent =
S.VGPRThresholdPercent;
2347 S.VGPRThresholdPercent = NewVGPRThresholdPercent;
2355 S.VGPRExcessLimit = SavedVGPRExcessLimit;
2356 S.VGPRCriticalLimit = SavedVGPRCriticalLimit;
2357 S.VGPRThresholdPercent = SavedVGPRThresholdPercent;
2364 LLVM_DEBUG(
dbgs() <<
"New pressure will result in more spilling.\n");
2376 "instruction number mismatch");
2377 if (MIOrder.
empty())
2390 if (MII != RegionEnd) {
2392 bool NonDebugReordered =
2393 !
MI->isDebugInstr() &&
2399 if (NonDebugReordered)
2400 DAG.LIS->handleMove(*
MI,
true);
2407 if (!
MI->isDebugInstr()) {
2409 SlotIndex PrevIdx =
DAG.LIS->getSlotIndexes()->getIndexBefore(*
MI);
2410 if (PrevIdx >= MIIdx)
2411 DAG.LIS->handleMove(*
MI,
true);
2415 if (
MI->isDebugInstr()) {
2422 DAG.ShouldTrackLaneMasks);
2441 if (RD->
getOpcode() == AMDGPU::AV_MOV_B32_IMM_PSEUDO ||
2442 RD->
getOpcode() == AMDGPU::AV_MOV_B64_IMM_PSEUDO)
2449bool RewriteMFMAFormStage::hasUseRequiringVGPR(
2451 const SmallPtrSetImpl<MachineInstr *> &RewriteSet) {
2452 for (SlotIndex RDIdx : Src2ReachingDefs) {
2453 const MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIdx);
2455 findReachingUses(RD,
DAG.LIS, ReachingUses);
2456 for (
const MachineOperand *UseMO : ReachingUses) {
2468void RewriteMFMAFormStage::resetRewriteCandsToVGPR(
2469 ArrayRef<std::pair<MachineInstr *, unsigned>> RewriteCands) {
2470 for (
auto [
MI, OriginalOpcode] : RewriteCands) {
2473 DAG.MRI.getRegClass(
MI->getOperand(0).getReg());
2475 DAG.MRI.setRegClass(
MI->getOperand(0).getReg(), VDefRC);
2476 MI->setDesc(
TII->get(OriginalOpcode));
2478 MachineOperand *
Src2 =
TII->getNamedOperand(*
MI, AMDGPU::OpName::src2);
2487 DAG.MRI.setRegClass(
Src2->getReg(), VUseRC);
2491bool RewriteMFMAFormStage::isRewriteCandidate(MachineInstr *
MI)
const {
2492 if (!
static_cast<const SIInstrInfo *
>(
DAG.TII)->isMAI(*
MI))
2497 Register DstReg =
MI->getOperand(0).getReg();
2498 for (
const MachineInstr &
UseMI :
DAG.MRI.use_nodbg_instructions(DstReg)) {
2505bool RewriteMFMAFormStage::initHeuristics(
2506 std::vector<std::pair<MachineInstr *, unsigned>> &RewriteCands,
2507 DenseMap<MachineBasicBlock *, std::set<Register>> &CopyForUse,
2508 SmallPtrSetImpl<MachineInstr *> &CopyForDef) {
2513 SmallPtrSet<MachineInstr *, 16> RewriteSet;
2514 DenseSet<Register> CandSrc2Regs;
2515 for (MachineBasicBlock &
MBB :
MF) {
2516 for (MachineInstr &
MI :
MBB) {
2517 if (!isRewriteCandidate(&
MI))
2520 MachineOperand *
Src2 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src2);
2521 if (Src2 &&
Src2->isReg())
2527 for (MachineBasicBlock &
MBB :
MF) {
2528 for (MachineInstr &
MI :
MBB) {
2529 if (!isRewriteCandidate(&
MI))
2533 assert(ReplacementOp != -1);
2535 RewriteCands.push_back({&
MI,
MI.getOpcode()});
2536 MI.setDesc(
TII->get(ReplacementOp));
2538 MachineOperand *
Src2 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src2);
2539 if (
Src2->isReg()) {
2541 findReachingDefs(*Src2,
DAG.LIS, Src2ReachingDefs);
2545 bool Src2NeedsVGPR = hasUseRequiringVGPR(Src2ReachingDefs, RewriteSet);
2546 Src2NeedsVGPRCache[&
MI] = Src2NeedsVGPR;
2548 for (SlotIndex RDIdx : Src2ReachingDefs) {
2549 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIdx);
2550 if (!Src2NeedsVGPR &&
2557 MachineOperand &Dst =
MI.getOperand(0);
2560 findReachingUses(&
MI,
DAG.LIS, DstReachingUses);
2562 for (MachineOperand *RUOp : DstReachingUses) {
2563 MachineInstr *UserMI = RUOp->getParent();
2565 if (
TII->isMAI(*UserMI) && RewriteSet.
contains(UserMI))
2571 CopyForUse[UserMI->
getParent()].insert(RUOp->getReg());
2573 if (
TII->isMAI(*UserMI))
2577 findReachingDefs(*RUOp,
DAG.LIS, DstUsesReachingDefs);
2579 for (SlotIndex RDIndex : DstUsesReachingDefs) {
2580 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIndex);
2581 if (
TII->isMAI(*RD))
2595 DAG.MRI.setRegClass(Dst.getReg(), ADefRC);
2596 if (
Src2->isReg()) {
2602 DAG.MRI.setRegClass(
Src2->getReg(), AUseRC);
2611int64_t RewriteMFMAFormStage::getRewriteCost(
2612 ArrayRef<std::pair<MachineInstr *, unsigned>> RewriteCands,
2613 const DenseMap<MachineBasicBlock *, std::set<Register>> &CopyForUse,
2614 const SmallPtrSetImpl<MachineInstr *> &CopyForDef) {
2615 MachineBlockFrequencyInfo *MBFI =
DAG.MBFI;
2617 int64_t BestSpillCost = 0;
2621 std::pair<unsigned, unsigned> MaxVectorRegs =
2622 ST.getMaxNumVectorRegs(
MF.getFunction());
2623 unsigned ArchVGPRThreshold = MaxVectorRegs.first;
2624 unsigned AGPRThreshold = MaxVectorRegs.second;
2625 unsigned CombinedThreshold =
ST.getMaxNumVGPRs(
MF);
2628 if (!RegionsWithExcessArchVGPR[Region])
2633 MF, ArchVGPRThreshold, AGPRThreshold, CombinedThreshold);
2641 MF, ArchVGPRThreshold, AGPRThreshold, CombinedThreshold);
2647 bool RelativeFreqIsDenom = EntryFreq > BlockFreq;
2648 uint64_t RelativeFreq = EntryFreq && BlockFreq
2649 ? (RelativeFreqIsDenom ? EntryFreq / BlockFreq
2650 : BlockFreq / EntryFreq)
2655 int64_t SpillCost = ((int)SpillCostAfter - (int)SpillCostBefore) * 2;
2658 if (RelativeFreqIsDenom)
2659 SpillCost /= (int64_t)RelativeFreq;
2661 SpillCost *= (int64_t)RelativeFreq;
2664 if (SpillCost > 0) {
2665 resetRewriteCandsToVGPR(RewriteCands);
2669 if (SpillCost < BestSpillCost)
2670 BestSpillCost = SpillCost;
2675 Cost = BestSpillCost;
2678 unsigned CopyCost = 0;
2682 for (MachineInstr *
DefMI : CopyForDef) {
2694 for (
auto &[UseBlock, UseRegs] : CopyForUse) {
2708 resetRewriteCandsToVGPR(RewriteCands);
2710 return Cost + CopyCost;
2713bool RewriteMFMAFormStage::rewrite(
2714 ArrayRef<std::pair<MachineInstr *, unsigned>> RewriteCands) {
2715 DenseMap<MachineInstr *, unsigned> FirstMIToRegion;
2716 DenseMap<MachineInstr *, unsigned> LastMIToRegion;
2724 if (
Entry.second !=
Entry.first->getParent()->end())
2767 DenseSet<Register> RewriteRegs;
2770 DenseMap<Register, Register> RedefMap;
2772 DenseMap<Register, DenseSet<MachineOperand *>>
ReplaceMap;
2774 DenseMap<Register, SmallPtrSet<MachineInstr *, 8>> ReachingDefCopyMap;
2777 DenseMap<unsigned, DenseMap<Register, SmallPtrSet<MachineOperand *, 8>>>
2782 SmallPtrSet<MachineInstr *, 16> RewriteCandsSet;
2783 DenseSet<Register> RewriteSrc2Regs;
2784 for (
auto &[
MI, OriginalOpcode] : RewriteCands) {
2786 MachineOperand *
Src2 =
TII->getNamedOperand(*
MI, AMDGPU::OpName::src2);
2787 if (Src2 &&
Src2->isReg())
2791 for (
auto &[
MI, OriginalOpcode] : RewriteCands) {
2793 if (ReplacementOp == -1)
2795 MI->setDesc(
TII->get(ReplacementOp));
2798 MachineOperand *
Src2 =
TII->getNamedOperand(*
MI, AMDGPU::OpName::src2);
2799 if (
Src2->isReg()) {
2806 findReachingDefs(*Src2,
DAG.LIS, Src2ReachingDefs);
2807 SmallSetVector<MachineInstr *, 8> Src2DefsReplace;
2811 bool Src2NeedsVGPR = Src2NeedsVGPRCache.lookup(
MI);
2813 for (SlotIndex RDIndex : Src2ReachingDefs) {
2814 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIndex);
2815 if (!Src2NeedsVGPR &&
2819 Src2DefsReplace.
insert(RD);
2822 if (!Src2DefsReplace.
empty()) {
2823 auto RI = RedefMap.
find(Src2Reg);
2824 if (RI != RedefMap.
end()) {
2825 MappedReg = RI->second;
2830 SRI->getEquivalentVGPRClass(Src2RC);
2833 MappedReg =
DAG.MRI.createVirtualRegister(VGPRRC);
2834 RedefMap[Src2Reg] = MappedReg;
2839 for (MachineInstr *RD : Src2DefsReplace) {
2841 if (ReachingDefCopyMap[Src2Reg].insert(RD).second) {
2842 MachineInstrBuilder VGPRCopy =
2845 .
addDef(MappedReg, {}, 0)
2846 .addUse(Src2Reg, {}, 0);
2847 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
2852 unsigned UpdateRegion = LastMIToRegion[RD];
2853 DAG.Regions[UpdateRegion].second = VGPRCopy;
2854 LastMIToRegion.
erase(RD);
2861 RewriteRegs.
insert(Src2Reg);
2871 MachineOperand *Dst = &
MI->getOperand(0);
2880 SmallVector<MachineInstr *, 8> DstUseDefsReplace;
2882 findReachingUses(
MI,
DAG.LIS, DstReachingUses);
2884 for (MachineOperand *RUOp : DstReachingUses) {
2885 MachineInstr *UserMI = RUOp->
getParent();
2887 if (
TII->isMAI(*UserMI) && RewriteCandsSet.
contains(UserMI))
2891 if (
find(DstReachingUseCopies, RUOp) == DstReachingUseCopies.
end())
2895 if (
TII->isMAI(*UserMI))
2899 findReachingDefs(*RUOp,
DAG.LIS, DstUsesReachingDefs);
2901 for (SlotIndex RDIndex : DstUsesReachingDefs) {
2902 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIndex);
2903 if (
TII->isMAI(*RD))
2908 if (
find(DstUseDefsReplace, RD) == DstUseDefsReplace.
end())
2913 if (!DstUseDefsReplace.
empty()) {
2914 auto RI = RedefMap.
find(DstReg);
2915 if (RI != RedefMap.
end()) {
2916 MappedReg = RI->second;
2923 MappedReg =
DAG.MRI.createVirtualRegister(VGPRRC);
2924 RedefMap[DstReg] = MappedReg;
2929 for (MachineInstr *RD : DstUseDefsReplace) {
2931 if (ReachingDefCopyMap[DstReg].insert(RD).second) {
2932 MachineInstrBuilder VGPRCopy =
2935 .
addDef(MappedReg, {}, 0)
2936 .addUse(DstReg, {}, 0);
2937 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
2941 auto LMI = LastMIToRegion.
find(RD);
2942 if (LMI != LastMIToRegion.
end()) {
2943 unsigned UpdateRegion = LMI->second;
2944 DAG.Regions[UpdateRegion].second = VGPRCopy;
2945 LastMIToRegion.
erase(RD);
2951 DenseSet<MachineOperand *> &DstRegSet =
ReplaceMap[DstReg];
2954 MachineInstr *EarliestSameBlockUse =
nullptr;
2955 for (MachineOperand *RU : DstReachingUseCopies) {
2956 MachineBasicBlock *RUBlock = RU->getParent()->getParent();
2959 if (RUBlock !=
MI->getParent()) {
2965 if (!SameBlockCopyReg.
isValid()) {
2968 SameBlockCopyReg =
DAG.MRI.createVirtualRegister(VGPRRC);
2972 MachineInstr *UseInst = RU->getParent();
2973 if (!EarliestSameBlockUse ||
2975 DAG.LIS->getInstructionIndex(*UseInst),
2976 DAG.LIS->getInstructionIndex(*EarliestSameBlockUse)))
2977 EarliestSameBlockUse = UseInst;
2978 RU->setReg(SameBlockCopyReg);
2982 if (SameBlockCopyReg.
isValid()) {
2983 MachineInstrBuilder VGPRCopy =
2986 TII->get(TargetOpcode::COPY), SameBlockCopyReg)
2988 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
2993 RewriteRegs.
insert(DstReg);
3003 std::pair<unsigned, DenseMap<Register, SmallPtrSet<MachineOperand *, 8>>>;
3004 for (RUBType RUBlockEntry : ReachingUseTracker) {
3005 using RUDType = std::pair<Register, SmallPtrSet<MachineOperand *, 8>>;
3006 for (RUDType RUDst : RUBlockEntry.second) {
3007 MachineOperand *OpBegin = *RUDst.second.begin();
3008 SlotIndex InstPt =
DAG.LIS->getInstructionIndex(*OpBegin->
getParent());
3011 for (MachineOperand *User : RUDst.second) {
3012 SlotIndex NewInstPt =
DAG.LIS->getInstructionIndex(*
User->getParent());
3019 Register NewUseReg =
DAG.MRI.createVirtualRegister(VGPRRC);
3020 MachineInstr *UseInst =
DAG.LIS->getInstructionFromIndex(InstPt);
3022 MachineInstrBuilder VGPRCopy =
3025 .
addDef(NewUseReg, {}, 0)
3026 .addUse(RUDst.first, {}, 0);
3027 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
3031 auto FI = FirstMIToRegion.
find(UseInst);
3032 if (FI != FirstMIToRegion.
end()) {
3033 unsigned UpdateRegion = FI->second;
3034 DAG.Regions[UpdateRegion].first = VGPRCopy;
3035 FirstMIToRegion.
erase(UseInst);
3039 for (MachineOperand *User : RUDst.second) {
3040 User->setReg(NewUseReg);
3051 for (std::pair<Register, Register> NewDef : RedefMap) {
3056 for (MachineOperand *ReplaceOp :
ReplaceMap[OldReg])
3057 ReplaceOp->setReg(NewReg);
3061 for (
Register RewriteReg : RewriteRegs) {
3062 Register RegToRewrite = RewriteReg;
3065 auto RI = RedefMap.find(RewriteReg);
3066 if (RI != RedefMap.end())
3067 RegToRewrite = RI->second;
3072 DAG.MRI.setRegClass(RegToRewrite, AGPRRC);
3076 DAG.LIS->reanalyze(
DAG.MF);
3078 RegionPressureMap LiveInUpdater(&
DAG,
false);
3079 LiveInUpdater.buildLiveRegMap();
3082 DAG.LiveIns[Region] = LiveInUpdater.getLiveRegsForRegionIdx(Region);
3089unsigned PreRARematStage::getStageTargetOccupancy()
const {
3090 return TargetOcc ? *TargetOcc :
MFI.getMinWavesPerEU();
3093bool PreRARematStage::setObjective() {
3097 unsigned MaxSGPRs =
ST.getMaxNumSGPRs(
F);
3098 unsigned MaxVGPRs =
ST.getMaxNumVGPRs(
F);
3099 bool HasVectorRegisterExcess =
false;
3100 for (
unsigned I = 0,
E =
DAG.Regions.size();
I !=
E; ++
I) {
3101 const GCNRegPressure &
RP =
DAG.Pressure[
I];
3102 GCNRPTarget &
Target = RPTargets.emplace_back(MaxSGPRs, MaxVGPRs,
MF, RP);
3104 TargetRegions.set(
I);
3105 HasVectorRegisterExcess |=
Target.hasVectorRegisterExcess();
3108 if (HasVectorRegisterExcess ||
DAG.MinOccupancy >=
MFI.getMaxWavesPerEU()) {
3111 TargetOcc = std::nullopt;
3115 TargetOcc =
DAG.MinOccupancy + 1;
3116 const unsigned VGPRBlockSize =
MFI.getDynamicVGPRBlockSize();
3117 MaxSGPRs =
ST.getMaxNumSGPRs(*TargetOcc,
false);
3118 MaxVGPRs =
ST.getMaxNumVGPRs(*TargetOcc, VGPRBlockSize);
3119 for (
auto [
I, Target] :
enumerate(RPTargets)) {
3120 Target.setTarget(MaxSGPRs, MaxVGPRs);
3122 TargetRegions.set(
I);
3126 return TargetRegions.any();
3129bool PreRARematStage::ScoredRemat::maybeBeneficial(
3131 for (
unsigned I : TargetRegions.set_bits()) {
3132 if (Live[
I] && RPTargets[
I].isSaveBeneficial(RPSave))
3145 const unsigned NumRegions =
DAG.Regions.size();
3149 for (
unsigned I = 0;
I < NumRegions; ++
I) {
3153 if (BlockFreq && BlockFreq <
MinFreq)
3162 if (
MinFreq >= ScaleFactor * ScaleFactor) {
3163 for (uint64_t &Freq :
Regions)
3164 Freq /= ScaleFactor;
3170void PreRARematStage::ScoredRemat::init(
const FreqInfo &Freq,
3175 assert(Reg.Uses.size() == 1 &&
"expected users in single region");
3176 const unsigned UseRegion = Reg.Uses.begin()->first;
3181 for (
unsigned I : Live.set_bits()) {
3184 if (!LiveIn[
I] || !LiveOut[
I] ||
I == UseRegion)
3185 UnpredictableRPSave.set(
I);
3192 int64_t DefOrMin = std::max(Freq.
Regions[Reg.DefRegion], Freq.
MinFreq);
3193 int64_t UseOrMax = Freq.
Regions[UseRegion];
3196 FreqDiff = DefOrMin - UseOrMax;
3199void PreRARematStage::ScoredRemat::update(
const BitVector &TargetRegions,
3201 const FreqInfo &FreqInfo,
3205 for (
unsigned I : TargetRegions.
set_bits()) {
3214 if (!NumRegsBenefit)
3218 RegionImpact += (UnpredictableRPSave[
I] ? 1 : 2) * NumRegsBenefit;
3221 uint64_t Freq = FreqInfo.
Regions[
I];
3222 if (UnpredictableRPSave[
I]) {
3227 MaxFreq = std::max(MaxFreq, Freq);
3232void PreRARematStage::ScoredRemat::rematerialize(
3233 Rematerializer &Remater)
const {
3234 const Rematerializer::Reg &
Reg = Remater.getReg(RegIdx);
3235 Rematerializer::DependencyReuseInfo DRI;
3236 for (RegisterIdx DepRegIdx :
Reg.Dependencies)
3237 DRI.
reuse(DepRegIdx);
3238 unsigned UseRegion =
Reg.Uses.begin()->first;
3239 Remater.rematerializeToRegion(RegIdx, UseRegion, DRI);
3242void PreRARematStage::updateRPTargets(
const BitVector &Regions,
3243 const GCNRegPressure &RPSave) {
3245 RPTargets[
I].saveRP(RPSave);
3246 if (TargetRegions[
I] && RPTargets[
I].satisfied()) {
3248 TargetRegions.reset(
I);
3253bool PreRARematStage::updateAndVerifyRPTargets(
const BitVector &Regions) {
3254 bool TooOptimistic =
false;
3256 GCNRPTarget &
Target = RPTargets[
I];
3262 if (!TargetRegions[
I] && !
Target.satisfied()) {
3264 TooOptimistic =
true;
3265 TargetRegions.set(
I);
3268 return TooOptimistic;
3271void PreRARematStage::removeFromLiveMaps(
Register Reg,
const BitVector &LiveIn,
3272 const BitVector &LiveOut) {
3274 LiveOut.
size() ==
DAG.Regions.size() &&
"region num mismatch");
3278 DAG.RegionLiveOuts.getLiveRegsForRegionIdx(
I).erase(
Reg);
3281void PreRARematStage::addToLiveMaps(
Register Reg, LaneBitmask Mask,
3282 const BitVector &LiveIn,
3283 const BitVector &LiveOut) {
3285 LiveOut.
size() ==
DAG.Regions.size() &&
"region num mismatch");
3286 std::pair<Register, LaneBitmask> LiveReg(
Reg, Mask);
3288 DAG.LiveIns[
I].insert(LiveReg);
3290 DAG.RegionLiveOuts.getLiveRegsForRegionIdx(
I).insert(LiveReg);
3302 if (
DAG.MinOccupancy >= *TargetOcc)
3306 for (
const auto &[
RegionIdx, OrigMIOrder, MaxPressure] : RegionReverts) {
3316 if (AchievedOcc >= *TargetOcc) {
3317 DAG.setTargetOccupancy(AchievedOcc);
3322 DAG.setTargetOccupancy(*TargetOcc - 1);
3327 assert(Rollback &&
"rollbacker should be defined");
3328 Rollback->Listener.rollback(Remater);
3329 for (
const auto &[RegIdx, LiveIn, LiveOut] : Rollback->LiveMapUpdates) {
3330 const Rematerializer::Reg &
Reg = Remater.getReg(RegIdx);
3331 addToLiveMaps(
Reg.getDefReg(),
Reg.Mask, LiveIn, LiveOut);
3334#ifdef EXPENSIVE_CHECKS
3339 for (
unsigned I : RescheduleRegions.set_bits())
3340 DAG.Pressure[
I] =
DAG.getRealRegPressure(
I);
3345void GCNScheduleDAGMILive::setTargetOccupancy(
unsigned TargetOccupancy) {
3346 MinOccupancy = TargetOccupancy;
3347 if (
MFI.getOccupancy() < TargetOccupancy)
3348 MFI.increaseOccupancy(
MF, MinOccupancy);
3350 MFI.limitOccupancy(MinOccupancy);
3367 if (HasIGLPInstrs) {
3368 SavedMutations.clear();
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static SUnit * pickOnlyChoice(SchedBoundary &Zone)
This file implements the BitVector class.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file defines the GCNRegPressure class, which tracks registry pressure by bookkeeping number of S...
static cl::opt< bool > GCNTrackers("amdgpu-use-amdgpu-trackers", cl::Hidden, cl::desc("Use the AMDGPU specific RPTrackers during scheduling"), cl::init(false))
static cl::opt< bool > DisableClusteredLowOccupancy("amdgpu-disable-clustered-low-occupancy-reschedule", cl::Hidden, cl::desc("Disable clustered low occupancy " "rescheduling for ILP scheduling stage."), cl::init(false))
#define REMAT_PREFIX
Allows to easily filter for this stage's debug output.
static MachineInstr * getLastMIForRegion(MachineBasicBlock::iterator RegionBegin, MachineBasicBlock::iterator RegionEnd)
static bool shouldCheckPending(SchedBoundary &Zone, const TargetSchedModel *SchedModel)
static cl::opt< bool > EnableLiveIntervalRPReschedule("amdgpu-lirp-reschedule", cl::Hidden, cl::desc("Enable live interval RP reschedule stage"), cl::init(true))
static cl::opt< bool > RelaxedOcc("amdgpu-schedule-relaxed-occupancy", cl::Hidden, cl::desc("Relax occupancy targets for kernels which are memory " "bound (amdgpu-membound-threshold), or " "Wave Limited (amdgpu-limit-wave-threshold)."), cl::init(false))
static cl::opt< bool > DisableUnclusterHighRP("amdgpu-disable-unclustered-high-rp-reschedule", cl::Hidden, cl::desc("Disable unclustered high register pressure " "reduction scheduling stage."), cl::init(false))
static void printScheduleModel(std::set< std::pair< MachineInstr *, unsigned >, EarlierIssuingCycle > &ReadyCycles)
static bool isReachingDefAGPRForm(MachineInstr *RD, const SmallPtrSetImpl< MachineInstr * > &RewriteSet, const DenseSet< Register > &CandSrc2Regs, const SIInstrInfo &TII)
Returns true if reaching def RD will be in AGPR form after the rewrite and so needs no bridge copy: a...
static cl::opt< bool > PrintMaxRPRegUsageAfterScheduler("amdgpu-print-max-reg-pressure-regusage-after-scheduler", cl::Hidden, cl::desc("Print a list of live registers along with their def/uses at the " "point of maximum register pressure after scheduling."), cl::init(false))
static bool hasIGLPInstrs(ScheduleDAGInstrs *DAG)
static cl::opt< bool > DisableRewriteMFMAFormSchedStage("amdgpu-disable-rewrite-mfma-form-sched-stage", cl::Hidden, cl::desc("Disable rewrite mfma rewrite scheduling stage"), cl::init(true))
static bool canUsePressureDiffs(const SUnit &SU)
Checks whether SU can use the cached DAG pressure diffs to compute the current register pressure.
static cl::opt< unsigned > LiveIntervalRPVGPRReduction("amdgpu-lirp-vgpr-reduction", cl::Hidden, cl::desc("Reduction factor (percent) for VGPR threshold during live interval RP " "reschedule stage"), cl::init(90))
static cl::opt< unsigned > PendingQueueLimit("amdgpu-scheduler-pending-queue-limit", cl::Hidden, cl::desc("Max (Available+Pending) size to inspect pending queue (0 disables)"), cl::init(256))
static cl::opt< bool > PrintMaxRPRegUsageBeforeScheduler("amdgpu-print-max-reg-pressure-regusage-before-scheduler", cl::Hidden, cl::desc("Print a list of live registers along with their def/uses at the " "point of maximum register pressure before scheduling."), cl::init(false))
static cl::opt< unsigned > LiveIntervalRPInstantLowerBound("amdgpu-lirp-instant-lower-bound", cl::Hidden, cl::desc("Lower bound (percent of the VGPR excess limit) on instant RP, " "below which a region is skipped"), cl::init(10))
static cl::opt< unsigned > ScheduleMetricBias("amdgpu-schedule-metric-bias", cl::Hidden, cl::desc("Sets the bias which adds weight to occupancy vs latency. Set it to " "100 to chase the occupancy only."), cl::init(10))
static cl::opt< unsigned > LiveIntervalRPThreshold("amdgpu-lirp-threshold", cl::Hidden, cl::desc("Percent increase of live interval RP over instant pressure to " "trigger rescheduling"), cl::init(10))
static Register UseReg(const MachineOperand &MO)
const HexagonInstrInfo * TII
static constexpr std::pair< StringLiteral, StringLiteral > ReplaceMap[]
iv Induction Variable Users
A common definition of LaneBitmask for use in TableGen and CodeGen.
Promote Memory to Register
MIR-level target-independent rematerialization helpers.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & front() const
Get the first element.
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
iterator_range< const_set_bits_iterator > set_bits() const
size_type size() const
Returns the number of bits in this bitvector.
uint64_t getFrequency() const
Returns the frequency as a fixpoint number scaled by the entry frequency.
bool initGCNSchedStage() override
bool shouldRevertScheduling(unsigned WavesAfter) override
bool initGCNRegion() override
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
iterator find(const_arg_type_t< KeyT > Val)
bool erase(const KeyT &Val)
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Implements a dense probed hash-table based set.
bool reset(const MachineInstr &MI, MachineBasicBlock::const_iterator End, const LiveRegSet *LiveRegs=nullptr)
Reset tracker to the point before the MI filling LiveRegs upon this point using LIS.
GCNRegPressure bumpDownwardPressure(const MachineInstr *MI, const SIRegisterInfo *TRI) const
Mostly copy/paste from CodeGen/RegisterPressure.cpp Calculate the impact MI will have on CurPressure ...
GCNMaxILPSchedStrategy(const MachineSchedContext *C)
bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const override
Apply a set of heuristics to a new candidate.
bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const override
GCNMaxMemoryClauseSchedStrategy tries best to clause memory instructions as much as possible.
GCNMaxMemoryClauseSchedStrategy(const MachineSchedContext *C)
GCNMaxOccupancySchedStrategy(const MachineSchedContext *C, bool IsLegacyScheduler=false)
void finalizeSchedule() override
Allow targets to perform final scheduling actions at the level of the whole MachineFunction.
void schedule() override
Orders nodes according to selected style.
GCNPostScheduleDAGMILive(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S, bool RemoveKillFlags)
Models a register pressure target, allowing to evaluate and track register savings against that targe...
unsigned getNumRegsBenefit(const GCNRegPressure &SaveRP) const
Returns the benefit towards achieving the RP target that saving SaveRP represents,...
GCNRegPressure getPressure() const
virtual bool initGCNRegion()
GCNRegPressure PressureBefore
bool isRegionWithExcessRP() const
void modifyRegionSchedule(unsigned RegionIdx, ArrayRef< MachineInstr * > MIOrder)
Sets the schedule of region RegionIdx to MIOrder.
bool mayCauseSpilling(unsigned WavesAfter)
ScheduleMetrics getScheduleMetrics(const std::vector< SUnit > &InputSchedule)
GCNScheduleDAGMILive & DAG
const GCNSchedStageID StageID
std::vector< MachineInstr * > Unsched
GCNRegPressure PressureAfter
virtual void finalizeGCNRegion()
SIMachineFunctionInfo & MFI
unsigned computeSUnitReadyCycle(const SUnit &SU, unsigned CurrCycle, DenseMap< unsigned, unsigned > &ReadyCycles, const TargetSchedModel &SM)
virtual void finalizeGCNSchedStage()
virtual bool initGCNSchedStage()
virtual bool shouldRevertScheduling(unsigned WavesAfter)
std::vector< std::unique_ptr< ScheduleDAGMutation > > SavedMutations
GCNSchedStage(GCNSchedStageID StageID, GCNScheduleDAGMILive &DAG)
MachineBasicBlock * CurrentMBB
This is a minimal scheduler strategy.
GCNDownwardRPTracker DownwardTracker
bool useGCNTrackers() const
void getRegisterPressures(bool AtTop, const RegPressureTracker &RPTracker, SUnit *SU, std::vector< unsigned > &Pressure, std::vector< unsigned > &MaxPressure, GCNDownwardRPTracker &DownwardTracker, GCNUpwardRPTracker &UpwardTracker, ScheduleDAGMI *DAG, const SIRegisterInfo *SRI)
GCNSchedStrategy(const MachineSchedContext *C)
SmallVector< GCNSchedStageID, 4 > SchedStages
unsigned SGPRCriticalLimit
unsigned VGPRThresholdPercent
std::vector< unsigned > MaxPressure
bool hasNextStage() const
SUnit * pickNodeBidirectional(bool &IsTopNode, bool &PickedPending)
GCNSchedStageID getCurrentStage()
bool tryPendingCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const
Evaluates instructions in the pending queue using a subset of scheduling heuristics.
SmallVectorImpl< GCNSchedStageID >::iterator CurrentStage
unsigned VGPRCriticalLimit
void schedNode(SUnit *SU, bool IsTopNode) override
Notify MachineSchedStrategy that ScheduleDAGMI has scheduled an instruction and updated scheduled/rem...
std::optional< bool > GCNTrackersOverride
GCNDownwardRPTracker * getDownwardTracker()
unsigned AGPRCriticalLimit
std::vector< unsigned > Pressure
void initialize(ScheduleDAGMI *DAG) override
Initialize the strategy after building the DAG for a new region.
GCNUpwardRPTracker UpwardTracker
void printCandidateDecision(const SchedCandidate &Current, const SchedCandidate &Preferred)
void pickNodeFromQueue(SchedBoundary &Zone, const CandPolicy &ZonePolicy, const RegPressureTracker &RPTracker, SchedCandidate &Cand, bool &IsPending, bool IsBottomUp)
void initCandidate(SchedCandidate &Cand, SUnit *SU, bool AtTop, const RegPressureTracker &RPTracker, const SIRegisterInfo *SRI, unsigned SGPRPressure, unsigned VGPRPressure, unsigned AGPRPressure, bool IsBottomUp)
SUnit * pickNode(bool &IsTopNode) override
Pick the next node to schedule, or return NULL.
GCNUpwardRPTracker * getUpwardTracker()
void finalizeSchedule() override
Allow targets to perform final scheduling actions at the level of the whole MachineFunction.
void schedule() override
Orders nodes according to selected style.
GCNScheduleDAGMILive(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S)
void recede(const MachineInstr &MI)
Move to the state of RP just before the MI .
void reset(const MachineInstr &MI)
Resets tracker to the point just after MI (in program order), which can be a debug instruction.
void compute(FunctionT &F)
Compute the cycle info for a function.
void traceCandidate(const SchedCandidate &Cand)
LLVM_ABI void setPolicy(CandPolicy &Policy, bool IsPostRA, SchedBoundary &CurrZone, SchedBoundary *OtherZone)
Set the CandPolicy given a scheduling zone given the current resources and latencies inside and outsi...
MachineSchedPolicy RegionPolicy
const TargetSchedModel * SchedModel
const MachineSchedContext * Context
const TargetRegisterInfo * TRI
SchedCandidate BotCand
Candidate last picked from Bot boundary.
SchedCandidate TopCand
Candidate last picked from Top boundary.
virtual bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const
Apply a set of heuristics to a new candidate.
void initialize(ScheduleDAGMI *dag) override
Initialize the strategy after building the DAG for a new region.
void schedNode(SUnit *SU, bool IsTopNode) override
Update the scheduler's state after scheduling a node.
GenericScheduler(const MachineSchedContext *C)
bool shouldRevertScheduling(unsigned WavesAfter) override
void resize(typename StorageT::size_type S)
void finalizeGCNRegion() override
bool initGCNRegion() override
bool initGCNSchedStage() override
LiveInterval - This class represents the liveness of a register, or stack slot.
bool hasSubRanges() const
Returns true if subregister liveness information is available.
iterator_range< subrange_iterator > subranges()
SlotIndex getInstructionIndex(const MachineInstr &Instr) const
Returns the base index of the given instruction.
SlotIndex getMBBEndIdx(const MachineBasicBlock *mbb) const
Return the last index in the given basic block.
LiveInterval & getInterval(Register Reg)
LLVM_ABI void dump() const
MachineBasicBlock * getMBBFromIndex(SlotIndex index) const
VNInfo * getVNInfoAt(SlotIndex Idx) const
getVNInfoAt - Return the VNInfo that is live at Idx, or NULL.
uint8_t getCopyCost() const
getCopyCost - Return the cost of copying a value between two registers in this class.
int getNumber() const
MachineBasicBlocks are uniquely numbered at the function level, unless they're not in a MachineFuncti...
succ_iterator succ_begin()
unsigned succ_size() const
iterator_range< pred_iterator > predecessors()
MachineInstrBundleIterator< MachineInstr > iterator
MachineBlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate machine basic b...
LLVM_ABI BlockFrequency getBlockFreq(const MachineBasicBlock *MBB) const
getblockFreq - Return block frequency.
LLVM_ABI BlockFrequency getEntryFreq() const
Divide a block's BlockFrequency::getFrequency() value by this value to obtain the entry block - relat...
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
unsigned getNumOperands() const
Retuns the total number of operands.
bool mayLoad(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read memory.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
const MachineOperand & getOperand(unsigned i) const
MachineOperand class - Representation of each machine instruction operand.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
Register getReg() const
getReg - Returns the register number.
bool shouldRevertScheduling(unsigned WavesAfter) override
bool shouldRevertScheduling(unsigned WavesAfter) override
bool shouldRevertScheduling(unsigned WavesAfter) override
void finalizeGCNRegion() override
bool initGCNRegion() override
bool initGCNSchedStage() override
Capture a change in pressure for a single pressure set.
Simple wrapper around std::function<void(raw_ostream&)>.
Helpers for implementing custom MachineSchedStrategy classes.
Track the current register pressure at some position in the instruction stream, and remember the high...
LLVM_ABI void advance()
Advance across the current instruction.
LLVM_ABI void getDownwardPressure(const MachineInstr *MI, std::vector< unsigned > &PressureResult, std::vector< unsigned > &MaxPressureResult)
Get the pressure of each PSet after traversing this instruction top-down.
const std::vector< unsigned > & getRegSetPressureAtPos() const
Get the register set pressure at the current position, which may be less than the pressure across the...
LLVM_ABI void getUpwardPressure(const MachineInstr *MI, std::vector< unsigned > &PressureResult, std::vector< unsigned > &MaxPressureResult)
Get the pressure of each PSet after traversing this instruction bottom-up.
GCNRPTracker::LiveRegSet & getLiveRegsForRegionIdx(unsigned RegionIdx)
static LLVM_ABI void restoreLivenessFlags(MachineInstr &MI, const TargetRegisterInfo &TRI, const MachineRegisterInfo &MRI, LiveIntervals &LIS, bool TrackLaneMasks=true, ArrayRef< Register > OnlyRegs={})
Clear potentially-stale read-undef flags on the defs of MI, then recompute them from LIS.
Wrapper class representing virtual and physical registers.
constexpr bool isValid() const
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
static constexpr bool isVirtualRegister(unsigned Reg)
Return true if the specified register number is in the virtual register namespace.
MIR-level target-independent rematerializer.
bool isIGLPMutationOnly(unsigned Opcode) const
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
unsigned getOccupancy() const
unsigned getDynamicVGPRBlockSize() const
unsigned getMinAllowedOccupancy() const
Scheduling unit. This is a node in the scheduling DAG.
bool isInstr() const
Returns true if this SUnit refers to a machine instruction as opposed to an SDNode.
unsigned TopReadyCycle
Cycle relative to start when node is ready.
unsigned NodeNum
Entry # of node in the node vector.
unsigned short Latency
Node latency.
bool isScheduled
True once scheduled.
unsigned ParentClusterIdx
The parent cluster id.
unsigned BotReadyCycle
Cycle relative to end when node is ready.
bool isBottomReady() const
SmallVector< SDep, 4 > Preds
All sunit predecessors.
MachineInstr * getInstr() const
Returns the representative MachineInstr for this SUnit.
Each Scheduling boundary is associated with ready queues.
LLVM_ABI void releasePending()
Release pending ready nodes in to the available queue.
LLVM_ABI unsigned getLatencyStallCycles(SUnit *SU)
Get the difference between the given SUnit's ready time and the current cycle.
LLVM_ABI SUnit * pickOnlyChoice()
Call this before applying any other heuristics to the Available queue.
LLVM_ABI void bumpCycle(unsigned NextCycle)
Move the boundary of scheduled code by one cycle.
unsigned getCurrMOps() const
Micro-ops issued in the current cycle.
unsigned getCurrCycle() const
Number of cycles to issue the instructions scheduled in this zone.
LLVM_ABI bool checkHazard(SUnit *SU)
Does this SU have a hazard within the current instruction group.
A ScheduleDAG for scheduling lists of MachineInstr.
bool ScheduleSingleMIRegions
True if regions with a single MI should be scheduled.
MachineBasicBlock::iterator RegionEnd
The end of the range to be scheduled.
virtual void finalizeSchedule()
Allow targets to perform final scheduling actions at the level of the whole MachineFunction.
virtual void exitRegion()
Called when the scheduler has finished scheduling the current region.
const MachineLoopInfo * MLI
bool RemoveKillFlags
True if the DAG builder should remove kill flags (in preparation for rescheduling).
MachineBasicBlock::iterator RegionBegin
The beginning of the range to be scheduled.
void schedule() override
Implement ScheduleDAGInstrs interface for scheduling a sequence of reorderable instructions.
ScheduleDAGMILive(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S)
RegPressureTracker RPTracker
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
void addMutation(std::unique_ptr< ScheduleDAGMutation > Mutation)
Add a postprocessing step to the DAG builder.
void schedule() override
Implement ScheduleDAGInstrs interface for scheduling a sequence of reorderable instructions.
ScheduleDAGMI(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S, bool RemoveKillFlags)
std::vector< std::unique_ptr< ScheduleDAGMutation > > Mutations
Ordered list of DAG postprocessing steps.
MachineRegisterInfo & MRI
Virtual/real register map.
const TargetInstrInfo * TII
Target instruction information.
MachineFunction & MF
Machine function.
static const unsigned ScaleFactor
unsigned getMetric() const
bool empty() const
Determine if the SetVector is empty or not.
bool insert(const value_type &X)
Insert a new element into the SetVector.
SlotIndex - An opaque wrapper around machine indexes.
static bool isSameInstr(SlotIndex A, SlotIndex B)
isSameInstr - Return true if A and B refer to the same instruction.
static bool isEarlierInstr(SlotIndex A, SlotIndex B)
isEarlierInstr - Return true if A refers to an instruction earlier than B.
SlotIndex getPrevSlot() const
Returns the previous slot in the index list.
SlotIndex getMBBStartIdx(const MachineBasicBlock *mbb) const
Returns the first index in the given basic block.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
SmallSet - This maintains a set of unique values, optimizing for the case when the set is small (less...
bool contains(const T &V) const
Check if the SmallSet contains the given element.
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Provide an instruction scheduling machine model to CodeGen passes.
LLVM_ABI bool hasInstrSchedModel() const
Return true if this machine model includes an instruction-level scheduling model.
unsigned getMicroOpBufferSize() const
Number of micro-ops that may be buffered for OOO execution.
bool initGCNSchedStage() override
bool initGCNRegion() override
void finalizeGCNSchedStage() override
bool shouldRevertScheduling(unsigned WavesAfter) override
VNInfo - Value Number Information.
SlotIndex def
The index of the defining instruction.
bool isPHIDef() const
Returns true if this value is defined by a PHI instruction (or was, PHI instructions may have been el...
LLVM Value Representation.
std::pair< iterator, bool > insert(const ValueT &V)
bool contains(const_arg_type_t< ValueT > V) const
Check if the set contains the given element.
self_iterator getIterator()
This class implements an extremely fast bulk output stream that can only output to a stream.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize)
unsigned getAllocatedNumVGPRBlocks(const MCSubtargetInfo &STI, unsigned NumVGPRs, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
LLVM_READONLY int32_t getAGPRFormOp(uint32_t Opcode)
initializer< Ty > init(const Ty &Val)
@ User
could "use" a pointer
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI int biasPhysReg(const SUnit *SU, bool isTop, bool BiasPRegsExtra=false)
Minimize physical register live ranges.
auto find(R &&Range, const T &Val)
Provide wrappers to std::find which take ranges instead of having to pass begin/end explicitly.
bool isEqual(const GCNRPTracker::LiveRegSet &S1, const GCNRPTracker::LiveRegSet &S2)
Printable print(const GCNRegPressure &RP, const GCNSubtarget *ST=nullptr, unsigned DynamicVGPRBlockSize=0)
LLVM_ABI unsigned getWeakLeft(const SUnit *SU, bool isTop)
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
GCNRegPressure getRegPressure(const MachineRegisterInfo &MRI, Range &&LiveRegs)
std::unique_ptr< ScheduleDAGMutation > createIGroupLPDAGMutation(AMDGPU::SchedulingPhase Phase)
Phase specifes whether or not this is a reentry into the IGroupLPDAGMutation.
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
std::pair< MachineBasicBlock::iterator, MachineBasicBlock::iterator > RegionBoundaries
A region's boundaries i.e.
IterT skipDebugInstructionsForward(IterT It, IterT End, bool SkipPseudoOp=true)
Increment It until it points to a non-debug instruction or to End and return the resulting iterator.
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI bool tryPressure(const PressureChange &TryP, const PressureChange &CandP, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason, const TargetRegisterInfo *TRI, const MachineFunction &MF)
@ UnclusteredHighRPReschedule
@ MemoryClauseInitialSchedule
@ LiveIntervalRPReschedule
@ ClusteredLowOccupancyReschedule
auto reverse(ContainerTy &&C)
void sort(IteratorTy Start, IteratorTy End)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
cl::opt< unsigned, false, VGPRThresholdParser > VGPRThresholdPercentOpt
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
LLVM_ABI bool shouldVerifyScheduling()
Returns whether -verify-misched is set.
iterator_range< ConstMIBundleOperands > const_mi_bundle_ops(const MachineInstr &MI)
LLVM_ABI bool tryLatency(GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, SchedBoundary &Zone)
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
IterT skipDebugInstructionsBackward(IterT It, IterT Begin, bool SkipPseudoOp=true)
Decrement It until it points to a non-debug instruction or to Begin and return the resulting iterator...
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
bool isTheSameCluster(unsigned A, unsigned B)
Return whether the input cluster ID's are the same and valid.
DWARFExpression::Operation Op
LLVM_ABI bool tryGreater(int TryVal, int CandVal, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason)
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
ArrayRef(const T &OneElt) -> ArrayRef< T >
OutputIt move(R &&Range, OutputIt Out)
Provide wrappers to std::move which take ranges instead of having to pass begin/end explicitly.
DenseMap< MachineInstr *, GCNRPTracker::LiveRegSet > getLiveRegMap(Range &&R, bool After, LiveIntervals &LIS)
creates a map MachineInstr -> LiveRegSet R - range of iterators on instructions After - upon entry or...
GCNRPTracker::LiveRegSet getLiveRegsBefore(const MachineInstr &MI, const LiveIntervals &LIS)
LLVM_ABI bool tryLess(int TryVal, int CandVal, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason)
Return true if this heuristic determines order.
LLVM_ABI void dumpMaxRegPressure(MachineFunction &MF, GCNRegPressure::RegKind Kind, LiveIntervals &LIS, const MachineLoopInfo *MLI)
unsigned estimateGreedyVGPRPressure(MachineBasicBlock::const_iterator RegionBegin, MachineBasicBlock::const_iterator RegionEnd, const GCNRPTracker::LiveRegSet &LiveIns, const LiveIntervals &LIS, const MachineRegisterInfo &MRI, const SIRegisterInfo &TRI)
Estimate VGPR pressure using greedy, non-splitting register allocation simulation,...
LLVM_ABI Printable printMBBReference(const MachineBasicBlock &MBB)
Prints a machine basic block reference.
MCRegisterClass TargetRegisterClass
Implement std::hash so that hash_code can be used in STL containers.
bool operator()(std::pair< MachineInstr *, unsigned > A, std::pair< MachineInstr *, unsigned > B) const
unsigned getArchVGPRNum() const
unsigned getAGPRNum() const
unsigned getSGPRNum() const
Policy for scheduling the next instruction in the candidate's zone.
Store the state used by GenericScheduler heuristics, required for the lifetime of one invocation of p...
void setBest(SchedCandidate &Best)
void reset(const CandPolicy &NewPolicy)
LLVM_ABI void initResourceDelta(const ScheduleDAGMI *DAG, const TargetSchedModel *SchedModel)
SchedResourceDelta ResDelta
Status of an instruction's critical resource consumption.
unsigned DemandedResources
constexpr bool any() const
static constexpr LaneBitmask getNone()
MachineSchedContext provides enough context from the MachineScheduler pass for the target to instanti...
Execution frequency information required by scoring heuristics.
SmallVector< uint64_t > Regions
Per-region execution frequencies. 0 when unknown.
uint64_t MinFreq
Minimum and maximum observed frequencies.
FreqInfo(MachineFunction &MF, const GCNScheduleDAGMILive &DAG)
PressureChange CriticalMax
PressureChange CurrentMax
DependencyReuseInfo & reuse(RegisterIdx DepIdx)
A rematerializable register, potentially defined by multiple instructions.
LLVM_ABI std::pair< MachineInstr *, MachineInstr * > getRegionUseBounds(unsigned UseRegion, const LiveIntervals &LIS) const
Returns the first and last user of the register in region UseRegion.
SmallVector< MachineInstr *, 1 > Defs
All instructions that define the register, in program order.
SmallDenseMap< unsigned, RegionUsers, 2 > Uses
Uses of the register, mapped by region.
MachineInstr * getLastDef() const
SmallVector< RegisterIdx, 2 > Dependencies
This register's rematerializable dependencies, one per unique rematerializable register operand over ...
bool parse(cl::Option &O, StringRef ArgName, StringRef Arg, unsigned &Value)