46#define DEBUG_TYPE "machine-scheduler"
51 "amdgpu-disable-unclustered-high-rp-reschedule",
cl::Hidden,
52 cl::desc(
"Disable unclustered high register pressure "
53 "reduction scheduling stage."),
57 "amdgpu-disable-clustered-low-occupancy-reschedule",
cl::Hidden,
58 cl::desc(
"Disable clustered low occupancy "
59 "rescheduling for ILP scheduling stage."),
65 "Sets the bias which adds weight to occupancy vs latency. Set it to "
66 "100 to chase the occupancy only."),
71 cl::desc(
"Relax occupancy targets for kernels which are memory "
72 "bound (amdgpu-membound-threshold), or "
73 "Wave Limited (amdgpu-limit-wave-threshold)."),
78 cl::desc(
"Use the AMDGPU specific RPTrackers during scheduling"),
82 "amdgpu-scheduler-pending-queue-limit",
cl::Hidden,
84 "Max (Available+Pending) size to inspect pending queue (0 disables)"),
87#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
88#define DUMP_MAX_REG_PRESSURE
90 "amdgpu-print-max-reg-pressure-regusage-before-scheduler",
cl::Hidden,
91 cl::desc(
"Print a list of live registers along with their def/uses at the "
92 "point of maximum register pressure before scheduling."),
96 "amdgpu-print-max-reg-pressure-regusage-after-scheduler",
cl::Hidden,
97 cl::desc(
"Print a list of live registers along with their def/uses at the "
98 "point of maximum register pressure after scheduling."),
103 "amdgpu-disable-rewrite-mfma-form-sched-stage",
cl::Hidden,
108struct VGPRThresholdParser :
public cl::parser<unsigned> {
111 bool parse(cl::Option &O, StringRef ArgName, StringRef Arg,
unsigned &
Value) {
113 return O.error(
"'" + Arg +
"' value invalid for uint argument!");
116 return O.error(
"'" + Arg +
"' value must be in the range [0, 100]!");
126 cl::desc(
"Percent of VGPR limits that we should use as RP threshold "
127 "during scheduling. We have two limits relevant to scheduling: "
128 "Critical (avoid decreasing occupancy), Excess (avoid spilling). "
129 "This flag scales both limits back by an equal percent: (0 = use "
130 " default calculation, 1-100 = use percentage), default: 0"),
150 Context->RegClassInfo->getNumAllocatableRegs(&AMDGPU::SGPR_32RegClass);
152 Context->RegClassInfo->getNumAllocatableRegs(&AMDGPU::VGPR_32RegClass);
154 Context->RegClassInfo->getNumAllocatableRegs(&AMDGPU::AGPR_32RegClass);
176 "VGPRCriticalLimit calculation method.\n");
180 unsigned Addressable =
183 VGPRBudget = std::max(VGPRBudget, Granule);
200 <<
". VGPRCriticalLimit: " << OriginalVGPRCriticalLimit
246 if (!
Op.isReg() ||
Op.isImplicit())
248 if (
Op.getReg().isPhysical() ||
249 (
Op.isDef() &&
Op.getSubReg() != AMDGPU::NoSubRegister))
284 Pressure[AMDGPU::RegisterPressureSets::VGPR_32] =
292 if (!Zone.
isTop() || !SU)
309 if (NextAvail > CurrCycle)
310 Stall = std::max(
Stall, NextAvail - CurrCycle);
329 unsigned SGPRPressure,
330 unsigned VGPRPressure,
331 unsigned AGPRPressure,
bool IsBottomUp) {
335 if (!
DAG->isTrackingPressure())
358 Pressure[AMDGPU::RegisterPressureSets::SReg_32] = SGPRPressure;
359 Pressure[AMDGPU::RegisterPressureSets::VGPR_32] = VGPRPressure;
360 Pressure[AMDGPU::RegisterPressureSets::AGPR_32] = AGPRPressure;
362 for (
const auto &Diff :
DAG->getPressureDiff(SU)) {
368 (IsBottomUp ? Diff.getUnitInc() : -Diff.getUnitInc());
371#ifdef EXPENSIVE_CHECKS
372 std::vector<unsigned> CheckPressure, CheckMaxPressure;
375 if (
Pressure[AMDGPU::RegisterPressureSets::SReg_32] !=
376 CheckPressure[AMDGPU::RegisterPressureSets::SReg_32] ||
377 Pressure[AMDGPU::RegisterPressureSets::VGPR_32] !=
378 CheckPressure[AMDGPU::RegisterPressureSets::VGPR_32] ||
379 Pressure[AMDGPU::RegisterPressureSets::AGPR_32] !=
380 CheckPressure[AMDGPU::RegisterPressureSets::AGPR_32]) {
381 errs() <<
"Register Pressure is inaccurate when calculated through "
383 <<
"SGPR got " <<
Pressure[AMDGPU::RegisterPressureSets::SReg_32]
385 << CheckPressure[AMDGPU::RegisterPressureSets::SReg_32] <<
"\n"
386 <<
"VGPR got " <<
Pressure[AMDGPU::RegisterPressureSets::VGPR_32]
388 << CheckPressure[AMDGPU::RegisterPressureSets::VGPR_32] <<
"\n"
389 <<
"AGPR got " <<
Pressure[AMDGPU::RegisterPressureSets::AGPR_32]
391 << CheckPressure[AMDGPU::RegisterPressureSets::AGPR_32] <<
"\n";
397 unsigned NewAGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::AGPR_32];
398 unsigned NewSGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::SReg_32];
399 unsigned NewVGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::VGPR_32];
409 const unsigned MaxVGPRPressureInc = 16;
410 bool ShouldTrackVGPRs = VGPRPressure + MaxVGPRPressureInc >=
VGPRExcessLimit;
413 bool ShouldTrackSGPRs =
414 !ShouldTrackVGPRs && !ShouldTrackAGPRs && SGPRPressure >=
SGPRExcessLimit;
449 : std::numeric_limits<int>::min();
451 if (SGPRDelta >= 0 || VGPRDelta >= 0 || AGPRDelta >= 0) {
454 if (VGPRDelta >= SGPRDelta && VGPRDelta >= AGPRDelta) {
458 }
else if (AGPRDelta >= SGPRDelta) {
472 bool HasBufferedModel =
491 dbgs() <<
"Prefer:\t\t";
492 DAG->dumpNode(*Preferred.
SU);
496 DAG->dumpNode(*Current.
SU);
499 dbgs() <<
"Reason:\t\t";
513 unsigned SGPRPressure = 0;
514 unsigned VGPRPressure = 0;
515 unsigned AGPRPressure = 0;
517 if (
DAG->isTrackingPressure()) {
519 SGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::SReg_32];
520 VGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::VGPR_32];
521 AGPRPressure =
Pressure[AMDGPU::RegisterPressureSets::AGPR_32];
526 SGPRPressure =
T->getPressure().getSGPRNum();
527 VGPRPressure =
T->getPressure().getArchVGPRNum();
528 AGPRPressure =
T->getPressure().getAGPRNum();
533 for (
SUnit *SU : AQ) {
537 VGPRPressure, AGPRPressure, IsBottomUp);
557 for (
SUnit *SU : PQ) {
561 VGPRPressure, AGPRPressure, IsBottomUp);
581 bool &PickedPending) {
601 bool BotPending =
false;
621 "Last pick result should correspond to re-picking right now");
626 bool TopPending =
false;
646 "Last pick result should correspond to re-picking right now");
656 PickedPending = BotPending && TopPending;
659 if (BotPending || TopPending) {
666 Cand.setBest(TryCand);
671 IsTopNode = Cand.AtTop;
678 if (
DAG->top() ==
DAG->bottom()) {
680 Bot.Available.empty() &&
Bot.Pending.empty() &&
"ReadyQ garbage");
686 PickedPending =
false;
720 if (ReadyCycle > CurrentCycle)
792 if (
DAG->isTrackingPressure() &&
798 if (
DAG->isTrackingPressure() &&
803 bool SameBoundary = Zone !=
nullptr;
827 if (IsLegacyScheduler)
846 if (
DAG->isTrackingPressure() &&
856 bool SameBoundary = Zone !=
nullptr;
891 bool CandIsClusterSucc =
893 bool TryCandIsClusterSucc =
895 if (
tryGreater(TryCandIsClusterSucc, CandIsClusterSucc, TryCand, Cand,
900 if (
DAG->isTrackingPressure() &&
906 if (
DAG->isTrackingPressure() &&
952 if (
DAG->isTrackingPressure()) {
968 bool CandIsClusterSucc =
970 bool TryCandIsClusterSucc =
972 if (
tryGreater(TryCandIsClusterSucc, CandIsClusterSucc, TryCand, Cand,
981 bool SameBoundary = Zone !=
nullptr;
998 if (TryMayLoad || CandMayLoad) {
999 bool TryLongLatency =
1001 bool CandLongLatency =
1005 Zone->
isTop() ? CandLongLatency : TryLongLatency, TryCand,
1023 if (
DAG->isTrackingPressure() &&
1042 !
Rem.IsAcyclicLatencyLimited &&
tryLatency(TryCand, Cand, *Zone))
1060 StartingOccupancy(MFI.getOccupancy()), MinOccupancy(StartingOccupancy),
1061 RegionLiveOuts(this,
true) {
1067 LLVM_DEBUG(
dbgs() <<
"Starting occupancy is " << StartingOccupancy <<
".\n");
1069 MinOccupancy = std::min(MFI.getMinAllowedOccupancy(), StartingOccupancy);
1070 if (MinOccupancy != StartingOccupancy)
1071 LLVM_DEBUG(
dbgs() <<
"Allowing Occupancy drops to " << MinOccupancy
1076std::unique_ptr<GCNSchedStage>
1078 switch (SchedStageID) {
1080 return std::make_unique<OccInitialScheduleStage>(SchedStageID, *
this);
1082 return std::make_unique<RewriteMFMAFormStage>(SchedStageID, *
this);
1084 return std::make_unique<UnclusteredHighRPStage>(SchedStageID, *
this);
1086 return std::make_unique<ClusteredLowOccStage>(SchedStageID, *
this);
1088 return std::make_unique<PreRARematStage>(SchedStageID, *
this);
1090 return std::make_unique<ILPInitialScheduleStage>(SchedStageID, *
this);
1092 return std::make_unique<MemoryClauseInitialScheduleStage>(SchedStageID,
1106GCNScheduleDAGMILive::getRealRegPressure(
unsigned RegionIdx)
const {
1107 if (Regions[RegionIdx].first == Regions[RegionIdx].second)
1111 &LiveIns[RegionIdx]);
1117 assert(RegionBegin != RegionEnd &&
"Region must not be empty");
1121void GCNScheduleDAGMILive::computeBlockPressure(
unsigned RegionIdx,
1133 const MachineBasicBlock *OnlySucc =
nullptr;
1136 if (!Candidate->empty() && Candidate->pred_size() == 1) {
1137 SlotIndexes *Ind =
LIS->getSlotIndexes();
1139 OnlySucc = Candidate;
1144 size_t CurRegion = RegionIdx;
1145 for (
size_t E = Regions.size(); CurRegion !=
E; ++CurRegion)
1146 if (Regions[CurRegion].first->getParent() !=
MBB)
1151 auto LiveInIt = MBBLiveIns.find(
MBB);
1152 auto &Rgn = Regions[CurRegion];
1154 if (LiveInIt != MBBLiveIns.end()) {
1155 auto LiveIn = std::move(LiveInIt->second);
1157 MBBLiveIns.erase(LiveInIt);
1160 auto LRS = BBLiveInMap.lookup(NonDbgMI);
1161#ifdef EXPENSIVE_CHECKS
1170 if (Regions[CurRegion].first ==
I || NonDbgMI ==
I) {
1171 LiveIns[CurRegion] =
RPTracker.getLiveRegs();
1175 if (Regions[CurRegion].second ==
I) {
1176 Pressure[CurRegion] =
RPTracker.moveMaxPressure();
1177 if (CurRegion-- == RegionIdx)
1179 auto &Rgn = Regions[CurRegion];
1192 MBBLiveIns[OnlySucc] =
RPTracker.moveLiveRegs();
1197GCNScheduleDAGMILive::getRegionLiveInMap()
const {
1198 assert(!Regions.empty());
1199 std::vector<MachineInstr *> RegionFirstMIs;
1200 RegionFirstMIs.reserve(Regions.size());
1202 RegionFirstMIs.push_back(
1209GCNScheduleDAGMILive::getRegionLiveOutMap()
const {
1210 assert(!Regions.empty());
1211 std::vector<MachineInstr *> RegionLastMIs;
1212 RegionLastMIs.reserve(Regions.size());
1223 IdxToInstruction.clear();
1226 IsLiveOut ? DAG->getRegionLiveOutMap() : DAG->getRegionLiveInMap();
1227 for (
unsigned I = 0;
I < DAG->Regions.size();
I++) {
1228 auto &[RegionBegin, RegionEnd] = DAG->Regions[
I];
1230 if (RegionBegin == RegionEnd)
1234 IdxToInstruction[
I] = RegionKey;
1242 LiveIns.resize(Regions.size());
1243 Pressure.resize(Regions.size());
1244 RegionsWithHighRP.resize(Regions.size());
1245 RegionsWithExcessRP.resize(Regions.size());
1246 RegionsWithIGLPInstrs.resize(Regions.size());
1247 RegionsWithHighRP.reset();
1248 RegionsWithExcessRP.reset();
1249 RegionsWithIGLPInstrs.reset();
1254void GCNScheduleDAGMILive::runSchedStages() {
1255 LLVM_DEBUG(
dbgs() <<
"All regions recorded, starting actual scheduling.\n");
1258 if (!Regions.
empty()) {
1259 BBLiveInMap = getRegionLiveInMap();
1264#ifdef DUMP_MAX_REG_PRESSURE
1274 if (!Stage->initGCNSchedStage())
1277 for (
auto Region : Regions) {
1281 if (!Stage->initGCNRegion()) {
1282 Stage->advanceRegion();
1288 const unsigned RegionIdx = Stage->getRegionIdx();
1291 MRI, RegionLiveOuts.getLiveRegsForRegionIdx(RegionIdx));
1295 Stage->finalizeGCNRegion();
1296 Stage->advanceRegion();
1300 Stage->finalizeGCNSchedStage();
1303#ifdef DUMP_MAX_REG_PRESSURE
1316 OS <<
"Max Occupancy Initial Schedule";
1319 OS <<
"Instruction Rewriting Reschedule";
1322 OS <<
"Unclustered High Register Pressure Reschedule";
1325 OS <<
"Clustered Low Occupancy Reschedule";
1328 OS <<
"Pre-RA Rematerialize";
1331 OS <<
"Max ILP Initial Schedule";
1334 OS <<
"Max memory clause Initial Schedule";
1354void RewriteMFMAFormStage::findReachingDefs(
1376 while (!Worklist.
empty()) {
1391 for (MachineBasicBlock *PredMBB : DefMBB->
predecessors()) {
1392 if (Visited.
insert(PredMBB).second)
1398void RewriteMFMAFormStage::findReachingUses(
1402 for (MachineOperand &UseMO :
1405 findReachingDefs(UseMO, LIS, ReachingDefIndexes);
1409 if (
any_of(ReachingDefIndexes, [DefIdx](SlotIndex RDIdx) {
1421 if (!
ST.hasGFX90AInsts() ||
MFI.getMinWavesPerEU() > 1)
1424 RegionsWithExcessArchVGPR.resize(
DAG.Regions.size());
1425 RegionsWithExcessArchVGPR.reset();
1429 RegionsWithExcessArchVGPR[
Region] =
true;
1432 if (RegionsWithExcessArchVGPR.none())
1435 TII =
ST.getInstrInfo();
1436 SRI =
ST.getRegisterInfo();
1438 std::vector<std::pair<MachineInstr *, unsigned>> RewriteCands;
1442 if (!initHeuristics(RewriteCands, CopyForUse, CopyForDef))
1445 int64_t
Cost = getRewriteCost(RewriteCands, CopyForUse, CopyForDef);
1452 return rewrite(RewriteCands);
1462 if (
DAG.RegionsWithHighRP.none() &&
DAG.RegionsWithExcessRP.none())
1469 InitialOccupancy =
DAG.MinOccupancy;
1472 TempTargetOccupancy =
MFI.getMaxWavesPerEU() >
DAG.MinOccupancy
1473 ? InitialOccupancy + 1
1475 IsAnyRegionScheduled =
false;
1476 S.SGPRLimitBias =
S.HighRPSGPRBias;
1477 S.VGPRLimitBias =
S.HighRPVGPRBias;
1481 <<
"Retrying function scheduling without clustering. "
1482 "Aggressively try to reduce register pressure to achieve occupancy "
1483 << TempTargetOccupancy <<
".\n");
1498 if (
DAG.StartingOccupancy <=
DAG.MinOccupancy)
1502 dbgs() <<
"Retrying function scheduling with lowest recorded occupancy "
1503 <<
DAG.MinOccupancy <<
".\n");
1508#define REMAT_PREFIX "[PreRARemat] "
1509#define REMAT_DEBUG(X) LLVM_DEBUG(dbgs() << REMAT_PREFIX; X;)
1511#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1512Printable PreRARematStage::ScoredRemat::print()
const {
1514 OS <<
'(' << MaxFreq <<
", " << FreqDiff <<
", " << RegionImpact <<
')';
1529 auto PrintTargetRegions = [&]() ->
void {
1530 if (TargetRegions.none()) {
1535 for (
unsigned I : TargetRegions.set_bits())
1542 dbgs() <<
"Analyzing ";
1543 MF.getFunction().printAsOperand(
dbgs(),
false);
1546 if (!setObjective()) {
1547 LLVM_DEBUG(
dbgs() <<
"no objective to achieve, occupancy is maximal at "
1548 <<
MFI.getMaxWavesPerEU() <<
'\n');
1553 dbgs() <<
"increase occupancy from " << *TargetOcc - 1 <<
'\n';
1555 dbgs() <<
"reduce spilling (minimum target occupancy is "
1556 <<
MFI.getMinWavesPerEU() <<
")\n";
1558 PrintTargetRegions();
1563 DAG.RegionLiveOuts.buildLiveRegMap();
1565 if (!Remater.analyze()) {
1579 for (
unsigned RegIdx = 0, E = Remater.getNumRegs(); RegIdx < E; ++RegIdx) {
1583 if (CandReg.
Uses.size() != 1)
1585 const auto [UseRegion,
Users] = *CandReg.
Uses.begin();
1604 "user must have at least one operand");
1611 assert(FirstUseMI &&
"there must be a user in the region");
1613 DAG.LIS->getInstructionIndex(*FirstUseMI).getRegSlot(
true);
1615 DAG.LIS->getInstructionIndex(*CandReg.
getLastDef()).getRegSlot(
true);
1617 const Rematerializer::Reg &DepReg = Remater.getReg(DepRegIdx);
1618 Register DepDefReg = DepReg.getDefReg();
1619 return MarkedRegs.contains(DepDefReg) ||
1620 !Remater.isRegIdenticalAtUses(DepDefReg, DepReg.Mask, RefIdx,
1625 [&](
const std::pair<Register, LaneBitmask> &RegAndMask) {
1626 const auto &[Reg, Mask] = RegAndMask;
1627 return !Remater.isRegIdenticalAtUses(Reg, Mask, RefIdx,
1632 MarkedRegs.
insert(CandReg.getDefReg());
1634 Cand.init(RegIdx, FreqInfo, Remater,
DAG);
1635 Cand.update(TargetRegions, RPTargets, FreqInfo, !TargetOcc);
1636 if (!Cand.hasNullScore())
1647 Rollback = std::make_unique<RollbackSupport>(Remater);
1654 RecomputeRP.reset();
1657 sort(CandidateOrder, [&](
unsigned LHSIndex,
unsigned RHSIndex) {
1658 return Candidates[LHSIndex] < Candidates[RHSIndex];
1662 dbgs() <<
"==== NEW REMAT ROUND ====\n"
1664 <<
"Candidates with non-null score, in rematerialization order:\n";
1665 for (
const ScoredRemat &Cand :
reverse(Candidates)) {
1667 << Remater.printRematReg(Cand.RegIdx) <<
'\n';
1669 PrintTargetRegions();
1675 while (!CandidateOrder.empty()) {
1676 const ScoredRemat &Cand = Candidates[CandidateOrder.back()];
1677 const Rematerializer::Reg &
Reg = Remater.getReg(Cand.RegIdx);
1685 if (!Cand.maybeBeneficial(TargetRegions, RPTargets)) {
1687 << Cand.print() <<
" | "
1688 << Remater.printRematReg(Cand.RegIdx));
1691 CandidateOrder.pop_back();
1693#ifdef EXPENSIVE_CHECKS
1696 for (
const MachineInstr *
DefMI :
Reg.Defs) {
1701 if (!MO.isReg() || !MO.getReg() || !MO.readsReg() || MO.isDef())
1708 LiveInterval &LI =
DAG.LIS->getInterval(
UseReg);
1709 LaneBitmask LM =
DAG.MRI.getMaxLaneMaskForVReg(MO.getReg());
1711 LM =
DAG.TRI->getSubRegIndexLaneMask(MO.getSubReg());
1713 const unsigned UseRegion =
Reg.Uses.begin()->first;
1714 LaneBitmask LiveInMask =
DAG.LiveIns[UseRegion].at(
UseReg);
1715 LaneBitmask UncoveredLanes = LM & ~(LiveInMask & LM);
1719 if (UncoveredLanes.
any()) {
1721 for (LiveInterval::SubRange &SR : LI.
subranges())
1722 assert((SR.LaneMask & UncoveredLanes).none());
1730 REMAT_DEBUG(
dbgs() <<
"** REMAT " << Remater.printRematReg(Cand.RegIdx)
1732 removeFromLiveMaps(
Reg.getDefReg(), Cand.LiveIn, Cand.LiveOut);
1734 Rollback->LiveMapUpdates.emplace_back(Cand.RegIdx, Cand.LiveIn,
1737 Cand.rematerialize(Remater);
1742 updateRPTargets(Cand.Live, Cand.RPSave);
1743 RecomputeRP |= Cand.UnpredictableRPSave;
1744 RescheduleRegions |= Cand.Live;
1745 if (!TargetRegions.any()) {
1751 if (!updateAndVerifyRPTargets(RecomputeRP) && !TargetRegions.any()) {
1760 unsigned NumUsefulCandidates = 0;
1761 for (
unsigned CandIdx : CandidateOrder) {
1762 ScoredRemat &Candidate = Candidates[CandIdx];
1763 Candidate.update(TargetRegions, RPTargets, FreqInfo, !TargetOcc);
1764 if (!Candidate.hasNullScore())
1765 CandidateOrder[NumUsefulCandidates++] = CandIdx;
1767 if (NumUsefulCandidates == 0) {
1768 REMAT_DEBUG(
dbgs() <<
"Stop on exhausted rematerialization candidates\n");
1771 CandidateOrder.truncate(NumUsefulCandidates);
1774 if (RescheduleRegions.none())
1780 unsigned DynamicVGPRBlockSize =
MFI.getDynamicVGPRBlockSize();
1781 for (
unsigned I : RescheduleRegions.set_bits()) {
1782 DAG.Pressure[
I] = RPTargets[
I].getCurrentRP();
1784 <<
DAG.Pressure[
I].getOccupancy(
ST, DynamicVGPRBlockSize)
1785 <<
" (" << RPTargets[
I] <<
")\n");
1787 AchievedOcc =
MFI.getMaxWavesPerEU();
1788 for (
const GCNRegPressure &RP :
DAG.Pressure) {
1790 std::min(AchievedOcc,
RP.getOccupancy(
ST, DynamicVGPRBlockSize));
1794 dbgs() <<
"Retrying function scheduling with new min. occupancy of "
1795 << AchievedOcc <<
" from rematerializing (original was "
1796 <<
DAG.MinOccupancy;
1798 dbgs() <<
", target was " << *TargetOcc;
1802 DAG.setTargetOccupancy(getStageTargetOccupancy());
1813 S.SGPRLimitBias =
S.VGPRLimitBias = 0;
1814 if (
DAG.MinOccupancy > InitialOccupancy) {
1815 assert(IsAnyRegionScheduled);
1817 <<
" stage successfully increased occupancy to "
1818 <<
DAG.MinOccupancy <<
'\n');
1819 }
else if (!IsAnyRegionScheduled) {
1820 assert(
DAG.MinOccupancy == InitialOccupancy);
1822 <<
": No regions scheduled, min occupancy stays at "
1823 <<
DAG.MinOccupancy <<
", MFI occupancy stays at "
1824 <<
MFI.getOccupancy() <<
".\n");
1832 if (
DAG.begin() ==
DAG.end())
1839 unsigned NumRegionInstrs = std::distance(
DAG.begin(),
DAG.end());
1843 if (
DAG.begin() == std::prev(
DAG.end()))
1849 <<
"\n From: " << *
DAG.begin() <<
" To: ";
1851 else dbgs() <<
"End";
1852 dbgs() <<
" RegionInstrs: " << NumRegionInstrs <<
'\n');
1860 for (
auto &
I :
DAG) {
1873 dbgs() <<
"Pressure before scheduling:\nRegion live-ins:"
1875 <<
"Region live-in pressure: "
1879 S.HasHighPressure =
false;
1901 unsigned DynamicVGPRBlockSize =
DAG.MFI.getDynamicVGPRBlockSize();
1904 unsigned CurrentTargetOccupancy =
1905 IsAnyRegionScheduled ?
DAG.MinOccupancy : TempTargetOccupancy;
1907 (CurrentTargetOccupancy <= InitialOccupancy ||
1908 DAG.Pressure[
RegionIdx].getOccupancy(
ST, DynamicVGPRBlockSize) !=
1915 if (!IsAnyRegionScheduled && IsSchedulingThisRegion) {
1916 IsAnyRegionScheduled =
true;
1917 if (
MFI.getMaxWavesPerEU() >
DAG.MinOccupancy)
1918 DAG.setTargetOccupancy(TempTargetOccupancy);
1920 return IsSchedulingThisRegion;
1936 return !RevertAllRegions && RescheduleRegions[
RegionIdx] &&
1956 if (
S.HasHighPressure)
1977 if (
DAG.MinOccupancy < *TargetOcc) {
1979 <<
" cannot meet occupancy target, interrupting "
1980 "re-scheduling in all regions\n");
1981 RevertAllRegions =
true;
1992 unsigned DynamicVGPRBlockSize =
DAG.MFI.getDynamicVGPRBlockSize();
2003 unsigned TargetOccupancy = std::min(
2004 S.getTargetOccupancy(),
ST.getOccupancyWithWorkGroupSizes(
MF).second);
2005 unsigned WavesAfter = std::min(
2006 TargetOccupancy,
PressureAfter.getOccupancy(
ST, DynamicVGPRBlockSize));
2007 unsigned WavesBefore = std::min(
2009 LLVM_DEBUG(
dbgs() <<
"Occupancy before scheduling: " << WavesBefore
2010 <<
", after " << WavesAfter <<
".\n");
2016 unsigned NewOccupancy = std::max(WavesAfter, WavesBefore);
2020 if (WavesAfter < WavesBefore && WavesAfter <
DAG.MinOccupancy &&
2021 WavesAfter >=
MFI.getMinAllowedOccupancy()) {
2022 LLVM_DEBUG(
dbgs() <<
"Function is memory bound, allow occupancy drop up to "
2023 <<
MFI.getMinAllowedOccupancy() <<
" waves\n");
2024 NewOccupancy = WavesAfter;
2027 if (NewOccupancy <
DAG.MinOccupancy) {
2028 DAG.MinOccupancy = NewOccupancy;
2029 MFI.limitOccupancy(
DAG.MinOccupancy);
2031 <<
DAG.MinOccupancy <<
".\n");
2035 unsigned MaxVGPRs =
ST.getMaxNumVGPRs(
MF);
2038 unsigned MaxArchVGPRs = std::min(MaxVGPRs,
ST.getAddressableNumArchVGPRs());
2039 unsigned MaxSGPRs =
ST.getMaxNumSGPRs(
MF);
2063 unsigned ReadyCycle = CurrCycle;
2064 for (
auto &
D : SU.
Preds) {
2065 if (
D.isAssignedRegDep()) {
2068 unsigned DefReady = ReadyCycles[
DAG.getSUnit(
DefMI)->NodeNum];
2069 ReadyCycle = std::max(ReadyCycle, DefReady +
Latency);
2072 ReadyCycles[SU.
NodeNum] = ReadyCycle;
2079 std::pair<MachineInstr *, unsigned>
B)
const {
2080 return A.second <
B.second;
2086 if (ReadyCycles.empty())
2088 unsigned BBNum = ReadyCycles.begin()->first->getParent()->getNumber();
2089 dbgs() <<
"\n################## Schedule time ReadyCycles for MBB : " << BBNum
2090 <<
" ##################\n# Cycle #\t\t\tInstruction "
2094 for (
auto &
I : ReadyCycles) {
2095 if (
I.second > IPrev + 1)
2096 dbgs() <<
"****************************** BUBBLE OF " <<
I.second - IPrev
2097 <<
" CYCLES DETECTED ******************************\n\n";
2098 dbgs() <<
"[ " <<
I.second <<
" ] : " << *
I.first <<
"\n";
2111 unsigned SumBubbles = 0;
2113 unsigned CurrCycle = 0;
2114 for (
auto &SU : InputSchedule) {
2115 unsigned ReadyCycle =
2117 SumBubbles += ReadyCycle - CurrCycle;
2119 ReadyCyclesSorted.insert(std::make_pair(SU.getInstr(), ReadyCycle));
2121 CurrCycle = ++ReadyCycle;
2144 unsigned SumBubbles = 0;
2146 unsigned CurrCycle = 0;
2147 for (
auto &
MI :
DAG) {
2151 unsigned ReadyCycle =
2153 SumBubbles += ReadyCycle - CurrCycle;
2155 ReadyCyclesSorted.insert(std::make_pair(SU->
getInstr(), ReadyCycle));
2157 CurrCycle = ++ReadyCycle;
2174 if (WavesAfter <
DAG.MinOccupancy)
2178 if (
DAG.MFI.isDynamicVGPREnabled()) {
2181 DAG.MFI.getDynamicVGPRBlockSize());
2184 if (BlocksAfter > BlocksBefore)
2221 <<
"\n\t *** In shouldRevertScheduling ***\n"
2222 <<
" *********** BEFORE UnclusteredHighRPStage ***********\n");
2226 <<
"\n *********** AFTER UnclusteredHighRPStage ***********\n");
2228 unsigned OldMetric = MBefore.
getMetric();
2229 unsigned NewMetric = MAfter.
getMetric();
2230 unsigned WavesBefore = std::min(
2231 S.getTargetOccupancy(),
2238 LLVM_DEBUG(
dbgs() <<
"\tMetric before " << MBefore <<
"\tMetric after "
2239 << MAfter <<
"Profit: " << Profit <<
"\n");
2270 unsigned WavesAfter) {
2277 LLVM_DEBUG(
dbgs() <<
"New pressure will result in more spilling.\n");
2289 "instruction number mismatch");
2290 if (MIOrder.
empty())
2303 if (MII != RegionEnd) {
2305 bool NonDebugReordered =
2306 !
MI->isDebugInstr() &&
2312 if (NonDebugReordered)
2313 DAG.LIS->handleMove(*
MI,
true);
2320 if (!
MI->isDebugInstr()) {
2322 SlotIndex PrevIdx =
DAG.LIS->getSlotIndexes()->getIndexBefore(*
MI);
2323 if (PrevIdx >= MIIdx)
2324 DAG.LIS->handleMove(*
MI,
true);
2328 if (
MI->isDebugInstr()) {
2335 Op.setIsUndef(
false);
2338 if (
DAG.ShouldTrackLaneMasks) {
2363 if (RD->
getOpcode() == AMDGPU::AV_MOV_B32_IMM_PSEUDO ||
2364 RD->
getOpcode() == AMDGPU::AV_MOV_B64_IMM_PSEUDO)
2371bool RewriteMFMAFormStage::hasUseRequiringVGPR(
2373 const SmallPtrSetImpl<MachineInstr *> &RewriteSet) {
2374 for (SlotIndex RDIdx : Src2ReachingDefs) {
2375 const MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIdx);
2377 findReachingUses(RD,
DAG.LIS, ReachingUses);
2378 for (
const MachineOperand *UseMO : ReachingUses) {
2390void RewriteMFMAFormStage::resetRewriteCandsToVGPR(
2391 ArrayRef<std::pair<MachineInstr *, unsigned>> RewriteCands) {
2392 for (
auto [
MI, OriginalOpcode] : RewriteCands) {
2395 DAG.MRI.getRegClass(
MI->getOperand(0).getReg());
2397 DAG.MRI.setRegClass(
MI->getOperand(0).getReg(), VDefRC);
2398 MI->setDesc(
TII->get(OriginalOpcode));
2400 MachineOperand *Src2 =
TII->getNamedOperand(*
MI, AMDGPU::OpName::src2);
2409 DAG.MRI.setRegClass(Src2->
getReg(), VUseRC);
2413bool RewriteMFMAFormStage::isRewriteCandidate(MachineInstr *
MI)
const {
2414 if (!
static_cast<const SIInstrInfo *
>(
DAG.TII)->isMAI(*
MI))
2419 Register DstReg =
MI->getOperand(0).getReg();
2420 for (
const MachineInstr &
UseMI :
DAG.MRI.use_nodbg_instructions(DstReg)) {
2427bool RewriteMFMAFormStage::initHeuristics(
2428 std::vector<std::pair<MachineInstr *, unsigned>> &RewriteCands,
2429 DenseMap<MachineBasicBlock *, std::set<Register>> &CopyForUse,
2430 SmallPtrSetImpl<MachineInstr *> &CopyForDef) {
2435 SmallPtrSet<MachineInstr *, 16> RewriteSet;
2436 DenseSet<Register> CandSrc2Regs;
2437 for (MachineBasicBlock &
MBB :
MF) {
2438 for (MachineInstr &
MI :
MBB) {
2439 if (!isRewriteCandidate(&
MI))
2442 MachineOperand *Src2 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src2);
2443 if (Src2 && Src2->
isReg())
2449 for (MachineBasicBlock &
MBB :
MF) {
2450 for (MachineInstr &
MI :
MBB) {
2451 if (!isRewriteCandidate(&
MI))
2455 assert(ReplacementOp != -1);
2457 RewriteCands.push_back({&
MI,
MI.getOpcode()});
2458 MI.setDesc(
TII->get(ReplacementOp));
2460 MachineOperand *Src2 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src2);
2461 if (Src2->
isReg()) {
2463 findReachingDefs(*Src2,
DAG.LIS, Src2ReachingDefs);
2467 bool Src2NeedsVGPR = hasUseRequiringVGPR(Src2ReachingDefs, RewriteSet);
2468 Src2NeedsVGPRCache[&
MI] = Src2NeedsVGPR;
2470 for (SlotIndex RDIdx : Src2ReachingDefs) {
2471 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIdx);
2472 if (!Src2NeedsVGPR &&
2479 MachineOperand &Dst =
MI.getOperand(0);
2482 findReachingUses(&
MI,
DAG.LIS, DstReachingUses);
2484 for (MachineOperand *RUOp : DstReachingUses) {
2485 MachineInstr *UserMI = RUOp->getParent();
2487 if (
TII->isMAI(*UserMI) && RewriteSet.
contains(UserMI))
2493 CopyForUse[UserMI->
getParent()].insert(RUOp->getReg());
2495 if (
TII->isMAI(*UserMI))
2499 findReachingDefs(*RUOp,
DAG.LIS, DstUsesReachingDefs);
2501 for (SlotIndex RDIndex : DstUsesReachingDefs) {
2502 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIndex);
2503 if (
TII->isMAI(*RD))
2517 DAG.MRI.setRegClass(Dst.getReg(), ADefRC);
2518 if (Src2->
isReg()) {
2524 DAG.MRI.setRegClass(Src2->
getReg(), AUseRC);
2533int64_t RewriteMFMAFormStage::getRewriteCost(
2534 ArrayRef<std::pair<MachineInstr *, unsigned>> RewriteCands,
2535 const DenseMap<MachineBasicBlock *, std::set<Register>> &CopyForUse,
2536 const SmallPtrSetImpl<MachineInstr *> &CopyForDef) {
2537 MachineBlockFrequencyInfo *MBFI =
DAG.MBFI;
2539 int64_t BestSpillCost = 0;
2543 std::pair<unsigned, unsigned> MaxVectorRegs =
2544 ST.getMaxNumVectorRegs(
MF.getFunction());
2545 unsigned ArchVGPRThreshold = MaxVectorRegs.first;
2546 unsigned AGPRThreshold = MaxVectorRegs.second;
2547 unsigned CombinedThreshold =
ST.getMaxNumVGPRs(
MF);
2550 if (!RegionsWithExcessArchVGPR[Region])
2555 MF, ArchVGPRThreshold, AGPRThreshold, CombinedThreshold);
2563 MF, ArchVGPRThreshold, AGPRThreshold, CombinedThreshold);
2569 bool RelativeFreqIsDenom = EntryFreq > BlockFreq;
2570 uint64_t RelativeFreq = EntryFreq && BlockFreq
2571 ? (RelativeFreqIsDenom ? EntryFreq / BlockFreq
2572 : BlockFreq / EntryFreq)
2577 int64_t SpillCost = ((int)SpillCostAfter - (int)SpillCostBefore) * 2;
2580 if (RelativeFreqIsDenom)
2581 SpillCost /= (int64_t)RelativeFreq;
2583 SpillCost *= (int64_t)RelativeFreq;
2586 if (SpillCost > 0) {
2587 resetRewriteCandsToVGPR(RewriteCands);
2591 if (SpillCost < BestSpillCost)
2592 BestSpillCost = SpillCost;
2597 Cost = BestSpillCost;
2600 unsigned CopyCost = 0;
2604 for (MachineInstr *
DefMI : CopyForDef) {
2616 for (
auto &[UseBlock, UseRegs] : CopyForUse) {
2630 resetRewriteCandsToVGPR(RewriteCands);
2632 return Cost + CopyCost;
2635bool RewriteMFMAFormStage::rewrite(
2636 ArrayRef<std::pair<MachineInstr *, unsigned>> RewriteCands) {
2637 DenseMap<MachineInstr *, unsigned> FirstMIToRegion;
2638 DenseMap<MachineInstr *, unsigned> LastMIToRegion;
2646 if (
Entry.second !=
Entry.first->getParent()->end())
2689 DenseSet<Register> RewriteRegs;
2692 DenseMap<Register, Register> RedefMap;
2694 DenseMap<Register, DenseSet<MachineOperand *>>
ReplaceMap;
2696 DenseMap<Register, SmallPtrSet<MachineInstr *, 8>> ReachingDefCopyMap;
2699 DenseMap<unsigned, DenseMap<Register, SmallPtrSet<MachineOperand *, 8>>>
2704 SmallPtrSet<MachineInstr *, 16> RewriteCandsSet;
2705 DenseSet<Register> RewriteSrc2Regs;
2706 for (
auto &[
MI, OriginalOpcode] : RewriteCands) {
2708 MachineOperand *Src2 =
TII->getNamedOperand(*
MI, AMDGPU::OpName::src2);
2709 if (Src2 && Src2->
isReg())
2713 for (
auto &[
MI, OriginalOpcode] : RewriteCands) {
2715 if (ReplacementOp == -1)
2717 MI->setDesc(
TII->get(ReplacementOp));
2720 MachineOperand *Src2 =
TII->getNamedOperand(*
MI, AMDGPU::OpName::src2);
2721 if (Src2->
isReg()) {
2728 findReachingDefs(*Src2,
DAG.LIS, Src2ReachingDefs);
2729 SmallSetVector<MachineInstr *, 8> Src2DefsReplace;
2733 bool Src2NeedsVGPR = Src2NeedsVGPRCache.lookup(
MI);
2735 for (SlotIndex RDIndex : Src2ReachingDefs) {
2736 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIndex);
2737 if (!Src2NeedsVGPR &&
2741 Src2DefsReplace.
insert(RD);
2744 if (!Src2DefsReplace.
empty()) {
2745 auto RI = RedefMap.
find(Src2Reg);
2746 if (RI != RedefMap.
end()) {
2747 MappedReg = RI->second;
2752 SRI->getEquivalentVGPRClass(Src2RC);
2755 MappedReg =
DAG.MRI.createVirtualRegister(VGPRRC);
2756 RedefMap[Src2Reg] = MappedReg;
2761 for (MachineInstr *RD : Src2DefsReplace) {
2763 if (ReachingDefCopyMap[Src2Reg].insert(RD).second) {
2764 MachineInstrBuilder VGPRCopy =
2767 .
addDef(MappedReg, {}, 0)
2768 .addUse(Src2Reg, {}, 0);
2769 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
2774 unsigned UpdateRegion = LastMIToRegion[RD];
2775 DAG.Regions[UpdateRegion].second = VGPRCopy;
2776 LastMIToRegion.
erase(RD);
2783 RewriteRegs.
insert(Src2Reg);
2793 MachineOperand *Dst = &
MI->getOperand(0);
2802 SmallVector<MachineInstr *, 8> DstUseDefsReplace;
2804 findReachingUses(
MI,
DAG.LIS, DstReachingUses);
2806 for (MachineOperand *RUOp : DstReachingUses) {
2807 MachineInstr *UserMI = RUOp->
getParent();
2809 if (
TII->isMAI(*UserMI) && RewriteCandsSet.
contains(UserMI))
2813 if (
find(DstReachingUseCopies, RUOp) == DstReachingUseCopies.
end())
2817 if (
TII->isMAI(*UserMI))
2821 findReachingDefs(*RUOp,
DAG.LIS, DstUsesReachingDefs);
2823 for (SlotIndex RDIndex : DstUsesReachingDefs) {
2824 MachineInstr *RD =
DAG.LIS->getInstructionFromIndex(RDIndex);
2825 if (
TII->isMAI(*RD))
2830 if (
find(DstUseDefsReplace, RD) == DstUseDefsReplace.
end())
2835 if (!DstUseDefsReplace.
empty()) {
2836 auto RI = RedefMap.
find(DstReg);
2837 if (RI != RedefMap.
end()) {
2838 MappedReg = RI->second;
2845 MappedReg =
DAG.MRI.createVirtualRegister(VGPRRC);
2846 RedefMap[DstReg] = MappedReg;
2851 for (MachineInstr *RD : DstUseDefsReplace) {
2853 if (ReachingDefCopyMap[DstReg].insert(RD).second) {
2854 MachineInstrBuilder VGPRCopy =
2857 .
addDef(MappedReg, {}, 0)
2858 .addUse(DstReg, {}, 0);
2859 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
2863 auto LMI = LastMIToRegion.
find(RD);
2864 if (LMI != LastMIToRegion.
end()) {
2865 unsigned UpdateRegion = LMI->second;
2866 DAG.Regions[UpdateRegion].second = VGPRCopy;
2867 LastMIToRegion.
erase(RD);
2873 DenseSet<MachineOperand *> &DstRegSet =
ReplaceMap[DstReg];
2876 MachineInstr *EarliestSameBlockUse =
nullptr;
2877 for (MachineOperand *RU : DstReachingUseCopies) {
2878 MachineBasicBlock *RUBlock = RU->getParent()->getParent();
2881 if (RUBlock !=
MI->getParent()) {
2887 if (!SameBlockCopyReg.
isValid()) {
2890 SameBlockCopyReg =
DAG.MRI.createVirtualRegister(VGPRRC);
2894 MachineInstr *UseInst = RU->getParent();
2895 if (!EarliestSameBlockUse ||
2897 DAG.LIS->getInstructionIndex(*UseInst),
2898 DAG.LIS->getInstructionIndex(*EarliestSameBlockUse)))
2899 EarliestSameBlockUse = UseInst;
2900 RU->setReg(SameBlockCopyReg);
2904 if (SameBlockCopyReg.
isValid()) {
2905 MachineInstrBuilder VGPRCopy =
2908 TII->get(TargetOpcode::COPY), SameBlockCopyReg)
2910 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
2915 RewriteRegs.
insert(DstReg);
2925 std::pair<unsigned, DenseMap<Register, SmallPtrSet<MachineOperand *, 8>>>;
2926 for (RUBType RUBlockEntry : ReachingUseTracker) {
2927 using RUDType = std::pair<Register, SmallPtrSet<MachineOperand *, 8>>;
2928 for (RUDType RUDst : RUBlockEntry.second) {
2929 MachineOperand *OpBegin = *RUDst.second.begin();
2930 SlotIndex InstPt =
DAG.LIS->getInstructionIndex(*OpBegin->
getParent());
2933 for (MachineOperand *User : RUDst.second) {
2934 SlotIndex NewInstPt =
DAG.LIS->getInstructionIndex(*
User->getParent());
2941 Register NewUseReg =
DAG.MRI.createVirtualRegister(VGPRRC);
2942 MachineInstr *UseInst =
DAG.LIS->getInstructionFromIndex(InstPt);
2944 MachineInstrBuilder VGPRCopy =
2947 .
addDef(NewUseReg, {}, 0)
2948 .addUse(RUDst.first, {}, 0);
2949 DAG.LIS->InsertMachineInstrInMaps(*VGPRCopy);
2953 auto FI = FirstMIToRegion.
find(UseInst);
2954 if (FI != FirstMIToRegion.
end()) {
2955 unsigned UpdateRegion = FI->second;
2956 DAG.Regions[UpdateRegion].first = VGPRCopy;
2957 FirstMIToRegion.
erase(UseInst);
2961 for (MachineOperand *User : RUDst.second) {
2962 User->setReg(NewUseReg);
2973 for (std::pair<Register, Register> NewDef : RedefMap) {
2978 for (MachineOperand *ReplaceOp :
ReplaceMap[OldReg])
2979 ReplaceOp->setReg(NewReg);
2983 for (
Register RewriteReg : RewriteRegs) {
2984 Register RegToRewrite = RewriteReg;
2987 auto RI = RedefMap.find(RewriteReg);
2988 if (RI != RedefMap.end())
2989 RegToRewrite = RI->second;
2994 DAG.MRI.setRegClass(RegToRewrite, AGPRRC);
2998 DAG.LIS->reanalyze(
DAG.MF);
3000 RegionPressureMap LiveInUpdater(&
DAG,
false);
3001 LiveInUpdater.buildLiveRegMap();
3004 DAG.LiveIns[Region] = LiveInUpdater.getLiveRegsForRegionIdx(Region);
3011unsigned PreRARematStage::getStageTargetOccupancy()
const {
3012 return TargetOcc ? *TargetOcc :
MFI.getMinWavesPerEU();
3015bool PreRARematStage::setObjective() {
3019 unsigned MaxSGPRs =
ST.getMaxNumSGPRs(
F);
3020 unsigned MaxVGPRs =
ST.getMaxNumVGPRs(
F);
3021 bool HasVectorRegisterExcess =
false;
3022 for (
unsigned I = 0,
E =
DAG.Regions.size();
I !=
E; ++
I) {
3023 const GCNRegPressure &
RP =
DAG.Pressure[
I];
3024 GCNRPTarget &
Target = RPTargets.emplace_back(MaxSGPRs, MaxVGPRs,
MF, RP);
3026 TargetRegions.set(
I);
3027 HasVectorRegisterExcess |=
Target.hasVectorRegisterExcess();
3030 if (HasVectorRegisterExcess ||
DAG.MinOccupancy >=
MFI.getMaxWavesPerEU()) {
3033 TargetOcc = std::nullopt;
3037 TargetOcc =
DAG.MinOccupancy + 1;
3038 const unsigned VGPRBlockSize =
MFI.getDynamicVGPRBlockSize();
3039 MaxSGPRs =
ST.getMaxNumSGPRs(*TargetOcc,
false);
3040 MaxVGPRs =
ST.getMaxNumVGPRs(*TargetOcc, VGPRBlockSize);
3041 for (
auto [
I, Target] :
enumerate(RPTargets)) {
3042 Target.setTarget(MaxSGPRs, MaxVGPRs);
3044 TargetRegions.set(
I);
3048 return TargetRegions.any();
3051bool PreRARematStage::ScoredRemat::maybeBeneficial(
3053 for (
unsigned I : TargetRegions.set_bits()) {
3054 if (Live[
I] && RPTargets[
I].isSaveBeneficial(RPSave))
3067 const unsigned NumRegions =
DAG.Regions.size();
3071 for (
unsigned I = 0;
I < NumRegions; ++
I) {
3075 if (BlockFreq && BlockFreq <
MinFreq)
3084 if (
MinFreq >= ScaleFactor * ScaleFactor) {
3085 for (uint64_t &Freq :
Regions)
3086 Freq /= ScaleFactor;
3092void PreRARematStage::ScoredRemat::init(RegisterIdx RegIdx,
3096 this->RegIdx = RegIdx;
3097 const unsigned NumRegions =
DAG.Regions.size();
3098 LiveIn.resize(NumRegions);
3099 LiveOut.resize(NumRegions);
3100 Live.resize(NumRegions);
3101 UnpredictableRPSave.resize(NumRegions);
3105 assert(Reg.Uses.size() == 1 &&
"expected users in single region");
3106 const unsigned UseRegion = Reg.Uses.begin()->first;
3109 for (
unsigned I = 0, E = NumRegions;
I != E; ++
I) {
3110 if (
DAG.LiveIns[
I].contains(DefReg))
3112 if (
DAG.RegionLiveOuts.getLiveRegsForRegionIdx(
I).contains(DefReg))
3117 if (!LiveIn[
I] || !LiveOut[
I] ||
I == UseRegion)
3118 UnpredictableRPSave.set(
I);
3127 int64_t DefOrMin = std::max(Freq.
Regions[Reg.DefRegion], Freq.
MinFreq);
3128 int64_t UseOrMax = Freq.
Regions[UseRegion];
3131 FreqDiff = DefOrMin - UseOrMax;
3134void PreRARematStage::ScoredRemat::update(
const BitVector &TargetRegions,
3136 const FreqInfo &FreqInfo,
3140 for (
unsigned I : TargetRegions.
set_bits()) {
3149 if (!NumRegsBenefit)
3153 RegionImpact += (UnpredictableRPSave[
I] ? 1 : 2) * NumRegsBenefit;
3156 uint64_t Freq = FreqInfo.
Regions[
I];
3157 if (UnpredictableRPSave[
I]) {
3162 MaxFreq = std::max(MaxFreq, Freq);
3167void PreRARematStage::ScoredRemat::rematerialize(
3168 Rematerializer &Remater)
const {
3169 const Rematerializer::Reg &
Reg = Remater.getReg(RegIdx);
3170 Rematerializer::DependencyReuseInfo DRI;
3171 for (RegisterIdx DepRegIdx :
Reg.Dependencies)
3172 DRI.
reuse(DepRegIdx);
3173 unsigned UseRegion =
Reg.Uses.begin()->first;
3174 Remater.rematerializeToRegion(RegIdx, UseRegion, DRI);
3177void PreRARematStage::updateRPTargets(
const BitVector &Regions,
3178 const GCNRegPressure &RPSave) {
3180 RPTargets[
I].saveRP(RPSave);
3181 if (TargetRegions[
I] && RPTargets[
I].satisfied()) {
3183 TargetRegions.reset(
I);
3188bool PreRARematStage::updateAndVerifyRPTargets(
const BitVector &Regions) {
3189 bool TooOptimistic =
false;
3191 GCNRPTarget &
Target = RPTargets[
I];
3197 if (!TargetRegions[
I] && !
Target.satisfied()) {
3199 TooOptimistic =
true;
3200 TargetRegions.set(
I);
3203 return TooOptimistic;
3206void PreRARematStage::removeFromLiveMaps(
Register Reg,
const BitVector &LiveIn,
3207 const BitVector &LiveOut) {
3209 LiveOut.
size() ==
DAG.Regions.size() &&
"region num mismatch");
3213 DAG.RegionLiveOuts.getLiveRegsForRegionIdx(
I).erase(
Reg);
3216void PreRARematStage::addToLiveMaps(
Register Reg, LaneBitmask Mask,
3217 const BitVector &LiveIn,
3218 const BitVector &LiveOut) {
3220 LiveOut.
size() ==
DAG.Regions.size() &&
"region num mismatch");
3221 std::pair<Register, LaneBitmask> LiveReg(
Reg, Mask);
3223 DAG.LiveIns[
I].insert(LiveReg);
3225 DAG.RegionLiveOuts.getLiveRegsForRegionIdx(
I).insert(LiveReg);
3237 if (
DAG.MinOccupancy >= *TargetOcc)
3241 for (
const auto &[
RegionIdx, OrigMIOrder, MaxPressure] : RegionReverts) {
3251 if (AchievedOcc >= *TargetOcc) {
3252 DAG.setTargetOccupancy(AchievedOcc);
3257 DAG.setTargetOccupancy(*TargetOcc - 1);
3262 assert(Rollback &&
"rollbacker should be defined");
3263 Rollback->Listener.rollback(Remater);
3264 for (
const auto &[RegIdx, LiveIn, LiveOut] : Rollback->LiveMapUpdates) {
3265 const Rematerializer::Reg &
Reg = Remater.getReg(RegIdx);
3266 addToLiveMaps(
Reg.getDefReg(),
Reg.Mask, LiveIn, LiveOut);
3269#ifdef EXPENSIVE_CHECKS
3274 for (
unsigned I : RescheduleRegions.set_bits())
3275 DAG.Pressure[
I] =
DAG.getRealRegPressure(
I);
3280void GCNScheduleDAGMILive::setTargetOccupancy(
unsigned TargetOccupancy) {
3281 MinOccupancy = TargetOccupancy;
3282 if (
MFI.getOccupancy() < TargetOccupancy)
3283 MFI.increaseOccupancy(
MF, MinOccupancy);
3285 MFI.limitOccupancy(MinOccupancy);
3302 if (HasIGLPInstrs) {
3303 SavedMutations.clear();
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static SUnit * pickOnlyChoice(SchedBoundary &Zone)
This file implements the BitVector class.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file defines the GCNRegPressure class, which tracks registry pressure by bookkeeping number of S...
static cl::opt< bool > GCNTrackers("amdgpu-use-amdgpu-trackers", cl::Hidden, cl::desc("Use the AMDGPU specific RPTrackers during scheduling"), cl::init(false))
static cl::opt< bool > DisableClusteredLowOccupancy("amdgpu-disable-clustered-low-occupancy-reschedule", cl::Hidden, cl::desc("Disable clustered low occupancy " "rescheduling for ILP scheduling stage."), cl::init(false))
#define REMAT_PREFIX
Allows to easily filter for this stage's debug output.
static cl::opt< unsigned, false, VGPRThresholdParser > VGPRThresholdPercentOpt("amdgpu-vgpr-threshold-percent", cl::Hidden, cl::desc("Percent of VGPR limits that we should use as RP threshold " "during scheduling. We have two limits relevant to scheduling: " "Critical (avoid decreasing occupancy), Excess (avoid spilling). " "This flag scales both limits back by an equal percent: (0 = use " " default calculation, 1-100 = use percentage), default: 0"), cl::init(0))
static MachineInstr * getLastMIForRegion(MachineBasicBlock::iterator RegionBegin, MachineBasicBlock::iterator RegionEnd)
static bool shouldCheckPending(SchedBoundary &Zone, const TargetSchedModel *SchedModel)
static cl::opt< bool > RelaxedOcc("amdgpu-schedule-relaxed-occupancy", cl::Hidden, cl::desc("Relax occupancy targets for kernels which are memory " "bound (amdgpu-membound-threshold), or " "Wave Limited (amdgpu-limit-wave-threshold)."), cl::init(false))
static cl::opt< bool > DisableUnclusterHighRP("amdgpu-disable-unclustered-high-rp-reschedule", cl::Hidden, cl::desc("Disable unclustered high register pressure " "reduction scheduling stage."), cl::init(false))
static void printScheduleModel(std::set< std::pair< MachineInstr *, unsigned >, EarlierIssuingCycle > &ReadyCycles)
static bool isReachingDefAGPRForm(MachineInstr *RD, const SmallPtrSetImpl< MachineInstr * > &RewriteSet, const DenseSet< Register > &CandSrc2Regs, const SIInstrInfo &TII)
Returns true if reaching def RD will be in AGPR form after the rewrite and so needs no bridge copy: a...
static cl::opt< bool > PrintMaxRPRegUsageAfterScheduler("amdgpu-print-max-reg-pressure-regusage-after-scheduler", cl::Hidden, cl::desc("Print a list of live registers along with their def/uses at the " "point of maximum register pressure after scheduling."), cl::init(false))
static bool hasIGLPInstrs(ScheduleDAGInstrs *DAG)
static cl::opt< bool > DisableRewriteMFMAFormSchedStage("amdgpu-disable-rewrite-mfma-form-sched-stage", cl::Hidden, cl::desc("Disable rewrite mfma rewrite scheduling stage"), cl::init(true))
static bool canUsePressureDiffs(const SUnit &SU)
Checks whether SU can use the cached DAG pressure diffs to compute the current register pressure.
static cl::opt< unsigned > PendingQueueLimit("amdgpu-scheduler-pending-queue-limit", cl::Hidden, cl::desc("Max (Available+Pending) size to inspect pending queue (0 disables)"), cl::init(256))
static cl::opt< bool > PrintMaxRPRegUsageBeforeScheduler("amdgpu-print-max-reg-pressure-regusage-before-scheduler", cl::Hidden, cl::desc("Print a list of live registers along with their def/uses at the " "point of maximum register pressure before scheduling."), cl::init(false))
static cl::opt< unsigned > ScheduleMetricBias("amdgpu-schedule-metric-bias", cl::Hidden, cl::desc("Sets the bias which adds weight to occupancy vs latency. Set it to " "100 to chase the occupancy only."), cl::init(10))
static Register UseReg(const MachineOperand &MO)
const HexagonInstrInfo * TII
static constexpr std::pair< StringLiteral, StringLiteral > ReplaceMap[]
iv Induction Variable Users
A common definition of LaneBitmask for use in TableGen and CodeGen.
static llvm::Error parse(GsymDataExtractor &Data, uint64_t BaseAddr, LineEntryCallback const &Callback)
Promote Memory to Register
MIR-level target-independent rematerialization helpers.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & front() const
Get the first element.
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
iterator_range< const_set_bits_iterator > set_bits() const
size_type size() const
Returns the number of bits in this bitvector.
uint64_t getFrequency() const
Returns the frequency as a fixpoint number scaled by the entry frequency.
bool initGCNSchedStage() override
bool shouldRevertScheduling(unsigned WavesAfter) override
bool initGCNRegion() override
iterator find(const_arg_type_t< KeyT > Val)
bool erase(const KeyT &Val)
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Implements a dense probed hash-table based set.
bool reset(const MachineInstr &MI, MachineBasicBlock::const_iterator End, const LiveRegSet *LiveRegs=nullptr)
Reset tracker to the point before the MI filling LiveRegs upon this point using LIS.
GCNRegPressure bumpDownwardPressure(const MachineInstr *MI, const SIRegisterInfo *TRI) const
Mostly copy/paste from CodeGen/RegisterPressure.cpp Calculate the impact MI will have on CurPressure ...
GCNMaxILPSchedStrategy(const MachineSchedContext *C)
bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const override
Apply a set of heuristics to a new candidate.
bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const override
GCNMaxMemoryClauseSchedStrategy tries best to clause memory instructions as much as possible.
GCNMaxMemoryClauseSchedStrategy(const MachineSchedContext *C)
GCNMaxOccupancySchedStrategy(const MachineSchedContext *C, bool IsLegacyScheduler=false)
void finalizeSchedule() override
Allow targets to perform final scheduling actions at the level of the whole MachineFunction.
void schedule() override
Orders nodes according to selected style.
GCNPostScheduleDAGMILive(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S, bool RemoveKillFlags)
Models a register pressure target, allowing to evaluate and track register savings against that targe...
unsigned getNumRegsBenefit(const GCNRegPressure &SaveRP) const
Returns the benefit towards achieving the RP target that saving SaveRP represents,...
GCNRegPressure getPressure() const
virtual bool initGCNRegion()
GCNRegPressure PressureBefore
bool isRegionWithExcessRP() const
void modifyRegionSchedule(unsigned RegionIdx, ArrayRef< MachineInstr * > MIOrder)
Sets the schedule of region RegionIdx to MIOrder.
bool mayCauseSpilling(unsigned WavesAfter)
ScheduleMetrics getScheduleMetrics(const std::vector< SUnit > &InputSchedule)
GCNScheduleDAGMILive & DAG
const GCNSchedStageID StageID
std::vector< MachineInstr * > Unsched
GCNRegPressure PressureAfter
virtual void finalizeGCNRegion()
SIMachineFunctionInfo & MFI
unsigned computeSUnitReadyCycle(const SUnit &SU, unsigned CurrCycle, DenseMap< unsigned, unsigned > &ReadyCycles, const TargetSchedModel &SM)
virtual void finalizeGCNSchedStage()
virtual bool initGCNSchedStage()
virtual bool shouldRevertScheduling(unsigned WavesAfter)
std::vector< std::unique_ptr< ScheduleDAGMutation > > SavedMutations
GCNSchedStage(GCNSchedStageID StageID, GCNScheduleDAGMILive &DAG)
MachineBasicBlock * CurrentMBB
This is a minimal scheduler strategy.
GCNDownwardRPTracker DownwardTracker
bool useGCNTrackers() const
void getRegisterPressures(bool AtTop, const RegPressureTracker &RPTracker, SUnit *SU, std::vector< unsigned > &Pressure, std::vector< unsigned > &MaxPressure, GCNDownwardRPTracker &DownwardTracker, GCNUpwardRPTracker &UpwardTracker, ScheduleDAGMI *DAG, const SIRegisterInfo *SRI)
GCNSchedStrategy(const MachineSchedContext *C)
SmallVector< GCNSchedStageID, 4 > SchedStages
unsigned SGPRCriticalLimit
std::vector< unsigned > MaxPressure
bool hasNextStage() const
SUnit * pickNodeBidirectional(bool &IsTopNode, bool &PickedPending)
GCNSchedStageID getCurrentStage()
bool tryPendingCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const
Evaluates instructions in the pending queue using a subset of scheduling heuristics.
SmallVectorImpl< GCNSchedStageID >::iterator CurrentStage
unsigned VGPRCriticalLimit
void schedNode(SUnit *SU, bool IsTopNode) override
Notify MachineSchedStrategy that ScheduleDAGMI has scheduled an instruction and updated scheduled/rem...
std::optional< bool > GCNTrackersOverride
GCNDownwardRPTracker * getDownwardTracker()
unsigned AGPRCriticalLimit
std::vector< unsigned > Pressure
void initialize(ScheduleDAGMI *DAG) override
Initialize the strategy after building the DAG for a new region.
GCNUpwardRPTracker UpwardTracker
void printCandidateDecision(const SchedCandidate &Current, const SchedCandidate &Preferred)
void pickNodeFromQueue(SchedBoundary &Zone, const CandPolicy &ZonePolicy, const RegPressureTracker &RPTracker, SchedCandidate &Cand, bool &IsPending, bool IsBottomUp)
unsigned getStructuralStallCycles(SchedBoundary &Zone, SUnit *SU) const
Estimate how many cycles SU must wait due to structural hazards at the current boundary cycle.
void initCandidate(SchedCandidate &Cand, SUnit *SU, bool AtTop, const RegPressureTracker &RPTracker, const SIRegisterInfo *SRI, unsigned SGPRPressure, unsigned VGPRPressure, unsigned AGPRPressure, bool IsBottomUp)
SUnit * pickNode(bool &IsTopNode) override
Pick the next node to schedule, or return NULL.
GCNUpwardRPTracker * getUpwardTracker()
GCNSchedStageID getNextStage() const
void finalizeSchedule() override
Allow targets to perform final scheduling actions at the level of the whole MachineFunction.
void schedule() override
Orders nodes according to selected style.
GCNScheduleDAGMILive(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S)
void recede(const MachineInstr &MI)
Move to the state of RP just before the MI .
void reset(const MachineInstr &MI)
Resets tracker to the point just after MI (in program order), which can be a debug instruction.
void compute(FunctionT &F)
Compute the cycle info for a function.
void traceCandidate(const SchedCandidate &Cand)
LLVM_ABI void setPolicy(CandPolicy &Policy, bool IsPostRA, SchedBoundary &CurrZone, SchedBoundary *OtherZone)
Set the CandPolicy given a scheduling zone given the current resources and latencies inside and outsi...
MachineSchedPolicy RegionPolicy
const TargetSchedModel * SchedModel
const MachineSchedContext * Context
const TargetRegisterInfo * TRI
SchedCandidate BotCand
Candidate last picked from Bot boundary.
SchedCandidate TopCand
Candidate last picked from Top boundary.
virtual bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const
Apply a set of heuristics to a new candidate.
void initialize(ScheduleDAGMI *dag) override
Initialize the strategy after building the DAG for a new region.
void schedNode(SUnit *SU, bool IsTopNode) override
Update the scheduler's state after scheduling a node.
GenericScheduler(const MachineSchedContext *C)
bool shouldRevertScheduling(unsigned WavesAfter) override
LiveInterval - This class represents the liveness of a register, or stack slot.
bool hasSubRanges() const
Returns true if subregister liveness information is available.
iterator_range< subrange_iterator > subranges()
SlotIndex getInstructionIndex(const MachineInstr &Instr) const
Returns the base index of the given instruction.
SlotIndex getMBBEndIdx(const MachineBasicBlock *mbb) const
Return the last index in the given basic block.
LiveInterval & getInterval(Register Reg)
LLVM_ABI void dump() const
MachineBasicBlock * getMBBFromIndex(SlotIndex index) const
VNInfo * getVNInfoAt(SlotIndex Idx) const
getVNInfoAt - Return the VNInfo that is live at Idx, or NULL.
uint8_t getCopyCost() const
getCopyCost - Return the cost of copying a value between two registers in this class.
int getNumber() const
MachineBasicBlocks are uniquely numbered at the function level, unless they're not in a MachineFuncti...
succ_iterator succ_begin()
unsigned succ_size() const
iterator_range< pred_iterator > predecessors()
MachineInstrBundleIterator< MachineInstr > iterator
MachineBlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate machine basic b...
LLVM_ABI BlockFrequency getBlockFreq(const MachineBasicBlock *MBB) const
getblockFreq - Return block frequency.
LLVM_ABI BlockFrequency getEntryFreq() const
Divide a block's BlockFrequency::getFrequency() value by this value to obtain the entry block - relat...
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
unsigned getNumOperands() const
Retuns the total number of operands.
bool mayLoad(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read memory.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
const MachineOperand & getOperand(unsigned i) const
MachineOperand class - Representation of each machine instruction operand.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
Register getReg() const
getReg - Returns the register number.
bool shouldRevertScheduling(unsigned WavesAfter) override
bool shouldRevertScheduling(unsigned WavesAfter) override
bool shouldRevertScheduling(unsigned WavesAfter) override
void finalizeGCNRegion() override
bool initGCNRegion() override
bool initGCNSchedStage() override
Capture a change in pressure for a single pressure set.
Simple wrapper around std::function<void(raw_ostream&)>.
Helpers for implementing custom MachineSchedStrategy classes.
Track the current register pressure at some position in the instruction stream, and remember the high...
LLVM_ABI void advance()
Advance across the current instruction.
LLVM_ABI void getDownwardPressure(const MachineInstr *MI, std::vector< unsigned > &PressureResult, std::vector< unsigned > &MaxPressureResult)
Get the pressure of each PSet after traversing this instruction top-down.
const std::vector< unsigned > & getRegSetPressureAtPos() const
Get the register set pressure at the current position, which may be less than the pressure across the...
LLVM_ABI void getUpwardPressure(const MachineInstr *MI, std::vector< unsigned > &PressureResult, std::vector< unsigned > &MaxPressureResult)
Get the pressure of each PSet after traversing this instruction bottom-up.
List of registers defined and used by a machine instruction.
LLVM_ABI void detectDeadDefs(const MachineInstr &MI, const LiveIntervals &LIS, const MachineRegisterInfo &MRI)
Use liveness information to find dead defs at MI's dead slot not marked with a dead flag and move the...
LLVM_ABI void adjustLaneLiveness(const LiveIntervals &LIS, const MachineRegisterInfo &MRI, SlotIndex Pos)
Use liveness information to find out which uses/defs are partially undefined/dead at Pos and adjust t...
LLVM_ABI void collect(const MachineInstr &MI, const TargetRegisterInfo &TRI, const MachineRegisterInfo &MRI, bool TrackLaneMasks, bool IgnoreDead)
Analyze the given instruction MI and fill in the Uses, Defs and DeadDefs list based on the MachineOpe...
Wrapper class representing virtual and physical registers.
constexpr bool isValid() const
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
MIR-level target-independent rematerializer.
bool isIGLPMutationOnly(unsigned Opcode) const
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
unsigned getOccupancy() const
unsigned getDynamicVGPRBlockSize() const
unsigned getMinAllowedOccupancy() const
Scheduling unit. This is a node in the scheduling DAG.
bool isInstr() const
Returns true if this SUnit refers to a machine instruction as opposed to an SDNode.
unsigned TopReadyCycle
Cycle relative to start when node is ready.
unsigned NodeNum
Entry # of node in the node vector.
unsigned short Latency
Node latency.
bool isScheduled
True once scheduled.
unsigned ParentClusterIdx
The parent cluster id.
unsigned BotReadyCycle
Cycle relative to end when node is ready.
bool hasReservedResource
Uses a reserved resource.
bool isBottomReady() const
SmallVector< SDep, 4 > Preds
All sunit predecessors.
MachineInstr * getInstr() const
Returns the representative MachineInstr for this SUnit.
Each Scheduling boundary is associated with ready queues.
LLVM_ABI void releasePending()
Release pending ready nodes in to the available queue.
LLVM_ABI unsigned getLatencyStallCycles(SUnit *SU)
Get the difference between the given SUnit's ready time and the current cycle.
LLVM_ABI SUnit * pickOnlyChoice()
Call this before applying any other heuristics to the Available queue.
LLVM_ABI void bumpCycle(unsigned NextCycle)
Move the boundary of scheduled code by one cycle.
unsigned getCurrMOps() const
Micro-ops issued in the current cycle.
unsigned getCurrCycle() const
Number of cycles to issue the instructions scheduled in this zone.
std::unique_ptr< ScheduleHazardRecognizer > HazardRec
LLVM_ABI bool checkHazard(SUnit *SU)
Does this SU have a hazard within the current instruction group.
LLVM_ABI std::pair< unsigned, unsigned > getNextResourceCycle(const MCSchedClassDesc *SC, unsigned PIdx, unsigned ReleaseAtCycle, unsigned AcquireAtCycle)
Compute the next cycle at which the given processor resource can be scheduled.
A ScheduleDAG for scheduling lists of MachineInstr.
bool ScheduleSingleMIRegions
True if regions with a single MI should be scheduled.
MachineBasicBlock::iterator RegionEnd
The end of the range to be scheduled.
virtual void finalizeSchedule()
Allow targets to perform final scheduling actions at the level of the whole MachineFunction.
virtual void exitRegion()
Called when the scheduler has finished scheduling the current region.
const MachineLoopInfo * MLI
bool RemoveKillFlags
True if the DAG builder should remove kill flags (in preparation for rescheduling).
MachineBasicBlock::iterator RegionBegin
The beginning of the range to be scheduled.
void schedule() override
Implement ScheduleDAGInstrs interface for scheduling a sequence of reorderable instructions.
ScheduleDAGMILive(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S)
RegPressureTracker RPTracker
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
void addMutation(std::unique_ptr< ScheduleDAGMutation > Mutation)
Add a postprocessing step to the DAG builder.
void schedule() override
Implement ScheduleDAGInstrs interface for scheduling a sequence of reorderable instructions.
ScheduleDAGMI(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S, bool RemoveKillFlags)
std::vector< std::unique_ptr< ScheduleDAGMutation > > Mutations
Ordered list of DAG postprocessing steps.
MachineRegisterInfo & MRI
Virtual/real register map.
const TargetInstrInfo * TII
Target instruction information.
MachineFunction & MF
Machine function.
static const unsigned ScaleFactor
unsigned getMetric() const
bool empty() const
Determine if the SetVector is empty or not.
bool insert(const value_type &X)
Insert a new element into the SetVector.
SlotIndex - An opaque wrapper around machine indexes.
static bool isSameInstr(SlotIndex A, SlotIndex B)
isSameInstr - Return true if A and B refer to the same instruction.
static bool isEarlierInstr(SlotIndex A, SlotIndex B)
isEarlierInstr - Return true if A refers to an instruction earlier than B.
SlotIndex getPrevSlot() const
Returns the previous slot in the index list.
SlotIndex getMBBStartIdx(const MachineBasicBlock *mbb) const
Returns the first index in the given basic block.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
SmallSet - This maintains a set of unique values, optimizing for the case when the set is small (less...
bool contains(const T &V) const
Check if the SmallSet contains the given element.
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Provide an instruction scheduling machine model to CodeGen passes.
LLVM_ABI bool hasInstrSchedModel() const
Return true if this machine model includes an instruction-level scheduling model.
unsigned getMicroOpBufferSize() const
Number of micro-ops that may be buffered for OOO execution.
bool initGCNSchedStage() override
bool initGCNRegion() override
void finalizeGCNSchedStage() override
bool shouldRevertScheduling(unsigned WavesAfter) override
VNInfo - Value Number Information.
SlotIndex def
The index of the defining instruction.
bool isPHIDef() const
Returns true if this value is defined by a PHI instruction (or was, PHI instructions may have been el...
std::pair< iterator, bool > insert(const ValueT &V)
bool contains(const_arg_type_t< ValueT > V) const
Check if the set contains the given element.
self_iterator getIterator()
This class implements an extremely fast bulk output stream that can only output to a stream.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize)
unsigned getAllocatedNumVGPRBlocks(const MCSubtargetInfo &STI, unsigned NumVGPRs, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
LLVM_READONLY int32_t getAGPRFormOp(uint32_t Opcode)
This namespace contains all of the command line option processing machinery.
initializer< Ty > init(const Ty &Val)
@ User
could "use" a pointer
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI int biasPhysReg(const SUnit *SU, bool isTop, bool BiasPRegsExtra=false)
Minimize physical register live ranges.
auto find(R &&Range, const T &Val)
Provide wrappers to std::find which take ranges instead of having to pass begin/end explicitly.
bool isEqual(const GCNRPTracker::LiveRegSet &S1, const GCNRPTracker::LiveRegSet &S2)
Printable print(const GCNRegPressure &RP, const GCNSubtarget *ST=nullptr, unsigned DynamicVGPRBlockSize=0)
LLVM_ABI unsigned getWeakLeft(const SUnit *SU, bool isTop)
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
GCNRegPressure getRegPressure(const MachineRegisterInfo &MRI, Range &&LiveRegs)
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
std::unique_ptr< ScheduleDAGMutation > createIGroupLPDAGMutation(AMDGPU::SchedulingPhase Phase)
Phase specifes whether or not this is a reentry into the IGroupLPDAGMutation.
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
std::pair< MachineBasicBlock::iterator, MachineBasicBlock::iterator > RegionBoundaries
A region's boundaries i.e.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
IterT skipDebugInstructionsForward(IterT It, IterT End, bool SkipPseudoOp=true)
Increment It until it points to a non-debug instruction or to End and return the resulting iterator.
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI bool tryPressure(const PressureChange &TryP, const PressureChange &CandP, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason, const TargetRegisterInfo *TRI, const MachineFunction &MF)
@ UnclusteredHighRPReschedule
@ MemoryClauseInitialSchedule
@ ClusteredLowOccupancyReschedule
auto reverse(ContainerTy &&C)
void sort(IteratorTy Start, IteratorTy End)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
LLVM_ABI cl::opt< bool > VerifyScheduling
LLVM_ABI bool tryLatency(GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, SchedBoundary &Zone)
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
IterT skipDebugInstructionsBackward(IterT It, IterT Begin, bool SkipPseudoOp=true)
Decrement It until it points to a non-debug instruction or to Begin and return the resulting iterator...
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
bool isTheSameCluster(unsigned A, unsigned B)
Return whether the input cluster ID's are the same and valid.
DWARFExpression::Operation Op
LLVM_ABI bool tryGreater(int TryVal, int CandVal, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason)
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
ArrayRef(const T &OneElt) -> ArrayRef< T >
OutputIt move(R &&Range, OutputIt Out)
Provide wrappers to std::move which take ranges instead of having to pass begin/end explicitly.
DenseMap< MachineInstr *, GCNRPTracker::LiveRegSet > getLiveRegMap(Range &&R, bool After, LiveIntervals &LIS)
creates a map MachineInstr -> LiveRegSet R - range of iterators on instructions After - upon entry or...
GCNRPTracker::LiveRegSet getLiveRegsBefore(const MachineInstr &MI, const LiveIntervals &LIS)
LLVM_ABI bool tryLess(int TryVal, int CandVal, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason)
Return true if this heuristic determines order.
LLVM_ABI void dumpMaxRegPressure(MachineFunction &MF, GCNRegPressure::RegKind Kind, LiveIntervals &LIS, const MachineLoopInfo *MLI)
LLVM_ABI Printable printMBBReference(const MachineBasicBlock &MBB)
Prints a machine basic block reference.
MCRegisterClass TargetRegisterClass
Implement std::hash so that hash_code can be used in STL containers.
bool operator()(std::pair< MachineInstr *, unsigned > A, std::pair< MachineInstr *, unsigned > B) const
unsigned getArchVGPRNum() const
unsigned getAGPRNum() const
unsigned getSGPRNum() const
Policy for scheduling the next instruction in the candidate's zone.
Store the state used by GenericScheduler heuristics, required for the lifetime of one invocation of p...
void setBest(SchedCandidate &Best)
void reset(const CandPolicy &NewPolicy)
LLVM_ABI void initResourceDelta(const ScheduleDAGMI *DAG, const TargetSchedModel *SchedModel)
SchedResourceDelta ResDelta
Status of an instruction's critical resource consumption.
unsigned DemandedResources
constexpr bool any() const
static constexpr LaneBitmask getNone()
Summarize the scheduling resources required for an instruction of a particular scheduling class.
Identify one of the processor resource kinds consumed by a particular scheduling class for the specif...
MachineSchedContext provides enough context from the MachineScheduler pass for the target to instanti...
Execution frequency information required by scoring heuristics.
SmallVector< uint64_t > Regions
Per-region execution frequencies. 0 when unknown.
uint64_t MinFreq
Minimum and maximum observed frequencies.
FreqInfo(MachineFunction &MF, const GCNScheduleDAGMILive &DAG)
PressureChange CriticalMax
PressureChange CurrentMax
DependencyReuseInfo & reuse(RegisterIdx DepIdx)
A rematerializable register, potentially defined by multiple instructions.
LLVM_ABI std::pair< MachineInstr *, MachineInstr * > getRegionUseBounds(unsigned UseRegion, const LiveIntervals &LIS) const
Returns the first and last user of the register in region UseRegion.
SmallVector< MachineInstr *, 1 > Defs
All instructions that define the register, in program order.
SmallDenseMap< unsigned, RegionUsers, 2 > Uses
Uses of the register, mapped by region.
MachineInstr * getLastDef() const
SmallVector< RegisterIdx, 2 > Dependencies
This register's rematerializable dependencies, one per unique rematerializable register operand over ...