LLVM 24.0.0git
GCNSchedStrategy.h
Go to the documentation of this file.
1//===-- GCNSchedStrategy.h - GCN Scheduler Strategy -*- C++ -*-------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10//
11//===----------------------------------------------------------------------===//
12
13#ifndef LLVM_LIB_TARGET_AMDGPU_GCNSCHEDSTRATEGY_H
14#define LLVM_LIB_TARGET_AMDGPU_GCNSCHEDSTRATEGY_H
15
16#include "GCNRegPressure.h"
17#include "llvm/ADT/DenseMap.h"
23
24namespace llvm {
25
27class SIRegisterInfo;
28class GCNSubtarget;
29class GCNSchedStage;
30
40
41#ifndef NDEBUG
42raw_ostream &operator<<(raw_ostream &OS, const GCNSchedStageID &StageID);
43#endif
44
45/// This is a minimal scheduler strategy. The main difference between this
46/// and the GenericScheduler is that GCNSchedStrategy uses different
47/// heuristics to determine excess/critical pressure sets.
49protected:
50 SUnit *pickNodeBidirectional(bool &IsTopNode, bool &PickedPending);
51
52 void pickNodeFromQueue(SchedBoundary &Zone, const CandPolicy &ZonePolicy,
53 const RegPressureTracker &RPTracker,
54 SchedCandidate &Cand, bool &IsPending,
55 bool IsBottomUp);
56
57 void initCandidate(SchedCandidate &Cand, SUnit *SU, bool AtTop,
58 const RegPressureTracker &RPTracker,
59 const SIRegisterInfo *SRI, unsigned SGPRPressure,
60 unsigned VGPRPressure, unsigned AGPRPressure,
61 bool IsBottomUp);
62
63 /// Estimate how many cycles \p SU must wait due to structural hazards at the
64 /// current boundary cycle. Returns zero when no stall is required.
65 unsigned getStructuralStallCycles(SchedBoundary &Zone, SUnit *SU) const;
66
67 /// Evaluates instructions in the pending queue using a subset of scheduling
68 /// heuristics.
69 ///
70 /// Instructions that cannot be issued due to hardware constraints are placed
71 /// in the pending queue rather than the available queue, making them normally
72 /// invisible to scheduling heuristics. However, in certain scenarios (such as
73 /// avoiding register spilling), it may be beneficial to consider scheduling
74 /// these not-yet-ready instructions.
76 SchedBoundary *Zone) const;
77
78 void printCandidateDecision(const SchedCandidate &Current,
79 const SchedCandidate &Preferred);
80
81 void getRegisterPressures(bool AtTop, const RegPressureTracker &RPTracker,
82 SUnit *SU, std::vector<unsigned> &Pressure,
83 std::vector<unsigned> &MaxPressure,
86 ScheduleDAGMI *DAG, const SIRegisterInfo *SRI);
87
88 std::vector<unsigned> Pressure;
89
90 std::vector<unsigned> MaxPressure;
91
93
95
97
99
101
102 // Scheduling stages for this strategy.
104
105 // Pointer to the current SchedStageID.
107
108 // GCN RP Tracker for top-down scheduling
110
111 // GCN RP Tracker for botttom-up scheduling
113
114 bool UseGCNTrackers = false;
115
116 std::optional<bool> GCNTrackersOverride;
117
118public:
119 // schedule() have seen register pressure over the critical limits and had to
120 // track register pressure for actual scheduling heuristics.
122
123 // Schedule known to have excess register pressure. Be more conservative in
124 // increasing ILP and preserving VGPRs.
125 bool KnownExcessRP = false;
126
127 // An error margin is necessary because of poor performance of the generic RP
128 // tracker and can be adjusted up for tuning heuristics to try and more
129 // aggressively reduce register pressure.
130 unsigned ErrorMargin = 3;
131
132 // Bias for SGPR limits under a high register pressure.
133 const unsigned HighRPSGPRBias = 7;
134
135 // Bias for VGPR limits under a high register pressure.
136 const unsigned HighRPVGPRBias = 7;
137
139
141
143
144 unsigned SGPRLimitBias = 0;
145
146 unsigned VGPRLimitBias = 0;
147
149
150 SUnit *pickNode(bool &IsTopNode) override;
151
152 void schedNode(SUnit *SU, bool IsTopNode) override;
153
154 void initialize(ScheduleDAGMI *DAG) override;
155
156 unsigned getTargetOccupancy() { return TargetOccupancy; }
157
158 void setTargetOccupancy(unsigned Occ) { TargetOccupancy = Occ; }
159
161
162 // Advances stage. Returns true if there are remaining stages.
163 bool advanceStage();
164
165 bool hasNextStage() const;
166
167 bool useGCNTrackers() const {
168 return GCNTrackersOverride.value_or(UseGCNTrackers);
169 }
170
172
174
176};
177
178/// The goal of this scheduling strategy is to maximize kernel occupancy (i.e.
179/// maximum number of waves per simd).
181public:
183 bool IsLegacyScheduler = false);
184};
185
186/// The goal of this scheduling strategy is to maximize ILP for a single wave
187/// (i.e. latency hiding).
189protected:
190 bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand,
191 SchedBoundary *Zone) const override;
192
193public:
195};
196
197/// The goal of this scheduling strategy is to maximize memory clause for a
198/// single wave.
200protected:
201 bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand,
202 SchedBoundary *Zone) const override;
203
204public:
206};
207
209 unsigned ScheduleLength;
210 unsigned BubbleCycles;
211
212public:
213 ScheduleMetrics() = default;
214 ScheduleMetrics(unsigned L, unsigned BC)
215 : ScheduleLength(L), BubbleCycles(BC) {}
216 unsigned getLength() const { return ScheduleLength; }
217 unsigned getBubbles() const { return BubbleCycles; }
218 unsigned getMetric() const {
219 unsigned Metric = (BubbleCycles * ScaleFactor) / ScheduleLength;
220 // Metric is zero if the amount of bubbles is less than 1% which is too
221 // small. So, return 1.
222 return Metric ? Metric : 1;
223 }
224 static const unsigned ScaleFactor;
225};
226
228 dbgs() << "\n Schedule Metric (scaled by " << ScheduleMetrics::ScaleFactor
229 << " ) is: " << Sm.getMetric() << " [ " << Sm.getBubbles() << "/"
230 << Sm.getLength() << " ]\n";
231 return OS;
232}
233
234class GCNScheduleDAGMILive;
237 // The live in/out pressure as indexed by the first or last MI in the region
238 // before scheduling.
240 // The mapping of RegionIDx to key instruction
241 DenseMap<unsigned, MachineInstr *> IdxToInstruction;
242 // Whether we are calculating LiveOuts or LiveIns
243 bool IsLiveOut;
244
245public:
246 RegionPressureMap() = default;
248 : DAG(GCNDAG), IsLiveOut(LiveOut) {}
249 // Build the Instr->LiveReg and RegionIdx->Instr maps
250 void buildLiveRegMap();
251
252 // Retrieve the LiveReg for a given RegionIdx
254 assert(IdxToInstruction.contains(RegionIdx));
255 MachineInstr *Key = IdxToInstruction[RegionIdx];
256 return RegionLiveRegMap[Key];
257 }
258};
259
260/// A region's boundaries i.e. a pair of instruction bundle iterators. The lower
261/// boundary is inclusive, the upper boundary is exclusive.
263 std::pair<MachineBasicBlock::iterator, MachineBasicBlock::iterator>;
264
266 friend class GCNSchedStage;
271 friend class PreRARematStage;
273 friend class RegionPressureMap;
274
275 const GCNSubtarget &ST;
276
278
279 // Occupancy target at the beginning of function scheduling cycle.
280 unsigned StartingOccupancy;
281
282 // Minimal real occupancy recorder for the function.
283 unsigned MinOccupancy;
284
285 // Vector of regions recorder for later rescheduling
287
288 // Record regions with high register pressure.
289 BitVector RegionsWithHighRP;
290
291 // Record regions with excess register pressure over the physical register
292 // limit. Register pressure in these regions usually will result in spilling.
293 BitVector RegionsWithExcessRP;
294
295 // Regions that have IGLP instructions (SCHED_GROUP_BARRIER or IGLP_OPT).
296 BitVector RegionsWithIGLPInstrs;
297
298 // Region live-in cache.
300
301 // Region pressure cache.
303
304 // Temporary basic block live-in cache.
306
307 // The map of the initial first region instruction to region live in registers
309
310 // Calculate the map of the initial first region instruction to region live in
311 // registers
313
314 // Calculate the map of the initial last region instruction to region live out
315 // registers
317 getRegionLiveOutMap() const;
318
319 // The live out registers per region. These are internally stored as a map of
320 // the initial last region instruction to region live out registers, but can
321 // be retreived with the regionIdx by calls to getLiveRegsForRegionIdx.
322 RegionPressureMap RegionLiveOuts;
323
324 // Return current region pressure.
325 GCNRegPressure getRealRegPressure(unsigned RegionIdx) const;
326
327 // Compute and cache live-ins and pressure for all regions in block.
328 void computeBlockPressure(unsigned RegionIdx, const MachineBasicBlock *MBB);
329
330 /// Makes the scheduler try to achieve an occupancy of \p TargetOccupancy.
331 void setTargetOccupancy(unsigned TargetOccupancy);
332
333 void runSchedStages();
334
335 std::unique_ptr<GCNSchedStage> createSchedStage(GCNSchedStageID SchedStageID);
336
337public:
339 std::unique_ptr<MachineSchedStrategy> S);
340
341 void schedule() override;
342
343 void finalizeSchedule() override;
344};
345
346// GCNSchedStrategy applies multiple scheduling stages to a function.
348protected:
350
352
354
356
358
360
361 // The current block being scheduled.
363
364 // Current region index.
365 unsigned RegionIdx = 0;
366
367 // Record the original order of instructions before scheduling.
368 std::vector<MachineInstr *> Unsched;
369
370 // RP before scheduling the current region.
372
373 // RP after scheduling the current region.
375
376 std::vector<std::unique_ptr<ScheduleDAGMutation>> SavedMutations;
377
379
380public:
381 // Initialize state for a scheduling stage. Returns false if the current stage
382 // should be skipped.
383 virtual bool initGCNSchedStage();
384
385 // Finalize state after finishing a scheduling pass on the function.
386 virtual void finalizeGCNSchedStage();
387
388 // Setup for scheduling a region. Returns false if the current region should
389 // be skipped.
390 virtual bool initGCNRegion();
391
392 // Finalize state after scheduling a region.
393 virtual void finalizeGCNRegion();
394
395 // Track whether a new region is also a new MBB.
396 void setupNewBlock();
397
398 // Check result of scheduling.
399 void checkScheduling();
400
401 // computes the given schedule virtual execution time in clocks
402 ScheduleMetrics getScheduleMetrics(const std::vector<SUnit> &InputSchedule);
404 unsigned computeSUnitReadyCycle(const SUnit &SU, unsigned CurrCycle,
405 DenseMap<unsigned, unsigned> &ReadyCycles,
406 const TargetSchedModel &SM);
407
408 // Returns true if scheduling should be reverted.
409 virtual bool shouldRevertScheduling(unsigned WavesAfter);
410
411 // Returns true if current region has known excess pressure.
412 bool isRegionWithExcessRP() const {
413 return DAG.RegionsWithExcessRP[RegionIdx];
414 }
415
416 // The region number this stage is currently working on
417 unsigned getRegionIdx() { return RegionIdx; }
418
419 // Returns true if the new schedule may result in more spilling.
420 bool mayCauseSpilling(unsigned WavesAfter);
421
422 /// Sets the schedule of region \p RegionIdx to \p MIOrder. The MIs in \p
423 /// MIOrder must be exactly the same as the ones currently existing inside the
424 /// region, only in a different order that honors def-use chains.
425 void modifyRegionSchedule(unsigned RegionIdx,
427
429
430 virtual ~GCNSchedStage() = default;
431};
432
440
442private:
443 // Record regions with excess archvgpr register pressure over the physical
444 // register limit. Register pressure in these regions usually will result in
445 // spilling.
446 BitVector RegionsWithExcessArchVGPR;
447
448 const SIInstrInfo *TII;
449 const SIRegisterInfo *SRI;
450
451 /// Per-candidate cache of the src2 "needs VGPR" decision, computed once
452 /// and reused on-demand.
453 DenseMap<const MachineInstr *, bool> Src2NeedsVGPRCache;
454
455 /// Do a speculative rewrite and collect copy locations. The speculative
456 /// rewrite allows us to calculate the RP of the code after the rewrite, and
457 /// the copy locations allow us to calculate the total cost of copies required
458 /// for the rewrite. Stores the rewritten instructions in \p RewriteCands ,
459 /// the copy locations for uses (of the MFMA result) in \p CopyForUse and the
460 /// copy locations for defs (of the MFMA operands) in \p CopyForDef
461 bool
462 initHeuristics(std::vector<std::pair<MachineInstr *, unsigned>> &RewriteCands,
463 DenseMap<MachineBasicBlock *, std::set<Register>> &CopyForUse,
465
466 /// Calculate the rewrite cost and undo the state change (e.g. rewriting) done
467 /// in initHeuristics. Uses \p CopyForUse and \p CopyForDef to calculate copy
468 /// costs, and \p RewriteCands to undo rewriting.
469 int64_t getRewriteCost(
470 ArrayRef<std::pair<MachineInstr *, unsigned>> RewriteCands,
471 const DenseMap<MachineBasicBlock *, std::set<Register>> &CopyForUse,
472 const SmallPtrSetImpl<MachineInstr *> &CopyForDef);
473
474 /// Do the final rewrite on \p RewriteCands and insert any needed copies.
475 bool rewrite(ArrayRef<std::pair<MachineInstr *, unsigned>> RewriteCands);
476
477 /// \returns true if this MI is a rewrite candidate.
478 bool isRewriteCandidate(MachineInstr *MI) const;
479
480 /// Resets all candidates in \p RewriteCands back to VGPR form.
481 void resetRewriteCandsToVGPR(
482 ArrayRef<std::pair<MachineInstr *, unsigned>> RewriteCands);
483
484 /// Finds all the reaching defs of \p UseMO and stores the SlotIndexes into \p
485 /// DefIdxs
486 void findReachingDefs(MachineOperand &UseMO, LiveIntervals *LIS,
488
489 /// Finds all the reaching uses of \p DefMI and stores the use operands in \p
490 /// ReachingUses
491 void findReachingUses(const MachineInstr *DefMI, LiveIntervals *LIS,
493
494 /// Returns true if the src2 register with reaching defs \p Src2ReachingDefs
495 /// has a use other than a group MFMA (in \p RewriteSet) or a copy, which
496 /// would keep it in VGPR form rather than let it be reclassified to AGPR.
497 bool hasUseRequiringVGPR(ArrayRef<SlotIndex> Src2ReachingDefs,
498 const SmallPtrSetImpl<MachineInstr *> &RewriteSet);
499
500public:
501 bool initGCNSchedStage() override;
502
505};
506
508private:
509 // Save the initial occupancy before starting this stage.
510 unsigned InitialOccupancy;
511 // Save the temporary target occupancy before starting this stage.
512 unsigned TempTargetOccupancy;
513 // Track whether any region was scheduled by this stage.
514 bool IsAnyRegionScheduled;
515
516public:
517 bool initGCNSchedStage() override;
518
519 void finalizeGCNSchedStage() override;
520
521 bool initGCNRegion() override;
522
523 bool shouldRevertScheduling(unsigned WavesAfter) override;
524
527};
528
529// Retry function scheduling if we found resulting occupancy and it is
530// lower than used for other scheduling passes. This will give more freedom
531// to schedule low register pressure blocks.
533public:
534 bool initGCNSchedStage() override;
535
536 bool initGCNRegion() override;
537
538 bool shouldRevertScheduling(unsigned WavesAfter) override;
539
542};
543
544/// Attempts to reduce function spilling or, if there is no spilling, to
545/// increase function occupancy by one with respect to register usage by sinking
546/// rematerializable instructions to their use. When the stage estimates that
547/// reducing spilling or increasing occupancy is possible, it tries to
548/// rematerialize as few registers as possible to reduce potential negative
549/// effects on function latency.
550///
551/// The stage only supports rematerializing registers that meet all of the
552/// following constraints.
553/// 1. The register is virtual and has a single defining instruction.
554/// 2. The single defining instruction is either deemed rematerializable by the
555/// target-independent logic, or if not, has no non-constant and
556/// non-ignorable physical register use.
557/// 3 The register has no virtual register use whose live range would be
558/// extended by the rematerialization.
559/// 4. The register has a single non-debug user in a different region from its
560/// defining region.
561/// 5. The register is not used by or using another register that is going to be
562/// rematerialized.
564private:
565 using RegisterIdx = Rematerializer::RegisterIdx;
566
567 /// A scored rematerialization candidate. Higher scores indicate more
568 /// beneficial rematerializations. A null score indicate the rematerialization
569 /// is not helpful to reduce RP in target regions.
570 struct ScoredRemat {
571 /// The register index handle in the rematerializer.
572 RegisterIdx RegIdx;
573 /// Regions in which the register is live-in/live-out/live anywhere.
574 BitVector LiveIn, LiveOut, Live;
575 /// Subset of \ref Live regions in which the rematerialization is not
576 /// guaranteed to reduce RP (i.e., regions in which the register is not
577 /// live-through and unused).
578 BitVector UnpredictableRPSave;
579 /// Expected register pressure decrease induced by rematerializing this
580 /// candidate.
581 GCNRegPressure RPSave;
582
583 /// Execution frequency information required by scoring heuristics.
584 /// Frequencies are scaled down if they are high to avoid overflow/underflow
585 /// when combining them.
586 struct FreqInfo {
587 /// Per-region execution frequencies. 0 when unknown.
589 /// Minimum and maximum observed frequencies.
591
593
594 private:
595 static const uint64_t ScaleFactor = 1024;
596 };
597
598 /// Initializes the candidate with state-independent characteristics for
599 /// rematerializable register with index handle \p RegIdx. This doesn't
600 /// update the actual score (call \ref update for this).
601 void init(RegisterIdx RegIdx, const FreqInfo &Freq,
602 const Rematerializer &Remater, GCNScheduleDAGMILive &DAG);
603
604 /// Rematerializes the candidate using the \p Remater.
605 void rematerialize(Rematerializer &Remater) const;
606
607 /// Determines whether this rematerialization may be beneficial in at least
608 /// one target region.
609 bool maybeBeneficial(const BitVector &TargetRegions,
610 ArrayRef<GCNRPTarget> RPTargets) const;
611
612 /// Rematerializes the candidate and returns the new MI. This removes the
613 /// rematerialized register from live-in/out lists in the \p DAG and updates
614 /// \p RPTargets in all affected regions. Regions in which RP savings are
615 /// not guaranteed are set in \p RecomputeRP.
616 MachineInstr *rematerialize(BitVector &RecomputeRP,
619
620 /// Updates the rematerialization's score w.r.t. the current \p RPTargets.
621 /// \p RegionFreq indicates the frequency of each region.
622 void update(const BitVector &TargetRegions, ArrayRef<GCNRPTarget> RPTargets,
623 const FreqInfo &Freq, bool ReduceSpill);
624
625 /// Returns whether the current score is null, indicating the
626 /// rematerialization is useless.
627 bool hasNullScore() const { return !RegionImpact; }
628
629 /// Compare score components of non-null scores pair-wise. Scores shouldn't
630 /// be null (as defined by \ref hasNullScore).
631 bool operator<(const ScoredRemat &O) const {
632 assert(!hasNullScore() && "this has null score");
633 assert(!O.hasNullScore() && "other has null score");
634 if (MaxFreq != O.MaxFreq)
635 return MaxFreq < O.MaxFreq;
636 if (FreqDiff != O.FreqDiff)
637 return FreqDiff < O.FreqDiff;
638 if (RegionImpact != O.RegionImpact)
639 return RegionImpact < O.RegionImpact;
640 // Break ties using register index handles. If the two registers are
641 // connected in some dependency DAG of rematerializable registers, this
642 // will tend to give a higher score to the register further from the
643 // dependency DAG's root. If the two registers are disconnected, this will
644 // give a higher score to the register with lower virtual register index.
645 // In general, within a region, this should prefer registers defined
646 // earlier that have longer live ranges in their defining region (since
647 // the registers we consider are always live-out in their defining
648 // region).
649 return RegIdx > O.RegIdx;
650 }
651
652#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
653 Printable print() const;
654#endif
655
656 private:
657 // The three members below are the scoring components, top to bottom from
658 // most important to least important when comparing candidates.
659
660 /// Frequency of impacted target region with highest known frequency. This
661 /// only matters when the stage is trying to reduce spilling, so it is
662 /// always 0 when it is not.
663 uint64_t MaxFreq;
664 /// Frequency difference between defining and using regions. Negative values
665 /// indicate we are rematerializing to higher frequency regions; positive
666 /// values indicate the contrary.
667 int64_t FreqDiff;
668 /// Expected number of target regions impacted by the rematerialization,
669 /// scaled by the size of the register being rematerialized.
670 unsigned RegionImpact;
671 };
672
673 /// Register pressure targets for all regions.
674 SmallVector<GCNRPTarget> RPTargets;
675 /// Regions which are above the stage's RP target.
676 BitVector TargetRegions;
677 /// The target occupancy the set is trying to achieve. Empty when the
678 /// objective is spilling reduction.
679 std::optional<unsigned> TargetOcc;
680 /// Achieved occupancy *only* through rematerializations (pre-rescheduling).
681 unsigned AchievedOcc;
682 /// After successful stage initialization, indicates which regions should be
683 /// rescheduled.
684 BitVector RescheduleRegions;
685
686 /// Underlying utilities to identify and perform rematerializations.
687 Rematerializer Remater;
688
689 struct RollbackSupport {
691 /// The register index handle in the rematerializer.
692 RegisterIdx RegIdx;
693 /// Regions in which the original register was live-in or live-out.
695
699 };
700
701 /// Rollback listener.
702 Rollbacker Listener;
703 /// Registers removed from live-maps along with bitvectors indicationg the
704 /// regions in which they were live-ins and live-outs.
705 SmallVector<LiveMapUpdate> LiveMapUpdates;
706
707 /// Attaches the rollback listener to the rematerializer.
708 RollbackSupport(Rematerializer &Remater) { Remater.addListener(&Listener); }
709 };
710
711 /// Rollback support. Maintained through a unique pointer because it is
712 /// optional and needs to persist between stage initialization and
713 /// finalization.
714 std::unique_ptr<RollbackSupport> Rollback;
715
716 /// State of a region pre-re-scheduling but post-rematerializations that we
717 /// must keep to be able to revert re-scheduling effects.
718 struct RegionSchedRevert {
719 /// Region number;
720 unsigned RegionIdx;
721 /// Original instruction order (both debug and non-debug MIs).
722 std::vector<MachineInstr *> OrigMIOrder;
723 /// Maximum pressure recorded in the region.
724 GCNRegPressure MaxPressure;
725
726 RegionSchedRevert(unsigned RegionIdx, ArrayRef<MachineInstr *> OrigMIOrder,
727 const GCNRegPressure &MaxPressure)
728 : RegionIdx(RegionIdx), OrigMIOrder(OrigMIOrder),
729 MaxPressure(MaxPressure) {}
730 };
731 /// After re-scheduling, contains pre-re-scheduling data for all re-scheduled
732 /// regions.
733 SmallVector<RegionSchedRevert> RegionReverts;
734 /// Whether we should revert all re-scheduled regions.
735 bool RevertAllRegions = false;
736
737 /// Returns the occupancy the stage is trying to achieve.
738 unsigned getStageTargetOccupancy() const;
739
740 /// Determines the stage's objective (increasing occupancy or reducing
741 /// spilling, set in \ref TargetOcc). Defines \ref RPTargets in all regions to
742 /// achieve that objective and mark those that don't achieve it in \ref
743 /// TargetRegions. Returns whether there is any target region.
744 bool setObjective();
745
746 /// In all regions set in \p Regions, saves pressure \p RPSave and clear it as
747 /// a target if its RP target has been reached.
748 void updateRPTargets(const BitVector &Regions, const GCNRegPressure &RPSave);
749
750 /// Fully recomputes RP from the DAG in \p Regions. Among those regions, sets
751 /// again all \ref TargetRegions that were optimistically marked as satisfied
752 /// but are actually not, and returns whether there were any such regions.
753 bool updateAndVerifyRPTargets(const BitVector &Regions);
754
755 /// Removes register \p Reg from the live-ins of regions set in \p LiveIn and
756 /// the live-outs of regions set in \p LiveOut.
757 void removeFromLiveMaps(Register Reg, const BitVector &LiveIn,
758 const BitVector &LiveOut);
759
760 /// Adds register \p Reg with mask \p Mask to the live-ins of regions set in
761 /// \p LiveIn and the live-outs of regions set in \p LiveOut.
762 void addToLiveMaps(Register Reg, LaneBitmask Mask, const BitVector &LiveIn,
763 const BitVector &LiveOut);
764
765 /// If remat alone did not increase occupancy to the target one, rollbacks all
766 /// rematerializations and resets live-ins/RP in all regions impacted by the
767 /// stage to their pre-stage values.
768 void finalizeGCNSchedStage() override;
769
770public:
771 bool initGCNSchedStage() override;
772
773 bool initGCNRegion() override;
774
775 void finalizeGCNRegion() override;
776
777 bool shouldRevertScheduling(unsigned WavesAfter) override;
778
780 : GCNSchedStage(StageID, DAG), TargetRegions(DAG.Regions.size()),
781 RescheduleRegions(DAG.Regions.size()),
782 Remater(MF, DAG.Regions, *DAG.LIS) {
783 const unsigned NumRegions = DAG.Regions.size();
784 RPTargets.reserve(NumRegions);
785 }
786};
787
795
804
806private:
807 std::vector<std::unique_ptr<ScheduleDAGMutation>> SavedMutations;
808
809 bool HasIGLPInstrs = false;
810
811public:
812 void schedule() override;
813
814 void finalizeSchedule() override;
815
817 std::unique_ptr<MachineSchedStrategy> S,
818 bool RemoveKillFlags);
819};
820
821} // End namespace llvm
822
823#endif // LLVM_LIB_TARGET_AMDGPU_GCNSCHEDSTRATEGY_H
MachineInstrBuilder MachineInstrBuilder & DefMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
MachineBasicBlock & MBB
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
This file defines the DenseMap class.
This file defines the GCNRegPressure class, which tracks registry pressure by bookkeeping number of S...
IRTranslator LLVM IR MI
Register Reg
Promote Memory to Register
Definition Mem2Reg.cpp:110
MIR-level target-independent rematerialization helpers.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
bool shouldRevertScheduling(unsigned WavesAfter) override
ClusteredLowOccStage(GCNSchedStageID StageID, GCNScheduleDAGMILive &DAG)
GCNMaxILPSchedStrategy(const MachineSchedContext *C)
bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const override
Apply a set of heuristics to a new candidate.
bool tryCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const override
GCNMaxMemoryClauseSchedStrategy tries best to clause memory instructions as much as possible.
GCNMaxMemoryClauseSchedStrategy(const MachineSchedContext *C)
GCNMaxOccupancySchedStrategy(const MachineSchedContext *C, bool IsLegacyScheduler=false)
void finalizeSchedule() override
Allow targets to perform final scheduling actions at the level of the whole MachineFunction.
void schedule() override
Orders nodes according to selected style.
GCNPostScheduleDAGMILive(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S, bool RemoveKillFlags)
DenseMap< unsigned, LaneBitmask > LiveRegSet
GCNSchedStrategy & S
GCNRegPressure PressureBefore
bool isRegionWithExcessRP() const
void modifyRegionSchedule(unsigned RegionIdx, ArrayRef< MachineInstr * > MIOrder)
Sets the schedule of region RegionIdx to MIOrder.
bool mayCauseSpilling(unsigned WavesAfter)
ScheduleMetrics getScheduleMetrics(const std::vector< SUnit > &InputSchedule)
GCNScheduleDAGMILive & DAG
const GCNSchedStageID StageID
std::vector< MachineInstr * > Unsched
GCNRegPressure PressureAfter
MachineFunction & MF
virtual void finalizeGCNRegion()
SIMachineFunctionInfo & MFI
unsigned computeSUnitReadyCycle(const SUnit &SU, unsigned CurrCycle, DenseMap< unsigned, unsigned > &ReadyCycles, const TargetSchedModel &SM)
virtual ~GCNSchedStage()=default
virtual void finalizeGCNSchedStage()
virtual bool initGCNSchedStage()
virtual bool shouldRevertScheduling(unsigned WavesAfter)
std::vector< std::unique_ptr< ScheduleDAGMutation > > SavedMutations
GCNSchedStage(GCNSchedStageID StageID, GCNScheduleDAGMILive &DAG)
MachineBasicBlock * CurrentMBB
const GCNSubtarget & ST
This is a minimal scheduler strategy.
const unsigned HighRPSGPRBias
GCNDownwardRPTracker DownwardTracker
void getRegisterPressures(bool AtTop, const RegPressureTracker &RPTracker, SUnit *SU, std::vector< unsigned > &Pressure, std::vector< unsigned > &MaxPressure, GCNDownwardRPTracker &DownwardTracker, GCNUpwardRPTracker &UpwardTracker, ScheduleDAGMI *DAG, const SIRegisterInfo *SRI)
GCNSchedStrategy(const MachineSchedContext *C)
SmallVector< GCNSchedStageID, 4 > SchedStages
std::vector< unsigned > MaxPressure
SUnit * pickNodeBidirectional(bool &IsTopNode, bool &PickedPending)
GCNSchedStageID getCurrentStage()
bool tryPendingCandidate(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone) const
Evaluates instructions in the pending queue using a subset of scheduling heuristics.
SmallVectorImpl< GCNSchedStageID >::iterator CurrentStage
void schedNode(SUnit *SU, bool IsTopNode) override
Notify MachineSchedStrategy that ScheduleDAGMI has scheduled an instruction and updated scheduled/rem...
std::optional< bool > GCNTrackersOverride
GCNDownwardRPTracker * getDownwardTracker()
std::vector< unsigned > Pressure
void initialize(ScheduleDAGMI *DAG) override
Initialize the strategy after building the DAG for a new region.
GCNUpwardRPTracker UpwardTracker
void printCandidateDecision(const SchedCandidate &Current, const SchedCandidate &Preferred)
const unsigned HighRPVGPRBias
void pickNodeFromQueue(SchedBoundary &Zone, const CandPolicy &ZonePolicy, const RegPressureTracker &RPTracker, SchedCandidate &Cand, bool &IsPending, bool IsBottomUp)
unsigned getStructuralStallCycles(SchedBoundary &Zone, SUnit *SU) const
Estimate how many cycles SU must wait due to structural hazards at the current boundary cycle.
void initCandidate(SchedCandidate &Cand, SUnit *SU, bool AtTop, const RegPressureTracker &RPTracker, const SIRegisterInfo *SRI, unsigned SGPRPressure, unsigned VGPRPressure, unsigned AGPRPressure, bool IsBottomUp)
void setTargetOccupancy(unsigned Occ)
SUnit * pickNode(bool &IsTopNode) override
Pick the next node to schedule, or return NULL.
GCNUpwardRPTracker * getUpwardTracker()
GCNSchedStageID getNextStage() const
void finalizeSchedule() override
Allow targets to perform final scheduling actions at the level of the whole MachineFunction.
void schedule() override
Orders nodes according to selected style.
GCNScheduleDAGMILive(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S)
ScheduleDAGMILive * DAG
GenericScheduler(const MachineSchedContext *C)
bool shouldRevertScheduling(unsigned WavesAfter) override
ILPInitialScheduleStage(GCNSchedStageID StageID, GCNScheduleDAGMILive &DAG)
Representation of each machine instruction.
MachineOperand class - Representation of each machine instruction operand.
bool shouldRevertScheduling(unsigned WavesAfter) override
MemoryClauseInitialScheduleStage(GCNSchedStageID StageID, GCNScheduleDAGMILive &DAG)
bool shouldRevertScheduling(unsigned WavesAfter) override
OccInitialScheduleStage(GCNSchedStageID StageID, GCNScheduleDAGMILive &DAG)
PreRARematStage(GCNSchedStageID StageID, GCNScheduleDAGMILive &DAG)
bool shouldRevertScheduling(unsigned WavesAfter) override
void finalizeGCNRegion() override
bool initGCNSchedStage() override
Simple wrapper around std::function<void(raw_ostream&)>.
Definition Printable.h:38
Track the current register pressure at some position in the instruction stream, and remember the high...
GCNRPTracker::LiveRegSet & getLiveRegsForRegionIdx(unsigned RegionIdx)
RegionPressureMap(GCNScheduleDAGMILive *GCNDAG, bool LiveOut)
MIR-level target-independent rematerializer.
unsigned RegisterIdx
Index type for rematerializable registers.
RewriteMFMAFormStage(GCNSchedStageID StageID, GCNScheduleDAGMILive &DAG)
Rematerializer listener with the ability to re-create deleted registers and rollback rematerializatio...
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
Scheduling unit. This is a node in the scheduling DAG.
Each Scheduling boundary is associated with ready queues.
bool RemoveKillFlags
True if the DAG builder should remove kill flags (in preparation for rescheduling).
ScheduleDAGMILive(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S)
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
ScheduleDAGMI(MachineSchedContext *C, std::unique_ptr< MachineSchedStrategy > S, bool RemoveKillFlags)
unsigned getBubbles() const
ScheduleMetrics(unsigned L, unsigned BC)
unsigned getLength() const
static const unsigned ScaleFactor
unsigned getMetric() const
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
typename SuperClass::iterator iterator
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Provide an instruction scheduling machine model to CodeGen passes.
UnclusteredHighRPStage(GCNSchedStageID StageID, GCNScheduleDAGMILive &DAG)
bool shouldRevertScheduling(unsigned WavesAfter) override
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
This is an optimization pass for GlobalISel generic memory operations.
bool operator<(int64_t V1, const APSInt &V2)
Definition APSInt.h:360
Printable print(const GCNRegPressure &RP, const GCNSubtarget *ST=nullptr, unsigned DynamicVGPRBlockSize=0)
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
Definition STLExtras.h:1669
std::pair< MachineBasicBlock::iterator, MachineBasicBlock::iterator > RegionBoundaries
A region's boundaries i.e.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
ArrayRef(const T &OneElt) -> ArrayRef< T >
Policy for scheduling the next instruction in the candidate's zone.
Store the state used by GenericScheduler heuristics, required for the lifetime of one invocation of p...
MachineSchedContext provides enough context from the MachineScheduler pass for the target to instanti...
BitVector LiveIn
Regions in which the original register was live-in or live-out.
LiveMapUpdate(RegisterIdx RegIdx, const BitVector &LiveIn, const BitVector &LiveOut)
RegisterIdx RegIdx
The register index handle in the rematerializer.
Execution frequency information required by scoring heuristics.
SmallVector< uint64_t > Regions
Per-region execution frequencies. 0 when unknown.
uint64_t MinFreq
Minimum and maximum observed frequencies.
FreqInfo(MachineFunction &MF, const GCNScheduleDAGMILive &DAG)