LLVM 24.0.0git
AMDGPUCoExecSchedStrategy.cpp
Go to the documentation of this file.
1//===- AMDGPUCoExecSchedStrategy.cpp - CoExec Scheduling Strategy ---------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// Coexecution-focused scheduling strategy for AMDGPU.
11//
12//===----------------------------------------------------------------------===//
13
15#include "AMDGPUIGroupLP.h"
16#include "GCNHazardRecognizer.h"
17#include "llvm/Support/Debug.h"
18
19using namespace llvm;
20using namespace llvm::AMDGPU;
21
22#define DEBUG_TYPE "machine-scheduler"
23
24namespace {
25
26// Used to disable post-RA scheduling with function level granularity.
27class GCNNoopPostScheduleDAG final : public ScheduleDAGInstrs {
28public:
29 explicit GCNNoopPostScheduleDAG(MachineSchedContext *C)
30 : ScheduleDAGInstrs(*C->MF, C->MLI, /*RemoveKillFlags=*/true) {}
31
32 // Do nothing.
33 void schedule() override {}
34};
35
36} // namespace
37
39 // pickOnlyChoice() releases pending instructions and checks for new hazards.
40 SUnit *OnlyChoice = Zone.pickOnlyChoice();
41 if (!Zone.Pending.empty())
42 return nullptr;
43
44 return OnlyChoice;
45}
46
48 const SIInstrInfo &SII) {
49 if (MI.isDebugInstr())
51
52 unsigned Opc = MI.getOpcode();
53
54 // Check for specific opcodes first.
55 if (Opc == AMDGPU::ATOMIC_FENCE || Opc == AMDGPU::S_WAIT_ASYNCCNT ||
56 Opc == AMDGPU::S_WAIT_TENSORCNT || Opc == AMDGPU::S_BARRIER_WAIT ||
57 Opc == AMDGPU::S_BARRIER_SIGNAL_IMM)
59
60 if (SII.isLDSDMA(MI))
62
63 if (SII.isMFMAorWMMA(MI))
65
66 if (SII.isTRANS(MI))
68
69 if (SII.isVALU(MI, /*AllowLDSDMA=*/true))
71
72 if (SII.isSMRD(MI))
74
75 if (SII.isDS(MI))
77
78 if (SII.isVMEM(MI))
80
81 if (SII.isSALU(MI))
83
85}
86
88 for (SUnit *PrioritySU : PrioritySUs) {
89 if (!PrioritySU->isTopReady())
90 return PrioritySU;
91 }
92
93 if (!LookDeep)
94 return nullptr;
95
96 unsigned MinDepth = std::numeric_limits<unsigned int>::max();
97 SUnit *TargetSU = nullptr;
98 for (auto *SU : AllSUs) {
99 if (SU->isScheduled)
100 continue;
101
102 if (SU->isTopReady())
103 continue;
104
105 if (SU->getDepth() < MinDepth) {
106 MinDepth = SU->getDepth();
107 TargetSU = SU;
108 }
109 }
110 return TargetSU;
111}
112
113void HardwareUnitInfo::insert(SUnit *SU, unsigned BlockingCycles) {
114 if (!AllSUs.insert(SU))
115 llvm_unreachable("HardwareUnit already contains SU!");
116
117 TotalCycles += BlockingCycles;
118
119 if (PrioritySUs.empty()) {
120 PrioritySUs.insert(SU);
121 return;
122 }
123 unsigned SUDepth = SU->getDepth();
124 unsigned CurrDepth = (*PrioritySUs.begin())->getDepth();
125 if (SUDepth > CurrDepth)
126 return;
127
128 if (SUDepth == CurrDepth) {
129 PrioritySUs.insert(SU);
130 return;
131 }
132
133 // SU is lower depth and should be prioritized.
134 PrioritySUs.clear();
135 PrioritySUs.insert(SU);
136}
137
138void HardwareUnitInfo::markScheduled(SUnit *SU, unsigned BlockingCycles) {
139 // We may want to ignore some HWUIs (e.g. InstructionFlavor::Other). To do so,
140 // we just clear the HWUI. However, we still have instructions which map to
141 // this HWUI. Don't bother managing the state for these HWUI.
142 if (TotalCycles == 0)
143 return;
144
145 ScheduledSUs.push_back(SU);
146 AllSUs.remove(SU);
147 PrioritySUs.remove(SU);
148
149 // BufferSize 0 is unlimited, while size 1 has no parallel buffering. In
150 // either case, each SU uses the HardwareUnit for BlockingCycles.
151 if (BufferSize <= 1 || (ScheduledSUs.size() % BufferSize == 0))
152 TotalCycles -= std::min(TotalCycles, BlockingCycles);
153
154 if (AllSUs.empty())
155 return;
156 if (PrioritySUs.empty()) {
157 for (auto SU : AllSUs) {
158 if (PrioritySUs.empty()) {
159 PrioritySUs.insert(SU);
160 continue;
161 }
162 unsigned SUDepth = SU->getDepth();
163 unsigned CurrDepth = (*PrioritySUs.begin())->getDepth();
164 if (SUDepth > CurrDepth)
165 continue;
166
167 if (SUDepth == CurrDepth) {
168 PrioritySUs.insert(SU);
169 continue;
170 }
171
172 // SU is lower depth and should be prioritized.
173 PrioritySUs.clear();
174 PrioritySUs.insert(SU);
175 }
176 }
177}
178
180 if (BufferSize == 0 || AllSUs.empty())
181 return;
182
183 // We estimate the amount of cycles it takes to free up a slot in the buffer
184 // as the average cycles per SU.
185 BufferCycles = TotalCycles / AllSUs.size();
186 // A single-entry buffer does not reduce TotalCycles.
187 if (BufferSize == 1)
188 return;
189
190 // The TotalCycles is normalized against the BufferSize.
191 // This provides an estimate of the TotalCycles which is not always accurate
192 // -- particularly in cases where we have fewer instructions than the
193 // BufferSize. For example, if we have 2 instructions which each take 50
194 // cycles and a BufferSize of 16, then a TotalCycles of 51 cycles would be
195 // somewhat accurate. This normalization calculates TotalCycles as 6. However,
196 // if we have 64 of these instructions, our normalized estimate of 200 is more
197 // reasonable, given the more accurate measure is 264. Having a completely
198 // accurate measure is not very important, since this metric is mainly used to
199 // compare the relative demand per HardwareUnit across the region. The simpler
200 // estimate makes managing the metric incrementally during scheduling much
201 // simpler.
202 TotalCycles /= BufferSize;
203}
204
207 for (HardwareUnitInfo &HWUICand : HWUInfo) {
208 if (HWUICand.getType() == Flavor) {
209 return &HWUICand;
210 }
211 }
212 return nullptr;
213}
214
216 assert(SchedModel && SchedModel->hasInstrSchedModel());
217 MachineInstr *MI = SU->getInstr();
218 if (SII->isDS(*MI))
219 return SchedModel->computeInstrLatency(MI);
220
221 unsigned ReleaseAtCycle = 0;
222 const MCSchedClassDesc *SC = DAG->getSchedClass(SU);
223 for (TargetSchedModel::ProcResIter PI = SchedModel->getWriteProcResBegin(SC),
224 PE = SchedModel->getWriteProcResEnd(SC);
225 PI != PE; ++PI) {
226 ReleaseAtCycle = std::max(ReleaseAtCycle, (unsigned)PI->ReleaseAtCycle);
227 }
228 return ReleaseAtCycle;
229}
230
237
240 const TargetRegisterInfo *TRI) {
241 DAG = SchedDAG;
243 assert(SchedModel && SchedModel->hasInstrSchedModel());
244
245 SRI = static_cast<const SIRegisterInfo *>(TRI);
246 SII = static_cast<const SIInstrInfo *>(DAG->TII);
247
249
250 for (unsigned I = 0; I < HWUInfo.size(); I++) {
251 HWUInfo[I].reset();
252 HWUInfo[I].setType(I);
253 }
254
255 HWUInfo[(int)InstructionFlavor::WMMA].setProducesCoexecWindow(true);
256 HWUInfo[(int)InstructionFlavor::MultiCycleVALU].setProducesCoexecWindow(true);
257 HWUInfo[(int)InstructionFlavor::TRANS].setProducesCoexecWindow(true);
259
261}
262
264 if (!SchedModel || !SchedModel->hasInstrSchedModel())
265 return;
266
267 for (auto &SU : DAG->SUnits) {
268 const InstructionFlavor Flavor = classifyFlavor(*SU.getInstr(), *SII);
269 HWUInfo[(int)(Flavor)].insert(&SU, getHWUICyclesForInst(&SU));
270 }
271
272 for (auto &HWUI : HWUInfo)
273 HWUI.finalizeCycles();
274
276}
277
279 MachineBasicBlock *BB = DAG->begin()->getParent();
280 dbgs() << "\n=== Region: " << DAG->MF.getName() << " BB" << BB->getNumber()
281 << " (" << DAG->SUnits.size() << " SUs) ===\n";
282
283 dbgs() << "\nHWUI Resource Pressure:\n";
284 for (auto &HWUI : HWUInfo) {
285 if (HWUI.getTotalCycles() == 0)
286 continue;
287
288 StringRef Name = getFlavorName(HWUI.getType());
289 dbgs() << " " << Name << ": " << HWUI.getTotalCycles() << " cycles, "
290 << HWUI.size() << " instrs\n";
291 }
292 dbgs() << "\n";
293}
294
296 // Highest priority should be first.
298 // Prefer CoexecWindow producers
299 if (A.producesCoexecWindow() != B.producesCoexecWindow())
300 return A.producesCoexecWindow();
301
302 // Prefer more demanded resources
303 if (A.getTotalCycles() != B.getTotalCycles())
304 return A.getTotalCycles() > B.getTotalCycles();
305
306 // In ties -- prefer the resource with more instructions
307 if (A.size() != B.size())
308 return A.size() < B.size();
309
310 // Default to Flavor order
311 return static_cast<unsigned>(A.getType()) <
312 static_cast<unsigned>(B.getType());
313 });
314}
315
319
320 auto HasPrioritySU = [this, &Cand, &TryCand](unsigned ResourceIdx) {
321 const HardwareUnitInfo &HWUI = HWUInfo[ResourceIdx];
322
323 auto CandFlavor = classifyFlavor(*Cand.SU->getInstr(), *SII);
324 auto TryCandFlavor = classifyFlavor(*TryCand.SU->getInstr(), *SII);
325 bool LookDeep = (CandFlavor == InstructionFlavor::DS ||
326 TryCandFlavor == InstructionFlavor::DS) &&
328 auto *TargetSU = HWUI.getNextTargetSU(LookDeep);
329
330 // If we do not have a TargetSU for this resource, then it is not critical.
331 if (!TargetSU)
332 return false;
333
334 return true;
335 };
336
337 auto TryEnablesResource = [&Cand, &TryCand, this](unsigned ResourceIdx) {
338 const HardwareUnitInfo &HWUI = HWUInfo[ResourceIdx];
339 auto CandFlavor = classifyFlavor(*Cand.SU->getInstr(), *SII);
340
341 // We want to ensure our DS order matches WMMA order.
342 bool LookDeep = CandFlavor == InstructionFlavor::DS &&
344 auto *TargetSU = HWUI.getNextTargetSU(LookDeep);
345
346 bool CandEnables =
347 TargetSU != Cand.SU && DAG->IsReachable(TargetSU, Cand.SU);
348 bool TryCandEnables =
349 TargetSU != TryCand.SU && DAG->IsReachable(TargetSU, TryCand.SU);
350
351 if (!CandEnables && !TryCandEnables)
352 return false;
353
354 if (CandEnables && !TryCandEnables) {
357
358 return true;
359 }
360
361 if (!CandEnables && TryCandEnables) {
363 return true;
364 }
365
366 // Both enable, prefer the critical path.
367 unsigned CandHeight = Cand.SU->getHeight();
368 unsigned TryCandHeight = TryCand.SU->getHeight();
369
370 if (CandHeight > TryCandHeight) {
373
374 return true;
375 }
376
377 if (CandHeight < TryCandHeight) {
379 return true;
380 }
381
382 // Same critical path, just prefer original candidate.
385
386 return true;
387 };
388
389 for (unsigned I = 0; I < HWUInfo.size(); I++) {
390 // If we have encountered a resource that is not critical, then neither
391 // candidate enables a critical resource
392 if (!HasPrioritySU(I))
393 continue;
394
395 bool Enabled = TryEnablesResource(I);
396 // If neither has enabled the resource, continue to the next resource
397 if (Enabled)
398 return true;
399 }
400 return false;
401}
402
406 for (unsigned I = 0; I < HWUInfo.size(); I++) {
407 const HardwareUnitInfo &HWUI = HWUInfo[I];
408
409 bool CandUsesCrit = HWUI.contains(Cand.SU);
410 bool TryCandUsesCrit = HWUI.contains(TryCand.SU);
411
412 if (!CandUsesCrit && !TryCandUsesCrit)
413 continue;
414
415 if (CandUsesCrit != TryCandUsesCrit) {
416 if (CandUsesCrit) {
419 return true;
420 }
422 return true;
423 }
424
425 // Otherwise, both use the critical resource
426 // For longer latency InstructionFlavors, we should prioritize first by
427 // their enablement of critical resources
428 if (HWUI.getType() == InstructionFlavor::DS) {
429 if (tryCriticalResourceDependency(TryCand, Cand, Zone))
430 return true;
431 }
432
433 // Prioritize based on HWUI priorities.
434 SUnit *Match = HWUI.getHigherPriority(Cand.SU, TryCand.SU);
435 if (Match) {
436 if (Match == Cand.SU) {
439 return true;
440 }
442 return true;
443 }
444 }
445
446 return false;
447}
448
458
461 unsigned NumRegionInstrs) {
465 "coexec scheduler only supports top-down scheduling");
466 RegionPolicy.OnlyTopDown = true;
467 RegionPolicy.OnlyBottomUp = false;
468 RegionPolicy.ShouldTrackLaneMasks = true;
469}
470
472 // Coexecution scheduling strategy is only done top-down to support new
473 // resource balancing heuristics.
474 RegionPolicy.OnlyTopDown = true;
475 RegionPolicy.OnlyBottomUp = false;
476
478 Heurs.initialize(DAG, SchedModel, TRI);
479
480 // Replace the default hazard recognizer with our PreRA one so that pre-RA
481 // scheduling accounts for WMMA co-execution slot constraints. This must
482 // happen after GCNSchedStrategy::initialize() because
483 // GenericScheduler::initialize() calls SchedBoundary::reset(), which deletes
484 // and recreates the hazard recognizer each region.
485 Top.HazardRec = std::make_unique<GCNHazardRecognizer>(
487}
488
490 Heurs.updateForScheduling(SU);
491 GCNSchedStrategy::schedNode(SU, IsTopNode);
492}
493
495 assert(RegionPolicy.OnlyTopDown && !RegionPolicy.OnlyBottomUp &&
496 "coexec scheduler only supports top-down scheduling");
497
498 if (DAG->top() == DAG->bottom()) {
499 assert(Top.Available.empty() && Top.Pending.empty() &&
500 Bot.Available.empty() && Bot.Pending.empty() && "ReadyQ garbage");
501 return nullptr;
502 }
503
504 bool PickedPending = false;
505 SUnit *SU = nullptr;
506#ifndef NDEBUG
507 SchedCandidate *PickedCand = nullptr;
508#endif
509 do {
510 PickedPending = false;
511 SU = pickOnlyChoice(Top);
512 if (!SU) {
513 CandPolicy NoPolicy;
514 TopCand.reset(NoPolicy);
515 pickNodeFromQueue(Top, NoPolicy, DAG->getTopRPTracker(), TopCand,
516 PickedPending, /*IsBottomUp=*/false);
517 assert(TopCand.Reason != NoCand && "failed to find a candidate");
518 SU = TopCand.SU;
519#ifndef NDEBUG
520 PickedCand = &TopCand;
521#endif
522 }
523 IsTopNode = true;
524 } while (SU->isScheduled);
525
526 LLVM_DEBUG(if (PickedCand) dumpPickSummary(SU, IsTopNode, *PickedCand));
527
528 if (PickedPending) {
529 unsigned ReadyCycle = SU->TopReadyCycle;
530 unsigned CurrentCycle = Top.getCurrCycle();
531 if (ReadyCycle > CurrentCycle)
532 Top.bumpCycle(ReadyCycle);
533
534 // checkHazard() does not expose the exact cycle where the hazard clears.
535 while (Top.checkHazard(SU))
536 Top.bumpCycle(Top.getCurrCycle() + 1);
537
538 Top.releasePending();
539 }
540
541 if (SU->isTopReady())
542 Top.removeReady(SU);
543 if (SU->isBottomReady())
544 Bot.removeReady(SU);
545
546 LLVM_DEBUG(dbgs() << "Scheduling SU(" << SU->NodeNum << ") "
547 << *SU->getInstr());
548
549 assert(IsTopNode && "coexec scheduler must only schedule from top boundary");
550 return SU;
551}
552
554 SchedBoundary &Zone, const CandPolicy &ZonePolicy,
555 const RegPressureTracker &RPTracker, SchedCandidate &Cand,
556 bool &PickedPending, bool IsBottomUp) {
557 assert(Zone.isTop() && "coexec scheduler only supports top boundary");
558 assert(!IsBottomUp && "coexec scheduler only supports top-down scheduling");
559
560 const SIRegisterInfo *SRI = static_cast<const SIRegisterInfo *>(TRI);
562 unsigned SGPRPressure = 0;
563 unsigned VGPRPressure = 0;
564 PickedPending = false;
565 if (DAG->isTrackingPressure()) {
566 if (!useGCNTrackers()) {
567 SGPRPressure = Pressure[AMDGPU::RegisterPressureSets::SReg_32];
568 VGPRPressure = Pressure[AMDGPU::RegisterPressureSets::VGPR_32];
569 } else {
570 SGPRPressure = DownwardTracker.getPressure().getSGPRNum();
571 VGPRPressure = DownwardTracker.getPressure().getArchVGPRNum();
572 }
573 }
574
575 auto EvaluateQueue = [&](ReadyQueue &Q, bool FromPending) {
576 for (SUnit *SU : Q) {
577 SchedCandidate TryCand(ZonePolicy);
578 initCandidate(TryCand, SU, Zone.isTop(), RPTracker, SRI, SGPRPressure,
579 VGPRPressure, IsBottomUp);
580 SchedBoundary *ZoneArg = Cand.AtTop == TryCand.AtTop ? &Zone : nullptr;
581 tryCandidateCoexec(Cand, TryCand, ZoneArg);
582 if (TryCand.Reason != NoCand) {
583 if (TryCand.ResDelta == SchedResourceDelta())
584 TryCand.initResourceDelta(Zone.DAG, SchedModel);
585 LLVM_DEBUG(printCandidateDecision(Cand, TryCand));
586 PickedPending = FromPending;
587 Cand.setBest(TryCand);
588 } else {
589 LLVM_DEBUG(printCandidateDecision(TryCand, Cand));
590 }
591 }
592 };
593
594 LLVM_DEBUG(dbgs() << "Available Q:\n");
595 EvaluateQueue(Zone.Available, /*FromPending=*/false);
596
597 LLVM_DEBUG(dbgs() << "Pending Q:\n");
598 EvaluateQueue(Zone.Pending, /*FromPending=*/true);
599}
600
601#ifndef NDEBUG
603 SchedCandidate &Cand) {
604 const SIInstrInfo *SII = static_cast<const SIInstrInfo *>(DAG->TII);
605 unsigned Cycle = IsTopNode ? Top.getCurrCycle() : Bot.getCurrCycle();
606
607 dbgs() << "=== Pick @ Cycle " << Cycle << " ===\n";
608
609 const InstructionFlavor Flavor = classifyFlavor(*SU->getInstr(), *SII);
610 dbgs() << "Picked: SU(" << SU->NodeNum << ") ";
611 SU->getInstr()->print(dbgs(), /*IsStandalone=*/true, /*SkipOpers=*/false,
612 /*SkipDebugLoc=*/true);
613 dbgs() << " [" << getFlavorName(Flavor) << "]\n";
614
615 dbgs() << " Reason: ";
618 else if (Cand.Reason != NoCand)
620 else
621 dbgs() << "Unknown";
622 dbgs() << "\n\n";
623
625}
626#endif
627
629 SchedCandidate &TryCand,
630 SchedBoundary *Zone) {
631 // Initialize the candidate if needed.
632 if (!Cand.isValid()) {
633 TryCand.Reason = FirstValid;
634 return true;
635 }
636
637 // Bias PhysReg Defs and copies to their uses and defined respectively.
638 if (tryGreater(biasPhysReg(TryCand.SU, TryCand.AtTop),
639 biasPhysReg(Cand.SU, Cand.AtTop), TryCand, Cand, PhysReg))
640 return TryCand.Reason != NoCand;
641
642 // Avoid exceeding the target's limit.
643 if (DAG->isTrackingPressure() &&
644 tryPressure(TryCand.RPDelta.Excess, Cand.RPDelta.Excess, TryCand, Cand,
645 RegExcess, TRI, DAG->MF))
646 return TryCand.Reason != NoCand;
647
648 // We only compare a subset of features when comparing nodes between
649 // Top and Bottom boundary. Some properties are simply incomparable, in many
650 // other instances we should only override the other boundary if something
651 // is a clear good pick on one boundary. Skip heuristics that are more
652 // "tie-breaking" in nature.
653 bool SameBoundary = Zone != nullptr;
654 if (SameBoundary) {
655 // Compare candidates by the stall they would introduce if
656 // scheduled in the current cycle.
657 if (tryEffectiveStall(Cand, TryCand, *Zone))
658 return TryCand.Reason != NoCand;
659
660 Heurs.sortHWUIResources();
661 if (Heurs.tryCriticalResource(TryCand, Cand, Zone)) {
663 return TryCand.Reason != NoCand;
664 }
665
666 if (Heurs.tryCriticalResourceDependency(TryCand, Cand, Zone)) {
668 return TryCand.Reason != NoCand;
669 }
670 }
671
672 // Keep clustered nodes together to encourage downstream peephole
673 // optimizations which may reduce resource requirements.
674 //
675 // This is a best effort to set things up for a post-RA pass. Optimizations
676 // like generating loads of multiple registers should ideally be done within
677 // the scheduler pass by combining the loads during DAG postprocessing.
678 unsigned CandZoneCluster = Cand.AtTop ? TopClusterID : BotClusterID;
679 unsigned TryCandZoneCluster = TryCand.AtTop ? TopClusterID : BotClusterID;
680 bool CandIsClusterSucc =
681 isTheSameCluster(CandZoneCluster, Cand.SU->ParentClusterIdx);
682 bool TryCandIsClusterSucc =
683 isTheSameCluster(TryCandZoneCluster, TryCand.SU->ParentClusterIdx);
684
685 if (tryGreater(TryCandIsClusterSucc, CandIsClusterSucc, TryCand, Cand,
686 Cluster))
687 return TryCand.Reason != NoCand;
688
689 if (SameBoundary) {
690 // Weak edges are for clustering and other constraints.
691 if (tryLess(getWeakLeft(TryCand.SU, TryCand.AtTop),
692 getWeakLeft(Cand.SU, Cand.AtTop), TryCand, Cand, Weak))
693 return TryCand.Reason != NoCand;
694 }
695
696 // Avoid increasing the max pressure of the entire region.
697 if (DAG->isTrackingPressure() &&
698 tryPressure(TryCand.RPDelta.CurrentMax, Cand.RPDelta.CurrentMax, TryCand,
699 Cand, RegMax, TRI, DAG->MF))
700 return TryCand.Reason != NoCand;
701
702 if (SameBoundary) {
703 // Avoid serializing long latency dependence chains.
704 // For acyclic path limited loops, latency was already checked above.
705 if (!RegionPolicy.DisableLatencyHeuristic && TryCand.Policy.ReduceLatency &&
706 !Rem.IsAcyclicLatencyLimited && tryLatency(TryCand, Cand, *Zone))
707 return TryCand.Reason != NoCand;
708
709 // Fall through to original instruction order.
710 if ((Zone->isTop() && TryCand.SU->NodeNum < Cand.SU->NodeNum) ||
711 (!Zone->isTop() && TryCand.SU->NodeNum > Cand.SU->NodeNum)) {
712 TryCand.Reason = NodeOrder;
713 return true;
714 }
715 }
716
717 return false;
718}
719
721 SchedCandidate &TryCand,
722 SchedBoundary &Zone) {
723 auto getBufferFullStalls = [this, &Zone](SUnit *SU) -> unsigned {
725 *SU->getInstr(), *static_cast<const SIInstrInfo *>(DAG->TII));
726 HardwareUnitInfo *HWUI = Heurs.getHWUIFromFlavor(Flavor);
727
728 // A BufferSize of 0 is unlimited, so it has no FIFO scheduling cost.
729 if (HWUI->getBufferSize() == 0)
730 return 0;
731
732 // getBufferAvailableCycle assumes top-down scheduling.
733 assert(Zone.isTop());
734 unsigned CurrCycle = Zone.getCurrCycle();
735 unsigned BufferReadyCycle = HWUI->getBufferAvailableCycle(CurrCycle);
736 if (BufferReadyCycle <= CurrCycle)
737 return 0;
738
739 return BufferReadyCycle - CurrCycle;
740 };
741
742 // Treat structural and latency stalls as a single scheduling cost for the
743 // current cycle.
744 struct StallCosts {
745 unsigned Ready = 0;
746 unsigned Structural = 0;
747 unsigned Latency = 0;
748 unsigned Effective = 0;
749 unsigned Buffer = 0;
750 };
751
752 unsigned CurrCycle = Zone.getCurrCycle();
753 auto GetStallCosts = [&](SUnit *SU) {
754 unsigned ReadyCycle = Zone.isTop() ? SU->TopReadyCycle : SU->BotReadyCycle;
755 StallCosts Costs;
756 Costs.Ready = ReadyCycle > CurrCycle ? ReadyCycle - CurrCycle : 0;
757 Costs.Structural = getStructuralStallCycles(Zone, SU);
758 Costs.Latency = Zone.getLatencyStallCycles(SU);
759 Costs.Buffer = getBufferFullStalls(SU);
760 Costs.Effective =
761 std::max({Costs.Ready, Costs.Structural, Costs.Latency, Costs.Buffer});
762 return Costs;
763 };
764
765 StallCosts TryCosts = GetStallCosts(TryCand.SU);
766 StallCosts CandCosts = GetStallCosts(Cand.SU);
767
768 LLVM_DEBUG(if (TryCosts.Effective || CandCosts.Effective) {
769 dbgs() << "Effective stalls: try=" << TryCosts.Effective
770 << " (ready=" << TryCosts.Ready << ", struct=" << TryCosts.Structural
771 << ", lat=" << TryCosts.Latency << ", buffer=" << TryCosts.Buffer
772 << ") cand=" << CandCosts.Effective << " (ready=" << CandCosts.Ready
773 << ", struct=" << CandCosts.Structural
774 << ", lat=" << CandCosts.Latency << ", buffer=" << CandCosts.Buffer
775 << ")\n";
776 });
777
778 return tryLess(TryCosts.Effective, CandCosts.Effective, TryCand, Cand, Stall);
779}
780
783 LLVM_DEBUG(dbgs() << "AMDGPU coexec preRA scheduler selected for "
784 << C->MF->getName() << '\n');
786 C, std::make_unique<AMDGPUCoExecSchedStrategy>(C));
788 return DAG;
789}
790
793 LLVM_DEBUG(dbgs() << "AMDGPU nop postRA scheduler selected for "
794 << C->MF->getName() << '\n');
795 return new GCNNoopPostScheduleDAG(C);
796}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static SUnit * pickOnlyChoice(SchedBoundary &Zone)
Coexecution-focused scheduling strategy for AMDGPU.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
IRTranslator LLVM IR MI
#define I(x, y, z)
Definition MD5.cpp:57
Register const TargetRegisterInfo * TRI
#define LLVM_DEBUG(...)
Definition Debug.h:119
void initPolicy(MachineBasicBlock::iterator Begin, MachineBasicBlock::iterator End, unsigned NumRegionInstrs) override
Optionally override the per-region scheduling policy.
SUnit * pickNode(bool &IsTopNode) override
Pick the next node to schedule, or return NULL.
void pickNodeFromQueue(SchedBoundary &Zone, const CandPolicy &ZonePolicy, const RegPressureTracker &RPTracker, SchedCandidate &Cand, bool &PickedPending, bool IsBottomUp)
bool tryEffectiveStall(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary &Zone)
void initialize(ScheduleDAGMI *DAG) override
Initialize the strategy after building the DAG for a new region.
void schedNode(SUnit *SU, bool IsTopNode) override
Notify MachineSchedStrategy that ScheduleDAGMI has scheduled an instruction and updated scheduled/rem...
AMDGPUCoExecSchedStrategy(const MachineSchedContext *C)
void dumpPickSummary(SUnit *SU, bool IsTopNode, SchedCandidate &Cand)
bool tryCandidateCoexec(SchedCandidate &Cand, SchedCandidate &TryCand, SchedBoundary *Zone)
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
void updateForScheduling(SUnit *SU)
Update the state to reflect that SU is going to be scheduled.
HardwareUnitInfo * getHWUIFromFlavor(AMDGPU::InstructionFlavor Flavor)
Given a Flavor , find the corresponding HardwareUnit.
void sortHWUIResources()
Sort the HardwarUnitInfo vector.
bool tryCriticalResource(GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, SchedBoundary *Zone) const
Check for critical resource consumption.
bool tryCriticalResourceDependency(GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, SchedBoundary *Zone) const
Check for dependencies of instructions that use prioritized HardwareUnits.
SmallVector< HardwareUnitInfo, 8 > HWUInfo
const TargetSchedModel * SchedModel
void collectHWUIPressure()
Walk over the region and collect total usage per HardwareUnit.
void initialize(ScheduleDAGMI *DAG, const TargetSchedModel *SchedModel, const TargetRegisterInfo *TRI)
unsigned getHWUICyclesForInst(SUnit *SU)
Compute the blocking cycles for the appropriate HardwareUnit given an SU.
GCNDownwardRPTracker DownwardTracker
GCNSchedStrategy(const MachineSchedContext *C)
SmallVector< GCNSchedStageID, 4 > SchedStages
void schedNode(SUnit *SU, bool IsTopNode) override
Notify MachineSchedStrategy that ScheduleDAGMI has scheduled an instruction and updated scheduled/rem...
std::vector< unsigned > Pressure
void initialize(ScheduleDAGMI *DAG) override
Initialize the strategy after building the DAG for a new region.
void printCandidateDecision(const SchedCandidate &Current, const SchedCandidate &Preferred)
unsigned getStructuralStallCycles(SchedBoundary &Zone, SUnit *SU) const
Estimate how many cycles SU must wait due to structural hazards at the current boundary cycle.
void initCandidate(SchedCandidate &Cand, SUnit *SU, bool AtTop, const RegPressureTracker &RPTracker, const SIRegisterInfo *SRI, unsigned SGPRPressure, unsigned VGPRPressure, bool IsBottomUp)
MachineSchedPolicy RegionPolicy
const TargetSchedModel * SchedModel
static const char * getReasonStr(GenericSchedulerBase::CandReason Reason)
const TargetRegisterInfo * TRI
SchedCandidate TopCand
Candidate last picked from Top boundary.
ScheduleDAGMILive * DAG
HardwareUnitInfo is a wrapper class which maps to some real hardware resource.
void markScheduled(SUnit *SU, unsigned BlockingCycles)
Update the state for SU being scheduled by removing it from the AllSUs and reducing its BlockingCycle...
SUnit * getNextTargetSU(bool LookDeep=false) const
void insert(SUnit *SU, unsigned BlockingCycles)
Insert the SU into AllSUs and account its BlockingCycles into the TotalCycles.
void finalizeCycles()
After we've collected all the region pressure for this HWUI, correct for any specifics of the behavio...
AMDGPU::InstructionFlavor getType() const
unsigned getBufferAvailableCycle(unsigned CurrCycle)
SUnit * getHigherPriority(SUnit *SU, SUnit *Other) const
int getNumber() const
MachineBasicBlocks are uniquely numbered at the function level, unless they're not in a MachineFuncti...
MachineInstrBundleIterator< MachineInstr > iterator
Representation of each machine instruction.
LLVM_ABI void print(raw_ostream &OS, bool IsStandalone=true, bool SkipOpers=false, bool SkipDebugLoc=false, bool AddNewLine=true, const TargetInstrInfo *TII=nullptr) const
Print this MI to OS.
virtual void initPolicy(MachineBasicBlock::iterator Begin, MachineBasicBlock::iterator End, unsigned NumRegionInstrs)
Optionally override the per-region scheduling policy.
Helpers for implementing custom MachineSchedStrategy classes.
Track the current register pressure at some position in the instruction stream, and remember the high...
const std::vector< unsigned > & getRegSetPressureAtPos() const
Get the register set pressure at the current position, which may be less than the pressure across the...
static bool isDS(const MachineInstr &MI)
static bool isVMEM(const MachineInstr &MI)
static bool isSMRD(const MachineInstr &MI)
static bool isSALU(const MachineInstr &MI)
static bool isMFMAorWMMA(const MachineInstr &MI)
static bool isVALU(const MachineInstr &MI, bool AllowLDSDMA)
static bool isTRANS(const MachineInstr &MI)
static bool isLDSDMA(const MachineInstr &MI)
Scheduling unit. This is a node in the scheduling DAG.
unsigned TopReadyCycle
Cycle relative to start when node is ready.
unsigned NodeNum
Entry # of node in the node vector.
unsigned getHeight() const
Returns the height of this node, which is the length of the maximum path down to any node which has n...
unsigned getDepth() const
Returns the depth of this node, which is the length of the maximum path up to any node which has no p...
bool isScheduled
True once scheduled.
unsigned ParentClusterIdx
The parent cluster id.
bool isBottomReady() const
bool isTopReady() const
MachineInstr * getInstr() const
Returns the representative MachineInstr for this SUnit.
Each Scheduling boundary is associated with ready queues.
LLVM_ABI unsigned getLatencyStallCycles(SUnit *SU)
Get the difference between the given SUnit's ready time and the current cycle.
LLVM_ABI SUnit * pickOnlyChoice()
Call this before applying any other heuristics to the Available queue.
unsigned getCurrCycle() const
Number of cycles to issue the instructions scheduled in this zone.
A ScheduleDAG for scheduling lists of MachineInstr.
ScheduleDAGMILive is an implementation of ScheduleDAGInstrs that schedules machine instructions while...
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
void addMutation(std::unique_ptr< ScheduleDAGMutation > Mutation)
Add a postprocessing step to the DAG builder.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
Provide an instruction scheduling machine model to CodeGen passes.
const MCWriteProcResEntry * ProcResIter
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
InstructionFlavor
Classification of instructions by execution characteristics.
constexpr StringRef getFlavorName(InstructionFlavor F)
InstructionFlavor classifyFlavor(const MachineInstr &MI, const SIInstrInfo &SII)
Classify MI into the execution flavor that drives both the scheduler's slot preferences and the hazar...
StringRef getReasonName(AMDGPUSchedReason R)
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI int biasPhysReg(const SUnit *SU, bool isTop, bool BiasPRegsExtra=false)
Minimize physical register live ranges.
LLVM_ABI unsigned getWeakLeft(const SUnit *SU, bool isTop)
std::unique_ptr< ScheduleDAGMutation > createIGroupLPDAGMutation(AMDGPU::SchedulingPhase Phase)
Phase specifes whether or not this is a reentry into the IGroupLPDAGMutation.
LLVM_ABI bool tryPressure(const PressureChange &TryP, const PressureChange &CandP, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason, const TargetRegisterInfo *TRI, const MachineFunction &MF)
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1636
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
ScheduleDAGInstrs * createGCNNoopPostMachineScheduler(MachineSchedContext *C)
LLVM_ABI bool tryLatency(GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, SchedBoundary &Zone)
ScheduleDAGInstrs * createGCNCoExecMachineScheduler(MachineSchedContext *C)
bool isTheSameCluster(unsigned A, unsigned B)
Return whether the input cluster ID's are the same and valid.
LLVM_ABI bool tryGreater(int TryVal, int CandVal, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason)
LLVM_ABI bool tryLess(int TryVal, int CandVal, GenericSchedulerBase::SchedCandidate &TryCand, GenericSchedulerBase::SchedCandidate &Cand, GenericSchedulerBase::CandReason Reason)
Return true if this heuristic determines order.
@ Enabled
Convert any .debug_str_offsets tables to DWARF64 if needed.
Definition DWP.h:31
LLVM_ABI cl::opt< MISched::Direction > PreRADirection
Policy for scheduling the next instruction in the candidate's zone.
Store the state used by GenericScheduler heuristics, required for the lifetime of one invocation of p...
LLVM_ABI void initResourceDelta(const ScheduleDAGMI *DAG, const TargetSchedModel *SchedModel)
Status of an instruction's critical resource consumption.
Summarize the scheduling resources required for an instruction of a particular scheduling class.
Definition MCSchedule.h:129
MachineSchedContext provides enough context from the MachineScheduler pass for the target to instanti...