LLVM 24.0.0git
AMDGPUBarrierLatency.cpp
Go to the documentation of this file.
1//===--- AMDGPUBarrierLatency.cpp - AMDGPU Barrier Latency ----------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file This file contains a DAG scheduling mutation to add latency to:
10/// 1. Barrier edges between ATOMIC_FENCE instructions and preceding
11/// memory accesses potentially affected by the fence.
12/// This encourages the scheduling of more instructions before
13/// ATOMIC_FENCE instructions. ATOMIC_FENCE instructions may
14/// introduce wait counting or indicate an impending S_BARRIER
15/// wait. Having more instructions in-flight across these
16/// constructs improves latency hiding.
17/// 2. Barrier edges from S_BARRIER_SIGNAL to S_BARRIER_WAIT.
18/// This encourages independent work to be scheduled between
19/// signal and wait, hiding barrier synchronization latency.
20//
21//===----------------------------------------------------------------------===//
22
24#include "GCNSubtarget.h"
25#include "SIInstrInfo.h"
29
30using namespace llvm;
31
33 "amdgpu-barrier-signal-wait-latency",
34 cl::desc("Synthetic latency between S_BARRIER_SIGNAL and S_BARRIER_WAIT "
35 "to encourage scheduling independent work between them"),
36 cl::init(16), cl::Hidden);
37
38namespace {
39
40class BarrierLatency : public ScheduleDAGMutation {
41private:
42 SmallSet<SyncScope::ID, 4> IgnoredScopes;
43
44public:
45 BarrierLatency(MachineFunction *MF) {
46 LLVMContext &Context = MF->getFunction().getContext();
47 const Triple &TT = MF->getSubtarget<GCNSubtarget>().getTargetTriple();
48 auto ScopeID = [&](AtomicScope Scope, bool OneAS) {
49 return Context.getOrInsertSyncScopeID(
50 *getAtomicScopeIRString(TT, Scope, OneAS));
51 };
52 IgnoredScopes.insert(SyncScope::SingleThread);
53 IgnoredScopes.insert(ScopeID(AtomicScope::Wavefront, /*OneAS=*/false));
54 IgnoredScopes.insert(ScopeID(AtomicScope::Wavefront, /*OneAS=*/true));
55 IgnoredScopes.insert(ScopeID(AtomicScope::Single, /*OneAS=*/true));
56
57 const GCNSubtarget &ST = MF->getSubtarget<GCNSubtarget>();
58 bool TgSplit =
59 ST.hasTgSplitSupport() && AMDGPU::isTgSplitEnabled(MF->getFunction());
60 if (!ST.requiresWaitOnWorkgroupReleaseFence(TgSplit)) {
61 // Prior to GFX10 workgroup scope does not normally require waitcnts
62 IgnoredScopes.insert(ScopeID(AtomicScope::Workgroup, /*OneAS=*/false));
63 }
64 }
65 void apply(ScheduleDAGInstrs *DAG) override;
66};
67
68void addLatencyToEdge(SDep &PredDep, SUnit &SU, unsigned Latency) {
69 SUnit *PredSU = PredDep.getSUnit();
70 SDep ForwardD = PredDep;
71 ForwardD.setSUnit(&SU);
72 for (SDep &SuccDep : PredSU->Succs) {
73 if (SuccDep == ForwardD) {
74 SuccDep.setLatency(SuccDep.getLatency() + Latency);
75 break;
76 }
77 }
78 PredDep.setLatency(PredDep.getLatency() + Latency);
79 PredSU->setDepthDirty();
80 SU.setDepthDirty();
81}
82
83void setLatencyForEdge(SDep &PredDep, SUnit &SU, unsigned Latency) {
84 SUnit *PredSU = PredDep.getSUnit();
85 SDep ForwardD = PredDep;
86 ForwardD.setSUnit(&SU);
87 for (SDep &SuccDep : PredSU->Succs) {
88 if (SuccDep == ForwardD) {
89 SuccDep.setLatency(Latency);
90 break;
91 }
92 }
93 PredDep.setLatency(Latency);
94 PredSU->setDepthDirty();
95 SU.setDepthDirty();
96}
97
98void BarrierLatency::apply(ScheduleDAGInstrs *DAG) {
99 const SIInstrInfo *TII = static_cast<const SIInstrInfo *>(DAG->TII);
100 constexpr unsigned FenceLatency = 2000;
101 const unsigned BarrierSignalWaitLatency = BarrierSignalWaitLatencyOpt;
102 SmallVector<SUnit *, 8> RegionTDM;
103 SmallVector<SUnit *, 8> RegionAsync;
104 const TargetSchedModel *SchedModel = DAG->getSchedModel();
105
106 for (SUnit &SU : DAG->SUnits) {
107 const MachineInstr *MI = SU.getInstr();
108 unsigned Op = MI->getOpcode();
109
110 if (Op == AMDGPU::ATOMIC_FENCE) {
111 // Update latency on barrier edges of ATOMIC_FENCE.
112 // Ignore scopes not expected to have any latency.
113 SyncScope::ID SSID =
114 static_cast<SyncScope::ID>(MI->getOperand(1).getImm());
115 if (IgnoredScopes.contains(SSID))
116 continue;
117
118 for (SDep &PredDep : SU.Preds) {
119 if (!PredDep.isBarrier())
120 continue;
121 SUnit *PredSU = PredDep.getSUnit();
122 MachineInstr *MI = PredSU->getInstr();
123 // Only consider memory loads
124 if (!MI->mayLoad() || MI->mayStore())
125 continue;
126
127 addLatencyToEdge(PredDep, SU,
128 SchedModel ? SchedModel->computeInstrLatency(MI, false)
129 : FenceLatency);
130 }
131 } else if (Op == AMDGPU::S_BARRIER_WAIT) {
132 for (SDep &PredDep : SU.Preds) {
133 SUnit *PredSU = PredDep.getSUnit();
134 const MachineInstr *PredMI = PredSU->getInstr();
135 if (TII->isBarrierStart(PredMI->getOpcode())) {
136 addLatencyToEdge(PredDep, SU, BarrierSignalWaitLatency);
137 }
138 }
139 } else if (TII->isLDSDMA(*MI)) {
141 RegionTDM.push_back(&SU);
143 RegionAsync.push_back(&SU);
144 } else if (Op == AMDGPU::S_WAIT_TENSORCNT ||
145 Op == AMDGPU::S_WAIT_ASYNCCNT) {
146 auto needWaitFor = [&](SmallVectorImpl<SUnit *> &RegionLDSDMA, SUnit *SU,
147 int64_t Count) {
148 if (RegionLDSDMA.size() <= static_cast<uint64_t>(Count)) {
149 return false;
150 }
151
152 int64_t Counter = 0;
153 auto I = RegionLDSDMA.rbegin(), E = RegionLDSDMA.rend();
154 for (; I != E; I++) {
155 if (Counter >= Count)
156 return true;
157
158 if (SU->NodeNum == (*I)->NodeNum)
159 return false;
160
161 ++Counter;
162 }
163 llvm_unreachable("Malformed RegionLDSDMA");
164 };
165
166 int64_t WaitVal = MI->getOperand(0).getImm();
167 for (SDep &PredDep : SU.Preds) {
168 if (PredDep.getKind() != SDep::Kind::Data)
169 continue;
170
171 Register DepReg = PredDep.getReg();
172 bool IsAsync = Op == AMDGPU::S_WAIT_ASYNCCNT;
173 Register LDSDMACnt = IsAsync ? AMDGPU::ASYNCcnt : AMDGPU::TENSORcnt;
174
175 if (DepReg != LDSDMACnt)
176 continue;
177
178 SUnit *PredSU = PredDep.getSUnit();
179
180 // The data dep can be carried by a non-LDSDMA SU
181 // (e.g. an intervening COPY or pseudo). Such predecessors are not
182 // tracked, so needWaitFor cannot reason about them.
183 const MachineInstr &PredMI = *PredSU->getInstr();
184 if (IsAsync ? !SIInstrFlags::usesASYNC_CNT(PredMI)
186 continue;
187
188 if (!needWaitFor(Op == AMDGPU::S_WAIT_ASYNCCNT ? RegionAsync
189 : RegionTDM,
190 PredSU, WaitVal)) {
191 setLatencyForEdge(PredDep, SU, 1);
192 }
193 }
194 }
195 }
196}
197
198} // end namespace
199
200std::unique_ptr<ScheduleDAGMutation>
202 return std::make_unique<BarrierLatency>(MF);
203}
static cl::opt< unsigned > BarrierSignalWaitLatencyOpt("amdgpu-barrier-signal-wait-latency", cl::desc("Synthetic latency between S_BARRIER_SIGNAL and S_BARRIER_WAIT " "to encourage scheduling independent work between them"), cl::init(16), cl::Hidden)
unsigned uint64_t
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
AMD GCN specific subclass of TargetSubtarget.
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
#define I(x, y, z)
Definition MD5.cpp:57
Promote Memory to Register
Definition Mem2Reg.cpp:110
Interface definition for SIInstrInfo.
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
Function & getFunction()
Return the LLVM function that this machine code represents.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
Scheduling dependency.
Definition ScheduleDAG.h:52
SUnit * getSUnit() const
Kind getKind() const
Returns an enum value representing the kind of the dependence.
void setLatency(unsigned Lat)
Sets the latency for this edge.
unsigned getLatency() const
Returns the latency value for this edge, which roughly means the minimum number of cycles that must e...
void setSUnit(SUnit *SU)
Register getReg() const
Returns the register associated with this edge.
bool isBarrier() const
Tests if this is an Order dependence that is marked as a barrier.
Scheduling unit. This is a node in the scheduling DAG.
SmallVector< SDep, 4 > Succs
All sunit successors.
LLVM_ABI void setDepthDirty()
Sets a flag in this node to indicate that its stored Depth value will require recomputation the next ...
SmallVector< SDep, 4 > Preds
All sunit predecessors.
MachineInstr * getInstr() const
Returns the representative MachineInstr for this SUnit.
A ScheduleDAG for scheduling lists of MachineInstr.
const TargetSchedModel * getSchedModel() const
Gets the machine model for instruction scheduling.
Mutate the DAG as a postpass after normal DAG building.
const TargetInstrInfo * TII
Target instruction information.
std::vector< SUnit > SUnits
The scheduling units.
SmallSet - This maintains a set of unique values, optimizing for the case when the set is small (less...
Definition SmallSet.h:134
bool contains(const T &V) const
Check if the SmallSet contains the given element.
Definition SmallSet.h:229
void push_back(const T &Elt)
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
bool isTgSplitEnabled(const Function &F)
constexpr bool usesTENSOR_CNT(const T &...O)
Definition SIDefines.h:310
constexpr bool usesASYNC_CNT(const T &...O)
Definition SIDefines.h:319
@ SingleThread
Synchronized with respect to signal handlers executing in the same thread.
Definition LLVMContext.h:55
void apply(Opt *O, const Mod &M, const Mods &... Ms)
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
std::unique_ptr< ScheduleDAGMutation > createAMDGPUBarrierLatencyDAGMutation(MachineFunction *MF)
AtomicScope
Target-neutral memory synchronization scopes.
Definition AtomicScope.h:23
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
DWARFExpression::Operation Op
std::optional< StringRef > getAtomicScopeIRString(const Triple &T, AtomicScope S, bool IsSingleAddressSpace=false)
Returns the LLVM IR syncscope string that T uses to spell S.
Definition AtomicScope.h:34