LLVM 24.0.0git
GCNSubtarget.cpp
Go to the documentation of this file.
1//===-- GCNSubtarget.cpp - GCN Subtarget Information ----------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// Implements the GCN specific subclass of TargetSubtarget.
11//
12//===----------------------------------------------------------------------===//
13
14#include "GCNSubtarget.h"
15#include "AMDGPUCallLowering.h"
17#include "AMDGPULegalizerInfo.h"
20#include "AMDGPUTargetMachine.h"
28#include "llvm/IR/MDBuilder.h"
30#include <algorithm>
31
32using namespace llvm;
33
34#define DEBUG_TYPE "gcn-subtarget"
35
36#define GET_SUBTARGETINFO_TARGET_DESC
37#define GET_SUBTARGETINFO_CTOR
38#define AMDGPUSubtarget GCNSubtarget
39#include "AMDGPUGenSubtargetInfo.inc"
40#undef AMDGPUSubtarget
41
43 "amdgpu-vgpr-index-mode",
44 cl::desc("Use GPR indexing mode instead of movrel for vector indexing"),
45 cl::init(false));
46
47static cl::opt<bool> UseAA("amdgpu-use-aa-in-codegen",
48 cl::desc("Enable the use of AA during codegen."),
49 cl::init(true));
50
52 NSAThreshold("amdgpu-nsa-threshold",
53 cl::desc("Number of addresses from which to enable MIMG NSA."),
55
57
59 // Legacy triples without a subarch default to the first target that supports
60 // flat addressing for HSA, otherwise the first amdgcn target.
61 if (TT.getSubArch() == Triple::NoSubArch)
62 return TT.getOS() == Triple::AMDHSA ? AMDGPUSubtarget::SEA_ISLANDS
64
65 switch (AMDGPU::getMajorSubArch(TT.getSubArch())) {
89 default:
90 reportFatalUsageError("invalid subarch for amdgpu");
91 }
92}
93
95 StringRef GPU,
96 StringRef FS) {
97 // Determine default and user-specified characteristics
98 //
99 // We want to be able to turn these off, but making this a subtarget feature
100 // for SI has the unhelpful behavior that it unsets everything else if you
101 // disable it.
102 //
103 // Similarly we want enable-prt-strict-null to be on by default and not to
104 // unset everything else if it is disabled
105
106 SmallString<256> FullFS("+load-store-opt,+enable-ds128,");
107
108 // Turn on features that HSA ABI requires. Also turn on FlatForGlobal by
109 // default
110 if (isAmdHsaOS())
111 FullFS += "+flat-for-global,+unaligned-access-mode,+trap-handler,";
112
113 FullFS += "+enable-prt-strict-null,"; // This is overridden by a disable in FS
114
115 // Disable mutually exclusive bits.
116 if (FS.contains_insensitive("+wavefrontsize")) {
117 if (!FS.contains_insensitive("wavefrontsize16"))
118 FullFS += "-wavefrontsize16,";
119 if (!FS.contains_insensitive("wavefrontsize32"))
120 FullFS += "-wavefrontsize32,";
121 if (!FS.contains_insensitive("wavefrontsize64"))
122 FullFS += "-wavefrontsize64,";
123 }
124
125 FullFS += FS;
126
127 ParseSubtargetFeatures(GPU, /*TuneCPU*/ GPU, FullFS);
128
129 // Implement the "generic" processors, which acts as the default when no
130 // generation features are enabled (e.g for -mcpu=''). HSA OS defaults to
131 // the first amdgcn target that supports flat addressing. Other OSes defaults
132 // to the first amdgcn target.
135 // Assume wave64 for the unknown target, if not explicitly set.
136 if (getWavefrontSizeLog2() == 0)
138 } else if (!hasFeature(AMDGPU::FeatureWavefrontSize32) &&
139 !hasFeature(AMDGPU::FeatureWavefrontSize64)) {
140 // If there is no default wave size it must be a generation before gfx10,
141 // these have FeatureWavefrontSize64 in their definition already. For gfx10+
142 // set wave32 as a default.
143 ToggleFeature(AMDGPU::FeatureWavefrontSize32);
145 }
146
147 // We don't support FP64 for EG/NI atm.
149
150 // Targets must either support 64-bit offsets for MUBUF instructions, and/or
151 // support flat operations, otherwise they cannot access a 64-bit global
152 // address space
153 assert(hasAddr64() || hasFlat());
154 // Unless +-flat-for-global is specified, turn on FlatForGlobal for targets
155 // that do not support ADDR64 variants of MUBUF instructions. Such targets
156 // cannot use a 64 bit offset with a MUBUF instruction to access the global
157 // address space
158 if (!hasAddr64() && !FS.contains("flat-for-global") && !UseFlatForGlobal) {
159 ToggleFeature(AMDGPU::FeatureUseFlatForGlobal);
160 UseFlatForGlobal = true;
161 }
162 // Unless +-flat-for-global is specified, use MUBUF instructions for global
163 // address space access if flat operations are not available.
164 if (!hasFlat() && !FS.contains("flat-for-global") && UseFlatForGlobal) {
165 ToggleFeature(AMDGPU::FeatureUseFlatForGlobal);
166 UseFlatForGlobal = false;
167 }
168
169 // Set defaults if needed.
170 if (MaxPrivateElementSize == 0)
172
173 if (LDSBankCount == 0)
174 LDSBankCount = 32;
175
178
179 if (FlatOffsetBitWidth == 0)
181
183 // LDS Allocation Granularity calculated in bytes from dwords
185 AMDGPU::getLdsDwGranularity(*this) * sizeof(uint32_t);
186
189
190 // InstCacheLineSize is set from TableGen subtarget features
191 // (FeatureInstCacheLineSize64 / FeatureInstCacheLineSize128).
192 // Fall back to 64 if no feature was specified (e.g. generic targets).
193 if (InstCacheLineSize == 0)
195
197 "InstCacheLineSize must be a power of 2");
198
199 return *this;
200}
201
203 LLVMContext &Ctx = F.getContext();
204 if (hasFeature(AMDGPU::FeatureWavefrontSize32) &&
205 hasFeature(AMDGPU::FeatureWavefrontSize64)) {
206 Ctx.diagnose(DiagnosticInfoUnsupported(
207 F, "must specify exactly one of wavefrontsize32 and wavefrontsize64"));
208 }
209}
210
211// TODO: Validate subarch for subtarget
212
214 const GCNTargetMachine &TM, bool BufferOOBRelaxed,
218 : // clang-format off
219 AMDGPUGenSubtargetInfo(TT, GPU, /*TuneCPU*/ GPU, FS),
220 AMDGPUSubtarget(TT),
221 TargetID(AMDGPU::createAMDGPUTargetID(*this, "")),
222 InstrItins(getInstrItineraryForCPU(GPU)),
225 InstrInfo(initializeSubtargetDependencies(TT, GPU, FS)),
226 TLInfo(TM, *this),
227 // Frame index expansion sometimes assumes the low bit of SP is 0
228 FrameLowering(TargetFrameLowering::StackGrowsUp, getStackAlignment(), 0,
229 /*TransAl=*/Align(4)) {
230
231 // clang-format on
232
233 // Apply the module flag's xnack setting if the target supports on/off modes.
234 // Targets without on/off mode support have xnack always on and ignore module
235 // flags.
236 if (hasXNACKOnOffModes())
237 TargetID.setXnackSetting(XnackSetting);
238
239 // Apply the module flag's sramecc setting if the target supports it.
240 if (supportsSRAMECC())
241 TargetID.setSramEccSetting(SramEccSetting);
242
243 LLVM_DEBUG(dbgs() << "xnack setting for subtarget: "
244 << TargetID.getXnackSetting() << '\n');
245 LLVM_DEBUG(dbgs() << "sramecc setting for subtarget: "
246 << TargetID.getSramEccSetting() << '\n');
247
250
251 TSInfo = std::make_unique<AMDGPUSelectionDAGInfo>();
252
253 CallLoweringInfo = std::make_unique<AMDGPUCallLowering>(*getTargetLowering());
254 InlineAsmLoweringInfo =
255 std::make_unique<InlineAsmLowering>(getTargetLowering());
256 Legalizer = std::make_unique<AMDGPULegalizerInfo>(*this, TM);
257 RegBankInfo = std::make_unique<AMDGPURegisterBankInfo>(*this);
258 InstSelector =
259 std::make_unique<AMDGPUInstructionSelector>(*this, *RegBankInfo);
260}
261
263 return TSInfo.get();
264}
265
266unsigned GCNSubtarget::getConstantBusLimit(unsigned Opcode) const {
267 if (getGeneration() < GFX10)
268 return 1;
269
270 switch (Opcode) {
271 case AMDGPU::V_LSHLREV_B64_e64:
272 case AMDGPU::V_LSHLREV_B64_gfx10:
273 case AMDGPU::V_LSHLREV_B64_e64_gfx11:
274 case AMDGPU::V_LSHLREV_B64_e32_gfx12:
275 case AMDGPU::V_LSHLREV_B64_e64_gfx12:
276 case AMDGPU::V_LSHL_B64_e64:
277 case AMDGPU::V_LSHRREV_B64_e64:
278 case AMDGPU::V_LSHRREV_B64_gfx10:
279 case AMDGPU::V_LSHRREV_B64_e64_gfx11:
280 case AMDGPU::V_LSHRREV_B64_e64_gfx12:
281 case AMDGPU::V_LSHR_B64_e64:
282 case AMDGPU::V_ASHRREV_I64_e64:
283 case AMDGPU::V_ASHRREV_I64_gfx10:
284 case AMDGPU::V_ASHRREV_I64_e64_gfx11:
285 case AMDGPU::V_ASHRREV_I64_e64_gfx12:
286 case AMDGPU::V_ASHR_I64_e64:
287 return 1;
288 }
289
290 return 2;
291}
292
293/// This list was mostly derived from experimentation.
294bool GCNSubtarget::zeroesHigh16BitsOfDest(unsigned Opcode) const {
295 switch (Opcode) {
296 case AMDGPU::V_CVT_F16_F32_e32:
297 case AMDGPU::V_CVT_F16_F32_e64:
298 case AMDGPU::V_CVT_F16_U16_e32:
299 case AMDGPU::V_CVT_F16_U16_e64:
300 case AMDGPU::V_CVT_F16_I16_e32:
301 case AMDGPU::V_CVT_F16_I16_e64:
302 case AMDGPU::V_RCP_F16_e64:
303 case AMDGPU::V_RCP_F16_e32:
304 case AMDGPU::V_RSQ_F16_e64:
305 case AMDGPU::V_RSQ_F16_e32:
306 case AMDGPU::V_SQRT_F16_e64:
307 case AMDGPU::V_SQRT_F16_e32:
308 case AMDGPU::V_LOG_F16_e64:
309 case AMDGPU::V_LOG_F16_e32:
310 case AMDGPU::V_EXP_F16_e64:
311 case AMDGPU::V_EXP_F16_e32:
312 case AMDGPU::V_SIN_F16_e64:
313 case AMDGPU::V_SIN_F16_e32:
314 case AMDGPU::V_COS_F16_e64:
315 case AMDGPU::V_COS_F16_e32:
316 case AMDGPU::V_FLOOR_F16_e64:
317 case AMDGPU::V_FLOOR_F16_e32:
318 case AMDGPU::V_CEIL_F16_e64:
319 case AMDGPU::V_CEIL_F16_e32:
320 case AMDGPU::V_TRUNC_F16_e64:
321 case AMDGPU::V_TRUNC_F16_e32:
322 case AMDGPU::V_RNDNE_F16_e64:
323 case AMDGPU::V_RNDNE_F16_e32:
324 case AMDGPU::V_FRACT_F16_e64:
325 case AMDGPU::V_FRACT_F16_e32:
326 case AMDGPU::V_FREXP_MANT_F16_e64:
327 case AMDGPU::V_FREXP_MANT_F16_e32:
328 case AMDGPU::V_FREXP_EXP_I16_F16_e64:
329 case AMDGPU::V_FREXP_EXP_I16_F16_e32:
330 case AMDGPU::V_LDEXP_F16_e64:
331 case AMDGPU::V_LDEXP_F16_e32:
332 case AMDGPU::V_LSHLREV_B16_e64:
333 case AMDGPU::V_LSHLREV_B16_e32:
334 case AMDGPU::V_LSHRREV_B16_e64:
335 case AMDGPU::V_LSHRREV_B16_e32:
336 case AMDGPU::V_ASHRREV_I16_e64:
337 case AMDGPU::V_ASHRREV_I16_e32:
338 case AMDGPU::V_ADD_U16_e64:
339 case AMDGPU::V_ADD_U16_e32:
340 case AMDGPU::V_SUB_U16_e64:
341 case AMDGPU::V_SUB_U16_e32:
342 case AMDGPU::V_SUBREV_U16_e64:
343 case AMDGPU::V_SUBREV_U16_e32:
344 case AMDGPU::V_MUL_LO_U16_e64:
345 case AMDGPU::V_MUL_LO_U16_e32:
346 case AMDGPU::V_ADD_F16_e64:
347 case AMDGPU::V_ADD_F16_e32:
348 case AMDGPU::V_SUB_F16_e64:
349 case AMDGPU::V_SUB_F16_e32:
350 case AMDGPU::V_SUBREV_F16_e64:
351 case AMDGPU::V_SUBREV_F16_e32:
352 case AMDGPU::V_MUL_F16_e64:
353 case AMDGPU::V_MUL_F16_e32:
354 case AMDGPU::V_MAX_F16_e64:
355 case AMDGPU::V_MAX_F16_e32:
356 case AMDGPU::V_MIN_F16_e64:
357 case AMDGPU::V_MIN_F16_e32:
358 case AMDGPU::V_MAX_U16_e64:
359 case AMDGPU::V_MAX_U16_e32:
360 case AMDGPU::V_MIN_U16_e64:
361 case AMDGPU::V_MIN_U16_e32:
362 case AMDGPU::V_MAX_I16_e64:
363 case AMDGPU::V_MAX_I16_e32:
364 case AMDGPU::V_MIN_I16_e64:
365 case AMDGPU::V_MIN_I16_e32:
366 case AMDGPU::V_MAD_F16_e64:
367 case AMDGPU::V_MAD_U16_e64:
368 case AMDGPU::V_MAD_I16_e64:
369 case AMDGPU::V_FMA_F16_e64:
370 case AMDGPU::V_DIV_FIXUP_F16_e64:
371 // On gfx10, all 16-bit instructions preserve the high bits.
373 case AMDGPU::V_MADAK_F16:
374 case AMDGPU::V_MADMK_F16:
375 case AMDGPU::V_MAC_F16_e64:
376 case AMDGPU::V_MAC_F16_e32:
377 case AMDGPU::V_FMAMK_F16:
378 case AMDGPU::V_FMAAK_F16:
379 case AMDGPU::V_FMAC_F16_e64:
380 case AMDGPU::V_FMAC_F16_e32:
381 // In gfx9, the preferred handling of the unused high 16-bits changed. Most
382 // instructions maintain the legacy behavior of 0ing. Some instructions
383 // changed to preserving the high bits.
385 case AMDGPU::V_MAD_MIXLO_F16:
386 case AMDGPU::V_MAD_MIXHI_F16:
387 default:
388 return false;
389 }
390}
391
393 const SchedRegion &Region) const {
394 // Track register pressure so the scheduler can try to decrease
395 // pressure once register usage is above the threshold defined by
396 // SIRegisterInfo::getRegPressureSetLimit()
397 Policy.ShouldTrackPressure = true;
398
399 const Function &F = Region.RegionBegin->getMF()->getFunction();
400 if (AMDGPU::getSchedStrategy(F) == "coexec") {
401 Policy.OnlyTopDown = true;
402 Policy.OnlyBottomUp = false;
403 return;
404 }
405
406 // Enabling both top down and bottom up scheduling seems to give us less
407 // register spills than just using one of these approaches on its own.
408 Policy.OnlyTopDown = false;
409 Policy.OnlyBottomUp = false;
410
411 // Enabling ShouldTrackLaneMasks crashes the SI Machine Scheduler.
412 if (!enableSIScheduler())
413 Policy.ShouldTrackLaneMasks = true;
414}
415
417 const SchedRegion &Region) const {
418 const Function &F = Region.RegionBegin->getMF()->getFunction();
419 Attribute PostRADirectionAttr = F.getFnAttribute("amdgpu-post-ra-direction");
420 if (!PostRADirectionAttr.isValid())
421 return;
422
423 StringRef PostRADirectionStr = PostRADirectionAttr.getValueAsString();
424 if (PostRADirectionStr == "topdown") {
425 Policy.OnlyTopDown = true;
426 Policy.OnlyBottomUp = false;
427 } else if (PostRADirectionStr == "bottomup") {
428 Policy.OnlyTopDown = false;
429 Policy.OnlyBottomUp = true;
430 } else if (PostRADirectionStr == "bidirectional") {
431 Policy.OnlyTopDown = false;
432 Policy.OnlyBottomUp = false;
433 } else {
435 F, F.getSubprogram(), "invalid value for postRA direction attribute");
436 F.getContext().diagnose(Diag);
437 }
438
439 LLVM_DEBUG({
440 const char *DirStr = "default";
441 if (Policy.OnlyTopDown && !Policy.OnlyBottomUp)
442 DirStr = "topdown";
443 else if (!Policy.OnlyTopDown && Policy.OnlyBottomUp)
444 DirStr = "bottomup";
445 else if (!Policy.OnlyTopDown && !Policy.OnlyBottomUp)
446 DirStr = "bidirectional";
447
448 dbgs() << "Post-MI-sched direction (" << F.getName() << "): " << DirStr
449 << '\n';
450 });
451}
452
454 if (isWave32()) {
455 // Fix implicit $vcc operands after MIParser has verified that they match
456 // the instruction definitions.
457 for (auto &MBB : MF) {
458 for (auto &MI : MBB)
459 InstrInfo.fixImplicitOperands(MI);
460 }
461 }
462}
463
465 return InstrInfo.pseudoToMCOpcode(AMDGPU::V_MAD_F16_e64) != -1;
466}
467
469 return hasVGPRIndexMode() && (!hasMovrel() || EnableVGPRIndexMode);
470}
471
472bool GCNSubtarget::useAA() const { return UseAA; }
473
474unsigned GCNSubtarget::getOccupancyWithNumSGPRs(unsigned SGPRs) const {
476}
477
478unsigned
480 unsigned DynamicVGPRBlockSize) const {
482 DynamicVGPRBlockSize);
483}
484
485unsigned
486GCNSubtarget::getBaseReservedNumSGPRs(const bool HasFlatScratch) const {
488 return 2; // VCC. FLAT_SCRATCH and XNACK are no longer in SGPRs.
489
490 if (HasFlatScratch || HasArchitectedFlatScratch) {
492 return 6; // FLAT_SCRATCH, XNACK, VCC (in that order).
494 return 4; // FLAT_SCRATCH, VCC (in that order).
495 }
496
497 if (isXNACKEnabled())
498 return 4; // XNACK, VCC (in that order).
499 return 2; // VCC.
500}
501
506
508 // In principle we do not need to reserve SGPR pair used for flat_scratch if
509 // we know flat instructions do not access the stack anywhere in the
510 // program. For now assume it's needed if we have flat instructions.
511 const bool KernelUsesFlatScratch = hasFlatAddressSpace();
512 return getBaseReservedNumSGPRs(KernelUsesFlatScratch);
513}
514
515std::pair<unsigned, unsigned>
517 unsigned NumSGPRs, unsigned NumVGPRs) const {
518 unsigned DynamicVGPRBlockSize = AMDGPU::getDynamicVGPRBlockSize(F);
519 // Temporarily check both the attribute and the subtarget feature until the
520 // latter is removed.
521 if (DynamicVGPRBlockSize == 0 && isDynamicVGPREnabled())
522 DynamicVGPRBlockSize = getDynamicVGPRBlockSize();
523
524 auto [MinOcc, MaxOcc] = getOccupancyWithWorkGroupSizes(LDSSize, F);
525 unsigned SGPROcc = getOccupancyWithNumSGPRs(NumSGPRs);
526 unsigned VGPROcc = getOccupancyWithNumVGPRs(NumVGPRs, DynamicVGPRBlockSize);
527
528 // Maximum occupancy may be further limited by high SGPR/VGPR usage.
529 MaxOcc = std::min({MaxOcc, SGPROcc, VGPROcc});
530 return {std::min(MinOcc, MaxOcc), MaxOcc};
531}
532
534 const Function &F, std::pair<unsigned, unsigned> WavesPerEU,
535 unsigned PreloadedSGPRs, unsigned ReservedNumSGPRs) const {
536 // Compute maximum number of SGPRs function can use using default/requested
537 // minimum number of waves per execution unit.
538 unsigned MaxNumSGPRs = getMaxNumSGPRs(WavesPerEU.first, false);
539 unsigned MaxAddressableNumSGPRs = getMaxNumSGPRs(WavesPerEU.first, true);
540
541 // Check if maximum number of SGPRs was explicitly requested using
542 // "amdgpu-num-sgpr" attribute.
543 unsigned Requested =
544 F.getFnAttributeAsParsedInteger("amdgpu-num-sgpr", MaxNumSGPRs);
545
546 if (Requested != MaxNumSGPRs) {
547 // Make sure requested value does not violate subtarget's specifications.
548 if (Requested && (Requested <= ReservedNumSGPRs))
549 Requested = 0;
550
551 // If more SGPRs are required to support the input user/system SGPRs,
552 // increase to accommodate them.
553 //
554 // FIXME: This really ends up using the requested number of SGPRs + number
555 // of reserved special registers in total. Theoretically you could re-use
556 // the last input registers for these special registers, but this would
557 // require a lot of complexity to deal with the weird aliasing.
558 unsigned InputNumSGPRs = PreloadedSGPRs;
559 if (Requested && Requested < InputNumSGPRs)
560 Requested = InputNumSGPRs;
561
562 // Make sure requested value is compatible with values implied by
563 // default/requested minimum/maximum number of waves per execution unit.
564 if (Requested && Requested > getMaxNumSGPRs(WavesPerEU.first, false))
565 Requested = 0;
566 if (WavesPerEU.second && Requested &&
567 Requested < getMinNumSGPRs(WavesPerEU.second))
568 Requested = 0;
569
570 if (Requested)
571 MaxNumSGPRs = Requested;
572 }
573
574 if (hasSGPRInitBug())
576
577 return std::min(MaxNumSGPRs - ReservedNumSGPRs, MaxAddressableNumSGPRs);
578}
579
581 const Function &F = MF.getFunction();
585}
586
588 using USI = GCNUserSGPRUsageInfo;
589 // Max number of user SGPRs
590 const unsigned MaxUserSGPRs =
591 USI::getNumUserSGPRForField(USI::PrivateSegmentBufferID) +
592 USI::getNumUserSGPRForField(USI::DispatchPtrID) +
593 USI::getNumUserSGPRForField(USI::QueuePtrID) +
594 USI::getNumUserSGPRForField(USI::KernargSegmentPtrID) +
595 USI::getNumUserSGPRForField(USI::DispatchIdID) +
596 USI::getNumUserSGPRForField(USI::FlatScratchInitID) +
597 USI::getNumUserSGPRForField(USI::ImplicitBufferPtrID);
598
599 // Max number of system SGPRs
600 const unsigned MaxSystemSGPRs = 1 + // WorkGroupIDX
601 1 + // WorkGroupIDY
602 1 + // WorkGroupIDZ
603 1 + // WorkGroupInfo
604 1; // private segment wave byte offset
605
606 // Max number of synthetic SGPRs
607 const unsigned SyntheticSGPRs = 1; // LDSKernelId
608
609 return MaxUserSGPRs + MaxSystemSGPRs + SyntheticSGPRs;
610}
611
616
618 const Function &F, std::pair<unsigned, unsigned> NumVGPRBounds) const {
619 const auto [Min, Max] = NumVGPRBounds;
620
621 // Check if maximum number of VGPRs was explicitly requested using
622 // "amdgpu-num-vgpr" attribute.
623
624 unsigned Requested = F.getFnAttributeAsParsedInteger("amdgpu-num-vgpr", Max);
625 if (Requested != Max && hasGFX90AInsts())
626 Requested *= 2;
627
628 // Make sure requested value is inside the range of possible VGPR usage.
629 return std::clamp(Requested, Min, Max);
630}
631
633 // Temporarily check both the attribute and the subtarget feature, until the
634 // latter is removed.
635 unsigned DynamicVGPRBlockSize = AMDGPU::getDynamicVGPRBlockSize(F);
636 if (DynamicVGPRBlockSize == 0 && isDynamicVGPREnabled())
637 DynamicVGPRBlockSize = getDynamicVGPRBlockSize();
638
639 std::pair<unsigned, unsigned> Waves = getWavesPerEU(F);
640 return getBaseMaxNumVGPRs(
641 F, {getMinNumVGPRs(Waves.second, DynamicVGPRBlockSize),
642 getMaxNumVGPRs(Waves.first, DynamicVGPRBlockSize)});
643}
644
646 return getMaxNumVGPRs(MF.getFunction());
647}
648
649std::pair<unsigned, unsigned>
651 const unsigned MaxVectorRegs = getMaxNumVGPRs(F);
652
653 unsigned MaxNumVGPRs = MaxVectorRegs;
654 unsigned MaxNumAGPRs = 0;
655 unsigned NumArchVGPRs = getAddressableNumArchVGPRs();
656
657 // On GFX90A, the number of VGPRs and AGPRs need not be equal. Theoretically,
658 // a wave may have up to 512 total vector registers combining together both
659 // VGPRs and AGPRs. Hence, in an entry function without calls and without
660 // AGPRs used within it, it is possible to use the whole vector register
661 // budget for VGPRs.
662 //
663 // TODO: it shall be possible to estimate maximum AGPR/VGPR pressure and split
664 // register file accordingly.
665 if (hasGFX90AInsts()) {
666 unsigned MinNumAGPRs = 0;
667 const unsigned TotalNumAGPRs = AMDGPU::AGPR_32RegClass.getNumRegs();
668
669 const std::pair<unsigned, unsigned> DefaultNumAGPR = {~0u, ~0u};
670
671 // TODO: The lower bound should probably force the number of required
672 // registers up, overriding amdgpu-waves-per-eu.
673 std::tie(MinNumAGPRs, MaxNumAGPRs) =
674 AMDGPU::getIntegerPairAttribute(F, "amdgpu-agpr-alloc", DefaultNumAGPR,
675 /*OnlyFirstRequired=*/true);
676
677 if (MinNumAGPRs == DefaultNumAGPR.first) {
678 // Default to splitting half the registers if AGPRs are required.
679 MinNumAGPRs = MaxNumAGPRs = MaxVectorRegs / 2;
680 } else {
681 // Align to accum_offset's allocation granularity.
682 MinNumAGPRs = alignTo(MinNumAGPRs, 4);
683
684 MinNumAGPRs = std::min(MinNumAGPRs, TotalNumAGPRs);
685 }
686
687 // Clamp values to be inbounds of our limits, and ensure min <= max.
688
689 MaxNumAGPRs = std::min(std::max(MinNumAGPRs, MaxNumAGPRs), MaxVectorRegs);
690 MinNumAGPRs = std::min({MinNumAGPRs, TotalNumAGPRs, MaxNumAGPRs});
691
692 MaxNumVGPRs = std::min(MaxVectorRegs - MinNumAGPRs, NumArchVGPRs);
693 MaxNumAGPRs = std::min(MaxVectorRegs - MaxNumVGPRs, MaxNumAGPRs);
694
695 assert(MaxNumVGPRs + MaxNumAGPRs <= MaxVectorRegs &&
696 MaxNumAGPRs <= TotalNumAGPRs && MaxNumVGPRs <= NumArchVGPRs &&
697 "invalid register counts");
698 } else if (hasMAIInsts()) {
699 // On gfx908 the number of AGPRs always equals the number of VGPRs.
700 MaxNumAGPRs = MaxNumVGPRs = MaxVectorRegs;
701 }
702
703 return std::pair(MaxNumVGPRs, MaxNumAGPRs);
704}
705
706// Check to which source operand UseOpIdx points to and return a pointer to the
707// operand of the corresponding source modifier.
708// Return nullptr if UseOpIdx either doesn't point to src0/1/2 or if there is no
709// operand for the corresponding source modifier.
710static const MachineOperand *
712 const SIInstrInfo &InstrInfo) {
713 AMDGPU::OpName UseName =
714 AMDGPU::getOperandIdxName(UseI.getOpcode(), UseOpIdx);
715 switch (UseName) {
716 case AMDGPU::OpName::src0:
717 return InstrInfo.getNamedOperand(UseI, AMDGPU::OpName::src0_modifiers);
718 case AMDGPU::OpName::src1:
719 return InstrInfo.getNamedOperand(UseI, AMDGPU::OpName::src1_modifiers);
720 case AMDGPU::OpName::src2:
721 return InstrInfo.getNamedOperand(UseI, AMDGPU::OpName::src2_modifiers);
722 default:
723 return nullptr;
724 }
725}
726
727// Get the subreg idx of the subreg that is used by the given instruction
728// operand, considering the given op_sel modifier.
729// Return 0 if the whole register is used or as a conservative fallback.
731 const SIInstrInfo &InstrInfo,
732 const MachineInstr &I,
733 const MachineOperand &Op) {
734 if (!InstrInfo.isVOP3P(I) || InstrInfo.isWMMA(I) || InstrInfo.isSWMMAC(I))
735 return AMDGPU::NoSubRegister;
736
737 const MachineOperand *OpMod =
738 getVOP3PSourceModifierFromOpIdx(I, Op.getOperandNo(), InstrInfo);
739 if (!OpMod)
740 return AMDGPU::NoSubRegister;
741
742 // Note: the FMA_MIX* and MAD_MIX* instructions have different semantics for
743 // the op_sel and op_sel_hi source modifiers:
744 // - op_sel: selects low/high operand bits as input to the operation;
745 // has only meaning for 16-bit source operands
746 // - op_sel_hi: specifies the size of the source operands (16 or 32 bits);
747 // a value of 0 indicates 32 bit, 1 indicates 16 bit
748 // For the other VOP3P instructions, the semantics are:
749 // - op_sel: selects low/high operand bits as input to the operation which
750 // results in the lower-half of the destination
751 // - op_sel_hi: selects the low/high operand bits as input to the operation
752 // which results in the higher-half of the destination
753 int64_t OpSel = OpMod->getImm() & SISrcMods::OP_SEL_0;
754 int64_t OpSelHi = OpMod->getImm() & SISrcMods::OP_SEL_1;
755
756 // Check if all parts of the register are being used (= op_sel and op_sel_hi
757 // differ for VOP3P or op_sel_hi=0 for VOP3PMix). In that case we can return
758 // early.
759 if ((!InstrInfo.isVOP3PMix(I) && (!OpSel || !OpSelHi) &&
760 (OpSel || OpSelHi)) ||
761 (InstrInfo.isVOP3PMix(I) && !OpSelHi))
762 return AMDGPU::NoSubRegister;
763
764 const MachineRegisterInfo &MRI = I.getParent()->getParent()->getRegInfo();
765 const TargetRegisterClass *RC = TRI.getRegClassForOperandReg(MRI, Op);
766
767 if (unsigned SubRegIdx = OpSel ? AMDGPU::sub1 : AMDGPU::sub0;
768 TRI.getSubClassWithSubReg(RC, SubRegIdx) == RC)
769 return SubRegIdx;
770 if (unsigned SubRegIdx = OpSel ? AMDGPU::hi16 : AMDGPU::lo16;
771 TRI.getSubClassWithSubReg(RC, SubRegIdx) == RC)
772 return SubRegIdx;
773
774 return AMDGPU::NoSubRegister;
775}
776
777Register GCNSubtarget::getRealSchedDependency(const MachineInstr &DefI,
778 int DefOpIdx,
779 const MachineInstr &UseI,
780 int UseOpIdx) const {
781 const SIRegisterInfo *TRI = getRegisterInfo();
782 const MachineOperand &DefOp = DefI.getOperand(DefOpIdx);
783 const MachineOperand &UseOp = UseI.getOperand(UseOpIdx);
784 Register DefReg = DefOp.getReg();
785 Register UseReg = UseOp.getReg();
786
787 // If the registers aren't restricted to a sub-register, there is no point in
788 // further analysis. This check makes only sense for virtual registers because
789 // physical registers may form a tuple and thus be part of a superregister
790 // although they are not a subregister themselves (vgpr0 is a "subreg" of
791 // vgpr0_vgpr1 without being a subreg in itself).
792 unsigned DefSubRegIdx = DefOp.getSubReg();
793 if (DefReg.isVirtual() && DefSubRegIdx == AMDGPU::NoSubRegister)
794 return DefReg;
795 unsigned UseSubRegIdx = getEffectiveSubRegIdx(*TRI, InstrInfo, UseI, UseOp);
796 if (UseReg.isVirtual() && UseSubRegIdx == AMDGPU::NoSubRegister)
797 return DefReg;
798
799 if (!TRI->checkSubRegInterference(DefReg, DefSubRegIdx, UseReg, UseSubRegIdx))
800 return Register(); // No real dependency
801
802 // UseReg might be smaller or larger than DefReg, depending on the subreg and
803 // on whether DefReg is a subreg, too. -> Find the smaller one. This does not
804 // apply to virtual registers because we cannot construct a subreg for them.
805 if (DefReg.isVirtual())
806 return DefReg;
807 MCRegister DefMCReg =
808 DefSubRegIdx ? TRI->getSubReg(DefReg, DefSubRegIdx) : DefReg.asMCReg();
809 MCRegister UseMCReg =
810 UseSubRegIdx ? TRI->getSubReg(UseReg, UseSubRegIdx) : UseReg.asMCReg();
811 return TRI->isSubRegisterEq(DefMCReg, UseMCReg) ? UseMCReg : DefMCReg;
812}
813
815 SUnit *Def, int DefOpIdx, SUnit *Use, int UseOpIdx, SDep &Dep,
816 const TargetSchedModel *SchedModel) const {
817 if (Dep.getKind() != SDep::Kind::Data || !Dep.getReg() || !Def->isInstr() ||
818 !Use->isInstr())
819 return;
820
821 MachineInstr *DefI = Def->getInstr();
822 MachineInstr *UseI = Use->getInstr();
823
824 // Check for false latency on $tensorcnt / $asynccnt dependencies
825 if (Dep.getReg() == AMDGPU::TENSORcnt || Dep.getReg() == AMDGPU::ASYNCcnt) {
826 unsigned UseOp = UseI->getOpcode();
827 // Do not adjust latency for load->s_wait
828 bool IsBarrierCase =
829 InstrInfo.isLDSDMA(*DefI) &&
830 (UseOp == AMDGPU::S_WAIT_TENSORCNT || UseOp == AMDGPU::S_WAIT_ASYNCCNT);
831 if (!IsBarrierCase) {
832 Dep.setLatency(1);
833 return;
834 }
835 }
836
837 if (Register Reg = getRealSchedDependency(*DefI, DefOpIdx, *UseI, UseOpIdx)) {
838 Dep.setReg(Reg);
839 } else {
840 Dep = SDep(Def, SDep::Artificial);
841 return; // This is not a data dependency anymore.
842 }
843
844 if (DefI->isBundle()) {
846 auto Reg = Dep.getReg();
849 unsigned Lat = 0;
850 for (++I; I != E && I->isBundledWithPred(); ++I) {
851 if (I->isMetaInstruction())
852 continue;
853 if (I->modifiesRegister(Reg, TRI))
854 Lat = InstrInfo.getInstrLatency(getInstrItineraryData(), *I);
855 else if (Lat)
856 --Lat;
857 }
858 Dep.setLatency(Lat);
859 } else if (UseI->isBundle()) {
861 auto Reg = Dep.getReg();
864 unsigned Lat = InstrInfo.getInstrLatency(getInstrItineraryData(), *DefI);
865 for (++I; I != E && I->isBundledWithPred() && Lat; ++I) {
866 if (I->isMetaInstruction())
867 continue;
868 if (I->readsRegister(Reg, TRI))
869 break;
870 --Lat;
871 }
872 Dep.setLatency(Lat);
873 } else if (Dep.getLatency() == 0 && Dep.getReg() == AMDGPU::VCC_LO) {
874 // Work around the fact that SIInstrInfo::fixImplicitOperands modifies
875 // implicit operands which come from the MCInstrDesc, which can fool
876 // ScheduleDAGInstrs::addPhysRegDataDeps into treating them as implicit
877 // pseudo operands.
878 Dep.setLatency(InstrInfo.getSchedModel().computeOperandLatency(
879 DefI, DefOpIdx, UseI, UseOpIdx));
880 }
881}
882
885 return 0; // Not MIMG encoding.
886
887 if (NSAThreshold.getNumOccurrences() > 0)
888 return std::max(NSAThreshold.getValue(), 2u);
889
891 "amdgpu-nsa-threshold", -1);
892 if (Value > 0)
893 return std::max(Value, 2);
894
895 return NSAThreshold;
896}
897
899 const GCNSubtarget &ST)
900 : ST(ST) {
901 const CallingConv::ID CC = F.getCallingConv();
902 const bool IsKernel =
904
905 if (IsKernel && (!F.arg_empty() || ST.getImplicitArgNumBytes(F) != 0))
906 KernargSegmentPtr = true;
907
908 bool IsAmdHsaOrMesa = ST.isAmdHsaOrMesa(F);
909 if (IsAmdHsaOrMesa && !ST.hasFlatScratchEnabled())
910 PrivateSegmentBuffer = true;
911 else if (ST.isMesaGfxShader(F))
912 ImplicitBufferPtr = true;
913
914 if (!AMDGPU::isGraphics(CC)) {
915 if (!F.hasFnAttribute("amdgpu-no-dispatch-ptr"))
916 DispatchPtr = true;
917
918 // FIXME: Can this always be disabled with < COv5?
919 if (!F.hasFnAttribute("amdgpu-no-queue-ptr"))
920 QueuePtr = true;
921
922 if (!F.hasFnAttribute("amdgpu-no-dispatch-id"))
923 DispatchID = true;
924 }
925
926 if (ST.hasFlatAddressSpace() && AMDGPU::isEntryFunctionCC(CC) &&
927 (IsAmdHsaOrMesa || ST.hasFlatScratchEnabled()) &&
928 // FlatScratchInit cannot be true for graphics CC if
929 // hasFlatScratchEnabled() is false.
930 (ST.hasFlatScratchEnabled() ||
931 (!AMDGPU::isGraphics(CC) &&
932 !F.hasFnAttribute("amdgpu-no-flat-scratch-init"))) &&
933 !ST.hasArchitectedFlatScratch()) {
934 FlatScratchInit = true;
935 }
936
938 NumUsedUserSGPRs += getNumUserSGPRForField(ImplicitBufferPtrID);
939
942
943 if (hasDispatchPtr())
944 NumUsedUserSGPRs += getNumUserSGPRForField(DispatchPtrID);
945
946 if (hasQueuePtr())
947 NumUsedUserSGPRs += getNumUserSGPRForField(QueuePtrID);
948
950 NumUsedUserSGPRs += getNumUserSGPRForField(KernargSegmentPtrID);
951
952 if (hasDispatchID())
953 NumUsedUserSGPRs += getNumUserSGPRForField(DispatchIdID);
954
955 if (hasFlatScratchInit())
956 NumUsedUserSGPRs += getNumUserSGPRForField(FlatScratchInitID);
957
959 NumUsedUserSGPRs += getNumUserSGPRForField(PrivateSegmentSizeID);
960}
961
963 assert(NumKernargPreloadSGPRs + NumSGPRs <= AMDGPU::getMaxNumUserSGPRs(ST));
964 NumKernargPreloadSGPRs += NumSGPRs;
965 NumUsedUserSGPRs += NumSGPRs;
966}
967
969 return AMDGPU::getMaxNumUserSGPRs(ST) - NumUsedUserSGPRs;
970}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static cl::opt< bool > UseAA("aarch64-use-aa", cl::init(true), cl::desc("Enable the use of AA during codegen."))
This file describes how to lower LLVM calls to machine code calls.
This file declares the targeting of the InstructionSelector class for AMDGPU.
This file declares the targeting of the Machinelegalizer class for AMDGPU.
This file declares the targeting of the RegisterBankInfo class for AMDGPU.
static cl::opt< bool > SramEccSetting("amdgpu-sramecc", cl::desc("Force amdgpu.sramecc for testing"), cl::ReallyHidden)
static cl::opt< bool > XnackSetting("amdgpu-xnack", cl::desc("Force amdgpu.xnack value for testing"), cl::ReallyHidden)
The AMDGPU TargetMachine interface definition for hw codegen targets.
MachineBasicBlock & MBB
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static AMDGPUSubtarget::Generation computeDefaultGeneration(const Triple &TT)
static cl::opt< unsigned > NSAThreshold("amdgpu-nsa-threshold", cl::desc("Number of addresses from which to enable MIMG NSA."), cl::init(2), cl::Hidden)
static cl::opt< bool > EnableVGPRIndexMode("amdgpu-vgpr-index-mode", cl::desc("Use GPR indexing mode instead of movrel for vector indexing"), cl::init(false))
static cl::opt< bool > UseAA("amdgpu-use-aa-in-codegen", cl::desc("Enable the use of AA during codegen."), cl::init(true))
static const MachineOperand * getVOP3PSourceModifierFromOpIdx(const MachineInstr &UseI, int UseOpIdx, const SIInstrInfo &InstrInfo)
static unsigned getEffectiveSubRegIdx(const SIRegisterInfo &TRI, const SIInstrInfo &InstrInfo, const MachineInstr &I, const MachineOperand &Op)
AMD GCN specific subclass of TargetSubtarget.
static Register UseReg(const MachineOperand &MO)
IRTranslator LLVM IR MI
This file describes how to lower LLVM inline asm to machine code INLINEASM.
static bool hasFeature(StringRef Feature, const FeatureBitset &FeatureBits, ArrayRef< SubtargetFeatureKV > ProcFeatures)
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
if(PassOpts->AAPipeline)
This file defines the SmallString class.
#define LLVM_DEBUG(...)
Definition Debug.h:119
std::pair< unsigned, unsigned > getWavesPerEU(const Function &F) const
std::pair< unsigned, unsigned > getOccupancyWithWorkGroupSizes(uint32_t LDSBytes, const Function &F) const
Subtarget's minimum/maximum occupancy, in number of waves per EU, that can be achieved when the only ...
unsigned getWavefrontSizeLog2() const
AMDGPUSubtarget(const Triple &TT)
unsigned AddressableLocalMemorySize
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:105
LLVM_ABI StringRef getValueAsString() const
Return the attribute's value as a string.
bool isValid() const
Return true if the attribute is any kind of attribute.
Definition Attributes.h:261
Diagnostic information for optimization failures.
Diagnostic information for unsupported feature in backend.
uint64_t getFnAttributeAsParsedInteger(StringRef Kind, uint64_t Default=0) const
For a string attribute Kind, parse attribute as an integer.
Definition Function.cpp:770
bool hasFlat() const
InstrItineraryData InstrItins
bool useVGPRIndexMode() const
void mirFileLoaded(MachineFunction &MF) const override
unsigned MaxPrivateElementSize
unsigned getAddressableNumArchVGPRs() const
unsigned getMinNumSGPRs(unsigned WavesPerEU) const
void ParseSubtargetFeatures(StringRef CPU, StringRef TuneCPU, StringRef FS)
unsigned getConstantBusLimit(unsigned Opcode) const
const InstrItineraryData * getInstrItineraryData() const override
void adjustSchedDependency(SUnit *Def, int DefOpIdx, SUnit *Use, int UseOpIdx, SDep &Dep, const TargetSchedModel *SchedModel) const override
void overridePostRASchedPolicy(MachineSchedPolicy &Policy, const SchedRegion &Region) const override
Align getStackAlignment() const
const bool BufferOOBRelaxed
bool hasMadF16() const
unsigned getMinNumVGPRs(unsigned WavesPerEU, unsigned DynamicVGPRBlockSize) const
bool isDynamicVGPREnabled() const
const SIRegisterInfo * getRegisterInfo() const override
unsigned getBaseMaxNumVGPRs(const Function &F, std::pair< unsigned, unsigned > NumVGPRBounds) const
bool zeroesHigh16BitsOfDest(unsigned Opcode) const
Returns if the result of this instruction with a 16-bit result returned in a 32-bit register implicit...
unsigned getBaseMaxNumSGPRs(const Function &F, std::pair< unsigned, unsigned > WavesPerEU, unsigned PreloadedSGPRs, unsigned ReservedNumSGPRs) const
unsigned getMaxNumPreloadedSGPRs() const
GCNSubtarget & initializeSubtargetDependencies(const Triple &TT, StringRef GPU, StringRef FS)
void overrideSchedPolicy(MachineSchedPolicy &Policy, const SchedRegion &Region) const override
std::pair< unsigned, unsigned > computeOccupancy(const Function &F, unsigned LDSSize=0, unsigned NumSGPRs=0, unsigned NumVGPRs=0) const
Subtarget's minimum/maximum occupancy, in number of waves per EU, that can be achieved when the only ...
unsigned getMaxNumVGPRs(unsigned WavesPerEU, unsigned DynamicVGPRBlockSize) const
AMDGPU::TargetID TargetID
const SITargetLowering * getTargetLowering() const override
unsigned getNSAThreshold(const MachineFunction &MF) const
GCNSubtarget(const Triple &TT, StringRef GPU, StringRef FS, const GCNTargetMachine &TM, bool BufferOOBRelaxed=false, bool TBufferOOBRelaxed=false, AMDGPU::TargetIDSetting XnackSetting=AMDGPU::TargetIDSetting::Any, AMDGPU::TargetIDSetting SramEccSetting=AMDGPU::TargetIDSetting::Any)
unsigned getReservedNumSGPRs(const MachineFunction &MF) const
const bool TBufferOOBRelaxed
bool useAA() const override
bool isWave32() const
unsigned getOccupancyWithNumVGPRs(unsigned VGPRs, unsigned DynamicVGPRBlockSize) const
Return the maximum number of waves per SIMD for kernels using VGPRs VGPRs.
unsigned InstCacheLineSize
unsigned getOccupancyWithNumSGPRs(unsigned SGPRs) const
Return the maximum number of waves per SIMD for kernels using SGPRs SGPRs.
Generation getGeneration() const
unsigned getMaxNumSGPRs(unsigned WavesPerEU, bool Addressable) const
std::pair< unsigned, unsigned > getMaxNumVectorRegs(const Function &F) const
Return a pair of maximum numbers of VGPRs and AGPRs that meet the number of waves per execution unit ...
bool isXNACKEnabled() const
unsigned getBaseReservedNumSGPRs(const bool HasFlatScratch) const
bool hasAddr64() const
unsigned getDynamicVGPRBlockSize() const
void checkSubtargetFeatures(const Function &F) const
Diagnose inconsistent subtarget features before attempting to codegen function F.
~GCNSubtarget() override
const SelectionDAGTargetInfo * getSelectionDAGInfo() const override
static unsigned getNumUserSGPRForField(UserSGPRID ID)
void allocKernargPreloadSGPRs(unsigned NumSGPRs)
bool hasPrivateSegmentBuffer() const
GCNUserSGPRUsageInfo(const Function &F, const GCNSubtarget &ST)
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Instructions::const_iterator const_instr_iterator
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
bool isBundle() const
const MachineOperand & getOperand(unsigned i) const
MachineOperand class - Representation of each machine instruction operand.
unsigned getSubReg() const
int64_t getImm() const
Register getReg() const
getReg - Returns the register number.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
Wrapper class representing virtual and physical registers.
Definition Register.h:20
MCRegister asMCReg() const
Utility to check-convert this value to a MCRegister.
Definition Register.h:107
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
Definition Register.h:79
Scheduling dependency.
Definition ScheduleDAG.h:52
Kind getKind() const
Returns an enum value representing the kind of the dependence.
@ Data
Regular data dependence (aka true-dependence).
Definition ScheduleDAG.h:56
void setLatency(unsigned Lat)
Sets the latency for this edge.
@ Artificial
Arbitrary strong DAG edge (no real dependence).
Definition ScheduleDAG.h:75
unsigned getLatency() const
Returns the latency value for this edge, which roughly means the minimum number of cycles that must e...
Register getReg() const
Returns the register associated with this edge.
void setReg(Register Reg)
Assigns the associated register for this edge.
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
std::pair< unsigned, unsigned > getWavesPerEU() const
GCNUserSGPRUsageInfo & getUserSGPRInfo()
Scheduling unit. This is a node in the scheduling DAG.
Targets can subclass this to parameterize the SelectionDAG lowering and instruction selection process...
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Information about stack frame layout on the target.
Provide an instruction scheduling machine model to CodeGen passes.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
@ AMDGPUSubArch9
Definition Triple.h:218
@ AMDGPUSubArch9_4
Definition Triple.h:231
@ AMDGPUSubArch6
Definition Triple.h:196
@ AMDGPUSubArch10_3
Definition Triple.h:241
@ AMDGPUSubArch90A
Definition Triple.h:229
@ AMDGPUSubArch810
Definition Triple.h:216
@ AMDGPUSubArch11
Definition Triple.h:250
@ AMDGPUSubArch7
Definition Triple.h:201
@ AMDGPUSubArch12_5
Definition Triple.h:270
@ AMDGPUSubArch10_1
Definition Triple.h:235
@ AMDGPUSubArch11_7
Definition Triple.h:261
@ AMDGPUSubArch8
Definition Triple.h:209
@ AMDGPUSubArch13
Definition Triple.h:274
@ AMDGPUSubArch12
Definition Triple.h:266
@ AMDGPUSubArch908
Definition Triple.h:228
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM Value Representation.
Definition Value.h:75
self_iterator getIterator()
Definition ilist_node.h:123
unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI, unsigned NumVGPRs, unsigned DynamicVGPRBlockSize)
unsigned getEUsPerCU(const MCSubtargetInfo &STI)
unsigned getOccupancyWithNumSGPRs(unsigned SGPRs, unsigned MaxWaves, unsigned TotalNumSGPRs, unsigned Granule, unsigned TrapReserve)
unsigned getMaxWavesPerEU(const MCSubtargetInfo &STI)
unsigned getLocalMemorySize(const MCSubtargetInfo &STI)
StringRef getSchedStrategy(const Function &F)
unsigned getMaxNumUserSGPRs(const MCSubtargetInfo &STI)
unsigned getLdsDwGranularity(const MCSubtargetInfo &ST)
LLVM_READNONE constexpr bool isEntryFunctionCC(CallingConv::ID CC)
unsigned getDynamicVGPRBlockSize(const Function &F)
LLVM_ABI Triple::SubArchType getMajorSubArch(Triple::SubArchType SubArch)
std::pair< unsigned, unsigned > getIntegerPairAttribute(const Function &F, StringRef Name, std::pair< unsigned, unsigned > Default, bool OnlyFirstRequired)
LLVM_READNONE constexpr bool isGraphics(CallingConv::ID CC)
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ SPIR_KERNEL
Used for SPIR kernel functions.
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
DWARFExpression::Operation Op
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Define a generic scheduling policy for targets that don't provide their own MachineSchedStrategy.
bool ShouldTrackLaneMasks
Track LaneMasks to allow reordering of independent subregister writes of the same vreg.
A region of an MBB for scheduling.