LLVM 24.0.0git
AMDGPUAsmPrinter.cpp
Go to the documentation of this file.
1//===-- AMDGPUAsmPrinter.cpp - AMDGPU assembly printer --------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10///
11/// The AMDGPUAsmPrinter is used to print both assembly string and also binary
12/// code. When passed an MCAsmStreamer it prints assembly and when passed
13/// an MCObjectStreamer it outputs binary code.
14//
15//===----------------------------------------------------------------------===//
16//
17
18#include "AMDGPUAsmPrinter.h"
19#include "AMDGPU.h"
23#include "AMDGPUTargetMachine.h"
24#include "GCNSubtarget.h"
29#include "R600AsmPrinter.h"
43#include "llvm/MC/MCAssembler.h"
44#include "llvm/MC/MCContext.h"
46#include "llvm/MC/MCStreamer.h"
47#include "llvm/MC/MCValue.h"
54
55using namespace llvm;
56using namespace llvm::AMDGPU;
57
58// This should get the default rounding mode from the kernel. We just set the
59// default here, but this could change if the OpenCL rounding mode pragmas are
60// used.
61//
62// The denormal mode here should match what is reported by the OpenCL runtime
63// for the CL_FP_DENORM bit from CL_DEVICE_{HALF|SINGLE|DOUBLE}_FP_CONFIG, but
64// can also be override to flush with the -cl-denorms-are-zero compiler flag.
65//
66// AMD OpenCL only sets flush none and reports CL_FP_DENORM for double
67// precision, and leaves single precision to flush all and does not report
68// CL_FP_DENORM for CL_DEVICE_SINGLE_FP_CONFIG. Mesa's OpenCL currently reports
69// CL_FP_DENORM for both.
70//
71// FIXME: It seems some instructions do not support single precision denormals
72// regardless of the mode (exp_*_f32, rcp_*_f32, rsq_*_f32, rsq_*f32, sqrt_f32,
73// and sin_f32, cos_f32 on most parts).
74
75// We want to use these instructions, and using fp32 denormals also causes
76// instructions to run at the double precision rate for the device so it's
77// probably best to just report no single precision denormals.
84
85static AsmPrinter *
87 std::unique_ptr<MCStreamer> &&Streamer) {
88 return new AMDGPUAsmPrinter(tm, std::move(Streamer));
89}
90
100
101namespace {
102class AMDGPUAsmPrinterHandler : public AsmPrinterHandler {
103protected:
104 AMDGPUAsmPrinter *Asm;
105
106public:
107 AMDGPUAsmPrinterHandler(AMDGPUAsmPrinter *A) : Asm(A) {}
108
109 void beginFunction(const MachineFunction *MF) override {}
110
111 void endFunction(const MachineFunction *MF) override { Asm->endFunction(MF); }
112
113 void endModule() override {}
114};
115} // End anonymous namespace
116
118 std::unique_ptr<MCStreamer> Streamer)
120 assert(OutStreamer && "AsmPrinter constructed without streamer");
123 if (auto *ResourceUsageW =
125 return &ResourceUsageW->getResourceInfo();
126 return nullptr;
127 };
128}
129
131 return "AMDGPU Assembly Printer";
132}
133
135 return &TM.getMCSubtargetInfo();
136}
137
139 if (!OutStreamer)
140 return nullptr;
141 return static_cast<AMDGPUTargetStreamer *>(OutStreamer->getTargetStreamer());
142}
143
147
148void AMDGPUAsmPrinter::initTargetStreamer(Module &M) {
150
151 // TODO: Which one is called first, emitStartOfAsmFile or
152 // emitFunctionBodyStart?
153 if (getTargetStreamer() && !getTargetStreamer()->getTargetID())
154 initializeTargetID(M);
155
156 const Triple &TT = M.getTargetTriple();
157 if (TT.getOS() != Triple::AMDHSA && TT.getOS() != Triple::AMDPAL)
158 return;
159
161
162 if (TT.getOS() == Triple::AMDHSA) {
164 CodeObjectVersion);
165 HSAMetadataStream->begin(M, *getTargetStreamer()->getTargetID());
166 }
167
168 if (TT.getOS() == Triple::AMDPAL)
170}
171
173 // Init target streamer if it has not yet happened
175 initTargetStreamer(M);
176
177 const Triple &TT = M.getTargetTriple();
178 if (TT.getOS() != Triple::AMDHSA)
180
181 // Emit HSA Metadata (NT_AMD_AMDGPU_HSA_METADATA).
182 // Emit HSA Metadata (NT_AMD_HSA_METADATA).
183 if (TT.getOS() == Triple::AMDHSA) {
184 HSAMetadataStream->end();
185 bool Success = HSAMetadataStream->emitTo(*getTargetStreamer());
186 (void)Success;
187 assert(Success && "Malformed HSA Metadata");
188 }
189}
190
192 const SIMachineFunctionInfo &MFI = *MF->getInfo<SIMachineFunctionInfo>();
193 const GCNSubtarget &STM = MF->getSubtarget<GCNSubtarget>();
194 const Function &F = MF->getFunction();
195
196 // TODO: We're checking this late, would be nice to check it earlier.
197 if (STM.requiresCodeObjectV6() && CodeObjectVersion < AMDGPU::AMDHSA_COV6) {
199 STM.getCPU() + " is only available on code object version 6 or better");
200 }
201
202 // TODO: Which one is called first, emitStartOfAsmFile or
203 // emitFunctionBodyStart?
204 if (!getTargetStreamer()->getTargetID())
205 initializeTargetID(*F.getParent());
206
207 if (!MFI.isEntryFunction())
208 return;
209
210 if (STM.isMesaKernel(F) &&
211 (F.getCallingConv() == CallingConv::AMDGPU_KERNEL ||
212 F.getCallingConv() == CallingConv::SPIR_KERNEL)) {
213 AMDGPUMCKernelCodeT KernelCode;
214 getAmdKernelCode(KernelCode, CurrentProgramInfo, *MF);
215 KernelCode.validate(&STM, MF->getContext());
217 }
218
219 if (STM.isAmdHsaOS())
220 HSAMetadataStream->emitKernel(*MF, CurrentProgramInfo);
221}
222
223/// Set bits in a kernel descriptor MCExpr field:
224/// return ((Dst & ~Mask) | (Value << Shift))
225static const MCExpr *setBits(const MCExpr *Dst, const MCExpr *Value,
226 uint32_t Mask, uint32_t Shift, MCContext &Ctx) {
227 const auto *Shft = MCConstantExpr::create(Shift, Ctx);
228 const auto *Msk = MCConstantExpr::create(Mask, Ctx);
229 Dst = MCBinaryExpr::createAnd(Dst, MCUnaryExpr::createNot(Msk, Ctx), Ctx);
231 Ctx);
232 return Dst;
233}
234
236 const SIMachineFunctionInfo &MFI = *MF->getInfo<SIMachineFunctionInfo>();
237 if (!MFI.isEntryFunction())
238 return;
239
240 assert(TM.getTargetTriple().getOS() == Triple::AMDHSA);
241
242 const GCNSubtarget &STM = MF->getSubtarget<GCNSubtarget>();
243 MCContext &Ctx = MF->getContext();
244
246 getAmdhsaKernelDescriptor(*MF, CurrentProgramInfo);
247
248 // Compute inst_pref_size using MCExpr label subtraction for exact code
249 // size. At this point .Lfunc_end has been emitted (by the base AsmPrinter)
250 // right after the function code, so (Lfunc_end - func_sym) gives the
251 // exact function code size in bytes.
252 if (STM.hasInstPrefSize()) {
253 const MCExpr *CodeSizeExpr = MCBinaryExpr::createSub(
256
257 uint32_t Mask, Shift, Width, CacheLineSize;
258 STM.getInstPrefSizeArgs(Mask, Shift, Width, CacheLineSize);
259 const MCExpr *InstPrefSize =
260 AMDGPUMCExpr::createInstPrefSize(CodeSizeExpr, Ctx);
262 setBits(KD.compute_pgm_rsrc3, InstPrefSize, Mask, Shift, Ctx);
263 }
264
266 auto &Context = Streamer.getContext();
267 auto &ObjectFileInfo = *Context.getObjectFileInfo();
268 auto &ReadOnlySection = *ObjectFileInfo.getReadOnlySection();
269
270 Streamer.pushSection();
271 Streamer.switchSection(&ReadOnlySection);
272
273 // CP microcode requires the kernel descriptor to be allocated on 64 byte
274 // alignment.
275 Streamer.emitValueToAlignment(Align(64), 0, 1, 0);
276 ReadOnlySection.ensureMinAlignment(Align(64));
277
278 SmallString<128> KernelName;
279 getNameWithPrefix(KernelName, &MF->getFunction());
281 STM, KernelName, KD, CurrentProgramInfo.NumVGPRsForWavesPerEU,
283 CurrentProgramInfo.NumSGPRsForWavesPerEU,
285 CurrentProgramInfo.VCCUsed, CurrentProgramInfo.FlatUsed,
286 getTargetStreamer()->getTargetID()->isXnackOnOrAny(), Context),
287 Context),
288 CurrentProgramInfo.VCCUsed, CurrentProgramInfo.FlatUsed);
289
290 Streamer.popSection();
291}
292
294 Register RegNo = MI->getOperand(0).getReg();
295
297 raw_svector_ostream OS(Str);
298 OS << "implicit-def: "
299 << printReg(RegNo, MF->getSubtarget().getRegisterInfo());
300
301 if (MI->getAsmPrinterFlags() & AMDGPU::SGPR_SPILL)
302 OS << " : SGPR spill to VGPR lane";
303
304 OutStreamer->AddComment(OS.str());
305 OutStreamer->addBlankLine();
306}
307
309 if (TM.getTargetTriple().getOS() == Triple::AMDHSA) {
311 return;
312 }
313
314 const SIMachineFunctionInfo *MFI = MF->getInfo<SIMachineFunctionInfo>();
315 const GCNSubtarget &STM = MF->getSubtarget<GCNSubtarget>();
316 if (MFI->isEntryFunction() && STM.isAmdHsaOrMesa(MF->getFunction())) {
317 SmallString<128> SymbolName;
318 getNameWithPrefix(SymbolName, &MF->getFunction()),
321 }
322 if (DumpCodeInstEmitter) {
323 // Disassemble function name label to text.
324 DisasmLines.push_back(MF->getName().str() + ":");
325 DisasmLineMaxLen = std::max(DisasmLineMaxLen, DisasmLines.back().size());
326 HexLines.emplace_back("");
327 }
328
330}
331
333 if (DumpCodeInstEmitter && !isBlockOnlyReachableByFallthrough(&MBB)) {
334 // Write a line for the basic block label if it is not only fallthrough.
335 DisasmLines.push_back((Twine("BB") + Twine(getFunctionNumber()) + "_" +
336 Twine(MBB.getNumber()) + ":")
337 .str());
338 DisasmLineMaxLen = std::max(DisasmLineMaxLen, DisasmLines.back().size());
339 HexLines.emplace_back("");
340 }
342}
343
346 if (GV->hasInitializer() && !isa<UndefValue>(GV->getInitializer())) {
347 OutContext.reportError({},
348 Twine(GV->getName()) +
349 ": unsupported initializer for address space");
350 return;
351 }
352
353 const Triple::OSType OS = TM.getTargetTriple().getOS();
354 if (OS == Triple::AMDHSA || OS == Triple::AMDPAL) {
356 return;
357 // With object linking, LDS definitions should have been externalized
358 // by earlier passes (e.g. LDS lowering, named barrier lowering).
359 // Only declarations reach here, emitted as SHN_AMDGPU_LDS symbols
360 // so the linker can assign their offsets.
361 assert(GV->isDeclaration() &&
362 "LDS definitions should have been externalized when object "
363 "linking is enabled");
364 }
365
366 MCSymbol *GVSym = getSymbol(GV);
367
368 GVSym->redefineIfPossible();
369 if (GVSym->isDefined() || GVSym->isVariable())
370 report_fatal_error("symbol '" + Twine(GVSym->getName()) +
371 "' is already defined");
372
373 const DataLayout &DL = GV->getDataLayout();
374 uint64_t Size = GV->getGlobalSize(DL);
375 Align Alignment = GV->getAlign().value_or(Align(4));
376
377 emitVisibility(GVSym, GV->getVisibility(), !GV->isDeclaration());
378 emitLinkage(GV, GVSym);
379 auto *TS = getTargetStreamer();
380 TS->emitAMDGPULDS(GVSym, Size, Alignment);
381 return;
382 }
383
385}
386
388 const Triple &TT = M.getTargetTriple();
389 if (TT.getSubArch() == Triple::NoSubArch) {
390 Triple::SubArchType SubArch =
392 if (SubArch != Triple::NoSubArch) {
393 Triple Fixed(TT);
394 Fixed.setArch(Triple::amdgpu, SubArch);
395 M.getContext().diagnose(DiagnosticInfoGeneric(
396 "codegen with no subarch in the target triple is deprecated and will "
397 "become an error; use the target triple '" +
398 Fixed.str() + "' instead",
399 DS_Warning));
400 } else {
401 M.getContext().diagnose(DiagnosticInfoGeneric(
402 "codegen with no subarch in the target triple is deprecated and will "
403 "become an error",
404 DS_Warning));
405 }
406 }
407
408 CodeObjectVersion = AMDGPU::getAMDHSACodeObjectVersion(M);
409
410 if (TT.getOS() == Triple::AMDHSA) {
411 switch (CodeObjectVersion) {
413 HSAMetadataStream = std::make_unique<HSAMD::MetadataStreamerMsgPackV4>();
414 break;
416 HSAMetadataStream = std::make_unique<HSAMD::MetadataStreamerMsgPackV5>();
417 break;
419 HSAMetadataStream = std::make_unique<HSAMD::MetadataStreamerMsgPackV6>();
420 break;
421 default:
422 reportFatalUsageError("unsupported code object version");
423 }
424
425 addAsmPrinterHandler(std::make_unique<AMDGPUAsmPrinterHandler>(this));
426 }
427
429}
430
431/// Mimics GCNSubtarget::computeOccupancy for MCExpr.
432///
433/// Remove dependency on GCNSubtarget and depend only only the necessary values
434/// for said occupancy computation. Should match computeOccupancy implementation
435/// without passing \p STM on.
436const AMDGPUMCExpr *createOccupancy(unsigned InitOcc, const MCExpr *NumSGPRs,
437 const MCExpr *NumVGPRs,
438 unsigned DynamicVGPRBlockSize,
439 const GCNSubtarget &STM, MCContext &Ctx) {
440 unsigned MaxWaves = STM.getMaxWavesPerEU();
441 unsigned Granule = IsaInfo::getVGPRAllocGranule(STM, DynamicVGPRBlockSize);
442 unsigned TargetTotalNumVGPRs = STM.getTotalNumVGPRs();
443
444 // Bake the per-function SGPR budget into the operands so the late-evaluated
445 // MCExpr stays arithmetic. The trap reservation in particular is implicit on
446 // amdhsa and lives on STM, not on the assembler's MCSubtargetInfo.
448 unsigned SGPRTotal = AMDGPU::getTotalNumSGPRs(Kind);
449 unsigned SGPRGranule = AMDGPU::getSGPRAllocGranule(Kind);
450 unsigned SGPRTrapReserve = STM.hasTrapHandler() ? IsaInfo::TRAP_NUM_SGPRS : 0;
451
452 auto CreateExpr = [&Ctx](unsigned Value) {
453 return MCConstantExpr::create(Value, Ctx);
454 };
455
456 // Zero SGPR count when SGPRs don't limit occupancy, so the MCExpr skips the
457 // SGPR term without having to test the generation itself.
458 const MCExpr *SGPRArg =
459 IsaInfo::isSGPROccupancyLimited(STM) ? NumSGPRs : CreateExpr(0);
460
462 {CreateExpr(MaxWaves), CreateExpr(Granule),
463 CreateExpr(TargetTotalNumVGPRs),
464 CreateExpr(InitOcc), CreateExpr(SGPRTotal),
465 CreateExpr(SGPRGranule),
466 CreateExpr(SGPRTrapReserve), SGPRArg, NumVGPRs},
467 Ctx);
468}
469
470void AMDGPUAsmPrinter::validateMCResourceInfo(Function &F) {
471 if (F.isDeclaration() || !AMDGPU::isModuleEntryFunctionCC(F.getCallingConv()))
472 return;
473
475 const GCNSubtarget &STM = TM.getSubtarget<GCNSubtarget>(F);
476 MCSymbol *FnSym = TM.getSymbol(&F);
477
478 auto TryGetMCExprValue = [](const MCExpr *Value, uint64_t &Res) -> bool {
479 int64_t Val;
480 if (Value->evaluateAsAbsolute(Val)) {
481 Res = Val;
482 return true;
483 }
484 return false;
485 };
486
487 const uint64_t MaxScratchPerWorkitem =
489 MCSymbol *ScratchSizeSymbol =
490 RI.getSymbol(FnSym->getName(), RIK::RIK_PrivateSegSize, OutContext);
491 uint64_t ScratchSize;
492 if (ScratchSizeSymbol->isVariable() &&
493 TryGetMCExprValue(ScratchSizeSymbol->getVariableValue(), ScratchSize) &&
494 ScratchSize > MaxScratchPerWorkitem) {
495 DiagnosticInfoStackSize DiagStackSize(F, ScratchSize, MaxScratchPerWorkitem,
496 DS_Error);
497 F.getContext().diagnose(DiagStackSize);
498 }
499
500 // Validate addressable scalar registers (i.e., prior to added implicit
501 // SGPRs).
502 MCSymbol *NumSGPRSymbol =
503 RI.getSymbol(FnSym->getName(), RIK::RIK_NumSGPR, OutContext);
505 !STM.hasSGPRInitBug()) {
506 unsigned MaxAddressableNumSGPRs = STM.getAddressableNumSGPRs();
507 uint64_t NumSgpr;
508 if (NumSGPRSymbol->isVariable() &&
509 TryGetMCExprValue(NumSGPRSymbol->getVariableValue(), NumSgpr) &&
510 NumSgpr > MaxAddressableNumSGPRs) {
511 F.getContext().diagnose(DiagnosticInfoResourceLimit(
512 F, "addressable scalar registers", NumSgpr, MaxAddressableNumSGPRs,
514 return;
515 }
516 }
517
518 MCSymbol *VCCUsedSymbol =
519 RI.getSymbol(FnSym->getName(), RIK::RIK_UsesVCC, OutContext);
520 MCSymbol *FlatUsedSymbol =
521 RI.getSymbol(FnSym->getName(), RIK::RIK_UsesFlatScratch, OutContext);
522 uint64_t VCCUsed, FlatUsed, NumSgpr;
523
524 if (NumSGPRSymbol->isVariable() && VCCUsedSymbol->isVariable() &&
525 FlatUsedSymbol->isVariable() &&
526 TryGetMCExprValue(NumSGPRSymbol->getVariableValue(), NumSgpr) &&
527 TryGetMCExprValue(VCCUsedSymbol->getVariableValue(), VCCUsed) &&
528 TryGetMCExprValue(FlatUsedSymbol->getVariableValue(), FlatUsed)) {
529
530 // Recomputes NumSgprs + implicit SGPRs but all symbols should now be
531 // resolvable.
532 NumSgpr += IsaInfo::getNumExtraSGPRs(
533 STM, VCCUsed, FlatUsed,
534 getTargetStreamer()->getTargetID()->isXnackOnOrAny());
536 STM.hasSGPRInitBug()) {
537 unsigned MaxAddressableNumSGPRs = STM.getAddressableNumSGPRs();
538 if (NumSgpr > MaxAddressableNumSGPRs) {
539 F.getContext().diagnose(DiagnosticInfoResourceLimit(
540 F, "scalar registers", NumSgpr, MaxAddressableNumSGPRs, DS_Error,
542 return;
543 }
544 }
545
546 MCSymbol *NumVgprSymbol =
547 RI.getSymbol(FnSym->getName(), RIK::RIK_NumVGPR, OutContext);
548 MCSymbol *NumAgprSymbol =
549 RI.getSymbol(FnSym->getName(), RIK::RIK_NumAGPR, OutContext);
550 uint64_t NumVgpr, NumAgpr;
551
552 MachineModuleInfo &MMI = *GetMMI();
553 MachineFunction *MF = MMI.getMachineFunction(F);
554 if (MF && NumVgprSymbol->isVariable() && NumAgprSymbol->isVariable() &&
555 TryGetMCExprValue(NumVgprSymbol->getVariableValue(), NumVgpr) &&
556 TryGetMCExprValue(NumAgprSymbol->getVariableValue(), NumAgpr)) {
557 const SIMachineFunctionInfo &MFI = *MF->getInfo<SIMachineFunctionInfo>();
558 unsigned MaxWaves = MFI.getMaxWavesPerEU();
559 uint64_t TotalNumVgpr =
560 getTotalNumVGPRs(STM.hasGFX90AInsts(), NumAgpr, NumVgpr);
561 uint64_t NumVGPRsForWavesPerEU =
562 std::max({TotalNumVgpr, (uint64_t)1,
564 MaxWaves, MFI.getDynamicVGPRBlockSize())});
565 uint64_t NumSGPRsForWavesPerEU = std::max(
566 {NumSgpr, (uint64_t)1, (uint64_t)STM.getMinNumSGPRs(MaxWaves)});
567 const MCExpr *OccupancyExpr = createOccupancy(
568 STM.getOccupancyWithWorkGroupSizes(*MF).second,
569 MCConstantExpr::create(NumSGPRsForWavesPerEU, OutContext),
570 MCConstantExpr::create(NumVGPRsForWavesPerEU, OutContext),
572 uint64_t Occupancy;
573
574 const auto [MinWEU, MaxWEU] = AMDGPU::getIntegerPairAttribute(
575 F, "amdgpu-waves-per-eu", {0, 0}, true);
576
577 if (TryGetMCExprValue(OccupancyExpr, Occupancy) && Occupancy < MinWEU) {
578 DiagnosticInfoOptimizationFailure Diag(
579 F, F.getSubprogram(),
580 "failed to meet occupancy target given by 'amdgpu-waves-per-eu' in "
581 "'" +
582 F.getName() + "': desired occupancy was " + Twine(MinWEU) +
583 ", final occupancy is " + Twine(Occupancy));
584 F.getContext().diagnose(Diag);
585 return;
586 }
587 }
588 }
589}
590
591static void appendTypeEncoding(std::string &Enc, Type *Ty, const DataLayout &DL,
592 bool IsReturnType) {
593 if (Ty->isVoidTy()) {
594 Enc += 'v';
595 return;
596 }
597 unsigned Bits = DL.getTypeSizeInBits(Ty);
598 // Zero-sized non-void types (e.g. `{}` or `[0 x i8]`) consume no ABI
599 // registers. For returns, emit the same no-result marker as void so the
600 // parameter encoding still has an explicit return-type prefix.
601 if (Bits == 0) {
602 if (IsReturnType)
603 Enc += 'v';
604 return;
605 }
606 if (Bits <= 32)
607 Enc += 'i';
608 else if (Bits <= 64)
609 Enc += 'l';
610 else
611 Enc.append(divideCeil(Bits, 32), 'i');
612}
613
614static std::string computeTypeId(const FunctionType *FTy,
615 const DataLayout &DL) {
616 std::string Enc;
617 appendTypeEncoding(Enc, FTy->getReturnType(), DL, /*IsReturnType=*/true);
618 for (Type *ParamTy : FTy->params())
619 appendTypeEncoding(Enc, ParamTy, DL, /*IsReturnType=*/false);
620 return Enc;
621}
622
623void AMDGPUAsmPrinter::collectCallEdge(const MachineInstr &MI) {
625 return;
626 const SIInstrInfo *TII = MF->getSubtarget<GCNSubtarget>().getInstrInfo();
627 const MachineOperand *Callee =
628 TII->getNamedOperand(MI, AMDGPU::OpName::callee);
629 if (!Callee || !Callee->isGlobal())
630 return;
631 DirectCallEdges.insert(
632 {getSymbol(&MF->getFunction()), getSymbol(Callee->getGlobal())});
633}
634
635void AMDGPUAsmPrinter::emitAMDGPUInfo(Module &M) {
637 return;
638
639 const NamedMDNode *LDSMD = M.getNamedMetadata("amdgpu.lds.uses");
640 bool HasLDSUses = LDSMD && LDSMD->getNumOperands() > 0;
641
642 const NamedMDNode *BarMD = M.getNamedMetadata("amdgpu.named_barrier.uses");
643 bool HasNamedBarriers = BarMD && BarMD->getNumOperands() > 0;
644
645 // Collect address-taken functions (with type IDs) and indirect call sites.
646 DenseMap<const Function *, std::string> AddrTakenTypeIds;
647 using IndirectCallInfo = std::pair<const Function *, std::string>;
649
650 for (const Function &F : M) {
651 bool IsKernel = AMDGPU::isKernel(F.getCallingConv());
652
653 if (!IsKernel && F.hasAddressTaken(/*PutOffender=*/nullptr,
654 /*IgnoreCallbackUses=*/false,
655 /*IgnoreAssumeLikeCalls=*/true,
656 /*IgnoreLLVMUsed=*/true)) {
657 AddrTakenTypeIds[&F] =
658 computeTypeId(F.getFunctionType(), M.getDataLayout());
659 }
660
661 if (F.isDeclaration())
662 continue;
663
664 StringSet<> SeenTypeIds;
665 for (const BasicBlock &BB : F) {
666 for (const Instruction &I : BB) {
667 const auto *CB = dyn_cast<CallBase>(&I);
668 if (!CB || !CB->isIndirectCall())
669 continue;
670 std::string TId =
671 computeTypeId(CB->getFunctionType(), M.getDataLayout());
672 if (SeenTypeIds.insert(TId).second)
673 IndirectCalls.push_back({&F, std::move(TId)});
674 }
675 }
676 }
677
678 if (FunctionInfos.empty() && DirectCallEdges.empty() && !HasLDSUses &&
679 !HasNamedBarriers && AddrTakenTypeIds.empty() && IndirectCalls.empty())
680 return;
681
682 AMDGPU::InfoSectionData Data;
683 Data.Funcs = std::move(FunctionInfos);
684
685 for (auto &[F, TypeId] : AddrTakenTypeIds) {
686 MCSymbol *Sym = getSymbol(F);
687 Data.TypeIds.push_back({Sym, TypeId});
688 }
689
690 for (auto &[CallerSym, CalleeSym] : DirectCallEdges)
691 Data.Calls.push_back({CallerSym, CalleeSym});
692 DirectCallEdges.clear();
693
694 if (HasLDSUses) {
695 for (const MDNode *N : LDSMD->operands()) {
696 auto *Func = mdconst::extract<Function>(N->getOperand(0));
697 auto *LdsVar = mdconst::extract<GlobalVariable>(N->getOperand(1));
698 Data.Uses.push_back({getSymbol(Func), getSymbol(LdsVar)});
699 }
700 }
701
702 if (HasNamedBarriers) {
703 for (const MDNode *N : BarMD->operands()) {
704 auto *BarVar = mdconst::extract<GlobalVariable>(N->getOperand(0));
705 MCSymbol *BarSym = getSymbol(BarVar);
706 for (unsigned I = 1, E = N->getNumOperands(); I < E; ++I) {
707 auto *Func = mdconst::extract<Function>(N->getOperand(I));
708 Data.Uses.push_back({getSymbol(Func), BarSym});
709 }
710 }
711 }
712
713 for (auto &[Caller, Enc] : IndirectCalls) {
714 MCSymbol *CallerSym = getSymbol(Caller);
715 Data.IndirectCalls.push_back({CallerSym, Enc});
716 }
717
719}
720
722 const Triple &TT = M.getTargetTriple();
723
724 // Pad with s_code_end to help tools and guard against instruction prefetch
725 // causing stale data in caches. Arguably this should be done by the linker,
726 // which is why this isn't done for Mesa.
727 // Don't do it if there is no code.
728 const MCSubtargetInfo &STI = *getGlobalSTI();
729 if ((AMDGPU::isGFX10Plus(STI) || AMDGPU::isGFX90A(STI)) &&
730 (TT.getOS() == Triple::AMDHSA || TT.getOS() == Triple::AMDPAL)) {
732 if (TextSect->hasInstructions()) {
733 OutStreamer->switchSection(TextSect);
735 }
736 }
737
738 // Emit the unified .amdgpu.info section (per-function resources, call graph,
739 // LDS/named-barrier use edges, indirect calls, and address-taken type IDs).
740 emitAMDGPUInfo(M);
741
742 // Assign expressions which can only be resolved when all other functions are
743 // known.
744 RI.finalize(OutContext);
745
746 // Switch section and emit all GPR maximums within the processed module.
747 OutStreamer->pushSection();
748 MCSectionELF *MaxGPRSection =
749 OutContext.getELFSection(".AMDGPU.gpr_maximums", ELF::SHT_PROGBITS, 0);
750 OutStreamer->switchSection(MaxGPRSection);
752 RI.getMaxVGPRSymbol(OutContext), RI.getMaxAGPRSymbol(OutContext),
753 RI.getMaxSGPRSymbol(OutContext), RI.getMaxNamedBarrierSymbol(OutContext));
754 OutStreamer->popSection();
755
756 // In the object-linking pipeline per-function resource MCExprs reference
757 // external callee symbols that cannot be evaluated here, so cross-TU limit
758 // checks would silently no-op for every non-leaf function. Defer resource
759 // sanity checking to the linker, which re-validates against the aggregated
760 // call graph in the combined .amdgpu.info metadata.
762 for (Function &F : M.functions())
763 validateMCResourceInfo(F);
764 }
765
766 RI.reset();
767
769}
770
771SmallString<128> AMDGPUAsmPrinter::getMCExprStr(const MCExpr *Value) {
773 raw_svector_ostream OSS(Str);
775 auto &Context = Streamer.getContext();
776 const MCExpr *New = foldAMDGPUMCExpr(Value, Context);
777 printAMDGPUMCExpr(New, OSS, &MAI);
778 return Str;
779}
780
781// Print comments that apply to both callable functions and entry points.
782void AMDGPUAsmPrinter::emitCommonFunctionComments(
783 const MCExpr *NumVGPR, const MCExpr *NumAGPR, const MCExpr *TotalNumVGPR,
784 const MCExpr *NumSGPR, const MCExpr *ScratchSize, uint64_t CodeSize,
785 const AMDGPUMachineFunctionInfo *MFI) {
786 OutStreamer->emitRawComment(" codeLenInByte = " + Twine(CodeSize), false);
787 OutStreamer->emitRawComment(" TotalNumSgprs: " + getMCExprStr(NumSGPR),
788 false);
789 OutStreamer->emitRawComment(" NumVgprs: " + getMCExprStr(NumVGPR), false);
790 if (NumAGPR && TotalNumVGPR) {
791 OutStreamer->emitRawComment(" NumAgprs: " + getMCExprStr(NumAGPR), false);
792 OutStreamer->emitRawComment(" TotalNumVgprs: " + getMCExprStr(TotalNumVGPR),
793 false);
794 }
795 OutStreamer->emitRawComment(" ScratchSize: " + getMCExprStr(ScratchSize),
796 false);
797 OutStreamer->emitRawComment(" MemoryBound: " + Twine(MFI->isMemoryBound()),
798 false);
799}
800
801const MCExpr *AMDGPUAsmPrinter::getAmdhsaKernelCodeProperties(
802 const MachineFunction &MF) const {
803 const SIMachineFunctionInfo &MFI = *MF.getInfo<SIMachineFunctionInfo>();
804 MCContext &Ctx = MF.getContext();
805 uint16_t KernelCodeProperties = 0;
806 const GCNUserSGPRUsageInfo &UserSGPRInfo = MFI.getUserSGPRInfo();
807
808 if (UserSGPRInfo.hasPrivateSegmentBuffer()) {
809 KernelCodeProperties |=
810 amdhsa::KERNEL_CODE_PROPERTY_ENABLE_SGPR_PRIVATE_SEGMENT_BUFFER;
811 }
812 if (UserSGPRInfo.hasDispatchPtr()) {
813 KernelCodeProperties |=
814 amdhsa::KERNEL_CODE_PROPERTY_ENABLE_SGPR_DISPATCH_PTR;
815 }
816 if (UserSGPRInfo.hasQueuePtr()) {
817 KernelCodeProperties |= amdhsa::KERNEL_CODE_PROPERTY_ENABLE_SGPR_QUEUE_PTR;
818 }
819 if (UserSGPRInfo.hasKernargSegmentPtr()) {
820 KernelCodeProperties |=
821 amdhsa::KERNEL_CODE_PROPERTY_ENABLE_SGPR_KERNARG_SEGMENT_PTR;
822 }
823 if (UserSGPRInfo.hasDispatchID()) {
824 KernelCodeProperties |=
825 amdhsa::KERNEL_CODE_PROPERTY_ENABLE_SGPR_DISPATCH_ID;
826 }
827 if (UserSGPRInfo.hasFlatScratchInit()) {
828 KernelCodeProperties |=
829 amdhsa::KERNEL_CODE_PROPERTY_ENABLE_SGPR_FLAT_SCRATCH_INIT;
830 }
831 if (UserSGPRInfo.hasPrivateSegmentSize()) {
832 KernelCodeProperties |=
833 amdhsa::KERNEL_CODE_PROPERTY_ENABLE_SGPR_PRIVATE_SEGMENT_SIZE;
834 }
835 const GCNSubtarget &STM = MF.getSubtarget<GCNSubtarget>();
836 if (STM.isWave32() &&
837 STM.getFeatureBits().test(AMDGPU::FeatureSupportsWave32) &&
838 STM.getFeatureBits().test(AMDGPU::FeatureSupportsWave64)) {
839 KernelCodeProperties |=
840 amdhsa::KERNEL_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32;
841 }
842
843 // CurrentProgramInfo.DynamicCallStack is a MCExpr and could be
844 // un-evaluatable at this point so it cannot be conditionally checked here.
845 // Instead, we'll directly shift the possibly unknown MCExpr into its place
846 // and bitwise-or it into KernelCodeProperties.
847 const MCExpr *KernelCodePropExpr =
848 MCConstantExpr::create(KernelCodeProperties, Ctx);
849 const MCExpr *OrValue = MCConstantExpr::create(
850 amdhsa::KERNEL_CODE_PROPERTY_USES_DYNAMIC_STACK_SHIFT, Ctx);
851 OrValue = MCBinaryExpr::createShl(CurrentProgramInfo.DynamicCallStack,
852 OrValue, Ctx);
853 KernelCodePropExpr = MCBinaryExpr::createOr(KernelCodePropExpr, OrValue, Ctx);
854
855 return KernelCodePropExpr;
856}
857
858MCKernelDescriptor
859AMDGPUAsmPrinter::getAmdhsaKernelDescriptor(const MachineFunction &MF,
860 const SIProgramInfo &PI) const {
861 const GCNSubtarget &STM = MF.getSubtarget<GCNSubtarget>();
862 const Function &F = MF.getFunction();
863 const SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
864 MCContext &Ctx = MF.getContext();
865
866 MCKernelDescriptor KernelDescriptor;
867
868 KernelDescriptor.group_segment_fixed_size =
870 KernelDescriptor.private_segment_fixed_size = PI.ScratchSize;
871
872 Align MaxKernArgAlign;
873 KernelDescriptor.kernarg_size = MCConstantExpr::create(
874 STM.getKernArgSegmentSize(F, MaxKernArgAlign), Ctx);
875
876 KernelDescriptor.compute_pgm_rsrc1 = PI.getComputePGMRSrc1(STM, Ctx);
877 KernelDescriptor.compute_pgm_rsrc2 = PI.getComputePGMRSrc2(STM, Ctx);
878 KernelDescriptor.kernel_code_properties = getAmdhsaKernelCodeProperties(MF);
879
880 int64_t PGM_Rsrc3 = 1;
881 bool EvaluatableRsrc3 =
882 CurrentProgramInfo.ComputePGMRSrc3->evaluateAsAbsolute(PGM_Rsrc3);
883 (void)PGM_Rsrc3;
884 (void)EvaluatableRsrc3;
886 STM.hasGFX90AInsts() || STM.hasGFX1250Insts() || !EvaluatableRsrc3 ||
887 static_cast<uint64_t>(PGM_Rsrc3) == 0);
888 KernelDescriptor.compute_pgm_rsrc3 = CurrentProgramInfo.ComputePGMRSrc3;
889
890 KernelDescriptor.kernarg_preload = MCConstantExpr::create(
891 AMDGPU::hasKernargPreload(STM) ? Info->getNumKernargPreloadedSGPRs() : 0,
892 Ctx);
893
894 return KernelDescriptor;
895}
896
898 // Init target streamer lazily on the first function so that previous passes
899 // can set metadata.
901 initTargetStreamer(*MF.getFunction().getParent());
902
903 ResourceUsage = GetResourceUsage(MF);
904 CurrentProgramInfo.reset(MF);
905
906 const AMDGPUMachineFunctionInfo *MFI =
907 MF.getInfo<AMDGPUMachineFunctionInfo>();
908 MCContext &Ctx = MF.getContext();
909
910 // The starting address of all shader programs must be 256 bytes aligned.
911 // Regular functions just need the basic required instruction alignment.
912 MF.ensureAlignment(MFI->isEntryFunction() ? Align(256) : Align(4));
913
915
916 const GCNSubtarget &STM = MF.getSubtarget<GCNSubtarget>();
918 // FIXME: This should be an explicit check for Mesa.
919 if (!STM.isAmdHsaOS() && !STM.isAmdPalOS()) {
920 MCSectionELF *ConfigSection =
921 Context.getELFSection(".AMDGPU.config", ELF::SHT_PROGBITS, 0);
922 OutStreamer->switchSection(ConfigSection);
923 }
924
925 RI.gatherResourceInfo(MF, *ResourceUsage, OutContext);
926
929 *ResourceUsage;
930 FunctionInfos.push_back(
931 {/*NumSGPR=*/static_cast<uint32_t>(RU.NumExplicitSGPR),
932 /*NumArchVGPR=*/static_cast<uint32_t>(RU.NumVGPR),
933 /*NumAccVGPR=*/static_cast<uint32_t>(RU.NumAGPR),
934 /*PrivateSegmentSize=*/static_cast<uint32_t>(RU.PrivateSegmentSize),
935 /*UsesVCC=*/RU.UsesVCC,
936 /*UsesFlatScratch=*/RU.UsesFlatScratch,
937 /*HasDynStack=*/RU.HasDynamicallySizedStack,
938 /*Sym=*/getSymbol(&MF.getFunction())});
939 }
940
941 if (MFI->isModuleEntryFunction()) {
942 getSIProgramInfo(CurrentProgramInfo, MF);
943 }
944
945 if (STM.isAmdPalOS()) {
946 if (MFI->isEntryFunction())
947 EmitPALMetadata(MF, CurrentProgramInfo);
948 else if (MFI->isModuleEntryFunction())
949 emitPALFunctionMetadata(MF);
950 } else if (!STM.isAmdHsaOS()) {
951 EmitProgramInfoSI(MF, CurrentProgramInfo);
952 }
953
954 DumpCodeInstEmitter = nullptr;
955 if (STM.dumpCode()) {
956 // For -dumpcode, get the assembler out of the streamer. This only works
957 // with -filetype=obj.
958 MCAssembler *Assembler = OutStreamer->getAssemblerPtr();
959 if (Assembler)
960 DumpCodeInstEmitter = Assembler->getEmitterPtr();
961 }
962
963 DisasmLines.clear();
964 HexLines.clear();
966
968
969 emitResourceUsageRemarks(MF, CurrentProgramInfo, MFI->isModuleEntryFunction(),
970 STM.hasMAIInsts());
971
972 {
975 RI.getSymbol(CurrentFnSym->getName(), RIK::RIK_NumVGPR, OutContext),
976 RI.getSymbol(CurrentFnSym->getName(), RIK::RIK_NumAGPR, OutContext),
977 RI.getSymbol(CurrentFnSym->getName(), RIK::RIK_NumSGPR, OutContext),
978 RI.getSymbol(CurrentFnSym->getName(), RIK::RIK_NumNamedBarrier,
979 OutContext),
980 RI.getSymbol(CurrentFnSym->getName(), RIK::RIK_PrivateSegSize,
981 OutContext),
982 RI.getSymbol(CurrentFnSym->getName(), RIK::RIK_UsesVCC, OutContext),
983 RI.getSymbol(CurrentFnSym->getName(), RIK::RIK_UsesFlatScratch,
984 OutContext),
985 RI.getSymbol(CurrentFnSym->getName(), RIK::RIK_HasDynSizedStack,
986 OutContext),
987 RI.getSymbol(CurrentFnSym->getName(), RIK::RIK_HasRecursion,
988 OutContext),
989 RI.getSymbol(CurrentFnSym->getName(), RIK::RIK_HasIndirectCall,
990 OutContext));
991 }
992
993 // Emit _dvgpr$ symbol when appropriate.
994 emitDVgprSymbol(MF);
995
996 if (isVerbose()) {
997 MCSectionELF *CommentSection =
998 Context.getELFSection(".AMDGPU.csdata", ELF::SHT_PROGBITS, 0);
999 OutStreamer->switchSection(CommentSection);
1000
1001 if (!MFI->isEntryFunction()) {
1003 OutStreamer->emitRawComment(" Function info:", false);
1004
1005 emitCommonFunctionComments(
1006 RI.getSymbol(CurrentFnSym->getName(), RIK::RIK_NumVGPR, OutContext)
1007 ->getVariableValue(),
1008 STM.hasMAIInsts() ? RI.getSymbol(CurrentFnSym->getName(),
1009 RIK::RIK_NumAGPR, OutContext)
1010 ->getVariableValue()
1011 : nullptr,
1012 RI.createTotalNumVGPRs(MF, Ctx),
1013 RI.createTotalNumSGPRs(
1014 MF,
1015 MF.getSubtarget<GCNSubtarget>().getTargetID().isXnackOnOrAny(),
1016 Ctx),
1017 RI.getSymbol(CurrentFnSym->getName(), RIK::RIK_PrivateSegSize,
1018 OutContext)
1019 ->getVariableValue(),
1020 CurrentProgramInfo.getFunctionCodeSize(MF), MFI);
1021 return false;
1022 }
1023
1024 OutStreamer->emitRawComment(" Kernel info:", false);
1025 emitCommonFunctionComments(
1026 CurrentProgramInfo.NumArchVGPR,
1027 STM.hasMAIInsts() ? CurrentProgramInfo.NumAccVGPR : nullptr,
1028 CurrentProgramInfo.NumVGPR, CurrentProgramInfo.NumSGPR,
1029 CurrentProgramInfo.ScratchSize,
1030 CurrentProgramInfo.getFunctionCodeSize(MF), MFI);
1031
1032 OutStreamer->emitRawComment(
1033 " FloatMode: " + Twine(CurrentProgramInfo.FloatMode), false);
1034 OutStreamer->emitRawComment(
1035 " IeeeMode: " + Twine(CurrentProgramInfo.IEEEMode), false);
1036 OutStreamer->emitRawComment(
1037 " LDSByteSize: " + Twine(CurrentProgramInfo.LDSSize) +
1038 " bytes/workgroup (compile time only)",
1039 false);
1040
1041 OutStreamer->emitRawComment(
1042 " SGPRBlocks: " + getMCExprStr(CurrentProgramInfo.SGPRBlocks), false);
1043
1044 OutStreamer->emitRawComment(
1045 " VGPRBlocks: " + getMCExprStr(CurrentProgramInfo.VGPRBlocks), false);
1046
1047 OutStreamer->emitRawComment(
1048 " NumSGPRsForWavesPerEU: " +
1049 getMCExprStr(CurrentProgramInfo.NumSGPRsForWavesPerEU),
1050 false);
1051 OutStreamer->emitRawComment(
1052 " NumVGPRsForWavesPerEU: " +
1053 getMCExprStr(CurrentProgramInfo.NumVGPRsForWavesPerEU),
1054 false);
1055
1056 if (STM.hasGFX90AInsts()) {
1057 const MCExpr *AdjustedAccum = MCBinaryExpr::createAdd(
1058 CurrentProgramInfo.AccumOffset, MCConstantExpr::create(1, Ctx), Ctx);
1059 AdjustedAccum = MCBinaryExpr::createMul(
1060 AdjustedAccum, MCConstantExpr::create(4, Ctx), Ctx);
1061 OutStreamer->emitRawComment(
1062 " AccumOffset: " + getMCExprStr(AdjustedAccum), false);
1063 }
1064
1065 if (STM.hasGFX1250Insts())
1066 OutStreamer->emitRawComment(
1067 " NamedBarCnt: " + getMCExprStr(CurrentProgramInfo.NamedBarCnt),
1068 false);
1069
1070 OutStreamer->emitRawComment(
1071 " Occupancy: " + getMCExprStr(CurrentProgramInfo.Occupancy), false);
1072
1073 OutStreamer->emitRawComment(
1074 " WaveLimiterHint : " + Twine(MFI->needsWaveLimiter()), false);
1075
1076 OutStreamer->emitRawComment(
1077 " COMPUTE_PGM_RSRC2:SCRATCH_EN: " +
1078 getMCExprStr(CurrentProgramInfo.ScratchEnable),
1079 false);
1080 OutStreamer->emitRawComment(" COMPUTE_PGM_RSRC2:USER_SGPR: " +
1081 Twine(CurrentProgramInfo.UserSGPR),
1082 false);
1083 OutStreamer->emitRawComment(" COMPUTE_PGM_RSRC2:TRAP_HANDLER: " +
1084 Twine(CurrentProgramInfo.TrapHandlerEnable),
1085 false);
1086 OutStreamer->emitRawComment(" COMPUTE_PGM_RSRC2:TGID_X_EN: " +
1087 Twine(CurrentProgramInfo.TGIdXEnable),
1088 false);
1089 OutStreamer->emitRawComment(" COMPUTE_PGM_RSRC2:TGID_Y_EN: " +
1090 Twine(CurrentProgramInfo.TGIdYEnable),
1091 false);
1092 OutStreamer->emitRawComment(" COMPUTE_PGM_RSRC2:TGID_Z_EN: " +
1093 Twine(CurrentProgramInfo.TGIdZEnable),
1094 false);
1095 OutStreamer->emitRawComment(" COMPUTE_PGM_RSRC2:TIDIG_COMP_CNT: " +
1096 Twine(CurrentProgramInfo.TIdIGCompCount),
1097 false);
1098
1099 [[maybe_unused]] int64_t PGMRSrc3;
1101 STM.hasGFX90AInsts() || STM.hasGFX1250Insts() ||
1102 (CurrentProgramInfo.ComputePGMRSrc3->evaluateAsAbsolute(PGMRSrc3) &&
1103 static_cast<uint64_t>(PGMRSrc3) == 0));
1104 if (STM.hasGFX90AInsts()) {
1105 OutStreamer->emitRawComment(
1106 " COMPUTE_PGM_RSRC3_GFX90A:ACCUM_OFFSET: " +
1107 getMCExprStr(MCKernelDescriptor::bits_get(
1108 CurrentProgramInfo.ComputePGMRSrc3,
1109 amdhsa::COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET_SHIFT,
1110 amdhsa::COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET, Ctx)),
1111 false);
1112 OutStreamer->emitRawComment(
1113 " COMPUTE_PGM_RSRC3_GFX90A:TG_SPLIT: " +
1114 getMCExprStr(MCKernelDescriptor::bits_get(
1115 CurrentProgramInfo.ComputePGMRSrc3,
1116 amdhsa::COMPUTE_PGM_RSRC3_GFX90A_TG_SPLIT_SHIFT,
1117 amdhsa::COMPUTE_PGM_RSRC3_GFX90A_TG_SPLIT, Ctx)),
1118 false);
1119 }
1120 }
1121
1122 if (DumpCodeInstEmitter) {
1123
1124 OutStreamer->switchSection(
1125 Context.getELFSection(".AMDGPU.disasm", ELF::SHT_PROGBITS, 0));
1126
1127 for (size_t i = 0; i < DisasmLines.size(); ++i) {
1128 std::string Comment = "\n";
1129 if (!HexLines[i].empty()) {
1130 Comment = std::string(DisasmLineMaxLen - DisasmLines[i].size(), ' ');
1131 Comment += " ; " + HexLines[i] + "\n";
1132 }
1133
1134 OutStreamer->emitBytes(StringRef(DisasmLines[i]));
1135 OutStreamer->emitBytes(StringRef(Comment));
1136 }
1137 }
1138
1139 return false;
1140}
1141
1142// When appropriate, add a _dvgpr$ symbol, with the value of the function
1143// symbol, plus an offset encoding one less than the number of VGPR blocks used
1144// by the function in bits 5..3 of the symbol value. A "VGPR block" can be
1145// either 16 VGPRs (for a max of 128), or 32 VGPRs (for a max of 256). This is
1146// used by a front-end to have functions that are chained rather than called,
1147// and a dispatcher that dynamically resizes the VGPR count before dispatching
1148// to a function.
1149void AMDGPUAsmPrinter::emitDVgprSymbol(MachineFunction &MF) {
1151 if (MFI.isDynamicVGPREnabled() &&
1153 MCContext &Ctx = MF.getContext();
1154 unsigned BlockSize = MFI.getDynamicVGPRBlockSize();
1155
1156 const MCExpr *EncodedBlocks;
1157 MCValue NumVGPRs;
1158 if (CurrentProgramInfo.NumVGPRsForWavesPerEU->evaluateAsRelocatable(
1159 NumVGPRs, nullptr) &&
1160 NumVGPRs.isAbsolute()) {
1161
1162 // Calculate number of VGPR blocks.
1163 // Treat 0 VGPRs as 1 VGPR to avoid underflowing.
1164 unsigned NumBlocks =
1165 divideCeil(std::max(unsigned(NumVGPRs.getConstant()), 1U), BlockSize);
1166
1167 if (NumBlocks > AMDGPU::IsaInfo::MaxDynamicVGPRBlocks) {
1169 {}, "DVGPR block count " + Twine(NumBlocks) +
1170 " exceeds maximum of " +
1172 " for __dvgpr$ symbol for '" +
1173 Twine(CurrentFnSym->getName()) + "'");
1174 return;
1175 }
1176 unsigned EncodedNumBlocks = (NumBlocks - 1) << 3;
1177 EncodedBlocks = MCConstantExpr::create(EncodedNumBlocks, Ctx);
1178 } else {
1179 // Value not yet available so build a symbolic MCExpr:
1180 // ((alignTo(max(NumVGPRs, 1), BlockSize) / BlockSize - 1) << 3
1181 const MCExpr *One = MCConstantExpr::create(1, Ctx);
1182 const MCExpr *BlockSizeConst = MCConstantExpr::create(BlockSize, Ctx);
1183 const MCExpr *MaxVGPRs = AMDGPUMCExpr::createMax(
1184 {CurrentProgramInfo.NumVGPRsForWavesPerEU, One}, Ctx);
1185 const MCExpr *NumBlocks = MCBinaryExpr::createDiv(
1186 AMDGPUMCExpr::createAlignTo(MaxVGPRs, BlockSizeConst, Ctx),
1187 BlockSizeConst, Ctx);
1188 EncodedBlocks =
1190 MCConstantExpr::create(3, Ctx), Ctx);
1191 }
1192
1193 // Add to function symbol to create _dvgpr$ symbol.
1194 const MCExpr *DVgprFuncVal = MCBinaryExpr::createAdd(
1195 MCSymbolRefExpr::create(CurrentFnSym, Ctx), EncodedBlocks, Ctx);
1196 MCSymbol *DVgprFuncSym =
1197 Ctx.getOrCreateSymbol(Twine("_dvgpr$") + CurrentFnSym->getName());
1198 OutStreamer->emitAssignment(DVgprFuncSym, DVgprFuncVal);
1199 emitVisibility(DVgprFuncSym, MF.getFunction().getVisibility());
1200 emitLinkage(&MF.getFunction(), DVgprFuncSym);
1201 }
1202}
1203
1204// TODO: Fold this into emitFunctionBodyStart.
1205void AMDGPUAsmPrinter::initializeTargetID(const Module &M) {
1207
1208 auto &TSTargetID = getTargetStreamer()->getTargetID();
1209
1210 // Error if -mattr specified xnack or sramecc.
1211 // TODO: Remove this when subtarget features removed.
1212 StringRef FeatureString = getGlobalSTI()->getFeatureString();
1213 if (FeatureString.contains("xnack")) {
1214 M.getContext().diagnose(DiagnosticInfoGeneric(
1215 "xnack/sramecc should be specified via module flags. "
1216 "Use module flag 'amdgpu.xnack' instead of subtarget feature",
1217 DS_Error));
1218 }
1219 if (FeatureString.contains("sramecc")) {
1220 M.getContext().diagnose(DiagnosticInfoGeneric(
1221 "xnack/sramecc should be specified via module flags. "
1222 "Use module flag 'amdgpu.sramecc' instead of subtarget feature",
1223 DS_Error));
1224 }
1225
1226 // Apply xnack/sramecc settings from module flags.
1227 if (getGlobalSTI()->getFeatureBits().test(AMDGPU::FeatureXNACKOnOffModes)) {
1228 AMDGPU::TargetIDSetting Setting =
1230 if (Setting != AMDGPU::TargetIDSetting::Any)
1231 TSTargetID->setXnackSetting(Setting);
1232 }
1233
1234 if (getGlobalSTI()->getFeatureBits().test(AMDGPU::FeatureSRAMECCOnOffModes)) {
1235 AMDGPU::TargetIDSetting Setting =
1237 if (Setting != AMDGPU::TargetIDSetting::Any)
1238 TSTargetID->setSramEccSetting(Setting);
1239 }
1240}
1241
1242// AccumOffset computed for the MCExpr equivalent of:
1243// alignTo(std::max(1, NumVGPR), 4) / 4 - 1;
1244static const MCExpr *computeAccumOffset(const MCExpr *NumVGPR, MCContext &Ctx) {
1245 const MCExpr *ConstFour = MCConstantExpr::create(4, Ctx);
1246 const MCExpr *ConstOne = MCConstantExpr::create(1, Ctx);
1247
1248 // Can't be lower than 1 for subsequent alignTo.
1249 const MCExpr *MaximumTaken =
1250 AMDGPUMCExpr::createMax({ConstOne, NumVGPR}, Ctx);
1251
1252 // Practically, it's computing divideCeil(MaximumTaken, 4).
1253 const MCExpr *DivCeil = MCBinaryExpr::createDiv(
1254 AMDGPUMCExpr::createAlignTo(MaximumTaken, ConstFour, Ctx), ConstFour,
1255 Ctx);
1256
1257 return MCBinaryExpr::createSub(DivCeil, ConstOne, Ctx);
1258}
1259
1260static unsigned getLDSEncodingGranule(const GCNSubtarget &ST) {
1261 unsigned Granule =
1262 AMDGPU::getLDSEncodingGranule(ST.getTargetID().getGPUKind());
1263 // The legacy generic targets have no encoding granularity feature. Preserve
1264 // the default used for code generation when no GPU is specified.
1265 return Granule ? Granule : 256;
1266}
1267
1268void AMDGPUAsmPrinter::getSIProgramInfo(SIProgramInfo &ProgInfo,
1269 const MachineFunction &MF) {
1270 const GCNSubtarget &STM = MF.getSubtarget<GCNSubtarget>();
1271 MCContext &Ctx = MF.getContext();
1272
1273 auto CreateExpr = [&Ctx](int64_t Value) {
1274 return MCConstantExpr::create(Value, Ctx);
1275 };
1276
1277 auto TryGetMCExprValue = [](const MCExpr *Value, uint64_t &Res) -> bool {
1278 int64_t Val;
1279 if (Value->evaluateAsAbsolute(Val)) {
1280 Res = Val;
1281 return true;
1282 }
1283 return false;
1284 };
1285
1286 auto GetSymRefExpr =
1287 [&](MCResourceInfo::ResourceInfoKind RIK) -> const MCExpr * {
1288 MCSymbol *Sym = RI.getSymbol(CurrentFnSym->getName(), RIK, OutContext);
1289 return MCSymbolRefExpr::create(Sym, Ctx);
1290 };
1291
1293 ProgInfo.NumArchVGPR = GetSymRefExpr(RIK::RIK_NumVGPR);
1294 ProgInfo.NumAccVGPR = GetSymRefExpr(RIK::RIK_NumAGPR);
1296 ProgInfo.NumAccVGPR, ProgInfo.NumArchVGPR, Ctx);
1297
1298 ProgInfo.AccumOffset = computeAccumOffset(ProgInfo.NumArchVGPR, Ctx);
1299 ProgInfo.TgSplit =
1300 STM.hasTgSplitSupport() && AMDGPU::isTgSplitEnabled(MF.getFunction());
1301 ProgInfo.NumSGPR = GetSymRefExpr(RIK::RIK_NumSGPR);
1302 ProgInfo.ScratchSize = GetSymRefExpr(RIK::RIK_PrivateSegSize);
1303 ProgInfo.VCCUsed = GetSymRefExpr(RIK::RIK_UsesVCC);
1304 ProgInfo.FlatUsed = GetSymRefExpr(RIK::RIK_UsesFlatScratch);
1305 ProgInfo.DynamicCallStack =
1306 MCBinaryExpr::createOr(GetSymRefExpr(RIK::RIK_HasDynSizedStack),
1307 GetSymRefExpr(RIK::RIK_HasRecursion), Ctx);
1308
1309 const MCExpr *BarBlkConst = MCConstantExpr::create(4, Ctx);
1310 const MCExpr *AlignToBlk = AMDGPUMCExpr::createAlignTo(
1311 GetSymRefExpr(RIK::RIK_NumNamedBarrier), BarBlkConst, Ctx);
1312 ProgInfo.NamedBarCnt = MCBinaryExpr::createDiv(AlignToBlk, BarBlkConst, Ctx);
1313
1314 const SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
1315
1316 // The calculations related to SGPR/VGPR blocks are
1317 // duplicated in part in AMDGPUAsmParser::calculateGPRBlocks, and could be
1318 // unified.
1319 const MCExpr *ExtraSGPRs = AMDGPUMCExpr::createExtraSGPRs(
1320 ProgInfo.VCCUsed, ProgInfo.FlatUsed,
1321 getTargetStreamer()->getTargetID()->isXnackOnOrAny(), Ctx);
1322
1323 // Check the addressable register limit before we add ExtraSGPRs.
1325 !STM.hasSGPRInitBug()) {
1326 unsigned MaxAddressableNumSGPRs = STM.getAddressableNumSGPRs();
1327 uint64_t NumSgpr;
1328 if (TryGetMCExprValue(ProgInfo.NumSGPR, NumSgpr) &&
1329 NumSgpr > MaxAddressableNumSGPRs) {
1330 // This can happen due to a compiler bug or when using inline asm.
1331 LLVMContext &Ctx = MF.getFunction().getContext();
1332 Ctx.diagnose(DiagnosticInfoResourceLimit(
1333 MF.getFunction(), "addressable scalar registers", NumSgpr,
1334 MaxAddressableNumSGPRs, DS_Error, DK_ResourceLimit));
1335 ProgInfo.NumSGPR = CreateExpr(MaxAddressableNumSGPRs - 1);
1336 }
1337 }
1338
1339 // Account for extra SGPRs and VGPRs reserved for debugger use.
1340 ProgInfo.NumSGPR = MCBinaryExpr::createAdd(ProgInfo.NumSGPR, ExtraSGPRs, Ctx);
1341
1342 const Function &F = MF.getFunction();
1343
1344 // Ensure there are enough SGPRs and VGPRs for wave dispatch, where wave
1345 // dispatch registers as function args.
1346 unsigned WaveDispatchNumSGPR = MFI->getNumWaveDispatchSGPRs(),
1347 WaveDispatchNumVGPR = MFI->getNumWaveDispatchVGPRs();
1348
1349 if (WaveDispatchNumSGPR) {
1351 {ProgInfo.NumSGPR,
1352 MCBinaryExpr::createAdd(CreateExpr(WaveDispatchNumSGPR), ExtraSGPRs,
1353 Ctx)},
1354 Ctx);
1355 }
1356
1357 if (WaveDispatchNumVGPR) {
1359 {ProgInfo.NumArchVGPR, CreateExpr(WaveDispatchNumVGPR)}, Ctx);
1360
1362 ProgInfo.NumAccVGPR, ProgInfo.NumArchVGPR, Ctx);
1363 }
1364
1365 // Adjust number of registers used to meet default/requested minimum/maximum
1366 // number of waves per execution unit request.
1367 unsigned MaxWaves = MFI->getMaxWavesPerEU();
1368 ProgInfo.NumSGPRsForWavesPerEU =
1369 AMDGPUMCExpr::createMax({ProgInfo.NumSGPR, CreateExpr(1ul),
1370 CreateExpr(STM.getMinNumSGPRs(MaxWaves))},
1371 Ctx);
1372 ProgInfo.NumVGPRsForWavesPerEU =
1373 AMDGPUMCExpr::createMax({ProgInfo.NumVGPR, CreateExpr(1ul),
1374 CreateExpr(STM.getMinNumVGPRs(
1375 MaxWaves, MFI->getDynamicVGPRBlockSize()))},
1376 Ctx);
1377
1379 STM.hasSGPRInitBug()) {
1380 unsigned MaxAddressableNumSGPRs = STM.getAddressableNumSGPRs();
1381 uint64_t NumSgpr;
1382 if (TryGetMCExprValue(ProgInfo.NumSGPR, NumSgpr) &&
1383 NumSgpr > MaxAddressableNumSGPRs) {
1384 // This can happen due to a compiler bug or when using inline asm to use
1385 // the registers which are usually reserved for vcc etc.
1386 LLVMContext &Ctx = MF.getFunction().getContext();
1387 Ctx.diagnose(DiagnosticInfoResourceLimit(
1388 MF.getFunction(), "scalar registers", NumSgpr, MaxAddressableNumSGPRs,
1390 ProgInfo.NumSGPR = CreateExpr(MaxAddressableNumSGPRs);
1391 ProgInfo.NumSGPRsForWavesPerEU = CreateExpr(MaxAddressableNumSGPRs);
1392 }
1393 }
1394
1395 if (STM.hasSGPRInitBug()) {
1396 ProgInfo.NumSGPR =
1398 ProgInfo.NumSGPRsForWavesPerEU =
1400 }
1401
1402 if (MFI->getNumUserSGPRs() > STM.getMaxNumUserSGPRs()) {
1403 LLVMContext &Ctx = MF.getFunction().getContext();
1404 Ctx.diagnose(DiagnosticInfoResourceLimit(
1405 MF.getFunction(), "user SGPRs", MFI->getNumUserSGPRs(),
1407 }
1408
1409 if (MFI->getLDSSize() > STM.getAddressableLocalMemorySize()) {
1410 LLVMContext &Ctx = MF.getFunction().getContext();
1411 Ctx.diagnose(DiagnosticInfoResourceLimit(
1412 MF.getFunction(), "local memory", MFI->getLDSSize(),
1414 }
1415
1416 // When dynamic VGPRs are enabled, entry functions are launched with a single
1417 // VGPR block. The register allocator enforces this constraint, but we also
1418 // need to catch explicit physical registers in inline asm and wave dispatch
1419 // VGPR arguments.
1420 if (MFI->isDynamicVGPREnabled() &&
1421 AMDGPU::isEntryFunctionCC(F.getCallingConv())) {
1422 unsigned BlockSize = MFI->getDynamicVGPRBlockSize();
1423 uint64_t NumVgpr;
1424 if (TryGetMCExprValue(ProgInfo.NumVGPRsForWavesPerEU, NumVgpr) &&
1425 NumVgpr > BlockSize) {
1426 LLVMContext &Ctx = F.getContext();
1427 Ctx.diagnose(DiagnosticInfoResourceLimit(
1428 F, "dynamic VGPR entry point vector registers", NumVgpr, BlockSize,
1430 }
1431 }
1432
1433 // The MCExpr equivalent of getNumSGPRBlocks/getNumVGPRBlocks:
1434 // (alignTo(max(1u, NumGPR), GPREncodingGranule) / GPREncodingGranule) - 1
1435 auto GetNumGPRBlocks = [&CreateExpr, &Ctx](const MCExpr *NumGPR,
1436 unsigned Granule) {
1437 const MCExpr *OneConst = CreateExpr(1ul);
1438 const MCExpr *GranuleConst = CreateExpr(Granule);
1439 const MCExpr *MaxNumGPR = AMDGPUMCExpr::createMax({NumGPR, OneConst}, Ctx);
1440 const MCExpr *AlignToGPR =
1441 AMDGPUMCExpr::createAlignTo(MaxNumGPR, GranuleConst, Ctx);
1442 const MCExpr *DivGPR =
1443 MCBinaryExpr::createDiv(AlignToGPR, GranuleConst, Ctx);
1444 const MCExpr *SubGPR = MCBinaryExpr::createSub(DivGPR, OneConst, Ctx);
1445 return SubGPR;
1446 };
1447 // GFX10+ will always allocate 128 SGPRs and this field must be 0
1449 ProgInfo.SGPRBlocks = CreateExpr(0ul);
1450 } else {
1451 ProgInfo.SGPRBlocks = GetNumGPRBlocks(ProgInfo.NumSGPRsForWavesPerEU,
1453 }
1454 ProgInfo.VGPRBlocks = GetNumGPRBlocks(ProgInfo.NumVGPRsForWavesPerEU,
1456
1457 const SIModeRegisterDefaults Mode = MFI->getMode();
1458
1459 // Set the value to initialize FP_ROUND and FP_DENORM parts of the mode
1460 // register.
1461 ProgInfo.FloatMode = getFPMode(Mode);
1462
1463 ProgInfo.IEEEMode = Mode.IEEE;
1464
1465 // Make clamp modifier on NaN input returns 0.
1466 ProgInfo.DX10Clamp = Mode.DX10Clamp;
1467 ProgInfo.SGPRSpill = MFI->getNumSpilledSGPRs();
1468 ProgInfo.VGPRSpill = MFI->getNumSpilledVGPRs();
1469
1470 ProgInfo.LDSSize = MFI->getLDSSize();
1471
1472 unsigned LDSGranularityBytes = getLDSEncodingGranule(STM);
1473 ProgInfo.LDSBlocks =
1474 alignTo(ProgInfo.LDSSize, LDSGranularityBytes) / LDSGranularityBytes;
1475
1476 // The MCExpr equivalent of divideCeil.
1477 auto DivideCeil = [&Ctx](const MCExpr *Numerator, const MCExpr *Denominator) {
1478 const MCExpr *Ceil =
1479 AMDGPUMCExpr::createAlignTo(Numerator, Denominator, Ctx);
1480 return MCBinaryExpr::createDiv(Ceil, Denominator, Ctx);
1481 };
1482
1483 // Scratch is allocated in 64-dword or 256-dword blocks.
1484 unsigned ScratchAlignShift =
1485 STM.getGeneration() >= AMDGPUSubtarget::GFX11 ? 8 : 10;
1486 // We need to program the hardware with the amount of scratch memory that
1487 // is used by the entire wave. ProgInfo.ScratchSize is the amount of
1488 // scratch memory used per thread.
1489 ProgInfo.ScratchBlocks = DivideCeil(
1491 CreateExpr(STM.getWavefrontSize()), Ctx),
1492 CreateExpr(1ULL << ScratchAlignShift));
1493
1494 if (STM.hasSupportsWGP()) {
1495 ProgInfo.WgpMode = STM.isCuModeEnabled() ? 0 : 1;
1496 }
1497
1498 if (getIsaVersion(getGlobalSTI()->getCPU()).Major >= 10) {
1499 ProgInfo.MemOrdered = 1;
1500 ProgInfo.FwdProgress = !F.hasFnAttribute("amdgpu-no-fwd-progress");
1501 }
1502
1503 // 0 = X, 1 = XY, 2 = XYZ
1504 unsigned TIDIGCompCnt = 0;
1505 if (MFI->hasWorkItemIDZ())
1506 TIDIGCompCnt = 2;
1507 else if (MFI->hasWorkItemIDY())
1508 TIDIGCompCnt = 1;
1509
1510 // The private segment wave byte offset is the last of the system SGPRs. We
1511 // initially assumed it was allocated, and may have used it. It shouldn't harm
1512 // anything to disable it if we know the stack isn't used here. We may still
1513 // have emitted code reading it to initialize scratch, but if that's unused
1514 // reading garbage should be OK.
1517 MCConstantExpr::create(0, Ctx), Ctx),
1518 ProgInfo.DynamicCallStack, Ctx);
1519
1520 ProgInfo.UserSGPR = MFI->getNumUserSGPRs();
1521 // For AMDHSA, TRAP_HANDLER must be zero, as it is populated by the CP.
1522 ProgInfo.TrapHandlerEnable = STM.isAmdHsaOS() ? 0 : STM.hasTrapHandler();
1523 ProgInfo.TGIdXEnable = MFI->hasWorkGroupIDX();
1524 ProgInfo.TGIdYEnable = MFI->hasWorkGroupIDY();
1525 ProgInfo.TGIdZEnable = MFI->hasWorkGroupIDZ();
1526 ProgInfo.TGSizeEnable = MFI->hasWorkGroupInfo();
1527 ProgInfo.TIdIGCompCount = TIDIGCompCnt;
1528 ProgInfo.EXCPEnMSB = 0;
1529 // For AMDHSA, LDS_SIZE must be zero, as it is populated by the CP.
1530 ProgInfo.LdsSize = STM.isAmdHsaOS() ? 0 : ProgInfo.LDSBlocks;
1531 ProgInfo.EXCPEnable = 0;
1532
1533 if (STM.hasGFX90AInsts()) {
1534 ProgInfo.ComputePGMRSrc3 =
1535 setBits(ProgInfo.ComputePGMRSrc3, ProgInfo.AccumOffset,
1536 amdhsa::COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET,
1537 amdhsa::COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET_SHIFT, Ctx);
1538 ProgInfo.ComputePGMRSrc3 =
1539 setBits(ProgInfo.ComputePGMRSrc3, CreateExpr(ProgInfo.TgSplit),
1540 amdhsa::COMPUTE_PGM_RSRC3_GFX90A_TG_SPLIT,
1541 amdhsa::COMPUTE_PGM_RSRC3_GFX90A_TG_SPLIT_SHIFT, Ctx);
1542 }
1543
1544 if (STM.hasGFX1250Insts())
1545 ProgInfo.ComputePGMRSrc3 =
1546 setBits(ProgInfo.ComputePGMRSrc3, ProgInfo.NamedBarCnt,
1547 amdhsa::COMPUTE_PGM_RSRC3_GFX125_NAMED_BAR_CNT,
1548 amdhsa::COMPUTE_PGM_RSRC3_GFX125_NAMED_BAR_CNT_SHIFT, Ctx);
1549
1550 ProgInfo.Occupancy = createOccupancy(
1551 STM.computeOccupancy(F, ProgInfo.LDSSize).second,
1553 MFI->getDynamicVGPRBlockSize(), STM, Ctx);
1554
1555 const auto [MinWEU, MaxWEU] =
1556 AMDGPU::getIntegerPairAttribute(F, "amdgpu-waves-per-eu", {0, 0}, true);
1557 uint64_t Occupancy;
1558 if (TryGetMCExprValue(ProgInfo.Occupancy, Occupancy) && Occupancy < MinWEU) {
1559 DiagnosticInfoOptimizationFailure Diag(
1560 F, F.getSubprogram(),
1561 "failed to meet occupancy target given by 'amdgpu-waves-per-eu' in "
1562 "'" +
1563 F.getName() + "': desired occupancy was " + Twine(MinWEU) +
1564 ", final occupancy is " + Twine(Occupancy));
1565 F.getContext().diagnose(Diag);
1566 }
1567}
1568
1569static unsigned getRsrcReg(CallingConv::ID CallConv) {
1570 switch (CallConv) {
1571 default:
1572 [[fallthrough]];
1587 }
1588}
1589
1590void AMDGPUAsmPrinter::EmitProgramInfoSI(
1591 const MachineFunction &MF, const SIProgramInfo &CurrentProgramInfo) {
1592 const SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
1593 const GCNSubtarget &STM = MF.getSubtarget<GCNSubtarget>();
1594 unsigned RsrcReg = getRsrcReg(MF.getFunction().getCallingConv());
1595 MCContext &Ctx = MF.getContext();
1596
1597 // (((Value) & Mask) << Shift)
1598 auto SetBits = [&Ctx](const MCExpr *Value, uint32_t Mask, uint32_t Shift) {
1599 const MCExpr *msk = MCConstantExpr::create(Mask, Ctx);
1600 const MCExpr *shft = MCConstantExpr::create(Shift, Ctx);
1602 shft, Ctx);
1603 };
1604
1605 auto EmitResolvedOrExpr = [this](const MCExpr *Value, unsigned Size) {
1606 int64_t Val;
1607 if (Value->evaluateAsAbsolute(Val))
1608 OutStreamer->emitIntValue(static_cast<uint64_t>(Val), Size);
1609 else
1610 OutStreamer->emitValue(Value, Size);
1611 };
1612
1613 if (AMDGPU::isCompute(MF.getFunction().getCallingConv())) {
1615
1616 EmitResolvedOrExpr(CurrentProgramInfo.getComputePGMRSrc1(STM, Ctx),
1617 /*Size=*/4);
1618
1620 EmitResolvedOrExpr(CurrentProgramInfo.getComputePGMRSrc2(STM, Ctx),
1621 /*Size=*/4);
1622
1624
1625 // Sets bits according to S_0286E8_WAVESIZE_* mask and shift values for the
1626 // appropriate generation.
1627 if (STM.getGeneration() >= AMDGPUSubtarget::GFX12) {
1628 EmitResolvedOrExpr(SetBits(CurrentProgramInfo.ScratchBlocks,
1629 /*Mask=*/0x3FFFF, /*Shift=*/12),
1630 /*Size=*/4);
1631 } else if (STM.getGeneration() == AMDGPUSubtarget::GFX11) {
1632 EmitResolvedOrExpr(SetBits(CurrentProgramInfo.ScratchBlocks,
1633 /*Mask=*/0x7FFF, /*Shift=*/12),
1634 /*Size=*/4);
1635 } else {
1636 EmitResolvedOrExpr(SetBits(CurrentProgramInfo.ScratchBlocks,
1637 /*Mask=*/0x1FFF, /*Shift=*/12),
1638 /*Size=*/4);
1639 }
1640
1641 // TODO: Should probably note flat usage somewhere. SC emits a "FlatPtr32 =
1642 // 0" comment but I don't see a corresponding field in the register spec.
1643 } else {
1644 OutStreamer->emitInt32(RsrcReg);
1645
1646 const MCExpr *GPRBlocks = MCBinaryExpr::createOr(
1647 SetBits(CurrentProgramInfo.VGPRBlocks, /*Mask=*/0x3F, /*Shift=*/0),
1648 SetBits(CurrentProgramInfo.SGPRBlocks, /*Mask=*/0x0F, /*Shift=*/6),
1649 MF.getContext());
1650 EmitResolvedOrExpr(GPRBlocks, /*Size=*/4);
1652
1653 // Sets bits according to S_0286E8_WAVESIZE_* mask and shift values for the
1654 // appropriate generation.
1655 if (STM.getGeneration() >= AMDGPUSubtarget::GFX12) {
1656 EmitResolvedOrExpr(SetBits(CurrentProgramInfo.ScratchBlocks,
1657 /*Mask=*/0x3FFFF, /*Shift=*/12),
1658 /*Size=*/4);
1659 } else if (STM.getGeneration() == AMDGPUSubtarget::GFX11) {
1660 EmitResolvedOrExpr(SetBits(CurrentProgramInfo.ScratchBlocks,
1661 /*Mask=*/0x7FFF, /*Shift=*/12),
1662 /*Size=*/4);
1663 } else {
1664 EmitResolvedOrExpr(SetBits(CurrentProgramInfo.ScratchBlocks,
1665 /*Mask=*/0x1FFF, /*Shift=*/12),
1666 /*Size=*/4);
1667 }
1668 }
1669
1670 if (MF.getFunction().getCallingConv() == CallingConv::AMDGPU_PS) {
1672 unsigned ExtraLDSSize = STM.getGeneration() >= AMDGPUSubtarget::GFX11
1673 ? divideCeil(CurrentProgramInfo.LDSBlocks, 2)
1674 : CurrentProgramInfo.LDSBlocks;
1675 OutStreamer->emitInt32(S_00B02C_EXTRA_LDS_SIZE(ExtraLDSSize));
1677 OutStreamer->emitInt32(MFI->getPSInputEnable());
1679 OutStreamer->emitInt32(MFI->getPSInputAddr());
1680 }
1681
1682 OutStreamer->emitInt32(R_SPILLED_SGPRS);
1683 OutStreamer->emitInt32(MFI->getNumSpilledSGPRs());
1684 OutStreamer->emitInt32(R_SPILLED_VGPRS);
1685 OutStreamer->emitInt32(MFI->getNumSpilledVGPRs());
1686}
1687
1688// Helper function to add common PAL Metadata 3.0+
1690 const SIProgramInfo &CurrentProgramInfo,
1691 CallingConv::ID CC, const GCNSubtarget &ST,
1692 unsigned DynamicVGPRBlockSize) {
1693 if (ST.hasFeature(AMDGPU::FeatureDX10ClampAndIEEEMode))
1694 MD->setHwStage(CC, ".ieee_mode", (bool)CurrentProgramInfo.IEEEMode);
1695
1696 MD->setHwStage(CC, ".wgp_mode", (bool)CurrentProgramInfo.WgpMode);
1697 MD->setHwStage(CC, ".mem_ordered", (bool)CurrentProgramInfo.MemOrdered);
1698 MD->setHwStage(CC, ".forward_progress", (bool)CurrentProgramInfo.FwdProgress);
1699
1700 if (AMDGPU::isCompute(CC)) {
1701 MD->setHwStage(CC, ".trap_present",
1702 (bool)CurrentProgramInfo.TrapHandlerEnable);
1703 MD->setHwStage(CC, ".excp_en", CurrentProgramInfo.EXCPEnable);
1704
1705 if (DynamicVGPRBlockSize != 0)
1706 MD->setComputeRegisters(".dynamic_vgpr_en", true);
1707 }
1708
1710 CC, ".lds_size",
1711 (unsigned)(CurrentProgramInfo.LdsSize * getLDSEncodingGranule(ST)));
1712}
1713
1714// This is the equivalent of EmitProgramInfoSI above, but for when the OS type
1715// is AMDPAL. It stores each compute/SPI register setting and other PAL
1716// metadata items into the PALMD::Metadata, combining with any provided by the
1717// frontend as LLVM metadata. Once all functions are written, the PAL metadata
1718// is then written as a single block in the .note section.
1719void AMDGPUAsmPrinter::EmitPALMetadata(
1720 const MachineFunction &MF, const SIProgramInfo &CurrentProgramInfo) {
1721 const SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
1722 auto CC = MF.getFunction().getCallingConv();
1723 auto *MD = getTargetStreamer()->getPALMetadata();
1724 auto &Ctx = MF.getContext();
1725
1726 MD->setEntryPoint(CC, MF.getFunction().getName());
1727 MD->setNumUsedVgprs(CC, CurrentProgramInfo.NumVGPRsForWavesPerEU, Ctx);
1728
1729 // For targets that support dynamic VGPRs, set the number of saved dynamic
1730 // VGPRs (if any) in the PAL metadata.
1731 const GCNSubtarget &STM = MF.getSubtarget<GCNSubtarget>();
1732 if (MFI->isDynamicVGPREnabled() &&
1734 MD->setHwStage(CC, ".dynamic_vgpr_saved_count",
1736
1737 // Only set AGPRs for supported devices
1738 if (STM.hasMAIInsts()) {
1739 MD->setNumUsedAgprs(CC, CurrentProgramInfo.NumAccVGPR);
1740 }
1741
1742 MD->setNumUsedSgprs(CC, CurrentProgramInfo.NumSGPRsForWavesPerEU, Ctx);
1743 if (MD->getPALMajorVersion() < 3) {
1744 MD->setRsrc1(CC, CurrentProgramInfo.getPGMRSrc1(CC, STM, Ctx), Ctx);
1745 if (AMDGPU::isCompute(CC)) {
1746 MD->setRsrc2(CC, CurrentProgramInfo.getComputePGMRSrc2(STM, Ctx), Ctx);
1747 } else {
1748 const MCExpr *HasScratchBlocks =
1749 MCBinaryExpr::createGT(CurrentProgramInfo.ScratchBlocks,
1750 MCConstantExpr::create(0, Ctx), Ctx);
1751 auto [Shift, Mask] = getShiftMask(C_00B84C_SCRATCH_EN);
1752 MD->setRsrc2(CC, maskShiftSet(HasScratchBlocks, Mask, Shift, Ctx), Ctx);
1753 }
1754 } else {
1755 MD->setHwStage(CC, ".debug_mode", (bool)CurrentProgramInfo.DebugMode);
1756 MD->setHwStage(CC, ".scratch_en", msgpack::Type::Boolean,
1757 CurrentProgramInfo.ScratchEnable);
1758 EmitPALMetadataCommon(MD, CurrentProgramInfo, CC, STM,
1760 }
1761
1762 // ScratchSize is in bytes, 16 aligned.
1763 MD->setScratchSize(
1764 CC,
1765 AMDGPUMCExpr::createAlignTo(CurrentProgramInfo.ScratchSize,
1766 MCConstantExpr::create(16, Ctx), Ctx),
1767 Ctx);
1768
1769 if (MF.getFunction().getCallingConv() == CallingConv::AMDGPU_PS) {
1770 unsigned ExtraLDSSize = STM.getGeneration() >= AMDGPUSubtarget::GFX11
1771 ? divideCeil(CurrentProgramInfo.LDSBlocks, 2)
1772 : CurrentProgramInfo.LDSBlocks;
1773 if (MD->getPALMajorVersion() < 3) {
1774 MD->setRsrc2(
1775 CC,
1777 Ctx);
1778 MD->setSpiPsInputEna(MFI->getPSInputEnable());
1779 MD->setSpiPsInputAddr(MFI->getPSInputAddr());
1780 } else {
1781 // Graphics registers
1782 const unsigned ExtraLdsDwGranularity =
1783 STM.getGeneration() >= AMDGPUSubtarget::GFX11 ? 256 : 128;
1784 MD->setGraphicsRegisters(
1785 ".ps_extra_lds_size",
1786 (unsigned)(ExtraLDSSize * ExtraLdsDwGranularity * sizeof(uint32_t)));
1787
1788 // Set PsInputEna and PsInputAddr .spi_ps_input_ena and .spi_ps_input_addr
1789 static StringLiteral const PsInputFields[] = {
1790 ".persp_sample_ena", ".persp_center_ena",
1791 ".persp_centroid_ena", ".persp_pull_model_ena",
1792 ".linear_sample_ena", ".linear_center_ena",
1793 ".linear_centroid_ena", ".line_stipple_tex_ena",
1794 ".pos_x_float_ena", ".pos_y_float_ena",
1795 ".pos_z_float_ena", ".pos_w_float_ena",
1796 ".front_face_ena", ".ancillary_ena",
1797 ".sample_coverage_ena", ".pos_fixed_pt_ena"};
1798 unsigned PSInputEna = MFI->getPSInputEnable();
1799 unsigned PSInputAddr = MFI->getPSInputAddr();
1800 for (auto [Idx, Field] : enumerate(PsInputFields)) {
1801 MD->setGraphicsRegisters(".spi_ps_input_ena", Field,
1802 (bool)((PSInputEna >> Idx) & 1));
1803 MD->setGraphicsRegisters(".spi_ps_input_addr", Field,
1804 (bool)((PSInputAddr >> Idx) & 1));
1805 }
1806 }
1807 }
1808
1809 // For version 3 and above the wave front size is already set in the metadata
1810 if (MD->getPALMajorVersion() < 3 && STM.isWave32())
1811 MD->setWave32(MF.getFunction().getCallingConv());
1812}
1813
1814void AMDGPUAsmPrinter::emitPALFunctionMetadata(const MachineFunction &MF) {
1815 auto *MD = getTargetStreamer()->getPALMetadata();
1816 const MachineFrameInfo &MFI = MF.getFrameInfo();
1817 StringRef FnName = MF.getFunction().getName();
1818 MD->setFunctionScratchSize(FnName, MFI.getStackSize());
1819 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
1820 MCContext &Ctx = MF.getContext();
1821
1822 if (MD->getPALMajorVersion() < 3) {
1823 // Set compute registers
1824 MD->setRsrc1(
1826 CurrentProgramInfo.getPGMRSrc1(CallingConv::AMDGPU_CS, ST, Ctx), Ctx);
1827 MD->setRsrc2(CallingConv::AMDGPU_CS,
1828 CurrentProgramInfo.getComputePGMRSrc2(ST, Ctx), Ctx);
1829 } else {
1831 MD, CurrentProgramInfo, CallingConv::AMDGPU_CS, ST,
1832 MF.getInfo<SIMachineFunctionInfo>()->getDynamicVGPRBlockSize());
1833 }
1834
1835 // Set optional info
1836 MD->setFunctionLdsSize(FnName, CurrentProgramInfo.LDSSize);
1837 MD->setFunctionNumUsedVgprs(FnName, CurrentProgramInfo.NumVGPRsForWavesPerEU);
1838 MD->setFunctionNumUsedSgprs(FnName, CurrentProgramInfo.NumSGPRsForWavesPerEU);
1839}
1840
1841// This is supposed to be log2(Size)
1843 switch (Size) {
1844 case 4:
1845 return AMD_ELEMENT_4_BYTES;
1846 case 8:
1847 return AMD_ELEMENT_8_BYTES;
1848 case 16:
1849 return AMD_ELEMENT_16_BYTES;
1850 default:
1851 llvm_unreachable("invalid private_element_size");
1852 }
1853}
1854
1855void AMDGPUAsmPrinter::getAmdKernelCode(AMDGPUMCKernelCodeT &Out,
1856 const SIProgramInfo &CurrentProgramInfo,
1857 const MachineFunction &MF) const {
1858 const Function &F = MF.getFunction();
1859 assert(F.getCallingConv() == CallingConv::AMDGPU_KERNEL ||
1860 F.getCallingConv() == CallingConv::SPIR_KERNEL);
1861
1862 const SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
1863 const GCNSubtarget &STM = MF.getSubtarget<GCNSubtarget>();
1864 MCContext &Ctx = MF.getContext();
1865
1866 Out.initDefault(STM, Ctx, /*InitMCExpr=*/false);
1867
1868 Out.compute_pgm_resource1_registers =
1869 CurrentProgramInfo.getComputePGMRSrc1(STM, Ctx);
1870 Out.compute_pgm_resource2_registers =
1871 CurrentProgramInfo.getComputePGMRSrc2(STM, Ctx);
1872 Out.code_properties |= AMD_CODE_PROPERTY_IS_PTR64;
1873
1874 Out.is_dynamic_callstack = CurrentProgramInfo.DynamicCallStack;
1875
1877 getElementByteSizeValue(STM.getMaxPrivateElementSize(true)));
1878
1879 const GCNUserSGPRUsageInfo &UserSGPRInfo = MFI->getUserSGPRInfo();
1880 if (UserSGPRInfo.hasPrivateSegmentBuffer()) {
1882 }
1883
1884 if (UserSGPRInfo.hasDispatchPtr())
1886
1887 if (UserSGPRInfo.hasQueuePtr())
1889
1890 if (UserSGPRInfo.hasKernargSegmentPtr())
1892
1893 if (UserSGPRInfo.hasDispatchID())
1895
1896 if (UserSGPRInfo.hasFlatScratchInit())
1898
1899 if (UserSGPRInfo.hasPrivateSegmentSize())
1901
1902 if (STM.isXNACKEnabled())
1903 Out.code_properties |= AMD_CODE_PROPERTY_IS_XNACK_SUPPORTED;
1904
1905 Align MaxKernArgAlign;
1906 Out.kernarg_segment_byte_size = STM.getKernArgSegmentSize(F, MaxKernArgAlign);
1907 Out.wavefront_sgpr_count = CurrentProgramInfo.NumSGPR;
1908 Out.workitem_vgpr_count = CurrentProgramInfo.NumVGPR;
1909 Out.workitem_private_segment_byte_size = CurrentProgramInfo.ScratchSize;
1910 Out.workgroup_group_segment_byte_size = CurrentProgramInfo.LDSSize;
1911
1912 // kernarg_segment_alignment is specified as log of the alignment.
1913 // The minimum alignment is 16.
1914 // FIXME: The metadata treats the minimum as 4?
1915 Out.kernarg_segment_alignment = Log2(std::max(Align(16), MaxKernArgAlign));
1916}
1917
1919 const char *ExtraCode, raw_ostream &O) {
1920 // First try the generic code, which knows about modifiers like 'c' and 'n'.
1921 if (!AsmPrinter::PrintAsmOperand(MI, OpNo, ExtraCode, O))
1922 return false;
1923
1924 if (ExtraCode && ExtraCode[0]) {
1925 if (ExtraCode[1] != 0)
1926 return true; // Unknown modifier.
1927
1928 switch (ExtraCode[0]) {
1929 case 'r':
1930 break;
1931 default:
1932 return true;
1933 }
1934 }
1935
1936 // TODO: Should be able to support other operand types like globals.
1937 const MachineOperand &MO = MI->getOperand(OpNo);
1938 if (MO.isReg()) {
1940 *MF->getSubtarget().getRegisterInfo());
1941 return false;
1942 }
1943 if (MO.isImm()) {
1944 int64_t Val = MO.getImm();
1946 O << Val;
1947 } else if (isUInt<16>(Val)) {
1948 O << format("0x%" PRIx16, static_cast<uint16_t>(Val));
1949 } else if (isUInt<32>(Val)) {
1950 O << format("0x%" PRIx32, static_cast<uint32_t>(Val));
1951 } else {
1952 O << format("0x%" PRIx64, static_cast<uint64_t>(Val));
1953 }
1954 return false;
1955 }
1956 return true;
1957}
1958
1966
1967void AMDGPUAsmPrinter::emitResourceUsageRemarks(
1968 const MachineFunction &MF, const SIProgramInfo &CurrentProgramInfo,
1969 bool isModuleEntryFunction, bool hasMAIInsts) {
1970 if (!ORE)
1971 return;
1972
1973 const char *Name = "kernel-resource-usage";
1974 const char *Indent = " ";
1975
1976 // If the remark is not specifically enabled, do not output to yaml
1978 if (!Ctx.getDiagHandlerPtr()->isAnalysisRemarkEnabled(Name))
1979 return;
1980
1981 // Currently non-kernel functions have no resources to emit.
1983 return;
1984
1985 auto EmitResourceUsageRemark = [&](StringRef RemarkName,
1986 StringRef RemarkLabel, auto Argument) {
1987 // Add an indent for every line besides the line with the kernel name. This
1988 // makes it easier to tell which resource usage go with which kernel since
1989 // the kernel name will always be displayed first.
1990 std::string LabelStr = RemarkLabel.str() + ": ";
1991 if (RemarkName != "FunctionName")
1992 LabelStr = Indent + LabelStr;
1993
1994 ORE->emit([&]() {
1995 return MachineOptimizationRemarkAnalysis(Name, RemarkName,
1997 &MF.front())
1998 << LabelStr << ore::NV(RemarkName, Argument);
1999 });
2000 };
2001
2002 // FIXME: Formatting here is pretty nasty because clang does not accept
2003 // newlines from diagnostics. This forces us to emit multiple diagnostic
2004 // remarks to simulate newlines. If and when clang does accept newlines, this
2005 // formatting should be aggregated into one remark with newlines to avoid
2006 // printing multiple diagnostic location and diag opts.
2007 EmitResourceUsageRemark("FunctionName", "Function Name",
2008 MF.getFunction().getName());
2009 EmitResourceUsageRemark("NumSGPR", "TotalSGPRs",
2010 getMCExprStr(CurrentProgramInfo.NumSGPR));
2011 EmitResourceUsageRemark("NumVGPR", "VGPRs",
2012 getMCExprStr(CurrentProgramInfo.NumArchVGPR));
2013 if (hasMAIInsts) {
2014 EmitResourceUsageRemark("NumAGPR", "AGPRs",
2015 getMCExprStr(CurrentProgramInfo.NumAccVGPR));
2016 }
2017 EmitResourceUsageRemark("ScratchSize", "ScratchSize [bytes/lane]",
2018 getMCExprStr(CurrentProgramInfo.ScratchSize));
2019 int64_t DynStack;
2020 bool DynStackEvaluatable =
2021 CurrentProgramInfo.DynamicCallStack->evaluateAsAbsolute(DynStack);
2022 StringRef DynamicStackStr =
2023 DynStackEvaluatable && DynStack ? "True" : "False";
2024 EmitResourceUsageRemark("DynamicStack", "Dynamic Stack", DynamicStackStr);
2025 EmitResourceUsageRemark("Occupancy", "Occupancy [waves/SIMD]",
2026 getMCExprStr(CurrentProgramInfo.Occupancy));
2027 EmitResourceUsageRemark("SGPRSpill", "SGPRs Spill",
2028 CurrentProgramInfo.SGPRSpill);
2029 EmitResourceUsageRemark("VGPRSpill", "VGPRs Spill",
2030 CurrentProgramInfo.VGPRSpill);
2031 if (isModuleEntryFunction)
2032 EmitResourceUsageRemark("BytesLDS", "LDS Size [bytes/block]",
2033 CurrentProgramInfo.LDSSize);
2034}
2035
2045
2061
2070
2071char AMDGPUAsmPrinter::ID = 0;
2072
2073INITIALIZE_PASS(AMDGPUAsmPrinter, "amdgpu-asm-printer",
2074 "AMDGPU Assembly Printer", false, false)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static void EmitPALMetadataCommon(AMDGPUPALMetadata *MD, const SIProgramInfo &CurrentProgramInfo, CallingConv::ID CC, const GCNSubtarget &ST, unsigned DynamicVGPRBlockSize)
const AMDGPUMCExpr * createOccupancy(unsigned InitOcc, const MCExpr *NumSGPRs, const MCExpr *NumVGPRs, unsigned DynamicVGPRBlockSize, const GCNSubtarget &STM, MCContext &Ctx)
Mimics GCNSubtarget::computeOccupancy for MCExpr.
static unsigned getRsrcReg(CallingConv::ID CallConv)
LLVM_ABI LLVM_EXTERNAL_VISIBILITY void LLVMInitializeAMDGPUAsmPrinter()
static amd_element_byte_size_t getElementByteSizeValue(unsigned Size)
static const MCExpr * setBits(const MCExpr *Dst, const MCExpr *Value, uint32_t Mask, uint32_t Shift, MCContext &Ctx)
Set bits in a kernel descriptor MCExpr field: return ((Dst & ~Mask) | (Value << Shift))
static uint32_t getFPMode(SIModeRegisterDefaults Mode)
static std::string computeTypeId(const FunctionType *FTy, const DataLayout &DL)
static const MCExpr * computeAccumOffset(const MCExpr *NumVGPR, MCContext &Ctx)
static void appendTypeEncoding(std::string &Enc, Type *Ty, const DataLayout &DL, bool IsReturnType)
static AsmPrinter * createAMDGPUAsmPrinterPass(TargetMachine &tm, std::unique_ptr< MCStreamer > &&Streamer)
AMDGPU Assembly printer class.
unsigned uint64_t
AMDGPU HSA Metadata Streamer.
AMDHSA kernel descriptor MCExpr struct for use in MC layer.
MC infrastructure to propagate the function level resource usage info.
Analyzes how many registers and other resources are used by functions.
The AMDGPU TargetMachine interface definition for hw codegen targets.
AMDHSA kernel descriptor definitions.
MC layer struct for AMDGPUMCKernelCodeT, provides MCExpr functionality where required.
amd_element_byte_size_t
The values used to define the number of bytes to use for the swizzle element size.
@ AMD_ELEMENT_8_BYTES
@ AMD_ELEMENT_16_BYTES
@ AMD_ELEMENT_4_BYTES
#define AMD_HSA_BITS_SET(dst, mask, val)
@ AMD_CODE_PROPERTY_ENABLE_SGPR_DISPATCH_ID
@ AMD_CODE_PROPERTY_PRIVATE_ELEMENT_SIZE
@ AMD_CODE_PROPERTY_ENABLE_SGPR_KERNARG_SEGMENT_PTR
@ AMD_CODE_PROPERTY_ENABLE_SGPR_QUEUE_PTR
@ AMD_CODE_PROPERTY_ENABLE_SGPR_PRIVATE_SEGMENT_SIZE
@ AMD_CODE_PROPERTY_ENABLE_SGPR_PRIVATE_SEGMENT_BUFFER
@ AMD_CODE_PROPERTY_ENABLE_SGPR_DISPATCH_PTR
@ AMD_CODE_PROPERTY_IS_XNACK_SUPPORTED
@ AMD_CODE_PROPERTY_ENABLE_SGPR_FLAT_SCRATCH_INIT
@ AMD_CODE_PROPERTY_IS_PTR64
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static const Function * getParent(const Value *V)
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
#define LLVM_ABI
Definition Compiler.h:215
#define LLVM_EXTERNAL_VISIBILITY
Definition Compiler.h:132
AMD GCN specific subclass of TargetSubtarget.
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
===- MachineOptimizationRemarkEmitter.h - Opt Diagnostics -*- C++ -*-—===//
modulo schedule test
OptimizedStructLayoutField Field
ModuleAnalysisManager MAM
#define INITIALIZE_PASS(passName, arg, name, cfg, analysis)
Definition PassSupport.h:56
R600 Assembly printer class.
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
#define R_00B028_SPI_SHADER_PGM_RSRC1_PS
Definition SIDefines.h:1378
#define R_0286E8_SPI_TMPRING_SIZE
Definition SIDefines.h:1520
#define FP_ROUND_MODE_DP(x)
Definition SIDefines.h:1502
#define C_00B84C_SCRATCH_EN
Definition SIDefines.h:1414
#define FP_ROUND_ROUND_TO_NEAREST
Definition SIDefines.h:1494
#define R_0286D0_SPI_PS_INPUT_ADDR
Definition SIDefines.h:1453
#define R_00B860_COMPUTE_TMPRING_SIZE
Definition SIDefines.h:1515
#define R_00B428_SPI_SHADER_PGM_RSRC1_HS
Definition SIDefines.h:1401
#define R_00B328_SPI_SHADER_PGM_RSRC1_ES
Definition SIDefines.h:1400
#define R_00B528_SPI_SHADER_PGM_RSRC1_LS
Definition SIDefines.h:1409
#define R_0286CC_SPI_PS_INPUT_ENA
Definition SIDefines.h:1452
#define R_00B128_SPI_SHADER_PGM_RSRC1_VS
Definition SIDefines.h:1387
#define FP_DENORM_MODE_DP(x)
Definition SIDefines.h:1513
#define R_00B848_COMPUTE_PGM_RSRC1
Definition SIDefines.h:1455
#define R_SPILLED_SGPRS
Definition SIDefines.h:1534
#define FP_ROUND_MODE_SP(x)
Definition SIDefines.h:1501
#define FP_DENORM_MODE_SP(x)
Definition SIDefines.h:1512
#define R_00B228_SPI_SHADER_PGM_RSRC1_GS
Definition SIDefines.h:1392
#define R_SPILLED_VGPRS
Definition SIDefines.h:1535
#define S_00B02C_EXTRA_LDS_SIZE(x)
Definition SIDefines.h:1386
#define R_00B84C_COMPUTE_PGM_RSRC2
Definition SIDefines.h:1411
#define R_00B02C_SPI_SHADER_PGM_RSRC2_PS
Definition SIDefines.h:1385
std::unique_ptr< MCStreamer > && Streamer
static const int BlockSize
Definition TarWriter.cpp:33
static cl::opt< unsigned > CacheLineSize("cache-line-size", cl::init(0), cl::Hidden, cl::desc("Use this to override the target cache line size when " "specified by the user."))
PreservedAnalyses run(Module &M, ModuleAnalysisManager &MAM)
PreservedAnalyses run(Module &M, ModuleAnalysisManager &MAM)
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
void emitFunctionEntryLabel() override
EmitFunctionEntryLabel - Emit the label that is the entrypoint for the function.
const MCSubtargetInfo * getGlobalSTI() const
void emitImplicitDef(const MachineInstr *MI) const override
Targets can override this to customize the output of IMPLICIT_DEF instructions in verbose mode.
std::vector< std::string > DisasmLines
std::function< const AMDGPUResourceUsageAnalysisImpl::SIFunctionResourceInfo *(MachineFunction &)> GetResourceUsage
void emitStartOfAsmFile(Module &M) override
This virtual method can be overridden by targets that want to emit something at the start of their fi...
void endFunction(const MachineFunction *MF)
StringRef getPassName() const override
getPassName - Return a nice clean name for a pass.
std::vector< std::string > HexLines
void emitGlobalVariable(const GlobalVariable *GV) override
Emit the specified global variable to the .s file.
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - This function should be overriden by passes that need analysis information to do t...
bool PrintAsmOperand(const MachineInstr *MI, unsigned OpNo, const char *ExtraCode, raw_ostream &O) override
Print the specified operand of MI, an INLINEASM instruction, using the specified assembler variant.
bool runOnMachineFunction(MachineFunction &MF) override
runOnMachineFunction - This method must be overloaded to perform the desired machine code transformat...
bool doFinalization(Module &M) override
doFinalization - Virtual method overriden by subclasses to do any necessary clean up after all passes...
void emitEndOfAsmFile(Module &M) override
This virtual method can be overridden by targets that want to emit something at the end of their file...
AMDGPUAsmPrinter(TargetMachine &TM, std::unique_ptr< MCStreamer > Streamer)
bool doInitialization(Module &M) override
doInitialization - Virtual method overridden by subclasses to do any necessary initialization before ...
void emitFunctionBodyStart() override
Targets can override this to emit stuff before the first basic block in the function.
void emitBasicBlockStart(const MachineBasicBlock &MBB) override
Targets can override this to emit stuff at the start of a basic block.
AMDGPUTargetStreamer * getTargetStreamer() const
static void printRegOperand(MCRegister Reg, raw_ostream &O, const MCRegisterInfo &MRI)
AMDGPU target specific MCExpr operations.
static const AMDGPUMCExpr * createInstPrefSize(const MCExpr *CodeSizeBytes, MCContext &Ctx)
Create an expression for instruction prefetch size computation: min(divideCeil(CodeSizeBytes,...
static const AMDGPUMCExpr * createMax(ArrayRef< const MCExpr * > Args, MCContext &Ctx)
static const AMDGPUMCExpr * createTotalNumVGPR(const MCExpr *NumAGPR, const MCExpr *NumVGPR, MCContext &Ctx)
static const AMDGPUMCExpr * create(VariantKind Kind, ArrayRef< const MCExpr * > Args, MCContext &Ctx)
static const AMDGPUMCExpr * createExtraSGPRs(const MCExpr *VCCUsed, const MCExpr *FlatScrUsed, bool XNACKUsed, MCContext &Ctx)
Allow delayed MCExpr resolve of ExtraSGPRs (in case VCCUsed or FlatScrUsed are unresolvable but neede...
static const AMDGPUMCExpr * createAlignTo(const MCExpr *Value, const MCExpr *Align, MCContext &Ctx)
void setHwStage(unsigned CC, StringRef field, unsigned Val)
void updateHwStageMaximum(unsigned CC, StringRef field, unsigned Val)
void setComputeRegisters(StringRef field, unsigned Val)
std::pair< unsigned, unsigned > getOccupancyWithWorkGroupSizes(uint32_t LDSBytes, const Function &F) const
Subtarget's minimum/maximum occupancy, in number of waves per EU, that can be achieved when the only ...
unsigned getAddressableLocalMemorySize() const
Return the maximum number of bytes of LDS that can be allocated to a single workgroup.
unsigned getKernArgSegmentSize(const Function &F, Align &MaxAlign) const
unsigned getWavefrontSize() const
virtual void EmitAmdhsaKernelDescriptor(const MCSubtargetInfo &STI, StringRef KernelName, const AMDGPU::MCKernelDescriptor &KernelDescriptor, const MCExpr *NextVGPR, const MCExpr *NextSGPR, const MCExpr *ReserveVCC, const MCExpr *ReserveFlatScr)
virtual void emitAMDGPUInfo(const AMDGPU::InfoSectionData &Data)
AMDGPUPALMetadata * getPALMetadata()
void initializeTargetID(const MCSubtargetInfo &STI, bool ApplyFeatureString=false)
virtual void EmitDirectiveAMDHSACodeObjectVersion(unsigned COV)
virtual void EmitMCResourceInfo(const MCSymbol *NumVGPR, const MCSymbol *NumAGPR, const MCSymbol *NumExplicitSGPR, const MCSymbol *NumNamedBarrier, const MCSymbol *PrivateSegmentSize, const MCSymbol *UsesVCC, const MCSymbol *UsesFlatScratch, const MCSymbol *HasDynamicallySizedStack, const MCSymbol *HasRecursion, const MCSymbol *HasIndirectCall)
virtual bool EmitCodeEnd(const MCSubtargetInfo &STI)
virtual void EmitAMDGPUSymbolType(StringRef SymbolName, unsigned Type)
const std::optional< AMDGPU::TargetID > & getTargetID() const
virtual void EmitAMDKernelCodeT(AMDGPU::AMDGPUMCKernelCodeT &Header)
virtual void EmitMCResourceMaximums(const MCSymbol *MaxVGPR, const MCSymbol *MaxAGPR, const MCSymbol *MaxSGPR, const MCSymbol *MaxNamedBarrier)
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
This class represents an incoming formal argument to a Function.
Definition Argument.h:32
Collects and handles AsmPrinter objects required to build debug or EH information.
This class is intended to be used as a driving class for all asm writers.
Definition AsmPrinter.h:91
const TargetLoweringObjectFile & getObjFileLowering() const
Return information about object file lowering.
MCSymbol * getSymbol(const GlobalValue *GV) const
virtual void emitGlobalVariable(const GlobalVariable *GV)
Emit the specified global variable to the .s file.
TargetMachine & TM
Target machine description.
Definition AsmPrinter.h:94
MachineFunction * MF
The current machine function.
Definition AsmPrinter.h:109
virtual void SetupMachineFunction(MachineFunction &MF)
This should be called when a new MachineFunction is being processed from runOnMachineFunction.
void emitFunctionBody()
This method emits the body and trailer for a function.
virtual bool isBlockOnlyReachableByFallthrough(const MachineBasicBlock *MBB) const
Return true if the basic block has exactly one predecessor and the control transfer mechanism between...
bool doInitialization(Module &M) override
Set up the AsmPrinter when we are working on a new module.
virtual void emitLinkage(const GlobalValue *GV, MCSymbol *GVSym) const
This emits linkage information about GVSym based on GV, if this is supported by the target.
void getAnalysisUsage(AnalysisUsage &AU) const override
Record analysis usage.
unsigned getFunctionNumber() const
Return a unique ID for the current function.
MachineOptimizationRemarkEmitter * ORE
Optimization remark emitter.
Definition AsmPrinter.h:124
AsmPrinter(TargetMachine &TM, std::unique_ptr< MCStreamer > Streamer, char &ID=AsmPrinter::ID)
MCSymbol * CurrentFnSym
The symbol for the current function.
Definition AsmPrinter.h:131
MachineModuleInfo * MMI
This is a pointer to the current MachineModuleInfo.
Definition AsmPrinter.h:112
MCContext & OutContext
This is the context for the output file that we are streaming.
Definition AsmPrinter.h:101
bool doFinalization(Module &M) override
Shut down the asmprinter.
virtual void emitBasicBlockStart(const MachineBasicBlock &MBB)
Targets can override this to emit stuff at the start of a basic block.
void emitVisibility(MCSymbol *Sym, unsigned Visibility, bool IsDefinition=true) const
This emits visibility information about symbol, if this is supported by the target.
bool runOnMachineFunction(MachineFunction &MF) override
Emit the specified function out to the OutStreamer.
Definition AsmPrinter.h:456
std::unique_ptr< MCStreamer > OutStreamer
This is the MCStreamer object for the file we are generating.
Definition AsmPrinter.h:106
const MCAsmInfo & MAI
Target Asm Printer information.
Definition AsmPrinter.h:97
std::function< MachineModuleInfo *()> GetMMI
Definition AsmPrinter.h:179
bool isVerbose() const
Return true if assembly output should contain comments.
Definition AsmPrinter.h:313
MCSymbol * getFunctionEnd() const
Definition AsmPrinter.h:323
void getNameWithPrefix(SmallVectorImpl< char > &Name, const GlobalValue *GV) const
virtual void emitFunctionEntryLabel()
EmitFunctionEntryLabel - Emit the label that is the entrypoint for the function.
void addAsmPrinterHandler(std::unique_ptr< AsmPrinterHandler > Handler)
virtual bool PrintAsmOperand(const MachineInstr *MI, unsigned OpNo, const char *ExtraCode, raw_ostream &OS)
Print the specified operand of MI, an INLINEASM instruction, using the specified assembler variant.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
bool empty() const
Definition DenseMap.h:717
DISubprogram * getSubprogram() const
Get the attached subprogram.
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
unsigned getMinNumSGPRs(unsigned WavesPerEU) const
unsigned getTotalNumVGPRs() const
unsigned getMinNumVGPRs(unsigned WavesPerEU, unsigned DynamicVGPRBlockSize) const
bool hasInstPrefSize() const
bool isCuModeEnabled() const
std::pair< unsigned, unsigned > computeOccupancy(const Function &F, unsigned LDSSize=0, unsigned NumSGPRs=0, unsigned NumVGPRs=0) const
Subtarget's minimum/maximum occupancy, in number of waves per EU, that can be achieved when the only ...
const AMDGPU::TargetID & getTargetID() const
bool isWave32() const
void getInstPrefSizeArgs(uint32_t &Mask, uint32_t &Shift, uint32_t &Width, uint32_t &CacheLineSize) const
unsigned getMaxNumUserSGPRs() const
unsigned getMaxWavesPerEU() const
Generation getGeneration() const
unsigned getAddressableNumSGPRs() const
unsigned getMaxWaveScratchSize() const
static AMDGPU::TargetIDSetting getTargetIDSettingFromModuleFlag(const Module &M, StringRef FlagName)
Get xnack/sramecc setting from module flag or cl::opt (for testing).
bool hasPrivateSegmentBuffer() const
VisibilityTypes getVisibility() const
LLVM_ABI bool isDeclaration() const
Return true if the primary definition of this global value is outside of the current translation unit...
Definition Globals.cpp:408
unsigned getAddressSpace() const
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this global belongs to.
Definition Globals.cpp:205
const Constant * getInitializer() const
getInitializer - Return the initializer for this global variable.
bool hasInitializer() const
Definitions have initializers, declarations don't.
MaybeAlign getAlign() const
Returns the alignment of the given variable.
LLVM_ABI uint64_t getGlobalSize(const DataLayout &DL) const
Get the size of this global variable in bytes.
Definition Globals.cpp:640
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LLVM_ABI void diagnose(const DiagnosticInfo &DI)
Report a message to the currently installed diagnostic handler.
MCCodeEmitter * getEmitterPtr() const
static const MCBinaryExpr * createAdd(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx, SMLoc Loc=SMLoc())
Definition MCExpr.h:342
static const MCBinaryExpr * createAnd(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:347
static const MCBinaryExpr * createOr(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:407
static const MCBinaryExpr * createLOr(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:377
static const MCBinaryExpr * createMul(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:397
static const MCBinaryExpr * createGT(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:362
static const MCBinaryExpr * createDiv(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:352
static const MCBinaryExpr * createShl(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:412
static const MCBinaryExpr * createSub(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:427
static LLVM_ABI const MCConstantExpr * create(int64_t Value, MCContext &Ctx, bool PrintInHex=false, unsigned SizeInBytes=0)
Definition MCExpr.cpp:212
Context object for machine code objects.
Definition MCContext.h:83
LLVM_ABI void reportError(SMLoc L, const Twine &Msg)
LLVM_ABI MCSymbol * getOrCreateSymbol(const Twine &Name)
Lookup the symbol inside with the specified Name.
Base class for the full range of assembler expressions which are needed for parsing.
Definition MCExpr.h:34
LLVM_ABI bool evaluateAsRelocatable(MCValue &Res, const MCAssembler *Asm) const
Try to evaluate the expression to a relocatable value, i.e.
Definition MCExpr.cpp:450
MCSection * getTextSection() const
MCContext & getContext() const
This represents a section on linux, lots of unix variants and some bare metal systems.
Instances of this class represent a uniqued identifier for a section in the current translation unit.
Definition MCSection.h:580
bool hasInstructions() const
Definition MCSection.h:676
Generic base class for all target subtargets.
StringRef getFeatureString() const
static const MCSymbolRefExpr * create(const MCSymbol *Symbol, MCContext &Ctx, SMLoc Loc=SMLoc())
Definition MCExpr.h:213
MCSymbol - Instances of this class represent a symbol name in the MC file, and MCSymbols are created ...
Definition MCSymbol.h:42
bool isDefined() const
isDefined - Check if this symbol is defined (i.e., it has an address).
Definition MCSymbol.h:233
StringRef getName() const
getName - Get the symbol name.
Definition MCSymbol.h:188
bool isVariable() const
isVariable - Check if this is a variable symbol.
Definition MCSymbol.h:267
void redefineIfPossible()
Prepare this symbol to be redefined.
Definition MCSymbol.h:212
const MCExpr * getVariableValue() const
Get the expression of the variable symbol.
Definition MCSymbol.h:270
MCStreamer & getStreamer()
Definition MCStreamer.h:103
static const MCUnaryExpr * createNot(const MCExpr *Expr, MCContext &Ctx, SMLoc Loc=SMLoc())
Definition MCExpr.h:272
uint64_t getStackSize() const
Return the number of bytes that must be allocated to hold all of the fixed size frame objects.
MCContext & getContext() const
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
const MachineBasicBlock & front() const
Representation of each machine instruction.
MachineOperand class - Representation of each machine instruction operand.
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
Register getReg() const
getReg - Returns the register number.
Diagnostic information for optimization analysis remarks.
LLVM_ABI void emit(DiagnosticInfoOptimizationBase &OptDiag)
Emit an optimization remark.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
LLVM_ABI unsigned getNumOperands() const
iterator_range< op_iterator > operands()
Definition Metadata.h:1893
AnalysisType * getAnalysisIfAvailable() const
getAnalysisIfAvailable<AnalysisType>() - Subclasses use this function to get analysis information tha...
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
Wrapper class representing virtual and physical registers.
Definition Register.h:20
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
GCNUserSGPRUsageInfo & getUserSGPRInfo()
SIModeRegisterDefaults getMode() const
unsigned getScratchReservedForDynamicVGPRs() const
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
void push_back(const T &Elt)
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
bool contains(StringRef Other) const
Return true if the given string is a substring of *this, and false otherwise.
Definition StringRef.h:446
std::pair< typename Base::iterator, bool > insert(StringRef key)
Definition StringSet.h:39
Primary interface to the complete machine description for the target machine.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
LLVM Value Representation.
Definition Value.h:75
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
A raw_ostream that writes to an SmallVector or SmallString.
StringRef str() const
Return a StringRef for the vector contents.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ LOCAL_ADDRESS
Address space for local memory.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
bool isSGPROccupancyLimited(const MCSubtargetInfo &STI)
unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI, std::optional< bool > EnableWavefrontSize32)
static constexpr unsigned MaxDynamicVGPRBlocks
Maximum number of VGPR blocks that can be allocated in dynamic VGPR mode.
unsigned getSGPREncodingGranule(const MCSubtargetInfo &STI)
unsigned getNumExtraSGPRs(const MCSubtargetInfo &STI, bool VCCUsed, bool FlatScrUsed, bool XNACKUsed)
unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
LLVM_ABI unsigned getLDSEncodingGranule(GPUKind AK)
void printAMDGPUMCExpr(const MCExpr *Expr, raw_ostream &OS, const MCAsmInfo *MAI)
LLVM_READNONE constexpr bool isModuleEntryFunctionCC(CallingConv::ID CC)
LLVM_ABI IsaVersion getIsaVersion(StringRef GPU)
LLVM_ABI unsigned getTotalNumVGPRs(GPUKind AK, bool IsWave32)
LLVM_ABI unsigned getTotalNumSGPRs(GPUKind AK)
const MCExpr * maskShiftSet(const MCExpr *Val, uint32_t Mask, uint32_t Shift, MCContext &Ctx)
Provided with the MCExpr * Val, uint32 Mask and Shift, will return the masked and left shifted,...
unsigned getAMDHSACodeObjectVersion(const Module &M)
bool isTgSplitEnabled(const Function &F)
GPUKind
GPU kinds supported by the AMDGPU target.
bool isGFX90A(const MCSubtargetInfo &STI)
LLVM_READNONE constexpr bool isKernel(CallingConv::ID CC)
LLVM_ABI unsigned getSGPRAllocGranule(GPUKind AK)
LLVM_READNONE constexpr bool isEntryFunctionCC(CallingConv::ID CC)
LLVM_READNONE constexpr bool isCompute(CallingConv::ID CC)
bool hasMAIInsts(const MCSubtargetInfo &STI)
LLVM_ABI Triple::SubArchType getSubArch(GPUKind AK)
LLVM_READNONE bool isInlinableIntLiteral(int64_t Literal)
Is this literal inlinable, and not one of the values intended for floating point values.
LLVM_ABI GPUKind parseArchAMDGCN(StringRef CPU)
const MCExpr * foldAMDGPUMCExpr(const MCExpr *Expr, MCContext &Ctx)
bool isGFX10Plus(const MCSubtargetInfo &STI)
constexpr std::pair< unsigned, unsigned > getShiftMask(unsigned Value)
Deduce the least significant bit aligned shift and mask values for a binary Complement Value (as they...
unsigned hasKernargPreload(const MCSubtargetInfo &STI)
std::pair< unsigned, unsigned > getIntegerPairAttribute(const Function &F, StringRef Name, std::pair< unsigned, unsigned > Default, bool OnlyFirstRequired)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
@ AMDGPU_VS
Used for Mesa vertex shaders, or AMDPAL last shader stage before rasterization (vertex shader if tess...
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ AMDGPU_HS
Used for Mesa/AMDPAL hull shaders (= tessellation control shaders).
@ AMDGPU_GS
Used for Mesa/AMDPAL geometry shaders.
@ AMDGPU_CS_Chain
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_PS
Used for Mesa/AMDPAL pixel shaders.
@ SPIR_KERNEL
Used for SPIR kernel functions.
@ AMDGPU_ES
Used for AMDPAL shader stage before geometry shader if geometry is in use.
@ AMDGPU_LS
Used for AMDPAL vertex shader if tessellation is in use.
@ SHT_PROGBITS
Definition ELF.h:1157
@ STT_AMDGPU_HSA_KERNEL
Definition ELF.h:1447
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract(Y &&MD)
Extract a Value from Metadata.
Definition Metadata.h:679
DiagnosticInfoOptimizationBase::Argument NV
NodeAddr< FuncNode * > Func
Definition RDFGraph.h:393
This is an optimization pass for GlobalISel generic memory operations.
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
Definition STLExtras.h:1685
OuterAnalysisManagerProxy< ModuleAnalysisManager, MachineFunction > ModuleAnalysisManagerMachineFunctionProxy
Provide the ModuleAnalysisManager to Function proxy.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
static StringRef getCPU(StringRef CPU)
Processes a CPU name.
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
Target & getTheR600Target()
The target for R600 GPUs.
@ DK_ResourceLimit
AsmPrinter * createR600AsmPrinterPass(TargetMachine &TM, std::unique_ptr< MCStreamer > &&Streamer)
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
LLVM_ABI void setupModuleAsmPrinter(Module &M, ModuleAnalysisManager &MAM, AsmPrinter &AsmPrinter)
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
format_object< Ts... > format(const char *Fmt, const Ts &... Vals)
These are helper functions used to produce formatted output.
Definition Format.h:102
@ Success
The lock was released successfully.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
Target & getTheGCNTarget()
The target for GCN GPUs.
OutputIt move(R &&Range, OutputIt Out)
Provide wrappers to std::move which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1933
LLVM_ABI void setupMachineFunctionAsmPrinter(MachineFunctionAnalysisManager &MFAM, MachineFunction &MF, AsmPrinter &AsmPrinter)
Target & getTheGCNLegacyTarget()
The target for GCN GPUs, registered under the legacy "amdgcn" architecture name for use with -march.
unsigned Log2(Align A)
Returns the log2 of the alignment.
Definition Alignment.h:197
LLVM_ABI Printable printReg(Register Reg, const TargetRegisterInfo *TRI=nullptr, unsigned SubIdx=0, const MachineRegisterInfo *MRI=nullptr)
Prints virtual and physical registers with or without a TRI instance.
AnalysisManager< Module > ModuleAnalysisManager
Convenience typedef for the Module analysis manager.
Definition MIRParser.h:39
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
Implement std::hash so that hash_code can be used in STL containers.
Definition BitVector.h:878
#define N
AMDGPUResourceUsageAnalysisImpl::SIFunctionResourceInfo FunctionResourceInfo
void validate(const MCSubtargetInfo *STI, MCContext &Ctx)
static const MCExpr * bits_get(const MCExpr *Src, uint32_t Shift, uint32_t Mask, MCContext &Ctx)
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Track resource usage for kernels / entry functions.
const MCExpr * NumSGPR
const MCExpr * NumArchVGPR
const MCExpr * VGPRBlocks
const MCExpr * ScratchBlocks
const MCExpr * ComputePGMRSrc3
const MCExpr * getComputePGMRSrc1(const GCNSubtarget &ST, MCContext &Ctx) const
Compute the value of the ComputePGMRsrc1 register.
const MCExpr * VCCUsed
const MCExpr * FlatUsed
const MCExpr * NamedBarCnt
const MCExpr * ScratchEnable
const MCExpr * AccumOffset
const MCExpr * NumAccVGPR
const MCExpr * DynamicCallStack
const MCExpr * SGPRBlocks
const MCExpr * NumVGPRsForWavesPerEU
const MCExpr * NumVGPR
const MCExpr * Occupancy
const MCExpr * ScratchSize
const MCExpr * NumSGPRsForWavesPerEU
const MCExpr * getComputePGMRSrc2(const GCNSubtarget &ST, MCContext &Ctx) const
Compute the value of the ComputePGMRsrc2 register.
static void RegisterAsmPrinter(Target &T, Target::AsmPrinterCtorTy Fn)
RegisterAsmPrinter - Register an AsmPrinter implementation for the given target.