LLVM 24.0.0git
AMDGPU.cpp
Go to the documentation of this file.
1//===- AMDGPU.cpp - AMDGPU ABI Implementation ----------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
11#include "llvm/ABI/TargetInfo.h"
12#include "llvm/ABI/Types.h"
17#include <algorithm>
18#include <cassert>
19#include <cstdint>
20
21namespace llvm {
22namespace abi {
23
24class AMDGPUTargetInfo final : public DefaultTargetInfo {
25private:
26 static const unsigned MaxNumRegsForArgsRet = 16;
27
28 ABICompatInfo CompatInfo;
29
30 /// HIP coerces a generic scalar-pointer kernel argument to the global
31 /// address space. Gated by the front end, which alone can see LangOpts.HIP.
32 bool CoerceGenericPtrArgToGlobal;
33
34 ArgInfo classifyReturnType(const Type *RetTy) const;
35 ArgInfo classifyKernelArgumentType(const Type *Ty) const;
36 ArgInfo classifyArgumentType(const Type *Ty, bool Variadic,
37 unsigned &NumRegsLeft) const;
38
39 /// Estimate number of registers the type will use when passed in registers.
40 uint64_t getNumRegsForType(const Type *Ty) const;
41
42public:
44 bool CoerceGenericPtrArgToGlobal)
45 : DefaultTargetInfo(TypeBuilder), CompatInfo(Compat),
46 CoerceGenericPtrArgToGlobal(CoerceGenericPtrArgToGlobal) {}
47
48 const ABICompatInfo &getABICompatInfo() const override { return CompatInfo; }
49
50 /// Indirect arguments live in the private (alloca) address space on AMDGPU.
51 unsigned getAllocaAddrSpace() const override {
53 }
54
55 void computeInfo(FunctionInfo &FI) const override;
56};
57
58uint64_t AMDGPUTargetInfo::getNumRegsForType(const Type *Ty) const {
59 uint64_t NumRegs = 0;
60
61 if (const auto *VT = dyn_cast<VectorType>(Ty)) {
62 // Compute from the number of elements. The reported size is based on the
63 // in-memory size, which includes the padding 4th element for 3-vectors.
64 const Type *EltTy = VT->getElementType();
65 uint64_t EltSize = EltTy->getSizeInBits().getFixedValue();
66 unsigned NumElts = VT->getNumElements().getFixedValue();
67
68 // 16-bit element vectors should be passed as packed.
69 if (EltSize == 16)
70 return (NumElts + 1) / 2;
71
72 uint64_t EltNumRegs = (EltSize + 31) / 32;
73 return EltNumRegs * NumElts;
74 }
75
76 if (const auto *RT = dyn_cast<RecordType>(Ty)) {
77 for (const FieldInfo &Field : RT->getFields())
78 NumRegs += getNumRegsForType(Field.FieldType);
79 return NumRegs;
80 }
81
82 return (Ty->getSizeInBits().getFixedValue() + 31) / 32;
83}
84
85ArgInfo AMDGPUTargetInfo::classifyReturnType(const Type *RetTy) const {
86 if (RetTy->isVoid())
87 return ArgInfo::getIgnore();
88
89 if (isAggregateTypeForABI(RetTy)) {
90 // Records with non-trivial destructors/copy-constructors should not be
91 // returned by value.
92 if (getRecordArgABI(RetTy) == RAA_Default) {
93 const auto *RT = dyn_cast<RecordType>(RetTy);
94
95 // Ignore empty structs/unions.
96 if (RT && RT->isEmpty())
97 return ArgInfo::getIgnore();
98
99 // Lower single-element structs to just return a regular value.
100 if (const Type *SeltTy = isSingleElementStruct(RetTy))
101 return ArgInfo::getDirect(SeltTy);
102
103 if (RT && RT->hasFlexibleArrayMember())
105
106 // Pack aggregates <= 4 bytes into single VGPR or pair.
107 uint64_t Size = RetTy->getSizeInBits().getFixedValue();
108 if (Size <= 16)
109 return ArgInfo::getDirect(TB.getIntegerType(16, Align(2), false));
110
111 if (Size <= 32)
112 return ArgInfo::getDirect(TB.getIntegerType(32, Align(4), false));
113
114 if (Size <= 64) {
115 const Type *I32Ty = TB.getIntegerType(32, Align(4), false);
116 return ArgInfo::getDirect(TB.getArrayType(I32Ty, 2, /*SizeInBits=*/64));
117 }
118
119 if (getNumRegsForType(RetTy) <= MaxNumRegsForArgsRet)
120 return ArgInfo::getDirect();
121 }
122 }
123
124 // Otherwise just do the default thing.
126}
127
128/// For kernels all parameters are really passed in a special buffer. It doesn't
129/// make sense to pass anything byval, so everything must be direct.
130ArgInfo AMDGPUTargetInfo::classifyKernelArgumentType(const Type *Ty) const {
132
133 if (const Type *SeltTy = isSingleElementStruct(Ty))
134 Ty = SeltTy;
135
136 // HIP passes a generic scalar pointer as a global pointer; a pointer is not
137 // an aggregate, so this stays on the direct path.
138 if (CoerceGenericPtrArgToGlobal) {
139 if (const auto *PtrTy = dyn_cast<PointerType>(Ty);
140 PtrTy && PtrTy->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS) {
141 const Type *Coerced =
142 TB.getPointerType(PtrTy->getSizeInBits().getFixedValue(),
143 PtrTy->getAlignment(), AMDGPUAS::GLOBAL_ADDRESS);
144 return ArgInfo::getDirect(Coerced, /*Offset=*/0, /*Align=*/std::nullopt,
145 /*CanBeFlattened=*/false);
146 }
147 }
148
149 // FIXME: This doesn't apply the optimization of coercing pointers in structs
150 // to global address space when using byref. This would require implementing a
151 // new kind of coercion of the in-memory type when for indirect arguments.
152 if (isAggregateTypeForABI(Ty))
154 Ty->getAlignment(),
155 /*AddrSpace=*/AMDGPUAS::CONSTANT_ADDRESS);
156
157 // CanBeFlattened=false keeps the struct intact.
158 return ArgInfo::getDirect(Ty, /*Offset=*/0, /*Align=*/std::nullopt,
159 /*CanBeFlattened=*/false);
160}
161
162ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
163 unsigned &NumRegsLeft) const {
164 assert(NumRegsLeft <= MaxNumRegsForArgsRet && "register estimate underflow");
165
167
168 // Variadic aggregates are kept intact rather than flattened into fields.
169 if (Variadic)
170 return ArgInfo::getDirect(/*T=*/nullptr, /*Offset=*/0,
171 /*Align=*/std::nullopt, /*CanBeFlattened=*/false);
172
173 if (isAggregateTypeForABI(Ty)) {
174 // Records with non-trivial destructors/copy-constructors should not be
175 // passed by value.
176 if (RecordArgABI RAA = getRecordArgABI(Ty); RAA != RAA_Default)
177 return ArgInfo::getIndirect(Ty->getAlignment(),
178 /*ByVal=*/RAA == RAA_DirectInMemory,
179 /*AddrSpace=*/AMDGPUAS::PRIVATE_ADDRESS);
180
181 // Ignore empty structs/unions.
182 if (Ty->isEmptyRecord())
183 return ArgInfo::getIgnore();
184
185 // Lower single-element structs to just pass a regular value.
186 if (const Type *SeltTy = isSingleElementStruct(Ty))
187 return ArgInfo::getDirect(SeltTy);
188
189 if (const auto *RT = dyn_cast<RecordType>(Ty);
190 RT && RT->hasFlexibleArrayMember())
192
193 // Pack aggregates <= 8 bytes into single VGPR or pair.
194 uint64_t Size = Ty->getSizeInBits().getFixedValue();
195 if (Size <= 64) {
196 unsigned NumRegs = (Size + 31) / 32;
197 NumRegsLeft -= std::min(NumRegsLeft, NumRegs);
198
199 if (Size <= 16)
200 return ArgInfo::getDirect(TB.getIntegerType(16, Align(2), false));
201
202 if (Size <= 32)
203 return ArgInfo::getDirect(TB.getIntegerType(32, Align(4), false));
204
205 const Type *I32Ty = TB.getIntegerType(32, Align(4), false);
206 return ArgInfo::getDirect(TB.getArrayType(I32Ty, 2, /*SizeInBits=*/64));
207 }
208
209 if (NumRegsLeft > 0) {
210 uint64_t NumRegs = getNumRegsForType(Ty);
211 if (NumRegsLeft >= NumRegs) {
212 NumRegsLeft -= NumRegs;
213 return ArgInfo::getDirect();
214 }
215 }
216
217 // Pass a struct argument by reference rather than by value.
218 return ArgInfo::getIndirectAliased(Ty->getAlignment(),
219 /*AddrSpace=*/AMDGPUAS::PRIVATE_ADDRESS);
220 }
221
222 // Otherwise just do the default thing.
224 if (!AI.isIndirect()) {
225 uint64_t NumRegs = getNumRegsForType(Ty);
226 NumRegsLeft -= std::min(NumRegs, uint64_t{NumRegsLeft});
227 }
228
229 return AI;
230}
231
234
235 // Non-trivial C++ records are returned indirectly
236 // in the flat address space.
238 FI.getReturnInfo() = classifyReturnType(FI.getReturnType());
239
240 unsigned ArgumentIndex = 0;
241 const unsigned NumFixedArguments = FI.getNumRequiredArgs();
242
243 unsigned NumRegsLeft = MaxNumRegsForArgsRet;
244 for (ArgEntry &Arg : FI.arguments()) {
245 if (CC == CallingConv::AMDGPU_KERNEL) {
246 Arg.Info = classifyKernelArgumentType(Arg.ABIType);
247 } else {
248 bool FixedArgument = ArgumentIndex++ < NumFixedArguments;
249 Arg.Info = classifyArgumentType(Arg.ABIType, !FixedArgument, NumRegsLeft);
250 }
251 }
252}
253
254std::unique_ptr<TargetInfo>
255createAMDGPUTargetInfo(TypeBuilder &TB, bool CoerceGenericPtrArgToGlobal) {
256 return std::make_unique<AMDGPUTargetInfo>(TB, ABICompatInfo(),
257 CoerceGenericPtrArgToGlobal);
258}
259
260} // namespace abi
261} // namespace llvm
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU address space definition.
unsigned uint64_t
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
Default ABI classification shared by targets without special rules.
OptimizedStructLayoutField Field
Target-specific ABI information and factory functions.
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
void computeInfo(FunctionInfo &FI) const override
Populate FI with the target's ABI-lowering decisions for each argument and return value.
Definition AMDGPU.cpp:232
const ABICompatInfo & getABICompatInfo() const override
Return this target's ABI compatibility flags.
Definition AMDGPU.cpp:48
AMDGPUTargetInfo(TypeBuilder &TypeBuilder, const ABICompatInfo &Compat, bool CoerceGenericPtrArgToGlobal)
Definition AMDGPU.cpp:43
unsigned getAllocaAddrSpace() const override
Indirect arguments live in the private (alloca) address space on AMDGPU.
Definition AMDGPU.cpp:51
Helper class to encapsulate information about how a specific type should be passed to or returned fro...
static ArgInfo getIgnore()
static ArgInfo getIndirectAliased(Align Align, unsigned AddrSpace, bool Realign=false)
An aliased indirect argument.
static ArgInfo getIndirect(Align Align, bool ByVal, unsigned AddrSpace=0, bool Realign=false)
Realign: the caller couldn't guarantee sufficient alignment - the callee must copy the argument to a ...
static ArgInfo getDirect(const Type *T=nullptr, unsigned Offset=0, MaybeAlign Align=std::nullopt, bool CanBeFlattened=true)
Self-consistent classification that conforms to no particular ABI.
ArgInfo classifyArgumentType(const Type *Ty) const
ArgInfo classifyReturnType(const Type *RetTy) const
ArrayRef< ArgEntry > arguments() const
unsigned getNumRequiredArgs() const
CallingConv::ID getCallingConvention() const
const Type * getReturnType() const
LLVM_ABI const Type * isSingleElementStruct(const Type *Ty) const
Returns the scalar a single-element struct reduces to, else null.
LLVM_ABI bool maybeCommonClassifyReturnType(FunctionInfo &FI) const
Apply rules for classifying return types that are common to all targets.
LLVM_ABI bool isAggregateTypeForABI(const Type *Ty) const
TypeBuilder & TB
Definition TargetInfo.h:67
LLVM_ABI const Type * useFirstFieldIfTransparentUnion(const Type *Ty) const
If Ty is a transparent union, return its first field type; otherwise return Ty unchanged.
LLVM_ABI RecordArgABI getRecordArgABI(const RecordType *RT) const
TypeBuilder manages the lifecycle of ABI types using bump pointer allocation.
Definition Types.h:474
Represents the ABI-specific view of a type in LLVM.
Definition Types.h:48
This file defines the type system for the LLVMABI library, which mirrors ABI-relevant aspects of fron...
@ CONSTANT_ADDRESS
Address space for constant memory (VTX2).
@ FLAT_ADDRESS
Address space for flat memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
@ PRIVATE_ADDRESS
Address space for private memory.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
LLVM_ABI std::unique_ptr< TargetInfo > createAMDGPUTargetInfo(TypeBuilder &TB, bool CoerceGenericPtrArgToGlobal=false)
Definition AMDGPU.cpp:255
@ RAA_DirectInMemory
Pass it on the stack using its defined layout.
Definition TargetInfo.h:34
@ RAA_Default
Pass it using the normal C aggregate rules for the ABI, potentially introducing extra copies and pass...
Definition TargetInfo.h:29
This is an optimization pass for GlobalISel generic memory operations.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
Flags controlling ABI compatibility behaviour that applies to every target.
Definition TargetInfo.h:43