1//===- AMDGPU.cpp - AMDGPU ABI Implementation ----------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
9#include "llvm/ABI/DefaultTargetInfo.h"
10#include "llvm/ABI/FunctionInfo.h"
11#include "llvm/ABI/TargetInfo.h"
12#include "llvm/ABI/Types.h"
13#include "llvm/Support/AMDGPUAddrSpace.h"
14#include "llvm/Support/Alignment.h"
15#include "llvm/Support/Casting.h"
16#include "llvm/Support/TypeSize.h"
17#include <algorithm>
18#include <cassert>
19#include <cstdint>
20
21namespace llvm {
22namespace abi {
23
24class AMDGPUTargetInfo final : public DefaultTargetInfo {
25private:
26 static const unsigned MaxNumRegsForArgsRet = 16;
27
28 ABICompatInfo CompatInfo;
29
30 /// HIP coerces a generic scalar-pointer kernel argument to the global
31 /// address space. Gated by the front end, which alone can see LangOpts.HIP.
32 bool CoerceGenericPtrArgToGlobal;
33
34 ArgInfo classifyReturnType(const Type *RetTy) const;
35 ArgInfo classifyKernelArgumentType(const Type *Ty) const;
36 ArgInfo classifyArgumentType(const Type *Ty, bool Variadic,
37 unsigned &NumRegsLeft) const;
38
39 /// Estimate number of registers the type will use when passed in registers.
40 uint64_t getNumRegsForType(const Type *Ty) const;
41
42public:
43 AMDGPUTargetInfo(TypeBuilder &TypeBuilder, const ABICompatInfo &Compat,
44 bool CoerceGenericPtrArgToGlobal)
45 : DefaultTargetInfo(TypeBuilder), CompatInfo(Compat),
46 CoerceGenericPtrArgToGlobal(CoerceGenericPtrArgToGlobal) {}
47
48 const ABICompatInfo &getABICompatInfo() const override { return CompatInfo; }
49
50 /// Indirect arguments live in the private (alloca) address space on AMDGPU.
51 unsigned getAllocaAddrSpace() const override {
52 return AMDGPUAS::PRIVATE_ADDRESS;
53 }
54
55 void computeInfo(FunctionInfo &FI) const override;
56};
57
58uint64_t AMDGPUTargetInfo::getNumRegsForType(const Type *Ty) const {
59 uint64_t NumRegs = 0;
60
61 if (const auto *VT = dyn_cast<VectorType>(Val: Ty)) {
62 // Compute from the number of elements. The reported size is based on the
63 // in-memory size, which includes the padding 4th element for 3-vectors.
64 const Type *EltTy = VT->getElementType();
65 uint64_t EltSize = EltTy->getSizeInBits().getFixedValue();
66 unsigned NumElts = VT->getNumElements().getFixedValue();
67
68 // 16-bit element vectors should be passed as packed.
69 if (EltSize == 16)
70 return (NumElts + 1) / 2;
71
72 uint64_t EltNumRegs = (EltSize + 31) / 32;
73 return EltNumRegs * NumElts;
74 }
75
76 if (const auto *RT = dyn_cast<RecordType>(Val: Ty)) {
77 for (const FieldInfo &Field : RT->getFields())
78 NumRegs += getNumRegsForType(Ty: Field.FieldType);
79 return NumRegs;
80 }
81
82 return (Ty->getSizeInBits().getFixedValue() + 31) / 32;
83}
84
85ArgInfo AMDGPUTargetInfo::classifyReturnType(const Type *RetTy) const {
86 if (RetTy->isVoid())
87 return ArgInfo::getIgnore();
88
89 if (isAggregateTypeForABI(Ty: RetTy)) {
90 // Records with non-trivial destructors/copy-constructors should not be
91 // returned by value.
92 if (getRecordArgABI(Ty: RetTy) == RAA_Default) {
93 const auto *RT = dyn_cast<RecordType>(Val: RetTy);
94
95 // Ignore empty structs/unions.
96 if (RT && RT->isEmpty())
97 return ArgInfo::getIgnore();
98
99 // Lower single-element structs to just return a regular value.
100 if (const Type *SeltTy = isSingleElementStruct(Ty: RetTy))
101 return ArgInfo::getDirect(T: SeltTy);
102
103 if (RT && RT->hasFlexibleArrayMember())
104 return DefaultTargetInfo::classifyReturnType(RetTy);
105
106 // Pack aggregates <= 4 bytes into single VGPR or pair.
107 uint64_t Size = RetTy->getSizeInBits().getFixedValue();
108 if (Size <= 16)
109 return ArgInfo::getDirect(T: TB.getIntegerType(BitWidth: 16, Align: Align(2), Signed: false));
110
111 if (Size <= 32)
112 return ArgInfo::getDirect(T: TB.getIntegerType(BitWidth: 32, Align: Align(4), Signed: false));
113
114 if (Size <= 64) {
115 const Type *I32Ty = TB.getIntegerType(BitWidth: 32, Align: Align(4), Signed: false);
116 return ArgInfo::getDirect(T: TB.getArrayType(ElementType: I32Ty, NumElements: 2, /*SizeInBits=*/64));
117 }
118
119 if (getNumRegsForType(Ty: RetTy) <= MaxNumRegsForArgsRet)
120 return ArgInfo::getDirect();
121 }
122 }
123
124 // Otherwise just do the default thing.
125 return DefaultTargetInfo::classifyReturnType(RetTy);
126}
127
128/// For kernels all parameters are really passed in a special buffer. It doesn't
129/// make sense to pass anything byval, so everything must be direct.
130ArgInfo AMDGPUTargetInfo::classifyKernelArgumentType(const Type *Ty) const {
131 Ty = useFirstFieldIfTransparentUnion(Ty);
132
133 if (const Type *SeltTy = isSingleElementStruct(Ty))
134 Ty = SeltTy;
135
136 // HIP passes a generic scalar pointer as a global pointer; a pointer is not
137 // an aggregate, so this stays on the direct path.
138 if (CoerceGenericPtrArgToGlobal) {
139 if (const auto *PtrTy = dyn_cast<PointerType>(Val: Ty);
140 PtrTy && PtrTy->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS) {
141 const Type *Coerced =
142 TB.getPointerType(Size: PtrTy->getSizeInBits().getFixedValue(),
143 Align: PtrTy->getAlignment(), Addrspace: AMDGPUAS::GLOBAL_ADDRESS);
144 return ArgInfo::getDirect(T: Coerced, /*Offset=*/0, /*Align=*/std::nullopt,
145 /*CanBeFlattened=*/false);
146 }
147 }
148
149 // FIXME: This doesn't apply the optimization of coercing pointers in structs
150 // to global address space when using byref. This would require implementing a
151 // new kind of coercion of the in-memory type when for indirect arguments.
152 if (isAggregateTypeForABI(Ty))
153 return ArgInfo::getIndirectAliased(
154 Align: Ty->getAlignment(),
155 /*AddrSpace=*/AMDGPUAS::CONSTANT_ADDRESS);
156
157 // CanBeFlattened=false keeps the struct intact.
158 return ArgInfo::getDirect(T: Ty, /*Offset=*/0, /*Align=*/std::nullopt,
159 /*CanBeFlattened=*/false);
160}
161
162ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic,
163 unsigned &NumRegsLeft) const {
164 assert(NumRegsLeft <= MaxNumRegsForArgsRet && "register estimate underflow");
165
166 Ty = useFirstFieldIfTransparentUnion(Ty);
167
168 // Variadic aggregates are kept intact rather than flattened into fields.
169 if (Variadic)
170 return ArgInfo::getDirect(/*T=*/nullptr, /*Offset=*/0,
171 /*Align=*/std::nullopt, /*CanBeFlattened=*/false);
172
173 if (isAggregateTypeForABI(Ty)) {
174 // Records with non-trivial destructors/copy-constructors should not be
175 // passed by value.
176 if (RecordArgABI RAA = getRecordArgABI(Ty); RAA != RAA_Default)
177 return ArgInfo::getIndirect(Align: Ty->getAlignment(),
178 /*ByVal=*/RAA == RAA_DirectInMemory,
179 /*AddrSpace=*/AMDGPUAS::PRIVATE_ADDRESS);
180
181 // Ignore empty structs/unions.
182 if (Ty->isEmptyRecord())
183 return ArgInfo::getIgnore();
184
185 // Lower single-element structs to just pass a regular value.
186 if (const Type *SeltTy = isSingleElementStruct(Ty))
187 return ArgInfo::getDirect(T: SeltTy);
188
189 if (const auto *RT = dyn_cast<RecordType>(Val: Ty);
190 RT && RT->hasFlexibleArrayMember())
191 return DefaultTargetInfo::classifyArgumentType(Ty);
192
193 // Pack aggregates <= 8 bytes into single VGPR or pair.
194 uint64_t Size = Ty->getSizeInBits().getFixedValue();
195 if (Size <= 64) {
196 unsigned NumRegs = (Size + 31) / 32;
197 NumRegsLeft -= std::min(a: NumRegsLeft, b: NumRegs);
198
199 if (Size <= 16)
200 return ArgInfo::getDirect(T: TB.getIntegerType(BitWidth: 16, Align: Align(2), Signed: false));
201
202 if (Size <= 32)
203 return ArgInfo::getDirect(T: TB.getIntegerType(BitWidth: 32, Align: Align(4), Signed: false));
204
205 const Type *I32Ty = TB.getIntegerType(BitWidth: 32, Align: Align(4), Signed: false);
206 return ArgInfo::getDirect(T: TB.getArrayType(ElementType: I32Ty, NumElements: 2, /*SizeInBits=*/64));
207 }
208
209 if (NumRegsLeft > 0) {
210 uint64_t NumRegs = getNumRegsForType(Ty);
211 if (NumRegsLeft >= NumRegs) {
212 NumRegsLeft -= NumRegs;
213 return ArgInfo::getDirect();
214 }
215 }
216
217 // Pass a struct argument by reference rather than by value.
218 return ArgInfo::getIndirectAliased(Align: Ty->getAlignment(),
219 /*AddrSpace=*/AMDGPUAS::PRIVATE_ADDRESS);
220 }
221
222 // Otherwise just do the default thing.
223 ArgInfo AI = DefaultTargetInfo::classifyArgumentType(Ty);
224 if (!AI.isIndirect()) {
225 uint64_t NumRegs = getNumRegsForType(Ty);
226 NumRegsLeft -= std::min(a: NumRegs, b: uint64_t{NumRegsLeft});
227 }
228
229 return AI;
230}
231
232void AMDGPUTargetInfo::computeInfo(FunctionInfo &FI) const {
233 CallingConv::ID CC = FI.getCallingConvention();
234
235 // Non-trivial C++ records are returned indirectly
236 // in the flat address space.
237 if (!maybeCommonClassifyReturnType(FI))
238 FI.getReturnInfo() = classifyReturnType(RetTy: FI.getReturnType());
239
240 unsigned ArgumentIndex = 0;
241 const unsigned NumFixedArguments = FI.getNumRequiredArgs();
242
243 unsigned NumRegsLeft = MaxNumRegsForArgsRet;
244 for (ArgEntry &Arg : FI.arguments()) {
245 if (CC == CallingConv::AMDGPU_KERNEL) {
246 Arg.Info = classifyKernelArgumentType(Ty: Arg.ABIType);
247 } else {
248 bool FixedArgument = ArgumentIndex++ < NumFixedArguments;
249 Arg.Info = classifyArgumentType(Ty: Arg.ABIType, Variadic: !FixedArgument, NumRegsLeft);
250 }
251 }
252}
253
254std::unique_ptr<TargetInfo>
255createAMDGPUTargetInfo(TypeBuilder &TB, bool CoerceGenericPtrArgToGlobal) {
256 return std::make_unique<AMDGPUTargetInfo>(args&: TB, args: ABICompatInfo(),
257 args&: CoerceGenericPtrArgToGlobal);
258}
259
260} // namespace abi
261} // namespace llvm
262