| 1 | //===- AMDGPU.cpp - AMDGPU ABI Implementation ----------------------------===// |
| 2 | // |
| 3 | // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. |
| 4 | // See https://llvm.org/LICENSE.txt for license information. |
| 5 | // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception |
| 6 | // |
| 7 | //===----------------------------------------------------------------------===// |
| 8 | |
| 9 | #include "llvm/ABI/DefaultTargetInfo.h" |
| 10 | #include "llvm/ABI/FunctionInfo.h" |
| 11 | #include "llvm/ABI/TargetInfo.h" |
| 12 | #include "llvm/ABI/Types.h" |
| 13 | #include "llvm/Support/AMDGPUAddrSpace.h" |
| 14 | #include "llvm/Support/Alignment.h" |
| 15 | #include "llvm/Support/Casting.h" |
| 16 | #include "llvm/Support/TypeSize.h" |
| 17 | #include <algorithm> |
| 18 | #include <cassert> |
| 19 | #include <cstdint> |
| 20 | |
| 21 | namespace llvm { |
| 22 | namespace abi { |
| 23 | |
| 24 | class AMDGPUTargetInfo final : public DefaultTargetInfo { |
| 25 | private: |
| 26 | static const unsigned MaxNumRegsForArgsRet = 16; |
| 27 | |
| 28 | ABICompatInfo CompatInfo; |
| 29 | |
| 30 | /// HIP coerces a generic scalar-pointer kernel argument to the global |
| 31 | /// address space. Gated by the front end, which alone can see LangOpts.HIP. |
| 32 | bool CoerceGenericPtrArgToGlobal; |
| 33 | |
| 34 | ArgInfo classifyReturnType(const Type *RetTy) const; |
| 35 | ArgInfo classifyKernelArgumentType(const Type *Ty) const; |
| 36 | ArgInfo classifyArgumentType(const Type *Ty, bool Variadic, |
| 37 | unsigned &NumRegsLeft) const; |
| 38 | |
| 39 | /// Estimate number of registers the type will use when passed in registers. |
| 40 | uint64_t getNumRegsForType(const Type *Ty) const; |
| 41 | |
| 42 | public: |
| 43 | AMDGPUTargetInfo(TypeBuilder &TypeBuilder, const ABICompatInfo &Compat, |
| 44 | bool CoerceGenericPtrArgToGlobal) |
| 45 | : DefaultTargetInfo(TypeBuilder), CompatInfo(Compat), |
| 46 | CoerceGenericPtrArgToGlobal(CoerceGenericPtrArgToGlobal) {} |
| 47 | |
| 48 | const ABICompatInfo &getABICompatInfo() const override { return CompatInfo; } |
| 49 | |
| 50 | /// Indirect arguments live in the private (alloca) address space on AMDGPU. |
| 51 | unsigned getAllocaAddrSpace() const override { |
| 52 | return AMDGPUAS::PRIVATE_ADDRESS; |
| 53 | } |
| 54 | |
| 55 | void computeInfo(FunctionInfo &FI) const override; |
| 56 | }; |
| 57 | |
| 58 | uint64_t AMDGPUTargetInfo::getNumRegsForType(const Type *Ty) const { |
| 59 | uint64_t NumRegs = 0; |
| 60 | |
| 61 | if (const auto *VT = dyn_cast<VectorType>(Val: Ty)) { |
| 62 | // Compute from the number of elements. The reported size is based on the |
| 63 | // in-memory size, which includes the padding 4th element for 3-vectors. |
| 64 | const Type *EltTy = VT->getElementType(); |
| 65 | uint64_t EltSize = EltTy->getSizeInBits().getFixedValue(); |
| 66 | unsigned NumElts = VT->getNumElements().getFixedValue(); |
| 67 | |
| 68 | // 16-bit element vectors should be passed as packed. |
| 69 | if (EltSize == 16) |
| 70 | return (NumElts + 1) / 2; |
| 71 | |
| 72 | uint64_t EltNumRegs = (EltSize + 31) / 32; |
| 73 | return EltNumRegs * NumElts; |
| 74 | } |
| 75 | |
| 76 | if (const auto *RT = dyn_cast<RecordType>(Val: Ty)) { |
| 77 | for (const FieldInfo &Field : RT->getFields()) |
| 78 | NumRegs += getNumRegsForType(Ty: Field.FieldType); |
| 79 | return NumRegs; |
| 80 | } |
| 81 | |
| 82 | return (Ty->getSizeInBits().getFixedValue() + 31) / 32; |
| 83 | } |
| 84 | |
| 85 | ArgInfo AMDGPUTargetInfo::classifyReturnType(const Type *RetTy) const { |
| 86 | if (RetTy->isVoid()) |
| 87 | return ArgInfo::getIgnore(); |
| 88 | |
| 89 | if (isAggregateTypeForABI(Ty: RetTy)) { |
| 90 | // Records with non-trivial destructors/copy-constructors should not be |
| 91 | // returned by value. |
| 92 | if (getRecordArgABI(Ty: RetTy) == RAA_Default) { |
| 93 | const auto *RT = dyn_cast<RecordType>(Val: RetTy); |
| 94 | |
| 95 | // Ignore empty structs/unions. |
| 96 | if (RT && RT->isEmpty()) |
| 97 | return ArgInfo::getIgnore(); |
| 98 | |
| 99 | // Lower single-element structs to just return a regular value. |
| 100 | if (const Type *SeltTy = isSingleElementStruct(Ty: RetTy)) |
| 101 | return ArgInfo::getDirect(T: SeltTy); |
| 102 | |
| 103 | if (RT && RT->hasFlexibleArrayMember()) |
| 104 | return DefaultTargetInfo::classifyReturnType(RetTy); |
| 105 | |
| 106 | // Pack aggregates <= 4 bytes into single VGPR or pair. |
| 107 | uint64_t Size = RetTy->getSizeInBits().getFixedValue(); |
| 108 | if (Size <= 16) |
| 109 | return ArgInfo::getDirect(T: TB.getIntegerType(BitWidth: 16, Align: Align(2), Signed: false)); |
| 110 | |
| 111 | if (Size <= 32) |
| 112 | return ArgInfo::getDirect(T: TB.getIntegerType(BitWidth: 32, Align: Align(4), Signed: false)); |
| 113 | |
| 114 | if (Size <= 64) { |
| 115 | const Type *I32Ty = TB.getIntegerType(BitWidth: 32, Align: Align(4), Signed: false); |
| 116 | return ArgInfo::getDirect(T: TB.getArrayType(ElementType: I32Ty, NumElements: 2, /*SizeInBits=*/64)); |
| 117 | } |
| 118 | |
| 119 | if (getNumRegsForType(Ty: RetTy) <= MaxNumRegsForArgsRet) |
| 120 | return ArgInfo::getDirect(); |
| 121 | } |
| 122 | } |
| 123 | |
| 124 | // Otherwise just do the default thing. |
| 125 | return DefaultTargetInfo::classifyReturnType(RetTy); |
| 126 | } |
| 127 | |
| 128 | /// For kernels all parameters are really passed in a special buffer. It doesn't |
| 129 | /// make sense to pass anything byval, so everything must be direct. |
| 130 | ArgInfo AMDGPUTargetInfo::classifyKernelArgumentType(const Type *Ty) const { |
| 131 | Ty = useFirstFieldIfTransparentUnion(Ty); |
| 132 | |
| 133 | if (const Type *SeltTy = isSingleElementStruct(Ty)) |
| 134 | Ty = SeltTy; |
| 135 | |
| 136 | // HIP passes a generic scalar pointer as a global pointer; a pointer is not |
| 137 | // an aggregate, so this stays on the direct path. |
| 138 | if (CoerceGenericPtrArgToGlobal) { |
| 139 | if (const auto *PtrTy = dyn_cast<PointerType>(Val: Ty); |
| 140 | PtrTy && PtrTy->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS) { |
| 141 | const Type *Coerced = |
| 142 | TB.getPointerType(Size: PtrTy->getSizeInBits().getFixedValue(), |
| 143 | Align: PtrTy->getAlignment(), Addrspace: AMDGPUAS::GLOBAL_ADDRESS); |
| 144 | return ArgInfo::getDirect(T: Coerced, /*Offset=*/0, /*Align=*/std::nullopt, |
| 145 | /*CanBeFlattened=*/false); |
| 146 | } |
| 147 | } |
| 148 | |
| 149 | // FIXME: This doesn't apply the optimization of coercing pointers in structs |
| 150 | // to global address space when using byref. This would require implementing a |
| 151 | // new kind of coercion of the in-memory type when for indirect arguments. |
| 152 | if (isAggregateTypeForABI(Ty)) |
| 153 | return ArgInfo::getIndirectAliased( |
| 154 | Align: Ty->getAlignment(), |
| 155 | /*AddrSpace=*/AMDGPUAS::CONSTANT_ADDRESS); |
| 156 | |
| 157 | // CanBeFlattened=false keeps the struct intact. |
| 158 | return ArgInfo::getDirect(T: Ty, /*Offset=*/0, /*Align=*/std::nullopt, |
| 159 | /*CanBeFlattened=*/false); |
| 160 | } |
| 161 | |
| 162 | ArgInfo AMDGPUTargetInfo::classifyArgumentType(const Type *Ty, bool Variadic, |
| 163 | unsigned &NumRegsLeft) const { |
| 164 | assert(NumRegsLeft <= MaxNumRegsForArgsRet && "register estimate underflow" ); |
| 165 | |
| 166 | Ty = useFirstFieldIfTransparentUnion(Ty); |
| 167 | |
| 168 | // Variadic aggregates are kept intact rather than flattened into fields. |
| 169 | if (Variadic) |
| 170 | return ArgInfo::getDirect(/*T=*/nullptr, /*Offset=*/0, |
| 171 | /*Align=*/std::nullopt, /*CanBeFlattened=*/false); |
| 172 | |
| 173 | if (isAggregateTypeForABI(Ty)) { |
| 174 | // Records with non-trivial destructors/copy-constructors should not be |
| 175 | // passed by value. |
| 176 | if (RecordArgABI RAA = getRecordArgABI(Ty); RAA != RAA_Default) |
| 177 | return ArgInfo::getIndirect(Align: Ty->getAlignment(), |
| 178 | /*ByVal=*/RAA == RAA_DirectInMemory, |
| 179 | /*AddrSpace=*/AMDGPUAS::PRIVATE_ADDRESS); |
| 180 | |
| 181 | // Ignore empty structs/unions. |
| 182 | if (Ty->isEmptyRecord()) |
| 183 | return ArgInfo::getIgnore(); |
| 184 | |
| 185 | // Lower single-element structs to just pass a regular value. |
| 186 | if (const Type *SeltTy = isSingleElementStruct(Ty)) |
| 187 | return ArgInfo::getDirect(T: SeltTy); |
| 188 | |
| 189 | if (const auto *RT = dyn_cast<RecordType>(Val: Ty); |
| 190 | RT && RT->hasFlexibleArrayMember()) |
| 191 | return DefaultTargetInfo::classifyArgumentType(Ty); |
| 192 | |
| 193 | // Pack aggregates <= 8 bytes into single VGPR or pair. |
| 194 | uint64_t Size = Ty->getSizeInBits().getFixedValue(); |
| 195 | if (Size <= 64) { |
| 196 | unsigned NumRegs = (Size + 31) / 32; |
| 197 | NumRegsLeft -= std::min(a: NumRegsLeft, b: NumRegs); |
| 198 | |
| 199 | if (Size <= 16) |
| 200 | return ArgInfo::getDirect(T: TB.getIntegerType(BitWidth: 16, Align: Align(2), Signed: false)); |
| 201 | |
| 202 | if (Size <= 32) |
| 203 | return ArgInfo::getDirect(T: TB.getIntegerType(BitWidth: 32, Align: Align(4), Signed: false)); |
| 204 | |
| 205 | const Type *I32Ty = TB.getIntegerType(BitWidth: 32, Align: Align(4), Signed: false); |
| 206 | return ArgInfo::getDirect(T: TB.getArrayType(ElementType: I32Ty, NumElements: 2, /*SizeInBits=*/64)); |
| 207 | } |
| 208 | |
| 209 | if (NumRegsLeft > 0) { |
| 210 | uint64_t NumRegs = getNumRegsForType(Ty); |
| 211 | if (NumRegsLeft >= NumRegs) { |
| 212 | NumRegsLeft -= NumRegs; |
| 213 | return ArgInfo::getDirect(); |
| 214 | } |
| 215 | } |
| 216 | |
| 217 | // Pass a struct argument by reference rather than by value. |
| 218 | return ArgInfo::getIndirectAliased(Align: Ty->getAlignment(), |
| 219 | /*AddrSpace=*/AMDGPUAS::PRIVATE_ADDRESS); |
| 220 | } |
| 221 | |
| 222 | // Otherwise just do the default thing. |
| 223 | ArgInfo AI = DefaultTargetInfo::classifyArgumentType(Ty); |
| 224 | if (!AI.isIndirect()) { |
| 225 | uint64_t NumRegs = getNumRegsForType(Ty); |
| 226 | NumRegsLeft -= std::min(a: NumRegs, b: uint64_t{NumRegsLeft}); |
| 227 | } |
| 228 | |
| 229 | return AI; |
| 230 | } |
| 231 | |
| 232 | void AMDGPUTargetInfo::computeInfo(FunctionInfo &FI) const { |
| 233 | CallingConv::ID CC = FI.getCallingConvention(); |
| 234 | |
| 235 | // Non-trivial C++ records are returned indirectly |
| 236 | // in the flat address space. |
| 237 | if (!maybeCommonClassifyReturnType(FI)) |
| 238 | FI.getReturnInfo() = classifyReturnType(RetTy: FI.getReturnType()); |
| 239 | |
| 240 | unsigned ArgumentIndex = 0; |
| 241 | const unsigned NumFixedArguments = FI.getNumRequiredArgs(); |
| 242 | |
| 243 | unsigned NumRegsLeft = MaxNumRegsForArgsRet; |
| 244 | for (ArgEntry &Arg : FI.arguments()) { |
| 245 | if (CC == CallingConv::AMDGPU_KERNEL) { |
| 246 | Arg.Info = classifyKernelArgumentType(Ty: Arg.ABIType); |
| 247 | } else { |
| 248 | bool FixedArgument = ArgumentIndex++ < NumFixedArguments; |
| 249 | Arg.Info = classifyArgumentType(Ty: Arg.ABIType, Variadic: !FixedArgument, NumRegsLeft); |
| 250 | } |
| 251 | } |
| 252 | } |
| 253 | |
| 254 | std::unique_ptr<TargetInfo> |
| 255 | createAMDGPUTargetInfo(TypeBuilder &TB, bool CoerceGenericPtrArgToGlobal) { |
| 256 | return std::make_unique<AMDGPUTargetInfo>(args&: TB, args: ABICompatInfo(), |
| 257 | args&: CoerceGenericPtrArgToGlobal); |
| 258 | } |
| 259 | |
| 260 | } // namespace abi |
| 261 | } // namespace llvm |
| 262 | |