1//===- SPIR.cpp -----------------------------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
9#include "ABIInfoImpl.h"
10#include "HLSLBufferLayoutBuilder.h"
11#include "TargetInfo.h"
12#include "clang/AST/DeclCXX.h"
13#include "clang/Basic/LangOptions.h"
14#include "clang/CodeGenUtils/TargetUtils.h"
15#include "llvm/IR/DerivedTypes.h"
16#include "llvm/IR/LLVMContext.h"
17
18#include <stdint.h>
19#include <utility>
20
21using namespace clang;
22using namespace clang::CodeGen;
23
24//===----------------------------------------------------------------------===//
25// Base ABI and target codegen info implementation common between SPIR and
26// SPIR-V.
27//===----------------------------------------------------------------------===//
28
29namespace {
30class CommonSPIRABIInfo : public DefaultABIInfo {
31public:
32 CommonSPIRABIInfo(CodeGenTypes &CGT) : DefaultABIInfo(CGT) { setCCs(); }
33
34private:
35 void setCCs();
36};
37
38class SPIRVABIInfo : public CommonSPIRABIInfo {
39public:
40 SPIRVABIInfo(CodeGenTypes &CGT) : CommonSPIRABIInfo(CGT) {}
41 void computeInfo(CGFunctionInfo &FI) const override;
42 RValue EmitVAArg(CodeGenFunction &CGF, Address VAListAddr, QualType Ty,
43 AggValueSlot Slot) const override;
44
45 llvm::FixedVectorType *
46 getOptimalVectorMemoryType(llvm::FixedVectorType *Ty,
47 const LangOptions &LangOpt) const override;
48
49private:
50 ABIArgInfo classifyKernelArgumentType(QualType Ty) const;
51};
52
53class AMDGCNSPIRVABIInfo : public SPIRVABIInfo {
54 // TODO: this should be unified / shared with AMDGPU, ideally we'd like to
55 // re-use AMDGPUABIInfo eventually, rather than duplicate.
56 static constexpr unsigned MaxNumRegsForArgsRet = 16; // 16 32-bit registers
57 mutable unsigned NumRegsLeft = 0;
58
59 uint64_t numRegsForType(QualType Ty) const;
60
61 bool isHomogeneousAggregateBaseType(QualType Ty) const override {
62 return true;
63 }
64 bool isHomogeneousAggregateSmallEnough(const Type *Base,
65 uint64_t Members) const override {
66 uint32_t NumRegs = (getContext().getTypeSize(T: Base) + 31) / 32;
67
68 // Homogeneous Aggregates may occupy at most 16 registers.
69 return Members * NumRegs <= MaxNumRegsForArgsRet;
70 }
71
72 // Coerce HIP scalar pointer arguments from generic pointers to global ones.
73 llvm::Type *coerceKernelArgumentType(llvm::Type *Ty, unsigned FromAS,
74 unsigned ToAS) const;
75
76 ABIArgInfo classifyReturnType(QualType RetTy) const;
77 ABIArgInfo classifyKernelArgumentType(QualType Ty) const;
78 ABIArgInfo classifyArgumentType(QualType Ty, bool Variadic) const;
79
80public:
81 AMDGCNSPIRVABIInfo(CodeGenTypes &CGT) : SPIRVABIInfo(CGT) {}
82 void computeInfo(CGFunctionInfo &FI) const override;
83
84 llvm::FixedVectorType *
85 getOptimalVectorMemoryType(llvm::FixedVectorType *Ty,
86 const LangOptions &LangOpt) const override;
87};
88} // end anonymous namespace
89namespace {
90class CommonSPIRTargetCodeGenInfo : public TargetCodeGenInfo {
91public:
92 CommonSPIRTargetCodeGenInfo(CodeGen::CodeGenTypes &CGT)
93 : TargetCodeGenInfo(std::make_unique<CommonSPIRABIInfo>(args&: CGT)) {}
94 CommonSPIRTargetCodeGenInfo(std::unique_ptr<ABIInfo> ABIInfo)
95 : TargetCodeGenInfo(std::move(ABIInfo)) {}
96
97 unsigned getDeviceKernelCallingConv() const override;
98 llvm::Type *getOpenCLType(CodeGenModule &CGM, const Type *T) const override;
99 llvm::Type *getHLSLType(CodeGenModule &CGM, const Type *Ty,
100 const CGHLSLOffsetInfo &OffsetInfo) const override;
101
102 llvm::Type *getHLSLPadding(CodeGenModule &CGM,
103 CharUnits NumBytes) const override {
104 unsigned Size = NumBytes.getQuantity();
105 return llvm::TargetExtType::get(Context&: CGM.getLLVMContext(), Name: "spirv.Padding", Types: {},
106 Ints: {Size});
107 }
108
109 bool isHLSLPadding(llvm::Type *Ty) const override {
110 if (auto *TET = dyn_cast<llvm::TargetExtType>(Val: Ty))
111 return TET->getName() == "spirv.Padding";
112 return false;
113 }
114
115 llvm::Type *getSPIRVImageTypeFromHLSLResource(
116 const HLSLAttributedResourceType::Attributes &attributes,
117 QualType SampledType, CodeGenModule &CGM) const;
118 void
119 setOCLKernelStubCallingConvention(const FunctionType *&FT) const override;
120 llvm::Constant *getNullPointer(const CodeGen::CodeGenModule &CGM,
121 llvm::PointerType *T,
122 QualType QT) const override;
123};
124class SPIRVTargetCodeGenInfo : public CommonSPIRTargetCodeGenInfo {
125public:
126 SPIRVTargetCodeGenInfo(CodeGen::CodeGenTypes &CGT)
127 : CommonSPIRTargetCodeGenInfo(
128 (CGT.getTarget().getTriple().getVendor() == llvm::Triple::AMD)
129 ? std::make_unique<AMDGCNSPIRVABIInfo>(args&: CGT)
130 : std::make_unique<SPIRVABIInfo>(args&: CGT)) {}
131 void setCUDAKernelCallingConvention(const FunctionType *&FT) const override;
132 LangAS getGlobalVarAddressSpace(CodeGenModule &CGM,
133 const VarDecl *D) const override;
134 void setTargetAttributes(const Decl *D, llvm::GlobalValue *GV,
135 CodeGen::CodeGenModule &M) const override;
136 StringRef getLLVMSyncScopeStr(const LangOptions &LangOpts, SyncScope Scope,
137 llvm::AtomicOrdering Ordering) const override;
138 void setTargetAtomicMetadata(CodeGenFunction &CGF,
139 llvm::Instruction &AtomicInst,
140 const AtomicExpr *Expr = nullptr) const override;
141 bool supportsLibCall() const override {
142 return getABIInfo().getTarget().getTriple().getVendor() !=
143 llvm::Triple::AMD;
144 }
145
146 LangAS getSRetAddrSpace(const CXXRecordDecl *RD) const override;
147};
148} // End anonymous namespace.
149
150void CommonSPIRABIInfo::setCCs() {
151 assert(getRuntimeCC() == llvm::CallingConv::C);
152 RuntimeCC = llvm::CallingConv::SPIR_FUNC;
153}
154
155ABIArgInfo SPIRVABIInfo::classifyKernelArgumentType(QualType Ty) const {
156 // Coerce pointer arguments with default address space to CrossWorkGroup
157 // pointers as default address space kernel
158 // arguments are not allowed. We use the opencl_global language address
159 // space which always maps to CrossWorkGroup.
160 llvm::Type *LTy = CGT.ConvertType(T: Ty);
161 auto DefaultAS = getContext().getTargetAddressSpace(AS: LangAS::Default);
162 auto GlobalAS = getContext().getTargetAddressSpace(AS: LangAS::opencl_global);
163 auto *PtrTy = llvm::dyn_cast<llvm::PointerType>(Val: LTy);
164 if (PtrTy && PtrTy->getAddressSpace() == DefaultAS) {
165 LTy = llvm::PointerType::get(C&: PtrTy->getContext(), AddressSpace: GlobalAS);
166 return ABIArgInfo::getDirect(T: LTy, Offset: 0, Padding: nullptr, CanBeFlattened: false);
167 }
168
169 if (getContext().getLangOpts().isTargetDevice() &&
170 isAggregateTypeForABI(T: Ty)) {
171 // Force copying aggregate type in kernel arguments by value when
172 // compiling CUDA targeting SPIR-V. This is required for the object
173 // copied to be valid on the device.
174 // This behavior follows the CUDA spec
175 // https://docs.nvidia.com/cuda/cuda-c-programming-guide/index.html#global-function-argument-processing,
176 // and matches the NVPTX implementation. TODO: hardcoding to 0 should be
177 // revisited if HIPSPV / byval starts making use of the AS of an indirect
178 // arg.
179 return getNaturalAlignIndirect(Ty, /*AddrSpace=*/0, /*byval=*/ByVal: true);
180 }
181 return classifyArgumentType(RetTy: Ty);
182}
183
184void SPIRVABIInfo::computeInfo(CGFunctionInfo &FI) const {
185 // The logic is same as in DefaultABIInfo with an exception on the kernel
186 // arguments handling.
187 llvm::CallingConv::ID CC = FI.getCallingConvention();
188
189 for (auto &&[ArgumentsCount, I] : llvm::enumerate(First: FI.arguments()))
190 I.info = ArgumentsCount < FI.getNumRequiredArgs()
191 ? classifyArgumentType(RetTy: I.type)
192 : ABIArgInfo::getDirect();
193
194 if (!getCXXABI().classifyReturnType(FI))
195 FI.getReturnInfo() = classifyReturnType(RetTy: FI.getReturnType());
196
197 for (auto &I : FI.arguments()) {
198 if (CC == llvm::CallingConv::SPIR_KERNEL) {
199 I.info = classifyKernelArgumentType(Ty: I.type);
200 } else {
201 I.info = classifyArgumentType(RetTy: I.type);
202 }
203 }
204}
205
206RValue SPIRVABIInfo::EmitVAArg(CodeGenFunction &CGF, Address VAListAddr,
207 QualType Ty, AggValueSlot Slot) const {
208 return emitVoidPtrVAArg(CGF, VAListAddr, ValueTy: Ty, /*IsIndirect=*/false,
209 ValueInfo: getContext().getTypeInfoInChars(T: Ty),
210 SlotSizeAndAlign: CharUnits::fromQuantity(Quantity: 1),
211 /*AllowHigherAlign=*/true, Slot);
212}
213
214uint64_t AMDGCNSPIRVABIInfo::numRegsForType(QualType Ty) const {
215 // This duplicates the AMDGPUABI computation.
216 uint64_t NumRegs = 0;
217
218 if (const VectorType *VT = Ty->getAs<VectorType>()) {
219 // Compute from the number of elements. The reported size is based on the
220 // in-memory size, which includes the padding 4th element for 3-vectors.
221 QualType EltTy = VT->getElementType();
222 uint64_t EltSize = getContext().getTypeSize(T: EltTy);
223
224 // 16-bit element vectors should be passed as packed.
225 if (EltSize == 16)
226 return (VT->getNumElements() + 1) / 2;
227
228 uint64_t EltNumRegs = (EltSize + 31) / 32;
229 return EltNumRegs * VT->getNumElements();
230 }
231
232 if (const auto *RD = Ty->getAsRecordDecl()) {
233 assert(!RD->hasFlexibleArrayMember());
234
235 for (const FieldDecl *Field : RD->fields()) {
236 QualType FieldTy = Field->getType();
237 NumRegs += numRegsForType(Ty: FieldTy);
238 }
239
240 return NumRegs;
241 }
242
243 return (getContext().getTypeSize(T: Ty) + 31) / 32;
244}
245
246llvm::Type *AMDGCNSPIRVABIInfo::coerceKernelArgumentType(llvm::Type *Ty,
247 unsigned FromAS,
248 unsigned ToAS) const {
249 // Single value types.
250 auto *PtrTy = llvm::dyn_cast<llvm::PointerType>(Val: Ty);
251 if (PtrTy && PtrTy->getAddressSpace() == FromAS)
252 return llvm::PointerType::get(C&: Ty->getContext(), AddressSpace: ToAS);
253 return Ty;
254}
255
256ABIArgInfo AMDGCNSPIRVABIInfo::classifyReturnType(QualType RetTy) const {
257 if (!isAggregateTypeForABI(T: RetTy) || getRecordArgABI(T: RetTy, CXXABI&: getCXXABI()))
258 return DefaultABIInfo::classifyReturnType(RetTy);
259
260 // Ignore empty structs/unions.
261 if (isEmptyRecord(Context&: getContext(), T: RetTy, AllowArrays: true))
262 return ABIArgInfo::getIgnore();
263
264 // Lower single-element structs to just return a regular value.
265 if (const Type *SeltTy = isSingleElementStruct(T: RetTy, Context&: getContext()))
266 return ABIArgInfo::getDirect(T: CGT.ConvertType(T: QualType(SeltTy, 0)));
267
268 if (const auto *RD = RetTy->getAsRecordDecl();
269 RD && RD->hasFlexibleArrayMember())
270 return DefaultABIInfo::classifyReturnType(RetTy);
271
272 // Pack aggregates <= 4 bytes into single VGPR or pair.
273 uint64_t Size = getContext().getTypeSize(T: RetTy);
274 if (Size <= 16)
275 return ABIArgInfo::getDirect(T: llvm::Type::getInt16Ty(C&: getVMContext()));
276
277 if (Size <= 32)
278 return ABIArgInfo::getDirect(T: llvm::Type::getInt32Ty(C&: getVMContext()));
279
280 // TODO: This carried over from AMDGPU oddity, we retain it to
281 // ensure consistency, but it might be reasonable to return Int64.
282 if (Size <= 64) {
283 llvm::Type *I32Ty = llvm::Type::getInt32Ty(C&: getVMContext());
284 return ABIArgInfo::getDirect(T: llvm::ArrayType::get(ElementType: I32Ty, NumElements: 2));
285 }
286
287 if (numRegsForType(Ty: RetTy) <= MaxNumRegsForArgsRet)
288 return ABIArgInfo::getDirect();
289 return DefaultABIInfo::classifyReturnType(RetTy);
290}
291
292/// For kernels all parameters are really passed in a special buffer. It doesn't
293/// make sense to pass anything byval, so everything must be direct.
294ABIArgInfo AMDGCNSPIRVABIInfo::classifyKernelArgumentType(QualType Ty) const {
295 Ty = useFirstFieldIfTransparentUnion(Ty);
296
297 // TODO: Can we omit empty structs?
298
299 if (const Type *SeltTy = isSingleElementStruct(T: Ty, Context&: getContext()))
300 Ty = QualType(SeltTy, 0);
301
302 llvm::Type *OrigLTy = CGT.ConvertType(T: Ty);
303 llvm::Type *LTy = OrigLTy;
304 if (getContext().getLangOpts().isTargetDevice()) {
305 LTy = coerceKernelArgumentType(
306 Ty: OrigLTy, /*FromAS=*/getContext().getTargetAddressSpace(AS: LangAS::Default),
307 /*ToAS=*/getContext().getTargetAddressSpace(AS: LangAS::opencl_global));
308 }
309
310 // FIXME: This doesn't apply the optimization of coercing pointers in structs
311 // to global address space when using byref. This would require implementing a
312 // new kind of coercion of the in-memory type when for indirect arguments.
313 if (LTy == OrigLTy && isAggregateTypeForABI(T: Ty)) {
314 return ABIArgInfo::getIndirectAliased(
315 Alignment: getContext().getTypeAlignInChars(T: Ty),
316 AddrSpace: getContext().getTargetAddressSpace(AS: LangAS::opencl_constant),
317 Realign: false /*Realign*/, Padding: nullptr /*Padding*/);
318 }
319
320 // TODO: inhibiting flattening is an AMDGPU workaround for Clover, which might
321 // be vestigial and should be revisited.
322 return ABIArgInfo::getDirect(T: LTy, Offset: 0, Padding: nullptr, CanBeFlattened: false);
323}
324
325ABIArgInfo AMDGCNSPIRVABIInfo::classifyArgumentType(QualType Ty,
326 bool Variadic) const {
327 assert(NumRegsLeft <= MaxNumRegsForArgsRet && "register estimate underflow");
328
329 Ty = useFirstFieldIfTransparentUnion(Ty);
330
331 if (Variadic) {
332 return ABIArgInfo::getDirect(/*T=*/nullptr,
333 /*Offset=*/0,
334 /*Padding=*/nullptr,
335 /*CanBeFlattened=*/false,
336 /*Align=*/0);
337 }
338
339 if (!isAggregateTypeForABI(T: Ty)) {
340 ABIArgInfo ArgInfo = DefaultABIInfo::classifyArgumentType(RetTy: Ty);
341 if (!ArgInfo.isIndirect()) {
342 uint64_t NumRegs = numRegsForType(Ty);
343 NumRegsLeft -= std::min(a: NumRegs, b: uint64_t{NumRegsLeft});
344 }
345
346 return ArgInfo;
347 }
348
349 // Records with non-trivial destructors/copy-constructors should not be
350 // passed by value.
351 if (auto RAA = getRecordArgABI(T: Ty, CXXABI&: getCXXABI()))
352 return getNaturalAlignIndirect(Ty, AddrSpace: getDataLayout().getAllocaAddrSpace(),
353 ByVal: RAA == CGCXXABI::RAA_DirectInMemory);
354
355 // Ignore empty structs/unions.
356 if (isEmptyRecord(Context&: getContext(), T: Ty, AllowArrays: true))
357 return ABIArgInfo::getIgnore();
358
359 // Lower single-element structs to just pass a regular value. TODO: We
360 // could do reasonable-size multiple-element structs too, using getExpand(),
361 // though watch out for things like bitfields.
362 if (const Type *SeltTy = isSingleElementStruct(T: Ty, Context&: getContext()))
363 return ABIArgInfo::getDirect(T: CGT.ConvertType(T: QualType(SeltTy, 0)));
364
365 if (const auto *RD = Ty->getAsRecordDecl();
366 RD && RD->hasFlexibleArrayMember())
367 return DefaultABIInfo::classifyArgumentType(RetTy: Ty);
368
369 uint64_t Size = getContext().getTypeSize(T: Ty);
370 if (Size <= 64) {
371 // Pack aggregates <= 8 bytes into single VGPR or pair.
372 unsigned NumRegs = (Size + 31) / 32;
373 NumRegsLeft -= std::min(a: NumRegsLeft, b: NumRegs);
374
375 if (Size <= 16)
376 return ABIArgInfo::getDirect(T: llvm::Type::getInt16Ty(C&: getVMContext()));
377
378 if (Size <= 32)
379 return ABIArgInfo::getDirect(T: llvm::Type::getInt32Ty(C&: getVMContext()));
380
381 // TODO: This is an AMDGPU oddity, and might be vestigial, we retain it to
382 // ensure consistency, but it should be revisited.
383 llvm::Type *I32Ty = llvm::Type::getInt32Ty(C&: getVMContext());
384 return ABIArgInfo::getDirect(T: llvm::ArrayType::get(ElementType: I32Ty, NumElements: 2));
385 }
386
387 if (NumRegsLeft > 0) {
388 uint64_t NumRegs = numRegsForType(Ty);
389 if (NumRegsLeft >= NumRegs) {
390 NumRegsLeft -= NumRegs;
391 return ABIArgInfo::getDirect();
392 }
393 }
394
395 // Use pass-by-reference in stead of pass-by-value for struct arguments in
396 // function ABI.
397 return ABIArgInfo::getIndirectAliased(
398 Alignment: getContext().getTypeAlignInChars(T: Ty),
399 AddrSpace: getContext().getTargetAddressSpace(AS: LangAS::opencl_private));
400}
401
402void AMDGCNSPIRVABIInfo::computeInfo(CGFunctionInfo &FI) const {
403 llvm::CallingConv::ID CC = FI.getCallingConvention();
404
405 if (!getCXXABI().classifyReturnType(FI))
406 FI.getReturnInfo() = classifyReturnType(RetTy: FI.getReturnType());
407
408 unsigned ArgumentIndex = 0;
409 const unsigned NumRequiredArgs = FI.getNumRequiredArgs();
410
411 NumRegsLeft = MaxNumRegsForArgsRet;
412 for (auto &I : FI.arguments()) {
413 if (CC == llvm::CallingConv::SPIR_KERNEL) {
414 I.info = classifyKernelArgumentType(Ty: I.type);
415 } else {
416 bool FixedArgument = ArgumentIndex++ < NumRequiredArgs;
417 I.info = classifyArgumentType(Ty: I.type, Variadic: !FixedArgument);
418 }
419 }
420}
421
422llvm::FixedVectorType *
423SPIRVABIInfo::getOptimalVectorMemoryType(llvm::FixedVectorType *Ty,
424 const LangOptions &LangOpt) const {
425 // For Logical SPIR-V, we don't know the underlying hardware or layout.
426 // This means we don't know which vector size is better, and also cannot
427 // assume a smaller vector size is stored in a larger vector size.
428 if (getTarget().getTriple().isSPIRVLogical())
429 return Ty;
430 return DefaultABIInfo::getOptimalVectorMemoryType(T: Ty, Opt: LangOpt);
431}
432
433llvm::FixedVectorType *AMDGCNSPIRVABIInfo::getOptimalVectorMemoryType(
434 llvm::FixedVectorType *Ty, const LangOptions &LangOpt) const {
435 // AMDGPU has legal instructions for 96-bit so 3x32 can be supported.
436 if (Ty->getNumElements() == 3 && getDataLayout().getTypeSizeInBits(Ty) == 96)
437 return Ty;
438 return DefaultABIInfo::getOptimalVectorMemoryType(T: Ty, Opt: LangOpt);
439}
440
441namespace clang {
442namespace CodeGen {
443void computeSPIRKernelABIInfo(CodeGenModule &CGM, CGFunctionInfo &FI) {
444 if (CGM.getTarget().getTriple().isSPIRV()) {
445 if (CGM.getTarget().getTriple().getVendor() == llvm::Triple::AMD)
446 AMDGCNSPIRVABIInfo(CGM.getTypes()).computeInfo(FI);
447 else
448 SPIRVABIInfo(CGM.getTypes()).computeInfo(FI);
449 } else {
450 CommonSPIRABIInfo(CGM.getTypes()).computeInfo(FI);
451 }
452}
453}
454}
455
456unsigned CommonSPIRTargetCodeGenInfo::getDeviceKernelCallingConv() const {
457 return llvm::CallingConv::SPIR_KERNEL;
458}
459
460LangAS SPIRVTargetCodeGenInfo::getSRetAddrSpace(const CXXRecordDecl *RD) const {
461 // Types with no viable copy/move must be constructed in-place, use the
462 // default AS so the sret pointer matches the "this" convention.
463 if (RD && !RD->canPassInRegisters())
464 return LangAS::Default;
465 return getLangASFromTargetAS(
466 TargetAS: getABIInfo().getDataLayout().getAllocaAddrSpace());
467}
468
469void SPIRVTargetCodeGenInfo::setCUDAKernelCallingConvention(
470 const FunctionType *&FT) const {
471 // Convert HIP kernels to SPIR-V kernels.
472 if (getABIInfo().getContext().getLangOpts().HIP) {
473 FT = getABIInfo().getContext().adjustFunctionType(
474 Fn: FT, EInfo: FT->getExtInfo().withCallingConv(cc: CC_DeviceKernel));
475 return;
476 }
477}
478
479void CommonSPIRTargetCodeGenInfo::setOCLKernelStubCallingConvention(
480 const FunctionType *&FT) const {
481 FT = getABIInfo().getContext().adjustFunctionType(
482 Fn: FT, EInfo: FT->getExtInfo().withCallingConv(cc: CC_C));
483}
484
485// LLVM currently assumes a null pointer has the bit pattern 0, but some GPU
486// targets use a non-zero encoding for null in certain address spaces.
487// Because SPIR(-V) is a generic target and the bit pattern of null in
488// non-generic AS is unspecified, materialize null in non-generic AS via an
489// addrspacecast from null in generic AS. This allows later lowering to
490// substitute the target's real sentinel value.
491llvm::Constant *
492CommonSPIRTargetCodeGenInfo::getNullPointer(const CodeGen::CodeGenModule &CGM,
493 llvm::PointerType *PT,
494 QualType QT) const {
495 if (!CodeGenUtils::spirNullPointerNeedsGenericCast(QT, Triple: CGM.getTriple()))
496 return llvm::ConstantPointerNull::get(T: PT);
497
498 auto &Ctx = CGM.getContext();
499 auto NPT = llvm::PointerType::get(
500 C&: PT->getContext(), AddressSpace: Ctx.getTargetAddressSpace(AS: LangAS::opencl_generic));
501 return llvm::ConstantExpr::getAddrSpaceCast(
502 C: llvm::ConstantPointerNull::get(T: NPT), Ty: PT);
503}
504
505LangAS
506SPIRVTargetCodeGenInfo::getGlobalVarAddressSpace(CodeGenModule &CGM,
507 const VarDecl *D) const {
508 assert(!CGM.getLangOpts().OpenCL &&
509 !(CGM.getLangOpts().CUDA && CGM.getLangOpts().CUDAIsDevice) &&
510 "Address space agnostic languages only");
511 // If we're here it means that we're using the SPIRDefIsGen ASMap, hence for
512 // the global AS we can rely on either cuda_device or sycl_global to be
513 // correct; however, since this is not a CUDA Device context, we use
514 // sycl_global to prevent confusion with the assertion.
515 LangAS DefaultGlobalAS = getLangASFromTargetAS(
516 TargetAS: CGM.getContext().getTargetAddressSpace(AS: LangAS::sycl_global));
517 if (!D)
518 return DefaultGlobalAS;
519
520 LangAS AddrSpace = D->getType().getAddressSpace();
521 if (AddrSpace != LangAS::Default)
522 return AddrSpace;
523
524 return DefaultGlobalAS;
525}
526
527void SPIRVTargetCodeGenInfo::setTargetAttributes(
528 const Decl *D, llvm::GlobalValue *GV, CodeGen::CodeGenModule &M) const {
529 if (GV->isDeclaration())
530 return;
531
532 const FunctionDecl *FD = dyn_cast_or_null<FunctionDecl>(Val: D);
533 if (!FD)
534 return;
535
536 llvm::Function *F = dyn_cast<llvm::Function>(Val: GV);
537 assert(F && "Expected GlobalValue to be a Function");
538
539 if (!M.getLangOpts().HIP ||
540 M.getTarget().getTriple().getVendor() != llvm::Triple::AMD)
541 return;
542
543 if (!FD->hasAttr<CUDAGlobalAttr>())
544 return;
545
546 unsigned N = M.getLangOpts().GPUMaxThreadsPerBlock;
547 if (auto FlatWGS = FD->getAttr<AMDGPUFlatWorkGroupSizeAttr>()) {
548 N = FlatWGS->getMax()->EvaluateKnownConstInt(Ctx: M.getContext()).getExtValue();
549 } else if (auto LB = FD->getAttr<CUDALaunchBoundsAttr>()) {
550 if (uint64_t MaxThreads = LB->getMaxThreads()
551 ->EvaluateKnownConstInt(Ctx: M.getContext())
552 .getExtValue())
553 N = MaxThreads;
554 }
555
556 // We encode the maximum flat WG size in the first component of the 3D
557 // max_work_group_size attribute, which will get reverse translated into the
558 // original AMDGPU attribute when targeting AMDGPU.
559 auto Int32Ty = llvm::IntegerType::getInt32Ty(C&: M.getLLVMContext());
560 llvm::Metadata *AttrMDArgs[] = {
561 llvm::ConstantAsMetadata::get(C: llvm::ConstantInt::get(Ty: Int32Ty, V: N)),
562 llvm::ConstantAsMetadata::get(C: llvm::ConstantInt::get(Ty: Int32Ty, V: 1)),
563 llvm::ConstantAsMetadata::get(C: llvm::ConstantInt::get(Ty: Int32Ty, V: 1))};
564
565 F->setMetadata(Kind: "max_work_group_size",
566 Node: llvm::MDNode::get(Context&: M.getLLVMContext(), MDs: AttrMDArgs));
567}
568
569StringRef SPIRVTargetCodeGenInfo::getLLVMSyncScopeStr(
570 const LangOptions &, SyncScope Scope, llvm::AtomicOrdering) const {
571 return *llvm::getAtomicScopeIRString(T: getABIInfo().getTarget().getTriple(),
572 S: getAtomicScope(S: Scope));
573}
574
575void SPIRVTargetCodeGenInfo::setTargetAtomicMetadata(
576 CodeGenFunction &CGF, llvm::Instruction &AtomicInst,
577 const AtomicExpr *AE) const {
578 if (CGF.CGM.getTriple().getVendor() != llvm::Triple::VendorType::AMD)
579 return;
580
581 auto *RMW = dyn_cast<llvm::AtomicRMWInst>(Val: &AtomicInst);
582 if (!RMW)
583 return;
584
585 AtomicOptions AO = CGF.CGM.getAtomicOpts();
586 llvm::MDNode *Empty = llvm::MDNode::get(Context&: CGF.getLLVMContext(), MDs: {});
587 if (!AO.getOption(Kind: clang::AtomicOptionKind::FineGrainedMemory))
588 RMW->setMetadata(Kind: "amdgpu.no.fine.grained.memory", Node: Empty);
589 if (!AO.getOption(Kind: clang::AtomicOptionKind::RemoteMemory))
590 RMW->setMetadata(Kind: "amdgpu.no.remote.memory", Node: Empty);
591 if (AO.getOption(Kind: clang::AtomicOptionKind::IgnoreDenormalMode) &&
592 RMW->getOperation() == llvm::AtomicRMWInst::FAdd &&
593 RMW->getType()->isFloatTy())
594 RMW->setMetadata(KindID: llvm::LLVMContext::MD_atomic_ignore_denormal_mode, Node: Empty);
595}
596
597/// Construct a SPIR-V target extension type for the given OpenCL image type.
598static llvm::Type *getSPIRVImageType(llvm::LLVMContext &Ctx, StringRef BaseType,
599 StringRef OpenCLName,
600 unsigned AccessQualifier) {
601 // These parameters compare to the operands of OpTypeImage (see
602 // https://registry.khronos.org/SPIR-V/specs/unified1/SPIRV.html#OpTypeImage
603 // for more details). The first 6 integer parameters all default to 0, and
604 // will be changed to 1 only for the image type(s) that set the parameter to
605 // one. The 7th integer parameter is the access qualifier, which is tacked on
606 // at the end.
607 SmallVector<unsigned, 7> IntParams = {0, 0, 0, 0, 0, 0};
608
609 // Choose the dimension of the image--this corresponds to the Dim enum in
610 // SPIR-V (first integer parameter of OpTypeImage).
611 if (OpenCLName.starts_with(Prefix: "image2d"))
612 IntParams[0] = 1;
613 else if (OpenCLName.starts_with(Prefix: "image3d"))
614 IntParams[0] = 2;
615 else if (OpenCLName == "image1d_buffer")
616 IntParams[0] = 5; // Buffer
617 else
618 assert(OpenCLName.starts_with("image1d") && "Unknown image type");
619
620 // Set the other integer parameters of OpTypeImage if necessary. Note that the
621 // OpenCL image types don't provide any information for the Sampled or
622 // Image Format parameters.
623 if (OpenCLName.contains(Other: "_depth"))
624 IntParams[1] = 1;
625 if (OpenCLName.contains(Other: "_array"))
626 IntParams[2] = 1;
627 if (OpenCLName.contains(Other: "_msaa"))
628 IntParams[3] = 1;
629
630 // Access qualifier
631 IntParams.push_back(Elt: AccessQualifier);
632
633 return llvm::TargetExtType::get(Context&: Ctx, Name: BaseType, Types: {llvm::Type::getVoidTy(C&: Ctx)},
634 Ints: IntParams);
635}
636
637llvm::Type *CommonSPIRTargetCodeGenInfo::getOpenCLType(CodeGenModule &CGM,
638 const Type *Ty) const {
639 llvm::LLVMContext &Ctx = CGM.getLLVMContext();
640 if (auto *PipeTy = dyn_cast<PipeType>(Val: Ty))
641 return llvm::TargetExtType::get(Context&: Ctx, Name: "spirv.Pipe", Types: {},
642 Ints: {!PipeTy->isReadOnly()});
643 if (auto *BuiltinTy = dyn_cast<BuiltinType>(Val: Ty)) {
644 enum AccessQualifier : unsigned { AQ_ro = 0, AQ_wo = 1, AQ_rw = 2 };
645 switch (BuiltinTy->getKind()) {
646#define IMAGE_TYPE(ImgType, Id, SingletonId, Access, Suffix) \
647 case BuiltinType::Id: \
648 return getSPIRVImageType(Ctx, "spirv.Image", #ImgType, AQ_##Suffix);
649#include "clang/Basic/OpenCLImageTypes.def"
650 case BuiltinType::OCLSampler:
651 return llvm::TargetExtType::get(Context&: Ctx, Name: "spirv.Sampler");
652 case BuiltinType::OCLEvent:
653 return llvm::TargetExtType::get(Context&: Ctx, Name: "spirv.Event");
654 case BuiltinType::OCLClkEvent:
655 return llvm::TargetExtType::get(Context&: Ctx, Name: "spirv.DeviceEvent");
656 case BuiltinType::OCLQueue:
657 return llvm::TargetExtType::get(Context&: Ctx, Name: "spirv.Queue");
658 case BuiltinType::OCLReserveID:
659 return llvm::TargetExtType::get(Context&: Ctx, Name: "spirv.ReserveId");
660#define INTEL_SUBGROUP_AVC_TYPE(Name, Id) \
661 case BuiltinType::OCLIntelSubgroupAVC##Id: \
662 return llvm::TargetExtType::get(Ctx, "spirv.Avc" #Id "INTEL");
663#include "clang/Basic/OpenCLExtensionTypes.def"
664 default:
665 return nullptr;
666 }
667 }
668
669 return nullptr;
670}
671
672// Gets a spirv.IntegralConstant or spirv.Literal. If IntegralType is present,
673// returns an IntegralConstant, otherwise returns a Literal.
674static llvm::Type *getInlineSpirvConstant(CodeGenModule &CGM,
675 llvm::Type *IntegralType,
676 llvm::APInt Value) {
677 llvm::LLVMContext &Ctx = CGM.getLLVMContext();
678
679 // Convert the APInt value to an array of uint32_t words
680 llvm::SmallVector<uint32_t> Words;
681
682 while (Value.ugt(RHS: 0)) {
683 uint32_t Word = Value.trunc(width: 32).getZExtValue();
684 Value.lshrInPlace(ShiftAmt: 32);
685
686 Words.push_back(Elt: Word);
687 }
688 if (Words.size() == 0)
689 Words.push_back(Elt: 0);
690
691 if (IntegralType)
692 return llvm::TargetExtType::get(Context&: Ctx, Name: "spirv.IntegralConstant",
693 Types: {IntegralType}, Ints: Words);
694 return llvm::TargetExtType::get(Context&: Ctx, Name: "spirv.Literal", Types: {}, Ints: Words);
695}
696
697static llvm::Type *getInlineSpirvType(CodeGenModule &CGM,
698 const HLSLInlineSpirvType *SpirvType) {
699 llvm::LLVMContext &Ctx = CGM.getLLVMContext();
700
701 llvm::SmallVector<llvm::Type *> Operands;
702
703 for (auto &Operand : SpirvType->getOperands()) {
704 using SpirvOperandKind = SpirvOperand::SpirvOperandKind;
705
706 llvm::Type *Result = nullptr;
707 switch (Operand.getKind()) {
708 case SpirvOperandKind::ConstantId: {
709 llvm::Type *IntegralType =
710 CGM.getTypes().ConvertType(T: Operand.getResultType());
711
712 Result = getInlineSpirvConstant(CGM, IntegralType, Value: Operand.getValue());
713 break;
714 }
715 case SpirvOperandKind::Literal: {
716 Result = getInlineSpirvConstant(CGM, IntegralType: nullptr, Value: Operand.getValue());
717 break;
718 }
719 case SpirvOperandKind::TypeId: {
720 QualType TypeOperand = Operand.getResultType();
721 if (const auto *RD = TypeOperand->getAsRecordDecl()) {
722 assert(RD->isCompleteDefinition() &&
723 "Type completion should have been required in Sema");
724
725 const FieldDecl *HandleField = RD->findFirstNamedDataMember();
726 if (HandleField) {
727 QualType ResourceType = HandleField->getType();
728 if (ResourceType->getAs<HLSLAttributedResourceType>()) {
729 TypeOperand = ResourceType;
730 }
731 }
732 }
733 Result = CGM.getTypes().ConvertType(T: TypeOperand);
734 break;
735 }
736 default:
737 llvm_unreachable("HLSLInlineSpirvType had invalid operand!");
738 break;
739 }
740
741 assert(Result);
742 Operands.push_back(Elt: Result);
743 }
744
745 return llvm::TargetExtType::get(Context&: Ctx, Name: "spirv.Type", Types: Operands,
746 Ints: {SpirvType->getOpcode(), SpirvType->getSize(),
747 SpirvType->getAlignment()});
748}
749
750llvm::Type *CommonSPIRTargetCodeGenInfo::getHLSLType(
751 CodeGenModule &CGM, const Type *Ty,
752 const CGHLSLOffsetInfo &OffsetInfo) const {
753 llvm::LLVMContext &Ctx = CGM.getLLVMContext();
754
755 if (auto *SpirvType = dyn_cast<HLSLInlineSpirvType>(Val: Ty))
756 return getInlineSpirvType(CGM, SpirvType);
757
758 auto *ResType = dyn_cast<HLSLAttributedResourceType>(Val: Ty);
759 if (!ResType)
760 return nullptr;
761
762 const HLSLAttributedResourceType::Attributes &ResAttrs = ResType->getAttrs();
763 switch (ResAttrs.ResourceClass) {
764 case llvm::dxil::ResourceClass::UAV:
765 case llvm::dxil::ResourceClass::SRV: {
766 // TypedBuffer and RawBuffer both need element type
767 QualType ContainedTy = ResType->getContainedType();
768 if (ContainedTy.isNull())
769 return nullptr;
770
771 assert(!ResAttrs.IsROV &&
772 "Rasterizer order views not implemented for SPIR-V yet");
773
774 if (!ResAttrs.RawBuffer) {
775 // convert element type
776 return getSPIRVImageTypeFromHLSLResource(attributes: ResAttrs, SampledType: ContainedTy, CGM);
777 }
778
779 if (ResAttrs.IsCounter) {
780 llvm::Type *ElemType = llvm::Type::getInt32Ty(C&: Ctx);
781 uint32_t StorageClass = /* StorageBuffer storage class */ 12;
782 return llvm::TargetExtType::get(Context&: Ctx, Name: "spirv.VulkanBuffer", Types: {ElemType},
783 Ints: {StorageClass, true});
784 }
785 llvm::Type *ElemType = CGM.getTypes().ConvertTypeForMem(T: ContainedTy);
786 llvm::ArrayType *RuntimeArrayType = llvm::ArrayType::get(ElementType: ElemType, NumElements: 0);
787 uint32_t StorageClass = /* StorageBuffer storage class */ 12;
788 bool IsWritable = ResAttrs.ResourceClass == llvm::dxil::ResourceClass::UAV;
789 return llvm::TargetExtType::get(Context&: Ctx, Name: "spirv.VulkanBuffer",
790 Types: {RuntimeArrayType},
791 Ints: {StorageClass, IsWritable});
792 }
793 case llvm::dxil::ResourceClass::CBuffer: {
794 QualType ContainedTy = ResType->getContainedType();
795 if (ContainedTy.isNull() || !ContainedTy->isStructureType())
796 return nullptr;
797
798 llvm::StructType *BufferLayoutTy =
799 HLSLBufferLayoutBuilder(CGM).layOutStruct(
800 StructType: ContainedTy->getAsCanonical<RecordType>(), OffsetInfo);
801 uint32_t StorageClass = /* Uniform storage class */ 2;
802 return llvm::TargetExtType::get(Context&: Ctx, Name: "spirv.VulkanBuffer", Types: {BufferLayoutTy},
803 Ints: {StorageClass, false});
804 break;
805 }
806 case llvm::dxil::ResourceClass::Sampler:
807 return llvm::TargetExtType::get(Context&: Ctx, Name: "spirv.Sampler");
808 }
809 return nullptr;
810}
811
812static unsigned
813getImageFormat(const LangOptions &LangOpts,
814 const HLSLAttributedResourceType::Attributes &attributes,
815 llvm::Type *SampledType, QualType Ty, unsigned NumChannels) {
816 // For images with `Sampled` operand equal to 2, there are restrictions on
817 // using the Unknown image format. To avoid these restrictions in common
818 // cases, we guess an image format for them based on the sampled type and the
819 // number of channels. This is intended to match the behaviour of DXC.
820 if (LangOpts.HLSLSpvUseUnknownImageFormat ||
821 attributes.ResourceClass != llvm::dxil::ResourceClass::UAV) {
822 return 0; // Unknown
823 }
824
825 if (SampledType->isIntegerTy(BitWidth: 32)) {
826 if (Ty->isSignedIntegerType()) {
827 if (NumChannels == 1)
828 return 24; // R32i
829 if (NumChannels == 2)
830 return 25; // Rg32i
831 if (NumChannels == 4)
832 return 21; // Rgba32i
833 } else {
834 if (NumChannels == 1)
835 return 33; // R32ui
836 if (NumChannels == 2)
837 return 35; // Rg32ui
838 if (NumChannels == 4)
839 return 30; // Rgba32ui
840 }
841 } else if (SampledType->isIntegerTy(BitWidth: 64)) {
842 if (NumChannels == 1) {
843 if (Ty->isSignedIntegerType()) {
844 return 41; // R64i
845 }
846 return 40; // R64ui
847 }
848 } else if (SampledType->isFloatTy()) {
849 if (NumChannels == 1)
850 return 3; // R32f
851 if (NumChannels == 2)
852 return 6; // Rg32f
853 if (NumChannels == 4)
854 return 1; // Rgba32f
855 }
856
857 return 0; // Unknown
858}
859
860llvm::Type *CommonSPIRTargetCodeGenInfo::getSPIRVImageTypeFromHLSLResource(
861 const HLSLAttributedResourceType::Attributes &attributes, QualType Ty,
862 CodeGenModule &CGM) const {
863 llvm::LLVMContext &Ctx = CGM.getLLVMContext();
864
865 unsigned NumChannels = 1;
866 Ty = Ty->getCanonicalTypeUnqualified();
867 if (const VectorType *V = dyn_cast<VectorType>(Val&: Ty)) {
868 NumChannels = V->getNumElements();
869 Ty = V->getElementType();
870 }
871 assert(!Ty->isVectorType() && "We still have a vector type.");
872
873 llvm::Type *SampledType = CGM.getTypes().ConvertTypeForMem(T: Ty);
874
875 assert((SampledType->isIntegerTy() || SampledType->isFloatingPointTy()) &&
876 "The element type for a SPIR-V resource must be a scalar integer or "
877 "floating point type.");
878
879 assert((!SampledType->isIntegerTy(64) || NumChannels <= 2) &&
880 "A 64-bit SPIR-V resource element can have at most 2 components.");
881
882 // SPIR-V has no 64-bit multi-component image format, so pack a 2-component
883 // 64-bit typed buffer into a 4-component 32-bit image. The backend
884 // reinterprets it with OpBitcast on load and store.
885 if (SampledType->isIntegerTy(BitWidth: 64) && NumChannels == 2) {
886 SampledType = llvm::Type::getInt32Ty(C&: Ctx);
887 NumChannels = 4;
888 }
889
890 // These parameters correspond to the operands to the OpTypeImage SPIR-V
891 // instruction. See
892 // https://registry.khronos.org/SPIR-V/specs/unified1/SPIRV.html#OpTypeImage.
893 SmallVector<unsigned, 6> IntParams(6, 0);
894
895 const char *Name =
896 Ty->isSignedIntegerType() ? "spirv.SignedImage" : "spirv.Image";
897
898 // Dim
899 switch (attributes.ResourceDimension) {
900 case llvm::dxil::ResourceDimension::Dim1D:
901 IntParams[0] = 0;
902 break;
903 case llvm::dxil::ResourceDimension::Dim2D:
904 IntParams[0] = 1;
905 break;
906 case llvm::dxil::ResourceDimension::Dim3D:
907 IntParams[0] = 2;
908 break;
909 case llvm::dxil::ResourceDimension::Cube:
910 IntParams[0] = 3;
911 break;
912 case llvm::dxil::ResourceDimension::Unknown:
913 IntParams[0] = 5;
914 break;
915 }
916
917 // Depth
918 // HLSL does not indicate if it is a depth texture or not, so we use unknown.
919 IntParams[1] = 2;
920
921 // Arrayed
922 IntParams[2] = static_cast<unsigned>(attributes.IsArray);
923
924 // MS
925 IntParams[3] = static_cast<unsigned>(attributes.isMultiSampled());
926
927 // Sampled
928 IntParams[4] =
929 attributes.ResourceClass == llvm::dxil::ResourceClass::UAV ? 2 : 1;
930
931 // Image format.
932 IntParams[5] = getImageFormat(LangOpts: CGM.getLangOpts(), attributes, SampledType, Ty,
933 NumChannels);
934
935 llvm::TargetExtType *ImageType =
936 llvm::TargetExtType::get(Context&: Ctx, Name, Types: {SampledType}, Ints: IntParams);
937 return ImageType;
938}
939
940std::unique_ptr<TargetCodeGenInfo>
941CodeGen::createCommonSPIRTargetCodeGenInfo(CodeGenModule &CGM) {
942 return std::make_unique<CommonSPIRTargetCodeGenInfo>(args&: CGM.getTypes());
943}
944
945std::unique_ptr<TargetCodeGenInfo>
946CodeGen::createSPIRVTargetCodeGenInfo(CodeGenModule &CGM) {
947 return std::make_unique<SPIRVTargetCodeGenInfo>(args&: CGM.getTypes());
948}
949