1//===-- AMDGPUSubtarget.cpp - AMDGPU Subtarget Information ----------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// Implements the AMDGPU specific subclass of TargetSubtarget.
11//
12//===----------------------------------------------------------------------===//
13
14#include "AMDGPUSubtarget.h"
15#include "AMDGPUCallLowering.h"
16#include "AMDGPUInstructionSelector.h"
17#include "AMDGPULegalizerInfo.h"
18#include "AMDGPURegisterBankInfo.h"
19#include "R600Subtarget.h"
20#include "SIMachineFunctionInfo.h"
21#include "Utils/AMDGPUBaseInfo.h"
22#include "llvm/CodeGen/MachineScheduler.h"
23#include "llvm/CodeGen/TargetFrameLowering.h"
24#include "llvm/IR/DiagnosticInfo.h"
25#include "llvm/IR/IntrinsicsAMDGPU.h"
26#include "llvm/IR/IntrinsicsR600.h"
27#include "llvm/IR/MDBuilder.h"
28#include <algorithm>
29
30using namespace llvm;
31
32#define DEBUG_TYPE "amdgpu-subtarget"
33
34// Returns the maximum per-workgroup LDS allocation size (in bytes) that still
35// allows the given function to achieve an occupancy of NWaves waves per
36// SIMD / EU, taking into account only the function's *maximum* workgroup size.
37unsigned
38AMDGPUSubtarget::getMaxLocalMemSizeWithWaveCount(unsigned NWaves,
39 const Function &F) const {
40 const unsigned WaveSize = getWavefrontSize();
41 const unsigned WorkGroupSize = getFlatWorkGroupSizes(F).second;
42 const unsigned WavesPerWorkgroup =
43 std::max(a: 1u, b: (WorkGroupSize + WaveSize - 1) / WaveSize);
44
45 const unsigned WorkGroupsPerCU =
46 std::max(a: 1u, b: (NWaves * getNumWorkGroupSIMDs()) / WavesPerWorkgroup);
47
48 const unsigned Granularity = std::max(a: LDSAllocationGranularity, b: 1u);
49 return alignDown(Value: std::min(a: getLocalMemorySize() / WorkGroupsPerCU,
50 b: getAddressableLocalMemorySize()),
51 Align: Granularity);
52}
53
54std::pair<unsigned, unsigned> AMDGPUSubtarget::getOccupancyWithWorkGroupSizes(
55 uint32_t LDSBytes, std::pair<unsigned, unsigned> FlatWorkGroupSizes) const {
56
57 // LDS granularity accounted for by aligning the queried LDS size to the
58 // allocation block size.
59 const unsigned Granularity = std::max(a: LDSAllocationGranularity, b: 1u);
60 LDSBytes = alignTo(Value: LDSBytes, Align: Granularity);
61 const unsigned MaxWGsLDS = getLocalMemorySize() / std::max(a: LDSBytes, b: 1u);
62
63 // Queried LDS size may be larger than available on a CU, in which case we
64 // consider the only achievable occupancy to be 1, in line with what we
65 // consider the occupancy to be when the number of requested registers in a
66 // particular bank is higher than the number of available ones in that bank.
67 if (!MaxWGsLDS)
68 return {1, 1};
69
70 const unsigned WaveSize = getWavefrontSize(), WavesPerEU = getMaxWavesPerEU();
71
72 auto PropsFromWGSize = [=](unsigned WGSize)
73 -> std::tuple<const unsigned, const unsigned, unsigned> {
74 unsigned WavesPerWG = divideCeil(Numerator: WGSize, Denominator: WaveSize);
75 unsigned WGsPerCU = std::min(a: getMaxWorkGroupsPerCU(FlatWorkGroupSize: WGSize), b: MaxWGsLDS);
76 return {WavesPerWG, WGsPerCU, WavesPerWG * WGsPerCU};
77 };
78
79 // The maximum group size will generally yield the minimum number of
80 // workgroups, maximum number of waves, and minimum occupancy. The opposite is
81 // generally true for the minimum group size. LDS or barrier ressource
82 // limitations can flip those minimums/maximums.
83 const auto [MinWGSize, MaxWGSize] = FlatWorkGroupSizes;
84 auto [MinWavesPerWG, MaxWGsPerCU, MaxWavesPerCU] = PropsFromWGSize(MinWGSize);
85 auto [MaxWavesPerWG, MinWGsPerCU, MinWavesPerCU] = PropsFromWGSize(MaxWGSize);
86
87 // It is possible that we end up with flipped minimum and maximum number of
88 // waves per CU when the number of minimum/maximum concurrent groups on the CU
89 // is limited by LDS usage or barrier resources.
90 if (MinWavesPerCU >= MaxWavesPerCU) {
91 std::swap(a&: MinWavesPerCU, b&: MaxWavesPerCU);
92 } else {
93 const unsigned WaveSlotsPerCU = WavesPerEU * getNumWorkGroupSIMDs();
94
95 // Look for a potential smaller group size than the maximum which decreases
96 // the concurrent number of waves on the CU for the same number of
97 // concurrent workgroups on the CU.
98 unsigned MinWavesPerCUForWGSize =
99 divideCeil(Numerator: WaveSlotsPerCU, Denominator: MinWGsPerCU + 1) * MinWGsPerCU;
100 if (MinWavesPerCU > MinWavesPerCUForWGSize) {
101 unsigned ExcessSlots = MinWavesPerCU - MinWavesPerCUForWGSize;
102 if (unsigned ExcessSlotsPerWG = ExcessSlots / MinWGsPerCU) {
103 // There may exist a smaller group size than the maximum that achieves
104 // the minimum number of waves per CU. This group size is the largest
105 // possible size that requires MaxWavesPerWG - E waves where E is
106 // maximized under the following constraints.
107 // 1. 0 <= E <= ExcessSlotsPerWG
108 // 2. (MaxWavesPerWG - E) * WaveSize >= MinWGSize
109 MinWavesPerCU -= MinWGsPerCU * std::min(a: ExcessSlotsPerWG,
110 b: MaxWavesPerWG - MinWavesPerWG);
111 }
112 }
113
114 // Look for a potential larger group size than the minimum which increases
115 // the concurrent number of waves on the CU for the same number of
116 // concurrent workgroups on the CU.
117 unsigned LeftoverSlots = WaveSlotsPerCU - MaxWGsPerCU * MinWavesPerWG;
118 if (unsigned LeftoverSlotsPerWG = LeftoverSlots / MaxWGsPerCU) {
119 // There may exist a larger group size than the minimum that achieves the
120 // maximum number of waves per CU. This group size is the smallest
121 // possible size that requires MinWavesPerWG + L waves where L is
122 // maximized under the following constraints.
123 // 1. 0 <= L <= LeftoverSlotsPerWG
124 // 2. (MinWavesPerWG + L - 1) * WaveSize <= MaxWGSize
125 MaxWavesPerCU += MaxWGsPerCU * std::min(a: LeftoverSlotsPerWG,
126 b: ((MaxWGSize - 1) / WaveSize) + 1 -
127 MinWavesPerWG);
128 }
129 }
130
131 // Return the minimum/maximum number of waves on any EU, assuming that all
132 // wavefronts are spread across all EUs as evenly as possible.
133 return {std::clamp(val: MinWavesPerCU / getNumWorkGroupSIMDs(), lo: 1U, hi: WavesPerEU),
134 std::clamp(val: divideCeil(Numerator: MaxWavesPerCU, Denominator: getNumWorkGroupSIMDs()), lo: 1U,
135 hi: WavesPerEU)};
136}
137
138std::pair<unsigned, unsigned> AMDGPUSubtarget::getOccupancyWithWorkGroupSizes(
139 const MachineFunction &MF) const {
140 const auto *MFI = MF.getInfo<SIMachineFunctionInfo>();
141 return getOccupancyWithWorkGroupSizes(LDSBytes: MFI->getLDSSize(), F: MF.getFunction());
142}
143
144std::pair<unsigned, unsigned>
145AMDGPUSubtarget::getDefaultFlatWorkGroupSize(CallingConv::ID CC) const {
146 switch (CC) {
147 case CallingConv::AMDGPU_VS:
148 case CallingConv::AMDGPU_LS:
149 case CallingConv::AMDGPU_HS:
150 case CallingConv::AMDGPU_ES:
151 case CallingConv::AMDGPU_GS:
152 case CallingConv::AMDGPU_PS:
153 return std::pair(1, getWavefrontSize());
154 default:
155 return std::pair(1u, getMaxFlatWorkGroupSize());
156 }
157}
158
159std::pair<unsigned, unsigned> AMDGPUSubtarget::getFlatWorkGroupSizes(
160 const Function &F) const {
161 // Default minimum/maximum flat work group sizes.
162 std::pair<unsigned, unsigned> Default =
163 getDefaultFlatWorkGroupSize(CC: F.getCallingConv());
164
165 // Requested minimum/maximum flat work group sizes.
166 std::pair<unsigned, unsigned> Requested = AMDGPU::getIntegerPairAttribute(
167 F, Name: "amdgpu-flat-work-group-size", Default);
168
169 // Make sure requested minimum is less than requested maximum.
170 if (Requested.first > Requested.second)
171 return Default;
172
173 // Make sure requested values do not violate subtarget's specifications.
174 if (Requested.first < getMinFlatWorkGroupSize())
175 return Default;
176 if (Requested.second > getMaxFlatWorkGroupSize())
177 return Default;
178
179 return Requested;
180}
181
182std::pair<unsigned, unsigned> AMDGPUSubtarget::getEffectiveWavesPerEU(
183 std::pair<unsigned, unsigned> RequestedWavesPerEU,
184 std::pair<unsigned, unsigned> FlatWorkGroupSizes, unsigned LDSBytes) const {
185 // Default minimum/maximum number of waves per EU. The range of flat workgroup
186 // sizes limits the achievable maximum, and we aim to support enough waves per
187 // EU so that we can concurrently execute all waves of a single workgroup of
188 // maximum size on a CU.
189 std::pair<unsigned, unsigned> Default = {
190 getWavesPerEUForWorkGroup(FlatWorkGroupSize: FlatWorkGroupSizes.second),
191 getOccupancyWithWorkGroupSizes(LDSBytes, FlatWorkGroupSizes).second};
192 Default.first = std::min(a: Default.first, b: Default.second);
193
194 // Make sure requested minimum is within the default range and lower than the
195 // requested maximum. The latter must not violate target specification.
196 if (RequestedWavesPerEU.first < Default.first ||
197 RequestedWavesPerEU.first > Default.second ||
198 RequestedWavesPerEU.first > RequestedWavesPerEU.second ||
199 RequestedWavesPerEU.second > getMaxWavesPerEU())
200 return Default;
201
202 // We cannot exceed maximum occupancy implied by flat workgroup size and LDS.
203 RequestedWavesPerEU.second =
204 std::min(a: RequestedWavesPerEU.second, b: Default.second);
205 return RequestedWavesPerEU;
206}
207
208std::pair<unsigned, unsigned>
209AMDGPUSubtarget::getWavesPerEU(const Function &F) const {
210 // Default/requested minimum/maximum flat work group sizes.
211 std::pair<unsigned, unsigned> FlatWorkGroupSizes = getFlatWorkGroupSizes(F);
212 // Minimum number of bytes allocated in the LDS.
213 unsigned LDSBytes =
214 AMDGPU::getIntegerPairAttribute(F, Name: "amdgpu-lds-size", Default: {0, UINT32_MAX},
215 /*OnlyFirstRequired=*/true)
216 .first;
217 return getWavesPerEU(FlatWorkGroupSizes, LDSBytes, F);
218}
219
220std::pair<unsigned, unsigned>
221AMDGPUSubtarget::getWavesPerEU(std::pair<unsigned, unsigned> FlatWorkGroupSizes,
222 unsigned LDSBytes, const Function &F) const {
223 // Default minimum/maximum number of waves per execution unit.
224 std::pair<unsigned, unsigned> Default(1, getMaxWavesPerEU());
225
226 // Requested minimum/maximum number of waves per execution unit.
227 std::pair<unsigned, unsigned> Requested =
228 AMDGPU::getIntegerPairAttribute(F, Name: "amdgpu-waves-per-eu", Default, OnlyFirstRequired: true);
229 return getEffectiveWavesPerEU(RequestedWavesPerEU: Requested, FlatWorkGroupSizes, LDSBytes);
230}
231
232std::optional<unsigned>
233AMDGPUSubtarget::getReqdWorkGroupSize(const Function &Kernel,
234 unsigned Dim) const {
235 auto *Node = Kernel.getMetadata(Kind: "reqd_work_group_size");
236 if (Node && Node->getNumOperands() == 3)
237 return mdconst::extract<ConstantInt>(MD: Node->getOperand(I: Dim))->getZExtValue();
238 return std::nullopt;
239}
240
241bool AMDGPUSubtarget::hasWavefrontsEvenlySplittingXDim(
242 const Function &F, bool RequiresUniformYZ) const {
243 auto *Node = F.getMetadata(Kind: "reqd_work_group_size");
244 if (!Node || Node->getNumOperands() != 3)
245 return false;
246 unsigned XLen =
247 mdconst::extract<ConstantInt>(MD: Node->getOperand(I: 0))->getZExtValue();
248 unsigned YLen =
249 mdconst::extract<ConstantInt>(MD: Node->getOperand(I: 1))->getZExtValue();
250 unsigned ZLen =
251 mdconst::extract<ConstantInt>(MD: Node->getOperand(I: 2))->getZExtValue();
252
253 bool Is1D = YLen <= 1 && ZLen <= 1;
254 bool IsXLargeEnough =
255 isPowerOf2_32(Value: XLen) && (!RequiresUniformYZ || XLen >= getWavefrontSize());
256 return Is1D || IsXLargeEnough;
257}
258
259bool AMDGPUSubtarget::isMesaKernel(const Function &F) const {
260 return isMesa3DOS() && !AMDGPU::isShader(CC: F.getCallingConv());
261}
262
263unsigned AMDGPUSubtarget::getMaxWorkitemID(const Function &Kernel,
264 unsigned Dimension) const {
265 std::optional<unsigned> ReqdSize = getReqdWorkGroupSize(Kernel, Dim: Dimension);
266 if (ReqdSize)
267 return *ReqdSize - 1;
268 return getFlatWorkGroupSizes(F: Kernel).second - 1;
269}
270
271bool AMDGPUSubtarget::isSingleLaneExecution(const Function &Func) const {
272 for (int I = 0; I < 3; ++I) {
273 if (getMaxWorkitemID(Kernel: Func, Dimension: I) > 0)
274 return false;
275 }
276
277 // If the function may call the WWM intrinsic, just return false as
278 // all threads will be active at some point
279 if (!Func.hasFnAttribute(Kind: "amdgpu-no-wwm"))
280 return false;
281
282 return true;
283}
284
285bool AMDGPUSubtarget::makeLIDRangeMetadata(Instruction *I) const {
286 Function *Kernel = I->getFunction();
287 unsigned MinSize = 0;
288 unsigned MaxSize = getFlatWorkGroupSizes(F: *Kernel).second;
289 bool IdQuery = false;
290
291 // If reqd_work_group_size is present it narrows value down.
292 if (auto *CI = dyn_cast<CallInst>(Val: I)) {
293 const Function *F = CI->getCalledFunction();
294 if (F) {
295 unsigned Dim = UINT_MAX;
296 switch (F->getIntrinsicID()) {
297 case Intrinsic::amdgcn_workitem_id_x:
298 case Intrinsic::r600_read_tidig_x:
299 IdQuery = true;
300 [[fallthrough]];
301 case Intrinsic::r600_read_local_size_x:
302 Dim = 0;
303 break;
304 case Intrinsic::amdgcn_workitem_id_y:
305 case Intrinsic::r600_read_tidig_y:
306 IdQuery = true;
307 [[fallthrough]];
308 case Intrinsic::r600_read_local_size_y:
309 Dim = 1;
310 break;
311 case Intrinsic::amdgcn_workitem_id_z:
312 case Intrinsic::r600_read_tidig_z:
313 IdQuery = true;
314 [[fallthrough]];
315 case Intrinsic::r600_read_local_size_z:
316 Dim = 2;
317 break;
318 default:
319 break;
320 }
321
322 if (Dim <= 3) {
323 std::optional<unsigned> ReqdSize = getReqdWorkGroupSize(Kernel: *Kernel, Dim);
324 if (ReqdSize)
325 MinSize = MaxSize = *ReqdSize;
326 }
327 }
328 }
329
330 if (!MaxSize)
331 return false;
332
333 // Range metadata is [Lo, Hi). For ID query we need to pass max size
334 // as Hi. For size query we need to pass Hi + 1.
335 if (IdQuery)
336 MinSize = 0;
337 else
338 ++MaxSize;
339
340 APInt Lower{32, MinSize};
341 APInt Upper{32, MaxSize};
342 if (auto *CI = dyn_cast<CallBase>(Val: I)) {
343 ConstantRange Range(Lower, Upper);
344 CI->addRangeRetAttr(CR: Range);
345 } else {
346 MDBuilder MDB(I->getContext());
347 MDNode *MaxWorkGroupSizeRange = MDB.createRange(Lo: Lower, Hi: Upper);
348 I->setMetadata(KindID: LLVMContext::MD_range, Node: MaxWorkGroupSizeRange);
349 }
350 return true;
351}
352
353unsigned AMDGPUSubtarget::getImplicitArgNumBytes(const Function &F) const {
354
355 // We don't allocate the segment if we know the implicit arguments weren't
356 // used, even if the ABI implies we need them.
357 if (F.hasFnAttribute(Kind: "amdgpu-no-implicitarg-ptr"))
358 return 0;
359
360 if (isMesaKernel(F))
361 return 16;
362
363 // Assume all implicit inputs are used by default
364 const Module *M = F.getParent();
365 unsigned NBytes =
366 AMDGPU::getAMDHSACodeObjectVersion(M: *M) >= AMDGPU::AMDHSA_COV5 ? 256 : 56;
367 return F.getFnAttributeAsParsedInteger(Kind: "amdgpu-implicitarg-num-bytes",
368 Default: NBytes);
369}
370
371uint64_t AMDGPUSubtarget::getExplicitKernArgSize(const Function &F,
372 Align &MaxAlign) const {
373 assert(F.getCallingConv() == CallingConv::AMDGPU_KERNEL ||
374 F.getCallingConv() == CallingConv::SPIR_KERNEL);
375
376 const DataLayout &DL = F.getDataLayout();
377 uint64_t ExplicitArgBytes = 0;
378 MaxAlign = Align(1);
379
380 for (const Argument &Arg : F.args()) {
381 if (Arg.hasAttribute(Kind: "amdgpu-hidden-argument"))
382 continue;
383
384 const bool IsByRef = Arg.hasByRefAttr();
385 Type *ArgTy = IsByRef ? Arg.getParamByRefType() : Arg.getType();
386 Align Alignment = DL.getValueOrABITypeAlignment(
387 Alignment: IsByRef ? Arg.getParamAlign() : std::nullopt, Ty: ArgTy);
388 uint64_t AllocSize = DL.getTypeAllocSize(Ty: ArgTy);
389 ExplicitArgBytes = alignTo(Size: ExplicitArgBytes, A: Alignment) + AllocSize;
390 MaxAlign = std::max(a: MaxAlign, b: Alignment);
391 }
392
393 return ExplicitArgBytes;
394}
395
396unsigned AMDGPUSubtarget::getKernArgSegmentSize(const Function &F,
397 Align &MaxAlign) const {
398 if (F.getCallingConv() != CallingConv::AMDGPU_KERNEL &&
399 F.getCallingConv() != CallingConv::SPIR_KERNEL)
400 return 0;
401
402 uint64_t ExplicitArgBytes = getExplicitKernArgSize(F, MaxAlign);
403
404 unsigned ExplicitOffset = getExplicitKernelArgOffset();
405
406 uint64_t TotalSize = ExplicitOffset + ExplicitArgBytes;
407 unsigned ImplicitBytes = getImplicitArgNumBytes(F);
408 if (ImplicitBytes != 0) {
409 const Align Alignment = getAlignmentForImplicitArgPtr();
410 TotalSize = alignTo(Size: ExplicitArgBytes, A: Alignment) + ImplicitBytes;
411 MaxAlign = std::max(a: MaxAlign, b: Alignment);
412 }
413
414 // Being able to dereference past the end is useful for emitting scalar loads.
415 return alignTo(Value: TotalSize, Align: 4);
416}
417
418AMDGPUDwarfFlavour AMDGPUSubtarget::getAMDGPUDwarfFlavour() const {
419 return getWavefrontSize() == 32 ? AMDGPUDwarfFlavour::Wave32
420 : AMDGPUDwarfFlavour::Wave64;
421}
422
423const AMDGPUSubtarget &AMDGPUSubtarget::get(const MachineFunction &MF) {
424 if (MF.getTarget().getTargetTriple().isAMDGCN())
425 return static_cast<const AMDGPUSubtarget&>(MF.getSubtarget<GCNSubtarget>());
426 return static_cast<const AMDGPUSubtarget &>(MF.getSubtarget<R600Subtarget>());
427}
428
429const AMDGPUSubtarget &AMDGPUSubtarget::get(const TargetMachine &TM, const Function &F) {
430 if (TM.getTargetTriple().isAMDGCN())
431 return static_cast<const AMDGPUSubtarget&>(TM.getSubtarget<GCNSubtarget>(F));
432 return static_cast<const AMDGPUSubtarget &>(
433 TM.getSubtarget<R600Subtarget>(F));
434}
435