1//=====-- AMDGPUSubtarget.h - Define Subtarget for AMDGPU -------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//==-----------------------------------------------------------------------===//
8//
9/// \file
10/// Base class for AMDGPU specific classes of TargetSubtarget.
11//
12//===----------------------------------------------------------------------===//
13
14#ifndef LLVM_LIB_TARGET_AMDGPU_AMDGPUSUBTARGET_H
15#define LLVM_LIB_TARGET_AMDGPU_AMDGPUSUBTARGET_H
16
17#include "llvm/ADT/SmallVector.h"
18#include "llvm/IR/CallingConv.h"
19#include "llvm/Support/Alignment.h"
20#include "llvm/TargetParser/Triple.h"
21
22namespace llvm {
23
24enum AMDGPUDwarfFlavour : unsigned;
25class Function;
26class Instruction;
27class MachineFunction;
28class TargetMachine;
29
30class AMDGPUSubtarget {
31public:
32 enum Generation {
33 INVALID = 0,
34 R600 = 1,
35 R700 = 2,
36 EVERGREEN = 3,
37 NORTHERN_ISLANDS = 4,
38 SOUTHERN_ISLANDS = 5,
39 SEA_ISLANDS = 6,
40 VOLCANIC_ISLANDS = 7,
41 GFX9 = 8,
42 GFX10 = 9,
43 GFX11 = 10,
44 GFX12 = 11,
45 GFX13 = 12,
46 };
47
48private:
49 const Triple &TargetTriple;
50
51protected:
52 bool HasMulI24 = true;
53 bool HasMulU24 = true;
54 bool HasSMulHi = false;
55 bool HasFminFmaxLegacy = true;
56
57 unsigned NumWorkGroupSIMDs = 4;
58 // Set from TableGen subtarget features; R600Subtarget sets it directly.
59 unsigned MaxWavesPerEU = 0;
60 unsigned LocalMemorySize = 0;
61 unsigned AddressableLocalMemorySize = 0;
62 unsigned LDSAllocationGranularity = 0;
63 unsigned LDSEncodingGranularity = 0;
64 char WavefrontSizeLog2 = 0;
65 unsigned FlatOffsetBitWidth = 0;
66
67public:
68 AMDGPUSubtarget(const Triple &TT) : TargetTriple(TT) {}
69
70 static const AMDGPUSubtarget &get(const MachineFunction &MF);
71 static const AMDGPUSubtarget &get(const TargetMachine &TM,
72 const Function &F);
73
74 /// \returns Default range flat work group size for a calling convention.
75 std::pair<unsigned, unsigned> getDefaultFlatWorkGroupSize(CallingConv::ID CC) const;
76
77 /// \returns Subtarget's default pair of minimum/maximum flat work group sizes
78 /// for function \p F, or minimum/maximum flat work group sizes explicitly
79 /// requested using "amdgpu-flat-work-group-size" attribute attached to
80 /// function \p F.
81 ///
82 /// \returns Subtarget's default values if explicitly requested values cannot
83 /// be converted to integer, or violate subtarget's specifications.
84 std::pair<unsigned, unsigned> getFlatWorkGroupSizes(const Function &F) const;
85
86 /// \returns The required size of workgroups that will be used to execute \p F
87 /// in the \p Dim dimension, if it is known (from `!reqd_work_group_size`
88 /// metadata. Otherwise, returns std::nullopt.
89 std::optional<unsigned> getReqdWorkGroupSize(const Function &F,
90 unsigned Dim) const;
91
92 /// \returns true if \p F will execute in a manner that leaves the X
93 /// dimensions of the workitem ID evenly tiling wavefronts - that is, if X /
94 /// wavefrontsize is uniform. This is true if either the Y and Z block
95 /// dimensions are known to always be 1 or if the X dimension will always be a
96 /// power of 2. If \p RequireUniformYZ is true, it also ensures that the Y and
97 /// Z workitem IDs will be uniform (so, while a (32, 2, 1) launch with
98 /// wavesize64 would ordinarily pass this test, it won't with
99 /// \pRequiresUniformYZ).
100 ///
101 /// This information is currently only gathered from the !reqd_work_group_size
102 /// metadata on \p F, but this may be improved in the future.
103 bool hasWavefrontsEvenlySplittingXDim(const Function &F,
104 bool RequiresUniformYZ = false) const;
105
106 /// \returns Subtarget's default pair of minimum/maximum number of waves per
107 /// execution unit for function \p F, or minimum/maximum number of waves per
108 /// execution unit explicitly requested using "amdgpu-waves-per-eu" attribute
109 /// attached to function \p F.
110 ///
111 /// \returns Subtarget's default values if explicitly requested values cannot
112 /// be converted to integer, violate subtarget's specifications, or are not
113 /// compatible with minimum/maximum number of waves limited by flat work group
114 /// size, register usage, and/or lds usage.
115 std::pair<unsigned, unsigned> getWavesPerEU(const Function &F) const;
116
117 /// Overload which uses the specified values for the flat workgroup sizes and
118 /// LDS space rather than querying the function itself. \p FlatWorkGroupSizes
119 /// should correspond to the function's value for getFlatWorkGroupSizes and \p
120 /// LDSBytes to the per-workgroup LDS allocation.
121 std::pair<unsigned, unsigned>
122 getWavesPerEU(std::pair<unsigned, unsigned> FlatWorkGroupSizes,
123 unsigned LDSBytes, const Function &F) const;
124
125 /// Returns the target minimum/maximum number of waves per EU. This is based
126 /// on the minimum/maximum number of \p RequestedWavesPerEU and further
127 /// limited by the maximum achievable occupancy derived from the range of \p
128 /// FlatWorkGroupSizes and number of \p LDSBytes per workgroup.
129 std::pair<unsigned, unsigned>
130 getEffectiveWavesPerEU(std::pair<unsigned, unsigned> RequestedWavesPerEU,
131 std::pair<unsigned, unsigned> FlatWorkGroupSizes,
132 unsigned LDSBytes) const;
133
134 /// Return the amount of LDS that can be used that will not restrict the
135 /// occupancy lower than WaveCount.
136 unsigned getMaxLocalMemSizeWithWaveCount(unsigned WaveCount,
137 const Function &) const;
138
139 /// Subtarget's minimum/maximum occupancy, in number of waves per EU, that can
140 /// be achieved when the only function running on a CU is \p F and each
141 /// workgroup running the function requires \p LDSBytes bytes of LDS space.
142 /// This notably depends on the range of allowed flat group sizes for the
143 /// function and hardware characteristics.
144 std::pair<unsigned, unsigned>
145 getOccupancyWithWorkGroupSizes(uint32_t LDSBytes, const Function &F) const {
146 return getOccupancyWithWorkGroupSizes(LDSBytes, FlatWorkGroupSizes: getFlatWorkGroupSizes(F));
147 }
148
149 /// Overload which uses the specified values for the flat work group sizes,
150 /// rather than querying the function itself. \p FlatWorkGroupSizes should
151 /// correspond to the function's value for getFlatWorkGroupSizes.
152 std::pair<unsigned, unsigned> getOccupancyWithWorkGroupSizes(
153 uint32_t LDSBytes,
154 std::pair<unsigned, unsigned> FlatWorkGroupSizes) const;
155
156 /// Subtarget's minimum/maximum occupancy, in number of waves per EU, that can
157 /// be achieved when the only function running on a CU is \p MF. This notably
158 /// depends on the range of allowed flat group sizes for the function, the
159 /// amount of per-workgroup LDS space required by the function, and hardware
160 /// characteristics.
161 std::pair<unsigned, unsigned>
162 getOccupancyWithWorkGroupSizes(const MachineFunction &MF) const;
163
164 bool isAmdHsaOS() const {
165 return TargetTriple.getOS() == Triple::AMDHSA;
166 }
167
168 bool isAmdPalOS() const {
169 return TargetTriple.getOS() == Triple::AMDPAL;
170 }
171
172 bool isMesa3DOS() const {
173 return TargetTriple.getOS() == Triple::Mesa3D;
174 }
175
176 bool isMesaKernel(const Function &F) const;
177
178 bool isAmdHsaOrMesa(const Function &F) const {
179 return isAmdHsaOS() || isMesaKernel(F);
180 }
181
182 bool isGCN() const { return TargetTriple.isAMDGCN(); }
183
184 //==---------------------------------------------------------------------===//
185 // TableGen-generated feature getters.
186 //==---------------------------------------------------------------------===//
187#define GET_SUBTARGETINFO_MACRO(ATTRIBUTE, DEFAULT, GETTER) \
188 virtual bool GETTER() const { return false; }
189#include "AMDGPUGenSubtargetInfo.inc"
190 //==---------------------------------------------------------------------===//
191
192 /// Return true if real (non-fake) variants of True16 instructions using
193 /// 16-bit registers should be code-generated. Fake True16 instructions are
194 /// identical to non-fake ones except that they take 32-bit registers as
195 /// operands and always use their low halves.
196 // TODO: Remove and use hasTrue16BitInsts() instead once True16 is fully
197 // supported and the support for fake True16 instructions is removed.
198 bool useRealTrue16Insts() const {
199 return hasTrue16BitInsts() && enableRealTrue16Insts();
200 }
201
202 bool hasMulI24() const {
203 return HasMulI24;
204 }
205
206 bool hasMulU24() const {
207 return HasMulU24;
208 }
209
210 bool hasSMulHi() const {
211 return HasSMulHi;
212 }
213
214 bool hasFminFmaxLegacy() const {
215 return HasFminFmaxLegacy;
216 }
217
218 unsigned getWavefrontSize() const {
219 return 1 << WavefrontSizeLog2;
220 }
221
222 unsigned getWavefrontSizeLog2() const {
223 return WavefrontSizeLog2;
224 }
225
226 /// Return the maximum number of bytes of LDS available for all workgroups
227 /// running on the same WGP or CU.
228 /// For GFX10-GFX12 in WGP mode this is 128k even though each workgroup is
229 /// limited to 64k.
230 unsigned getLocalMemorySize() const {
231 return LocalMemorySize;
232 }
233
234 /// Return the maximum number of bytes of LDS that can be allocated to a
235 /// single workgroup.
236 /// For GFX10-GFX12 in WGP mode this is limited to 64k even though the WGP has
237 /// 128k in total.
238 unsigned getAddressableLocalMemorySize() const {
239 return AddressableLocalMemorySize;
240 }
241
242 /// \returns Number of SIMDs a work-group's waves run on: all of the block's
243 /// SIMDs in full-SIMD mode, half of them otherwise.
244 unsigned getNumWorkGroupSIMDs() const { return NumWorkGroupSIMDs; }
245
246 Align getAlignmentForImplicitArgPtr() const {
247 return isAmdHsaOS() ? Align(8) : Align(4);
248 }
249
250 /// Returns the offset in bytes from the start of the input buffer
251 /// of the first explicit kernel argument.
252 unsigned getExplicitKernelArgOffset() const {
253 switch (TargetTriple.getOS()) {
254 case Triple::AMDHSA:
255 case Triple::AMDPAL:
256 case Triple::Mesa3D:
257 return 0;
258 case Triple::UnknownOS:
259 default:
260 // For legacy reasons unknown/other is treated as a different version of
261 // mesa.
262 return 36;
263 }
264
265 llvm_unreachable("invalid triple OS");
266 }
267
268 /// \returns Maximum number of work groups per compute unit supported by the
269 /// subtarget and limited by given \p FlatWorkGroupSize.
270 virtual unsigned getMaxWorkGroupsPerCU(unsigned FlatWorkGroupSize) const = 0;
271
272 /// \returns Minimum flat work group size supported by the subtarget.
273 virtual unsigned getMinFlatWorkGroupSize() const = 0;
274
275 /// \returns Maximum flat work group size supported by the subtarget.
276 virtual unsigned getMaxFlatWorkGroupSize() const = 0;
277
278 /// \returns Number of waves per execution unit required to support the given
279 /// \p FlatWorkGroupSize.
280 virtual unsigned
281 getWavesPerEUForWorkGroup(unsigned FlatWorkGroupSize) const = 0;
282
283 /// \returns Minimum number of waves per execution unit supported by the
284 /// subtarget.
285 virtual unsigned getMinWavesPerEU() const = 0;
286
287 /// \returns Maximum number of waves per execution unit supported by the
288 /// subtarget without any kind of limitation.
289 unsigned getMaxWavesPerEU() const { return MaxWavesPerEU; }
290
291 /// Return the maximum workitem ID value in the function, for the given (0, 1,
292 /// 2) dimension.
293 unsigned getMaxWorkitemID(const Function &Kernel, unsigned Dimension) const;
294
295 /// Return true if only a single workitem can be active in a wave.
296 bool isSingleLaneExecution(const Function &Kernel) const;
297
298 /// Creates value range metadata on an workitemid.* intrinsic call or load.
299 bool makeLIDRangeMetadata(Instruction *I) const;
300
301 /// \returns Number of bytes of arguments that are passed to a shader or
302 /// kernel in addition to the explicit ones declared for the function.
303 unsigned getImplicitArgNumBytes(const Function &F) const;
304 uint64_t getExplicitKernArgSize(const Function &F, Align &MaxAlign) const;
305 unsigned getKernArgSegmentSize(const Function &F, Align &MaxAlign) const;
306
307 /// \returns Corresponding DWARF register number mapping flavour for the
308 /// \p WavefrontSize.
309 AMDGPUDwarfFlavour getAMDGPUDwarfFlavour() const;
310
311 virtual ~AMDGPUSubtarget() = default;
312};
313
314} // end namespace llvm
315
316#endif // LLVM_LIB_TARGET_AMDGPU_AMDGPUSUBTARGET_H
317