1//===-- GCNSubtarget.cpp - GCN Subtarget Information ----------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// Implements the GCN specific subclass of TargetSubtarget.
11//
12//===----------------------------------------------------------------------===//
13
14#include "GCNSubtarget.h"
15#include "AMDGPUCallLowering.h"
16#include "AMDGPUInstructionSelector.h"
17#include "AMDGPULegalizerInfo.h"
18#include "AMDGPURegisterBankInfo.h"
19#include "AMDGPUSelectionDAGInfo.h"
20#include "AMDGPUTargetMachine.h"
21#include "SIMachineFunctionInfo.h"
22#include "Utils/AMDGPUBaseInfo.h"
23#include "llvm/ADT/SmallString.h"
24#include "llvm/CodeGen/GlobalISel/InlineAsmLowering.h"
25#include "llvm/CodeGen/MachinePipeliner.h"
26#include "llvm/CodeGen/MachineScheduler.h"
27#include "llvm/CodeGen/TargetFrameLowering.h"
28#include "llvm/IR/DiagnosticInfo.h"
29#include "llvm/IR/MDBuilder.h"
30#include "llvm/TargetParser/AMDGPUTargetParser.h"
31#include <algorithm>
32
33using namespace llvm;
34
35#define DEBUG_TYPE "gcn-subtarget"
36
37#define GET_SUBTARGETINFO_TARGET_DESC
38#define GET_SUBTARGETINFO_CTOR
39#define AMDGPUSubtarget GCNSubtarget
40#include "AMDGPUGenSubtargetInfo.inc"
41#undef AMDGPUSubtarget
42
43static cl::opt<bool> EnableVGPRIndexMode(
44 "amdgpu-vgpr-index-mode",
45 cl::desc("Use GPR indexing mode instead of movrel for vector indexing"),
46 cl::init(Val: false));
47
48static cl::opt<bool> UseAA("amdgpu-use-aa-in-codegen",
49 cl::desc("Enable the use of AA during codegen."),
50 cl::init(Val: true));
51
52static cl::opt<unsigned>
53 NSAThreshold("amdgpu-nsa-threshold",
54 cl::desc("Number of addresses from which to enable MIMG NSA."),
55 cl::init(Val: 2), cl::Hidden);
56
57GCNSubtarget::~GCNSubtarget() = default;
58
59static AMDGPUSubtarget::Generation computeDefaultGeneration(const Triple &TT) {
60 // Legacy triples without a subarch default to the first target that supports
61 // flat addressing for HSA, otherwise the first amdgcn target.
62 if (TT.getSubArch() == Triple::NoSubArch)
63 return TT.getOS() == Triple::AMDHSA ? AMDGPUSubtarget::SEA_ISLANDS
64 : AMDGPUSubtarget::SOUTHERN_ISLANDS;
65
66 switch (AMDGPU::getMajorSubArch(SubArch: TT.getSubArch())) {
67 case Triple::AMDGPUSubArch6:
68 return AMDGPUSubtarget::SOUTHERN_ISLANDS;
69 case Triple::AMDGPUSubArch7:
70 return AMDGPUSubtarget::SEA_ISLANDS;
71 case Triple::AMDGPUSubArch8:
72 case Triple::AMDGPUSubArch810:
73 return AMDGPUSubtarget::VOLCANIC_ISLANDS;
74 case Triple::AMDGPUSubArch9:
75 case Triple::AMDGPUSubArch908:
76 case Triple::AMDGPUSubArch90A:
77 case Triple::AMDGPUSubArch9_4:
78 return AMDGPUSubtarget::GFX9;
79 case Triple::AMDGPUSubArch10_1:
80 case Triple::AMDGPUSubArch10_3:
81 return AMDGPUSubtarget::GFX10;
82 case Triple::AMDGPUSubArch11:
83 case Triple::AMDGPUSubArch11_7:
84 return AMDGPUSubtarget::GFX11;
85 case Triple::AMDGPUSubArch12:
86 case Triple::AMDGPUSubArch12_5:
87 case Triple::AMDGPUSubArch1250S:
88 return AMDGPUSubtarget::GFX12;
89 case Triple::AMDGPUSubArch13:
90 return AMDGPUSubtarget::GFX13;
91 default:
92 reportFatalUsageError(reason: "invalid subarch for amdgpu");
93 }
94}
95
96GCNSubtarget &GCNSubtarget::initializeSubtargetDependencies(
97 const Triple &TT, StringRef GPU, StringRef FS,
98 AMDGPU::TargetIDSetting XnackSetting,
99 AMDGPU::TargetIDSetting SramEccSetting) {
100 // Determine default and user-specified characteristics
101 //
102 // We want to be able to turn these off, but making this a subtarget feature
103 // for SI has the unhelpful behavior that it unsets everything else if you
104 // disable it.
105 //
106 // Similarly we want enable-prt-strict-null to be on by default and not to
107 // unset everything else if it is disabled
108
109 SmallString<256> FullFS("+load-store-opt,+enable-ds128,");
110
111 // Turn on features that HSA ABI requires. Also turn on FlatForGlobal by
112 // default
113 if (isAmdHsaOS())
114 FullFS += "+flat-for-global,+unaligned-access-mode,+trap-handler,";
115
116 FullFS += "+enable-prt-strict-null,"; // This is overridden by a disable in FS
117
118 // Disable mutually exclusive bits.
119 if (FS.contains_insensitive(Other: "+wavefrontsize")) {
120 if (!FS.contains_insensitive(Other: "wavefrontsize16"))
121 FullFS += "-wavefrontsize16,";
122 if (!FS.contains_insensitive(Other: "wavefrontsize32"))
123 FullFS += "-wavefrontsize32,";
124 if (!FS.contains_insensitive(Other: "wavefrontsize64"))
125 FullFS += "-wavefrontsize64,";
126 }
127
128 FullFS += FS;
129
130 ParseSubtargetFeatures(CPU: GPU, /*TuneCPU*/ GPU, FS: FullFS);
131
132 // Implement the "generic" processors, which acts as the default when no
133 // generation features are enabled (e.g for -mcpu=''). HSA OS defaults to
134 // the first amdgcn target that supports flat addressing. Other OSes defaults
135 // to the first amdgcn target.
136 if (Gen == AMDGPUSubtarget::INVALID) {
137 Gen = computeDefaultGeneration(TT);
138 // Assume wave64 for the unknown target, if not explicitly set.
139 if (getWavefrontSizeLog2() == 0)
140 WavefrontSizeLog2 = 6;
141 } else if (!hasFeature(Feature: AMDGPU::FeatureWavefrontSize32) &&
142 !hasFeature(Feature: AMDGPU::FeatureWavefrontSize64)) {
143 // If there is no default wave size it must be a generation before gfx10,
144 // these have FeatureWavefrontSize64 in their definition already. For gfx10+
145 // set wave32 as a default.
146 ToggleFeature(FB: AMDGPU::FeatureWavefrontSize32);
147 WavefrontSizeLog2 = getGeneration() >= AMDGPUSubtarget::GFX10 ? 5 : 6;
148 }
149
150 // We don't support FP64 for EG/NI atm.
151 assert(!hasFP64() || (getGeneration() >= AMDGPUSubtarget::SOUTHERN_ISLANDS));
152
153 // Targets must either support 64-bit offsets for MUBUF instructions, and/or
154 // support flat operations, otherwise they cannot access a 64-bit global
155 // address space
156 assert(hasAddr64() || hasFlat());
157 // Unless +-flat-for-global is specified, turn on FlatForGlobal for targets
158 // that do not support ADDR64 variants of MUBUF instructions. Such targets
159 // cannot use a 64 bit offset with a MUBUF instruction to access the global
160 // address space
161 if (!hasAddr64() && !FS.contains(Other: "flat-for-global") && !UseFlatForGlobal) {
162 ToggleFeature(FB: AMDGPU::FeatureUseFlatForGlobal);
163 UseFlatForGlobal = true;
164 }
165 // Unless +-flat-for-global is specified, use MUBUF instructions for global
166 // address space access if flat operations are not available.
167 if (!hasFlat() && !FS.contains(Other: "flat-for-global") && UseFlatForGlobal) {
168 ToggleFeature(FB: AMDGPU::FeatureUseFlatForGlobal);
169 UseFlatForGlobal = false;
170 }
171
172 // Set defaults if needed.
173 if (MaxPrivateElementSize == 0)
174 MaxPrivateElementSize = 4;
175
176 if (LDSBankCount == 0)
177 LDSBankCount = 32;
178
179 if (MaxWavesPerEU == 0)
180 MaxWavesPerEU = 10;
181
182 if (FlatOffsetBitWidth == 0)
183 FlatOffsetBitWidth = 13;
184
185 LocalMemorySize =
186 AMDGPU::getLocalMemorySize(AK: getTargetID().getGPUKind(), FullSIMDMode: isFullSIMDMode());
187 AddressableLocalMemorySize = AMDGPU::getAddressableLocalMemorySize(
188 AK: getTargetID().getGPUKind(), FullSIMDMode: isFullSIMDMode());
189 // LDS allocation granularity is in bytes.
190 LDSAllocationGranularity =
191 AMDGPU::getLDSAllocGranule(AK: getTargetID().getGPUKind());
192
193 HasFminFmaxLegacy = getGeneration() < AMDGPUSubtarget::VOLCANIC_ISLANDS;
194 HasSMulHi = getGeneration() >= AMDGPUSubtarget::GFX9;
195
196 // InstCacheLineSize is set from TableGen subtarget features
197 // (FeatureInstCacheLineSize64 / FeatureInstCacheLineSize128).
198 // Fall back to 64 if no feature was specified (e.g. generic targets).
199 if (InstCacheLineSize == 0)
200 InstCacheLineSize = 64;
201
202 assert(llvm::isPowerOf2_32(InstCacheLineSize) &&
203 "InstCacheLineSize must be a power of 2");
204
205 // Apply the module flag's xnack setting if the target supports on/off modes.
206 // Targets without on/off mode support have xnack always on and ignore module
207 // flags.
208 if (hasXNACKOnOffModes())
209 TargetID.setXnackSetting(XnackSetting);
210
211 // Apply the module flag's sramecc setting if the target supports on/off
212 // modes. Targets with sramecc hardwired on ignore module flags.
213 if (hasSRAMECCOnOffModes())
214 TargetID.setSramEccSetting(SramEccSetting);
215
216 return *this;
217}
218
219void GCNSubtarget::checkSubtargetFeatures(const Function &F) const {
220 LLVMContext &Ctx = F.getContext();
221 if (hasFeature(Feature: AMDGPU::FeatureWavefrontSize32) &&
222 hasFeature(Feature: AMDGPU::FeatureWavefrontSize64)) {
223 Ctx.diagnose(DI: DiagnosticInfoUnsupported(
224 F, "must specify exactly one of wavefrontsize32 and wavefrontsize64"));
225 }
226}
227
228// TODO: Validate subarch for subtarget
229
230GCNSubtarget::GCNSubtarget(const Triple &TT, StringRef GPU, StringRef FS,
231 const GCNTargetMachine &TM, bool BufferOOBRelaxed,
232 bool TBufferOOBRelaxed,
233 AMDGPU::TargetIDSetting XnackSetting,
234 AMDGPU::TargetIDSetting SramEccSetting)
235 : // clang-format off
236 AMDGPUGenSubtargetInfo(TT, GPU, /*TuneCPU*/ GPU, FS),
237 AMDGPUSubtarget(TT),
238 TargetID(AMDGPU::createAMDGPUTargetID(STI: *this, FeatureString: "")),
239 InstrItins(getInstrItineraryForCPU(CPU: GPU)),
240 BufferOOBRelaxed(BufferOOBRelaxed),
241 TBufferOOBRelaxed(TBufferOOBRelaxed),
242 InstrInfo(initializeSubtargetDependencies(TT, GPU, FS, XnackSetting,
243 SramEccSetting)),
244 TLInfo(TM, *this),
245 // Frame index expansion sometimes assumes the low bit of SP is 0
246 FrameLowering(TargetFrameLowering::StackGrowsUp, getStackAlignment(), 0,
247 /*TransAl=*/Align(4)) {
248
249 // clang-format on
250
251 LLVM_DEBUG(dbgs() << "xnack setting for subtarget: "
252 << TargetID.getXnackSetting() << '\n');
253 LLVM_DEBUG(dbgs() << "sramecc setting for subtarget: "
254 << TargetID.getSramEccSetting() << '\n');
255
256 NumWorkGroupSIMDs = AMDGPU::getNumWorkGroupSIMDs(FullSIMDMode: isFullSIMDMode());
257
258 TSInfo = std::make_unique<AMDGPUSelectionDAGInfo>();
259
260 CallLoweringInfo = std::make_unique<AMDGPUCallLowering>(args: *getTargetLowering());
261 InlineAsmLoweringInfo =
262 std::make_unique<InlineAsmLowering>(args: getTargetLowering());
263 Legalizer = std::make_unique<AMDGPULegalizerInfo>(args&: *this, args: TM);
264 RegBankInfo = std::make_unique<AMDGPURegisterBankInfo>(args&: *this);
265 InstSelector =
266 std::make_unique<AMDGPUInstructionSelector>(args&: *this, args&: *RegBankInfo);
267}
268
269const SelectionDAGTargetInfo *GCNSubtarget::getSelectionDAGInfo() const {
270 return TSInfo.get();
271}
272
273unsigned GCNSubtarget::getConstantBusLimit(unsigned Opcode) const {
274 if (getGeneration() < GFX10)
275 return 1;
276
277 switch (Opcode) {
278 case AMDGPU::V_LSHLREV_B64_e64:
279 case AMDGPU::V_LSHLREV_B64_gfx10:
280 case AMDGPU::V_LSHLREV_B64_e64_gfx11:
281 case AMDGPU::V_LSHLREV_B64_e32_gfx12:
282 case AMDGPU::V_LSHLREV_B64_e64_gfx12:
283 case AMDGPU::V_LSHL_B64_e64:
284 case AMDGPU::V_LSHRREV_B64_e64:
285 case AMDGPU::V_LSHRREV_B64_gfx10:
286 case AMDGPU::V_LSHRREV_B64_e64_gfx11:
287 case AMDGPU::V_LSHRREV_B64_e64_gfx12:
288 case AMDGPU::V_LSHR_B64_e64:
289 case AMDGPU::V_ASHRREV_I64_e64:
290 case AMDGPU::V_ASHRREV_I64_gfx10:
291 case AMDGPU::V_ASHRREV_I64_e64_gfx11:
292 case AMDGPU::V_ASHRREV_I64_e64_gfx12:
293 case AMDGPU::V_ASHR_I64_e64:
294 return 1;
295 }
296
297 return 2;
298}
299
300/// This list was mostly derived from experimentation.
301bool GCNSubtarget::zeroesHigh16BitsOfDest(unsigned Opcode) const {
302 switch (Opcode) {
303 case AMDGPU::V_CVT_F16_F32_e32:
304 case AMDGPU::V_CVT_F16_F32_e64:
305 case AMDGPU::V_CVT_F16_U16_e32:
306 case AMDGPU::V_CVT_F16_U16_e64:
307 case AMDGPU::V_CVT_F16_I16_e32:
308 case AMDGPU::V_CVT_F16_I16_e64:
309 case AMDGPU::V_RCP_F16_e64:
310 case AMDGPU::V_RCP_F16_e32:
311 case AMDGPU::V_RSQ_F16_e64:
312 case AMDGPU::V_RSQ_F16_e32:
313 case AMDGPU::V_SQRT_F16_e64:
314 case AMDGPU::V_SQRT_F16_e32:
315 case AMDGPU::V_LOG_F16_e64:
316 case AMDGPU::V_LOG_F16_e32:
317 case AMDGPU::V_EXP_F16_e64:
318 case AMDGPU::V_EXP_F16_e32:
319 case AMDGPU::V_SIN_F16_e64:
320 case AMDGPU::V_SIN_F16_e32:
321 case AMDGPU::V_COS_F16_e64:
322 case AMDGPU::V_COS_F16_e32:
323 case AMDGPU::V_FLOOR_F16_e64:
324 case AMDGPU::V_FLOOR_F16_e32:
325 case AMDGPU::V_CEIL_F16_e64:
326 case AMDGPU::V_CEIL_F16_e32:
327 case AMDGPU::V_TRUNC_F16_e64:
328 case AMDGPU::V_TRUNC_F16_e32:
329 case AMDGPU::V_RNDNE_F16_e64:
330 case AMDGPU::V_RNDNE_F16_e32:
331 case AMDGPU::V_FRACT_F16_e64:
332 case AMDGPU::V_FRACT_F16_e32:
333 case AMDGPU::V_FREXP_MANT_F16_e64:
334 case AMDGPU::V_FREXP_MANT_F16_e32:
335 case AMDGPU::V_FREXP_EXP_I16_F16_e64:
336 case AMDGPU::V_FREXP_EXP_I16_F16_e32:
337 case AMDGPU::V_LDEXP_F16_e64:
338 case AMDGPU::V_LDEXP_F16_e32:
339 case AMDGPU::V_LSHLREV_B16_e64:
340 case AMDGPU::V_LSHLREV_B16_e32:
341 case AMDGPU::V_LSHRREV_B16_e64:
342 case AMDGPU::V_LSHRREV_B16_e32:
343 case AMDGPU::V_ASHRREV_I16_e64:
344 case AMDGPU::V_ASHRREV_I16_e32:
345 case AMDGPU::V_ADD_U16_e64:
346 case AMDGPU::V_ADD_U16_e32:
347 case AMDGPU::V_SUB_U16_e64:
348 case AMDGPU::V_SUB_U16_e32:
349 case AMDGPU::V_SUBREV_U16_e64:
350 case AMDGPU::V_SUBREV_U16_e32:
351 case AMDGPU::V_MUL_LO_U16_e64:
352 case AMDGPU::V_MUL_LO_U16_e32:
353 case AMDGPU::V_ADD_F16_e64:
354 case AMDGPU::V_ADD_F16_e32:
355 case AMDGPU::V_SUB_F16_e64:
356 case AMDGPU::V_SUB_F16_e32:
357 case AMDGPU::V_SUBREV_F16_e64:
358 case AMDGPU::V_SUBREV_F16_e32:
359 case AMDGPU::V_MUL_F16_e64:
360 case AMDGPU::V_MUL_F16_e32:
361 case AMDGPU::V_MAX_F16_e64:
362 case AMDGPU::V_MAX_F16_e32:
363 case AMDGPU::V_MIN_F16_e64:
364 case AMDGPU::V_MIN_F16_e32:
365 case AMDGPU::V_MAX_U16_e64:
366 case AMDGPU::V_MAX_U16_e32:
367 case AMDGPU::V_MIN_U16_e64:
368 case AMDGPU::V_MIN_U16_e32:
369 case AMDGPU::V_MAX_I16_e64:
370 case AMDGPU::V_MAX_I16_e32:
371 case AMDGPU::V_MIN_I16_e64:
372 case AMDGPU::V_MIN_I16_e32:
373 case AMDGPU::V_MAD_F16_e64:
374 case AMDGPU::V_MAD_U16_e64:
375 case AMDGPU::V_MAD_I16_e64:
376 case AMDGPU::V_FMA_F16_e64:
377 case AMDGPU::V_DIV_FIXUP_F16_e64:
378 // On gfx10, all 16-bit instructions preserve the high bits.
379 return getGeneration() <= AMDGPUSubtarget::GFX9;
380 case AMDGPU::V_MADAK_F16:
381 case AMDGPU::V_MADMK_F16:
382 case AMDGPU::V_MAC_F16_e64:
383 case AMDGPU::V_MAC_F16_e32:
384 case AMDGPU::V_FMAMK_F16:
385 case AMDGPU::V_FMAAK_F16:
386 case AMDGPU::V_FMAC_F16_e64:
387 case AMDGPU::V_FMAC_F16_e32:
388 // In gfx9, the preferred handling of the unused high 16-bits changed. Most
389 // instructions maintain the legacy behavior of 0ing. Some instructions
390 // changed to preserving the high bits.
391 return getGeneration() == AMDGPUSubtarget::VOLCANIC_ISLANDS;
392 case AMDGPU::V_MAD_MIXLO_F16:
393 case AMDGPU::V_MAD_MIXHI_F16:
394 default:
395 return false;
396 }
397}
398
399void GCNSubtarget::overrideSchedPolicy(MachineSchedPolicy &Policy,
400 const SchedRegion &Region) const {
401 // Track register pressure so the scheduler can try to decrease
402 // pressure once register usage is above the threshold defined by
403 // SIRegisterInfo::getRegPressureSetLimit()
404 Policy.ShouldTrackPressure = true;
405
406 const Function &F = Region.RegionBegin->getMF()->getFunction();
407 if (AMDGPU::getSchedStrategy(F) == "coexec") {
408 Policy.OnlyTopDown = true;
409 Policy.OnlyBottomUp = false;
410 return;
411 }
412
413 // Enabling both top down and bottom up scheduling seems to give us less
414 // register spills than just using one of these approaches on its own.
415 Policy.OnlyTopDown = false;
416 Policy.OnlyBottomUp = false;
417
418 // Enabling ShouldTrackLaneMasks crashes the SI Machine Scheduler.
419 if (!enableSIScheduler())
420 Policy.ShouldTrackLaneMasks = true;
421}
422
423void GCNSubtarget::overridePostRASchedPolicy(MachineSchedPolicy &Policy,
424 const SchedRegion &Region) const {
425 const Function &F = Region.RegionBegin->getMF()->getFunction();
426 Attribute PostRADirectionAttr = F.getFnAttribute(Kind: "amdgpu-post-ra-direction");
427 if (!PostRADirectionAttr.isValid())
428 return;
429
430 StringRef PostRADirectionStr = PostRADirectionAttr.getValueAsString();
431 if (PostRADirectionStr == "topdown") {
432 Policy.OnlyTopDown = true;
433 Policy.OnlyBottomUp = false;
434 } else if (PostRADirectionStr == "bottomup") {
435 Policy.OnlyTopDown = false;
436 Policy.OnlyBottomUp = true;
437 } else if (PostRADirectionStr == "bidirectional") {
438 Policy.OnlyTopDown = false;
439 Policy.OnlyBottomUp = false;
440 } else {
441 DiagnosticInfoOptimizationFailure Diag(
442 F, F.getSubprogram(), "invalid value for postRA direction attribute");
443 F.getContext().diagnose(DI: Diag);
444 }
445
446 LLVM_DEBUG({
447 const char *DirStr = "default";
448 if (Policy.OnlyTopDown && !Policy.OnlyBottomUp)
449 DirStr = "topdown";
450 else if (!Policy.OnlyTopDown && Policy.OnlyBottomUp)
451 DirStr = "bottomup";
452 else if (!Policy.OnlyTopDown && !Policy.OnlyBottomUp)
453 DirStr = "bidirectional";
454
455 dbgs() << "Post-MI-sched direction (" << F.getName() << "): " << DirStr
456 << '\n';
457 });
458}
459
460void GCNSubtarget::overridePipelinerPolicy(
461 MachinePipelinerPolicy &Policy) const {
462 Policy.ShouldLimitRegPressure = true;
463}
464
465void GCNSubtarget::mirFileLoaded(MachineFunction &MF) const {
466 if (isWave32()) {
467 // Fix implicit $vcc operands after MIParser has verified that they match
468 // the instruction definitions.
469 for (auto &MBB : MF) {
470 for (auto &MI : MBB)
471 InstrInfo.fixImplicitOperands(MI);
472 }
473 }
474}
475
476bool GCNSubtarget::hasMadF16() const {
477 return InstrInfo.pseudoToMCOpcode(Opcode: AMDGPU::V_MAD_F16_e64) != -1;
478}
479
480bool GCNSubtarget::useVGPRIndexMode() const {
481 return hasVGPRIndexMode() && (!hasMovrel() || EnableVGPRIndexMode);
482}
483
484bool GCNSubtarget::useAA() const { return UseAA; }
485
486unsigned GCNSubtarget::getOccupancyWithNumSGPRs(unsigned SGPRs) const {
487 return AMDGPU::IsaInfo::getOccupancyWithNumSGPRs(STI: *this, SGPRs);
488}
489
490unsigned
491GCNSubtarget::getOccupancyWithNumVGPRs(unsigned NumVGPRs,
492 unsigned DynamicVGPRBlockSize) const {
493 return AMDGPU::IsaInfo::getNumWavesPerEUWithNumVGPRs(STI: *this, NumVGPRs,
494 DynamicVGPRBlockSize);
495}
496
497unsigned
498GCNSubtarget::getBaseReservedNumSGPRs(const bool HasFlatScratch) const {
499 if (getGeneration() >= AMDGPUSubtarget::GFX10)
500 return 2; // VCC. FLAT_SCRATCH and XNACK are no longer in SGPRs.
501
502 if (HasFlatScratch || HasArchitectedFlatScratch) {
503 if (getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS)
504 return 6; // FLAT_SCRATCH, XNACK, VCC (in that order).
505 if (getGeneration() == AMDGPUSubtarget::SEA_ISLANDS)
506 return 4; // FLAT_SCRATCH, VCC (in that order).
507 }
508
509 if (isXNACKEnabled())
510 return 4; // XNACK, VCC (in that order).
511 return 2; // VCC.
512}
513
514unsigned GCNSubtarget::getReservedNumSGPRs(const MachineFunction &MF) const {
515 const SIMachineFunctionInfo &MFI = *MF.getInfo<SIMachineFunctionInfo>();
516 return getBaseReservedNumSGPRs(HasFlatScratch: MFI.getUserSGPRInfo().hasFlatScratchInit());
517}
518
519unsigned GCNSubtarget::getReservedNumSGPRs(const Function &F) const {
520 // In principle we do not need to reserve SGPR pair used for flat_scratch if
521 // we know flat instructions do not access the stack anywhere in the
522 // program. For now assume it's needed if we have flat instructions.
523 const bool KernelUsesFlatScratch = hasFlatAddressSpace();
524 return getBaseReservedNumSGPRs(HasFlatScratch: KernelUsesFlatScratch);
525}
526
527std::pair<unsigned, unsigned>
528GCNSubtarget::computeOccupancy(const Function &F, unsigned LDSSize,
529 unsigned NumSGPRs, unsigned NumVGPRs) const {
530 unsigned DynamicVGPRBlockSize = AMDGPU::getDynamicVGPRBlockSize(F);
531 auto [MinOcc, MaxOcc] = getOccupancyWithWorkGroupSizes(LDSBytes: LDSSize, F);
532 unsigned SGPROcc = getOccupancyWithNumSGPRs(SGPRs: NumSGPRs);
533 unsigned VGPROcc = getOccupancyWithNumVGPRs(NumVGPRs, DynamicVGPRBlockSize);
534
535 // Maximum occupancy may be further limited by high SGPR/VGPR usage.
536 MaxOcc = std::min(l: {MaxOcc, SGPROcc, VGPROcc});
537 return {std::min(a: MinOcc, b: MaxOcc), MaxOcc};
538}
539
540unsigned GCNSubtarget::getBaseMaxNumSGPRs(
541 const Function &F, std::pair<unsigned, unsigned> WavesPerEU,
542 unsigned PreloadedSGPRs, unsigned ReservedNumSGPRs) const {
543 // Compute maximum number of SGPRs function can use using default/requested
544 // minimum number of waves per execution unit.
545 unsigned MaxNumSGPRs = getMaxNumSGPRs(WavesPerEU: WavesPerEU.first, Addressable: false);
546 unsigned MaxAddressableNumSGPRs = getMaxNumSGPRs(WavesPerEU: WavesPerEU.first, Addressable: true);
547
548 // Check if maximum number of SGPRs was explicitly requested using
549 // "amdgpu-num-sgpr" attribute.
550 unsigned Requested =
551 F.getFnAttributeAsParsedInteger(Kind: "amdgpu-num-sgpr", Default: MaxNumSGPRs);
552
553 if (Requested != MaxNumSGPRs) {
554 // Make sure requested value does not violate subtarget's specifications.
555 if (Requested && (Requested <= ReservedNumSGPRs))
556 Requested = 0;
557
558 // If more SGPRs are required to support the input user/system SGPRs,
559 // increase to accommodate them.
560 //
561 // FIXME: This really ends up using the requested number of SGPRs + number
562 // of reserved special registers in total. Theoretically you could re-use
563 // the last input registers for these special registers, but this would
564 // require a lot of complexity to deal with the weird aliasing.
565 unsigned InputNumSGPRs = PreloadedSGPRs;
566 if (Requested && Requested < InputNumSGPRs)
567 Requested = InputNumSGPRs;
568
569 // Make sure requested value is compatible with values implied by
570 // default/requested minimum/maximum number of waves per execution unit.
571 if (Requested && Requested > getMaxNumSGPRs(WavesPerEU: WavesPerEU.first, Addressable: false))
572 Requested = 0;
573 if (WavesPerEU.second && Requested &&
574 Requested < getMinNumSGPRs(WavesPerEU: WavesPerEU.second))
575 Requested = 0;
576
577 if (Requested)
578 MaxNumSGPRs = Requested;
579 }
580
581 if (hasSGPRInitBug())
582 MaxNumSGPRs = AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG;
583
584 return std::min(a: MaxNumSGPRs - ReservedNumSGPRs, b: MaxAddressableNumSGPRs);
585}
586
587unsigned GCNSubtarget::getMaxNumSGPRs(const MachineFunction &MF) const {
588 const Function &F = MF.getFunction();
589 const SIMachineFunctionInfo &MFI = *MF.getInfo<SIMachineFunctionInfo>();
590 return getBaseMaxNumSGPRs(F, WavesPerEU: MFI.getWavesPerEU(), PreloadedSGPRs: MFI.getNumPreloadedSGPRs(),
591 ReservedNumSGPRs: getReservedNumSGPRs(MF));
592}
593
594unsigned GCNSubtarget::getMaxNumPreloadedSGPRs() const {
595 using USI = GCNUserSGPRUsageInfo;
596 // Max number of user SGPRs
597 const unsigned MaxUserSGPRs =
598 USI::getNumUserSGPRForField(ID: USI::PrivateSegmentBufferID) +
599 USI::getNumUserSGPRForField(ID: USI::DispatchPtrID) +
600 USI::getNumUserSGPRForField(ID: USI::QueuePtrID) +
601 USI::getNumUserSGPRForField(ID: USI::KernargSegmentPtrID) +
602 USI::getNumUserSGPRForField(ID: USI::DispatchIdID) +
603 USI::getNumUserSGPRForField(ID: USI::FlatScratchInitID) +
604 USI::getNumUserSGPRForField(ID: USI::ImplicitBufferPtrID);
605
606 // Max number of system SGPRs
607 const unsigned MaxSystemSGPRs = 1 + // WorkGroupIDX
608 1 + // WorkGroupIDY
609 1 + // WorkGroupIDZ
610 1 + // WorkGroupInfo
611 1; // private segment wave byte offset
612
613 // Max number of synthetic SGPRs
614 const unsigned SyntheticSGPRs = 1; // LDSKernelId
615
616 return MaxUserSGPRs + MaxSystemSGPRs + SyntheticSGPRs;
617}
618
619unsigned GCNSubtarget::getMaxNumSGPRs(const Function &F) const {
620 return getBaseMaxNumSGPRs(F, WavesPerEU: getWavesPerEU(F), PreloadedSGPRs: getMaxNumPreloadedSGPRs(),
621 ReservedNumSGPRs: getReservedNumSGPRs(F));
622}
623
624unsigned GCNSubtarget::getBaseMaxNumVGPRs(
625 const Function &F, std::pair<unsigned, unsigned> NumVGPRBounds) const {
626 const auto [Min, Max] = NumVGPRBounds;
627
628 // Check if maximum number of VGPRs was explicitly requested using
629 // "amdgpu-num-vgpr" attribute.
630
631 unsigned Requested = F.getFnAttributeAsParsedInteger(Kind: "amdgpu-num-vgpr", Default: Max);
632 if (Requested != Max && hasGFX90AInsts())
633 Requested *= 2;
634
635 // Make sure requested value is inside the range of possible VGPR usage.
636 return std::clamp(val: Requested, lo: Min, hi: Max);
637}
638
639unsigned GCNSubtarget::getMaxNumVGPRs(const Function &F) const {
640 unsigned DynamicVGPRBlockSize = AMDGPU::getDynamicVGPRBlockSize(F);
641 std::pair<unsigned, unsigned> Waves = getWavesPerEU(F);
642
643 unsigned MaxNumVGPRs = getBaseMaxNumVGPRs(
644 F, NumVGPRBounds: {getMinNumVGPRs(WavesPerEU: Waves.second, DynamicVGPRBlockSize),
645 getMaxNumVGPRs(WavesPerEU: Waves.first, DynamicVGPRBlockSize)});
646
647 // In DVGPR mode, a wave launches with a single VGPR block allocated. Applied
648 // after getBaseMaxNumVGPRs so "amdgpu-num-vgpr" cannot raise it back up.
649 if (DynamicVGPRBlockSize != 0 &&
650 AMDGPU::isEntryFunctionCC(CC: F.getCallingConv()))
651 MaxNumVGPRs = std::min(a: MaxNumVGPRs, b: DynamicVGPRBlockSize);
652
653 return MaxNumVGPRs;
654}
655
656unsigned GCNSubtarget::getMaxNumVGPRs(const MachineFunction &MF) const {
657 return getMaxNumVGPRs(F: MF.getFunction());
658}
659
660std::pair<unsigned, unsigned>
661GCNSubtarget::getMaxNumVectorRegs(const Function &F) const {
662 const unsigned MaxVectorRegs = getMaxNumVGPRs(F);
663
664 unsigned MaxNumVGPRs = MaxVectorRegs;
665 unsigned MaxNumAGPRs = 0;
666 unsigned NumArchVGPRs = getAddressableNumArchVGPRs();
667
668 // On GFX90A, the number of VGPRs and AGPRs need not be equal. Theoretically,
669 // a wave may have up to 512 total vector registers combining together both
670 // VGPRs and AGPRs. Hence, in an entry function without calls and without
671 // AGPRs used within it, it is possible to use the whole vector register
672 // budget for VGPRs.
673 //
674 // TODO: it shall be possible to estimate maximum AGPR/VGPR pressure and split
675 // register file accordingly.
676 if (hasGFX90AInsts()) {
677 unsigned MinNumAGPRs = 0;
678 const unsigned TotalNumAGPRs = AMDGPU::AGPR_32RegClass.getNumRegs();
679
680 const std::pair<unsigned, unsigned> DefaultNumAGPR = {~0u, ~0u};
681
682 // TODO: The lower bound should probably force the number of required
683 // registers up, overriding amdgpu-waves-per-eu.
684 std::tie(args&: MinNumAGPRs, args&: MaxNumAGPRs) =
685 AMDGPU::getIntegerPairAttribute(F, Name: "amdgpu-agpr-alloc", Default: DefaultNumAGPR,
686 /*OnlyFirstRequired=*/true);
687
688 if (MinNumAGPRs == DefaultNumAGPR.first) {
689 // Default to splitting half the registers if AGPRs are required.
690 MinNumAGPRs = MaxNumAGPRs = MaxVectorRegs / 2;
691 } else {
692 // Align to accum_offset's allocation granularity.
693 MinNumAGPRs = alignTo(Value: MinNumAGPRs, Align: 4);
694
695 MinNumAGPRs = std::min(a: MinNumAGPRs, b: TotalNumAGPRs);
696 }
697
698 // Clamp values to be inbounds of our limits, and ensure min <= max.
699
700 MaxNumAGPRs = std::min(a: std::max(a: MinNumAGPRs, b: MaxNumAGPRs), b: MaxVectorRegs);
701 MinNumAGPRs = std::min(l: {MinNumAGPRs, TotalNumAGPRs, MaxNumAGPRs});
702
703 MaxNumVGPRs = std::min(a: MaxVectorRegs - MinNumAGPRs, b: NumArchVGPRs);
704 MaxNumAGPRs = std::min(a: MaxVectorRegs - MaxNumVGPRs, b: MaxNumAGPRs);
705
706 assert(MaxNumVGPRs + MaxNumAGPRs <= MaxVectorRegs &&
707 MaxNumAGPRs <= TotalNumAGPRs && MaxNumVGPRs <= NumArchVGPRs &&
708 "invalid register counts");
709 } else if (hasMAIInsts()) {
710 // On gfx908 the number of AGPRs always equals the number of VGPRs.
711 MaxNumAGPRs = MaxNumVGPRs = MaxVectorRegs;
712 }
713
714 return std::pair(MaxNumVGPRs, MaxNumAGPRs);
715}
716
717// Check to which source operand UseOpIdx points to and return a pointer to the
718// operand of the corresponding source modifier.
719// Return nullptr if UseOpIdx either doesn't point to src0/1/2 or if there is no
720// operand for the corresponding source modifier.
721static const MachineOperand *
722getVOP3PSourceModifierFromOpIdx(const MachineInstr &UseI, int UseOpIdx,
723 const SIInstrInfo &InstrInfo) {
724 AMDGPU::OpName UseName =
725 AMDGPU::getOperandIdxName(Opcode: UseI.getOpcode(), Idx: UseOpIdx);
726 switch (UseName) {
727 case AMDGPU::OpName::src0:
728 return InstrInfo.getNamedOperand(MI: UseI, OperandName: AMDGPU::OpName::src0_modifiers);
729 case AMDGPU::OpName::src1:
730 return InstrInfo.getNamedOperand(MI: UseI, OperandName: AMDGPU::OpName::src1_modifiers);
731 case AMDGPU::OpName::src2:
732 return InstrInfo.getNamedOperand(MI: UseI, OperandName: AMDGPU::OpName::src2_modifiers);
733 default:
734 return nullptr;
735 }
736}
737
738// Get the subreg idx of the subreg that is used by the given instruction
739// operand, considering the given op_sel modifier.
740// Return 0 if the whole register is used or as a conservative fallback.
741static unsigned getEffectiveSubRegIdx(const SIRegisterInfo &TRI,
742 const SIInstrInfo &InstrInfo,
743 const MachineInstr &I,
744 const MachineOperand &Op) {
745 if (!InstrInfo.isVOP3P(MI: I) || InstrInfo.isWMMA(MI: I) || InstrInfo.isSWMMAC(MI: I))
746 return AMDGPU::NoSubRegister;
747
748 const MachineOperand *OpMod =
749 getVOP3PSourceModifierFromOpIdx(UseI: I, UseOpIdx: Op.getOperandNo(), InstrInfo);
750 if (!OpMod)
751 return AMDGPU::NoSubRegister;
752
753 // Note: the FMA_MIX* and MAD_MIX* instructions have different semantics for
754 // the op_sel and op_sel_hi source modifiers:
755 // - op_sel: selects low/high operand bits as input to the operation;
756 // has only meaning for 16-bit source operands
757 // - op_sel_hi: specifies the size of the source operands (16 or 32 bits);
758 // a value of 0 indicates 32 bit, 1 indicates 16 bit
759 // For the other VOP3P instructions, the semantics are:
760 // - op_sel: selects low/high operand bits as input to the operation which
761 // results in the lower-half of the destination
762 // - op_sel_hi: selects the low/high operand bits as input to the operation
763 // which results in the higher-half of the destination
764 int64_t OpSel = OpMod->getImm() & SISrcMods::OP_SEL_0;
765 int64_t OpSelHi = OpMod->getImm() & SISrcMods::OP_SEL_1;
766
767 // Check if all parts of the register are being used (= op_sel and op_sel_hi
768 // differ for VOP3P or op_sel_hi=0 for VOP3PMix). In that case we can return
769 // early.
770 if ((!InstrInfo.isVOP3PMix(MI: I) && (!OpSel || !OpSelHi) &&
771 (OpSel || OpSelHi)) ||
772 (InstrInfo.isVOP3PMix(MI: I) && !OpSelHi))
773 return AMDGPU::NoSubRegister;
774
775 const MachineRegisterInfo &MRI = I.getParent()->getParent()->getRegInfo();
776 const TargetRegisterClass *RC = TRI.getRegClassForOperandReg(MRI, MO: Op);
777
778 if (unsigned SubRegIdx = OpSel ? AMDGPU::sub1 : AMDGPU::sub0;
779 TRI.getSubClassWithSubReg(RC, SubRegIdx) == RC)
780 return SubRegIdx;
781 if (unsigned SubRegIdx = OpSel ? AMDGPU::hi16 : AMDGPU::lo16;
782 TRI.getSubClassWithSubReg(RC, SubRegIdx) == RC)
783 return SubRegIdx;
784
785 return AMDGPU::NoSubRegister;
786}
787
788Register GCNSubtarget::getRealSchedDependency(const MachineInstr &DefI,
789 int DefOpIdx,
790 const MachineInstr &UseI,
791 int UseOpIdx) const {
792 const SIRegisterInfo *TRI = getRegisterInfo();
793 const MachineOperand &DefOp = DefI.getOperand(i: DefOpIdx);
794 const MachineOperand &UseOp = UseI.getOperand(i: UseOpIdx);
795 Register DefReg = DefOp.getReg();
796 Register UseReg = UseOp.getReg();
797
798 // If the registers aren't restricted to a sub-register, there is no point in
799 // further analysis. This check makes only sense for virtual registers because
800 // physical registers may form a tuple and thus be part of a superregister
801 // although they are not a subregister themselves (vgpr0 is a "subreg" of
802 // vgpr0_vgpr1 without being a subreg in itself).
803 unsigned DefSubRegIdx = DefOp.getSubReg();
804 if (DefReg.isVirtual() && DefSubRegIdx == AMDGPU::NoSubRegister)
805 return DefReg;
806 unsigned UseSubRegIdx = getEffectiveSubRegIdx(TRI: *TRI, InstrInfo, I: UseI, Op: UseOp);
807 if (UseReg.isVirtual() && UseSubRegIdx == AMDGPU::NoSubRegister)
808 return DefReg;
809
810 if (!TRI->checkSubRegInterference(RegA: DefReg, SubA: DefSubRegIdx, RegB: UseReg, SubB: UseSubRegIdx))
811 return Register(); // No real dependency
812
813 // UseReg might be smaller or larger than DefReg, depending on the subreg and
814 // on whether DefReg is a subreg, too. -> Find the smaller one. This does not
815 // apply to virtual registers because we cannot construct a subreg for them.
816 if (DefReg.isVirtual())
817 return DefReg;
818 MCRegister DefMCReg =
819 DefSubRegIdx ? TRI->getSubReg(Reg: DefReg, Idx: DefSubRegIdx) : DefReg.asMCReg();
820 MCRegister UseMCReg =
821 UseSubRegIdx ? TRI->getSubReg(Reg: UseReg, Idx: UseSubRegIdx) : UseReg.asMCReg();
822 return TRI->isSubRegisterEq(RegA: DefMCReg, RegB: UseMCReg) ? UseMCReg : DefMCReg;
823}
824
825void GCNSubtarget::adjustSchedDependency(
826 SUnit *Def, int DefOpIdx, SUnit *Use, int UseOpIdx, SDep &Dep,
827 const TargetSchedModel *SchedModel) const {
828 if (Dep.getKind() != SDep::Kind::Data || !Dep.getReg() || !Def->isInstr() ||
829 !Use->isInstr())
830 return;
831
832 MachineInstr *DefI = Def->getInstr();
833 MachineInstr *UseI = Use->getInstr();
834
835 // Check for false latency on $tensorcnt / $asynccnt dependencies
836 if (Dep.getReg() == AMDGPU::TENSORcnt || Dep.getReg() == AMDGPU::ASYNCcnt) {
837 unsigned UseOp = UseI->getOpcode();
838 // Do not adjust latency for load->s_wait
839 bool IsBarrierCase =
840 InstrInfo.isLDSDMA(MI: *DefI) &&
841 (UseOp == AMDGPU::S_WAIT_TENSORCNT || UseOp == AMDGPU::S_WAIT_ASYNCCNT);
842 if (!IsBarrierCase) {
843 Dep.setLatency(1);
844 return;
845 }
846 }
847
848 if (Register Reg = getRealSchedDependency(DefI: *DefI, DefOpIdx, UseI: *UseI, UseOpIdx)) {
849 Dep.setReg(Reg);
850 } else {
851 Dep = SDep(Def, SDep::Artificial);
852 return; // This is not a data dependency anymore.
853 }
854
855 if (DefI->isBundle()) {
856 const SIRegisterInfo *TRI = getRegisterInfo();
857 auto Reg = Dep.getReg();
858 MachineBasicBlock::const_instr_iterator I(DefI->getIterator());
859 MachineBasicBlock::const_instr_iterator E(DefI->getParent()->instr_end());
860 unsigned Lat = 0;
861 for (++I; I != E && I->isBundledWithPred(); ++I) {
862 if (I->isMetaInstruction())
863 continue;
864 if (I->modifiesRegister(Reg, TRI))
865 Lat = InstrInfo.getInstrLatency(ItinData: getInstrItineraryData(), MI: *I);
866 else if (Lat)
867 --Lat;
868 }
869 Dep.setLatency(Lat);
870 } else if (UseI->isBundle()) {
871 const SIRegisterInfo *TRI = getRegisterInfo();
872 auto Reg = Dep.getReg();
873 MachineBasicBlock::const_instr_iterator I(UseI->getIterator());
874 MachineBasicBlock::const_instr_iterator E(UseI->getParent()->instr_end());
875 unsigned Lat = InstrInfo.getInstrLatency(ItinData: getInstrItineraryData(), MI: *DefI);
876 for (++I; I != E && I->isBundledWithPred() && Lat; ++I) {
877 if (I->isMetaInstruction())
878 continue;
879 if (I->readsRegister(Reg, TRI))
880 break;
881 --Lat;
882 }
883 Dep.setLatency(Lat);
884 } else if (Dep.getLatency() == 0 && Dep.getReg() == AMDGPU::VCC_LO) {
885 // Work around the fact that SIInstrInfo::fixImplicitOperands modifies
886 // implicit operands which come from the MCInstrDesc, which can fool
887 // ScheduleDAGInstrs::addPhysRegDataDeps into treating them as implicit
888 // pseudo operands.
889 Dep.setLatency(InstrInfo.getSchedModel().computeOperandLatency(
890 DefMI: DefI, DefOperIdx: DefOpIdx, UseMI: UseI, UseOperIdx: UseOpIdx));
891 }
892}
893
894unsigned GCNSubtarget::getNSAThreshold(const MachineFunction &MF) const {
895 if (getGeneration() >= AMDGPUSubtarget::GFX12)
896 return 0; // Not MIMG encoding.
897
898 if (NSAThreshold.getNumOccurrences() > 0)
899 return std::max(a: NSAThreshold.getValue(), b: 2u);
900
901 int Value = MF.getFunction().getFnAttributeAsParsedInteger(
902 Kind: "amdgpu-nsa-threshold", Default: -1);
903 if (Value > 0)
904 return std::max(a: Value, b: 2);
905
906 return NSAThreshold;
907}
908
909GCNUserSGPRUsageInfo::GCNUserSGPRUsageInfo(const Function &F,
910 const GCNSubtarget &ST)
911 : ST(ST) {
912 const CallingConv::ID CC = F.getCallingConv();
913 const bool IsKernel =
914 CC == CallingConv::AMDGPU_KERNEL || CC == CallingConv::SPIR_KERNEL;
915
916 if (IsKernel && (!F.arg_empty() || ST.getImplicitArgNumBytes(F) != 0))
917 KernargSegmentPtr = true;
918
919 bool IsAmdHsaOrMesa = ST.isAmdHsaOrMesa(F);
920 if (IsAmdHsaOrMesa && !ST.hasFlatScratchEnabled())
921 PrivateSegmentBuffer = true;
922 else if (ST.isMesaGfxShader(F))
923 ImplicitBufferPtr = true;
924
925 if (!AMDGPU::isGraphics(CC)) {
926 if (!F.hasFnAttribute(Kind: "amdgpu-no-dispatch-ptr"))
927 DispatchPtr = true;
928
929 // FIXME: Can this always be disabled with < COv5?
930 if (!F.hasFnAttribute(Kind: "amdgpu-no-queue-ptr"))
931 QueuePtr = true;
932
933 if (!F.hasFnAttribute(Kind: "amdgpu-no-dispatch-id"))
934 DispatchID = true;
935 }
936
937 if (ST.hasFlatAddressSpace() && AMDGPU::isEntryFunctionCC(CC) &&
938 (IsAmdHsaOrMesa || ST.hasFlatScratchEnabled()) &&
939 // FlatScratchInit cannot be true for graphics CC if
940 // hasFlatScratchEnabled() is false.
941 (ST.hasFlatScratchEnabled() ||
942 (!AMDGPU::isGraphics(CC) &&
943 !F.hasFnAttribute(Kind: "amdgpu-no-flat-scratch-init"))) &&
944 !ST.hasArchitectedFlatScratch()) {
945 FlatScratchInit = true;
946 }
947
948 if (hasImplicitBufferPtr())
949 NumUsedUserSGPRs += getNumUserSGPRForField(ID: ImplicitBufferPtrID);
950
951 if (hasPrivateSegmentBuffer())
952 NumUsedUserSGPRs += getNumUserSGPRForField(ID: PrivateSegmentBufferID);
953
954 if (hasDispatchPtr())
955 NumUsedUserSGPRs += getNumUserSGPRForField(ID: DispatchPtrID);
956
957 if (hasQueuePtr())
958 NumUsedUserSGPRs += getNumUserSGPRForField(ID: QueuePtrID);
959
960 if (hasKernargSegmentPtr())
961 NumUsedUserSGPRs += getNumUserSGPRForField(ID: KernargSegmentPtrID);
962
963 if (hasDispatchID())
964 NumUsedUserSGPRs += getNumUserSGPRForField(ID: DispatchIdID);
965
966 if (hasFlatScratchInit())
967 NumUsedUserSGPRs += getNumUserSGPRForField(ID: FlatScratchInitID);
968
969 if (hasPrivateSegmentSize())
970 NumUsedUserSGPRs += getNumUserSGPRForField(ID: PrivateSegmentSizeID);
971}
972
973void GCNUserSGPRUsageInfo::allocKernargPreloadSGPRs(unsigned NumSGPRs) {
974 assert(NumKernargPreloadSGPRs + NumSGPRs <= AMDGPU::getMaxNumUserSGPRs(ST));
975 NumKernargPreloadSGPRs += NumSGPRs;
976 NumUsedUserSGPRs += NumSGPRs;
977}
978
979unsigned GCNUserSGPRUsageInfo::getNumFreeUserSGPRs() {
980 return AMDGPU::getMaxNumUserSGPRs(STI: ST) - NumUsedUserSGPRs;
981}
982