1//=====-- GCNSubtarget.h - Define GCN Subtarget for AMDGPU ------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//==-----------------------------------------------------------------------===//
8//
9/// \file
10/// AMD GCN specific subclass of TargetSubtarget.
11//
12//===----------------------------------------------------------------------===//
13
14#ifndef LLVM_LIB_TARGET_AMDGPU_GCNSUBTARGET_H
15#define LLVM_LIB_TARGET_AMDGPU_GCNSUBTARGET_H
16
17#include "AMDGPUCallLowering.h"
18#include "AMDGPURegisterBankInfo.h"
19#include "AMDGPUSubtarget.h"
20#include "SIFrameLowering.h"
21#include "SIISelLowering.h"
22#include "SIInstrInfo.h"
23#include "Utils/AMDGPUBaseInfo.h"
24#include "llvm/Support/AMDHSAKernelDescriptor.h"
25#include "llvm/Support/ErrorHandling.h"
26#include <optional>
27
28#define GET_SUBTARGETINFO_HEADER
29#include "AMDGPUGenSubtargetInfo.inc"
30
31namespace llvm {
32
33class GCNTargetMachine;
34
35/// Module flag names controlling out-of-bounds buffer access semantics.
36/// Each flag is an i32 with Module::Max merge behaviour and tri-state values:
37/// 0 = any (absent/default - backend currently treats as strict)
38/// 1 = relaxed
39/// 2 = strict
40namespace AMDGPUOOBMode {
41inline constexpr StringLiteral BufferFlag("amdgpu.buffer.oob.mode");
42inline constexpr StringLiteral TBufferFlag("amdgpu.tbuffer.oob.mode");
43} // namespace AMDGPUOOBMode
44
45class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
46 public AMDGPUSubtarget {
47public:
48 using AMDGPUSubtarget::getMaxWavesPerEU;
49
50 // Following 2 enums are documented at:
51 // - https://llvm.org/docs/AMDGPUUsage.html#trap-handler-abi
52 enum class TrapHandlerAbi {
53 NONE = 0x00,
54 AMDHSA = 0x01,
55 };
56
57 enum class TrapID {
58 LLVMAMDHSATrap = 0x02,
59 LLVMAMDHSADebugTrap = 0x03,
60 };
61
62private:
63 /// SelectionDAGISel related APIs.
64 std::unique_ptr<const SelectionDAGTargetInfo> TSInfo;
65
66 /// GlobalISel related APIs.
67 std::unique_ptr<AMDGPUCallLowering> CallLoweringInfo;
68 std::unique_ptr<InlineAsmLowering> InlineAsmLoweringInfo;
69 std::unique_ptr<InstructionSelector> InstSelector;
70 std::unique_ptr<LegalizerInfo> Legalizer;
71 std::unique_ptr<AMDGPURegisterBankInfo> RegBankInfo;
72
73protected:
74 // Basic subtarget description.
75 AMDGPU::TargetID TargetID;
76 unsigned Gen = INVALID;
77 InstrItineraryData InstrItins;
78 int LDSBankCount = 0;
79 unsigned MaxPrivateElementSize = 0;
80
81 // Instruction cache line size in bytes; set from TableGen subtarget features.
82 unsigned InstCacheLineSize = 0;
83
84 // Data (VMEM) cache line size in bytes; set from TableGen subtarget features.
85 unsigned DataCacheLineSize = 0;
86
87 /// The width, in bits, of the num_records field of a buffer resource (V#),
88 /// set from tablegen subtarget features, 0 is unknown.
89 unsigned BufferResourceNumRecordsWidth = 0;
90
91 // Dynamically set bits that enable features.
92 const bool BufferOOBRelaxed;
93 const bool TBufferOOBRelaxed;
94
95 /// The maximum number of instructions that may be placed within an S_CLAUSE,
96 /// which is one greater than the maximum argument to S_CLAUSE. A value of 0
97 /// indicates a lack of S_CLAUSE support.
98 unsigned MaxHardClauseLength = 0;
99
100#define GET_SUBTARGETINFO_MACRO(ATTRIBUTE, DEFAULT, GETTER) \
101 bool ATTRIBUTE = DEFAULT;
102#include "AMDGPUGenSubtargetInfo.inc"
103
104private:
105 SIInstrInfo InstrInfo;
106 SITargetLowering TLInfo;
107 SIFrameLowering FrameLowering;
108
109 /// Get the register that represents the actual dependency between the
110 /// definition and the use. The definition might only affect a subregister
111 /// that is not actually used. Works for both virtual and physical registers.
112 /// Note: Currently supports VOP3P instructions (without WMMA an SWMMAC).
113 /// Returns the definition register if there is a real dependency and no
114 /// better match is found.
115 Register getRealSchedDependency(const MachineInstr &DefI, int DefOpIdx,
116 const MachineInstr &UseI, int UseOpIdx) const;
117
118public:
119 GCNSubtarget(
120 const Triple &TT, StringRef GPU, StringRef FS, const GCNTargetMachine &TM,
121 bool BufferOOBRelaxed = false, bool TBufferOOBRelaxed = false,
122 AMDGPU::TargetIDSetting XnackSetting = AMDGPU::TargetIDSetting::Any,
123 AMDGPU::TargetIDSetting SramEccSetting = AMDGPU::TargetIDSetting::Any);
124 ~GCNSubtarget() override;
125
126 GCNSubtarget &
127 initializeSubtargetDependencies(const Triple &TT, StringRef GPU, StringRef FS,
128 AMDGPU::TargetIDSetting XnackSetting,
129 AMDGPU::TargetIDSetting SramEccSetting);
130
131 /// Diagnose inconsistent subtarget features before attempting to codegen
132 /// function \p F.
133 void checkSubtargetFeatures(const Function &F) const;
134
135 const SIInstrInfo *getInstrInfo() const override { return &InstrInfo; }
136
137 const SIFrameLowering *getFrameLowering() const override {
138 return &FrameLowering;
139 }
140
141 const SITargetLowering *getTargetLowering() const override { return &TLInfo; }
142
143 const SIRegisterInfo *getRegisterInfo() const override {
144 return &InstrInfo.getRegisterInfo();
145 }
146
147 const SelectionDAGTargetInfo *getSelectionDAGInfo() const override;
148
149 const CallLowering *getCallLowering() const override {
150 return CallLoweringInfo.get();
151 }
152
153 const InlineAsmLowering *getInlineAsmLowering() const override {
154 return InlineAsmLoweringInfo.get();
155 }
156
157 InstructionSelector *getInstructionSelector() const override {
158 return InstSelector.get();
159 }
160
161 const LegalizerInfo *getLegalizerInfo() const override {
162 return Legalizer.get();
163 }
164
165 const AMDGPURegisterBankInfo *getRegBankInfo() const override {
166 return RegBankInfo.get();
167 }
168
169 const AMDGPU::TargetID &getTargetID() const { return TargetID; }
170
171 const InstrItineraryData *getInstrItineraryData() const override {
172 return &InstrItins;
173 }
174
175 void ParseSubtargetFeatures(StringRef CPU, StringRef TuneCPU, StringRef FS);
176
177 Generation getGeneration() const { return (Generation)Gen; }
178
179 bool isGFX11Plus() const { return getGeneration() >= GFX11; }
180
181#define GET_SUBTARGETINFO_MACRO(ATTRIBUTE, DEFAULT, GETTER) \
182 bool GETTER() const override { return ATTRIBUTE; }
183#include "AMDGPUGenSubtargetInfo.inc"
184
185 unsigned getMaxWaveScratchSize() const {
186 // See COMPUTE_TMPRING_SIZE.WAVESIZE.
187 if (getGeneration() >= GFX12) {
188 // 18-bit field in units of 64-dword.
189 return (64 * 4) * ((1 << 18) - 1);
190 }
191 if (getGeneration() == GFX11) {
192 // 15-bit field in units of 64-dword.
193 return (64 * 4) * ((1 << 15) - 1);
194 }
195 // 13-bit field in units of 256-dword.
196 return (256 * 4) * ((1 << 13) - 1);
197 }
198
199 /// Return the number of high bits known to be zero for a frame index.
200 unsigned getKnownHighZeroBitsForFrameIndex() const {
201 return llvm::countl_zero(Val: getMaxWaveScratchSize()) + getWavefrontSizeLog2();
202 }
203
204 int getLDSBankCount() const { return LDSBankCount; }
205
206 /// Instruction cache line size in bytes (64 for pre-GFX11, 128 for GFX11+).
207 unsigned getInstCacheLineSize() const { return InstCacheLineSize; }
208
209 /// Data (VMEM) cache line size in bytes (128 for gfx12), has no use before
210 /// GFX12.
211 unsigned getDataCacheLineSize() const { return DataCacheLineSize; }
212
213 unsigned getMaxPrivateElementSize(bool ForBufferRSrc = false) const {
214 return (ForBufferRSrc || !hasFlatScratchEnabled()) ? MaxPrivateElementSize
215 : 16;
216 }
217
218 unsigned getConstantBusLimit(unsigned Opcode) const;
219
220 /// Returns if the result of this instruction with a 16-bit result returned in
221 /// a 32-bit register implicitly zeroes the high 16-bits, rather than preserve
222 /// the original value.
223 bool zeroesHigh16BitsOfDest(unsigned Opcode) const;
224
225 bool hasAddr64() const {
226 return (getGeneration() < AMDGPUSubtarget::VOLCANIC_ISLANDS);
227 }
228
229 bool hasFlat() const {
230 return (getGeneration() > AMDGPUSubtarget::SOUTHERN_ISLANDS);
231 }
232
233 // Return true if the target only has the reverse operand versions of VALU
234 // shift instructions (e.g. v_lshrrev_b32, and no v_lshr_b32).
235 bool hasOnlyRevVALUShifts() const {
236 return getGeneration() >= VOLCANIC_ISLANDS;
237 }
238
239 bool hasFractBug() const { return getGeneration() == SOUTHERN_ISLANDS; }
240
241 bool hasMed3_16() const { return getGeneration() >= AMDGPUSubtarget::GFX9; }
242
243 bool hasMin3Max3_16() const {
244 return getGeneration() >= AMDGPUSubtarget::GFX9;
245 }
246
247 bool hasSwap() const { return HasGFX9Insts; }
248
249 bool hasScalarPackInsts() const { return HasGFX9Insts; }
250
251 bool hasScalarMulHiInsts() const { return HasGFX9Insts; }
252
253 bool hasScalarSubwordLoads() const { return getGeneration() >= GFX12; }
254
255 bool hasAsyncMark() const { return hasVMemToLDSLoad() || HasAsynccnt; }
256
257 TrapHandlerAbi getTrapHandlerAbi() const {
258 return isAmdHsaOS() ? TrapHandlerAbi::AMDHSA : TrapHandlerAbi::NONE;
259 }
260
261 bool supportsGetDoorbellID() const {
262 // The S_GETREG DOORBELL_ID is supported by all GFX9 onward targets.
263 return getGeneration() >= GFX9;
264 }
265
266 /// True if the offset field of DS instructions works as expected. On SI, the
267 /// offset uses a 16-bit adder and does not always wrap properly.
268 bool hasUsableDSOffset() const { return getGeneration() >= SEA_ISLANDS; }
269
270 bool unsafeDSOffsetFoldingEnabled() const {
271 return EnableUnsafeDSOffsetFolding;
272 }
273
274 /// Condition output from div_scale is usable.
275 bool hasUsableDivScaleConditionOutput() const {
276 return getGeneration() != SOUTHERN_ISLANDS;
277 }
278
279 /// Extra wait hazard is needed in some cases before
280 /// s_cbranch_vccnz/s_cbranch_vccz.
281 bool hasReadVCCZBug() const { return getGeneration() <= SEA_ISLANDS; }
282
283 /// Writes to VCC_LO/VCC_HI update the VCCZ flag.
284 bool partialVCCWritesUpdateVCCZ() const { return getGeneration() >= GFX10; }
285
286 /// A read of an SGPR by SMRD instruction requires 4 wait states when the SGPR
287 /// was written by a VALU instruction.
288 bool hasSMRDReadVALUDefHazard() const {
289 return getGeneration() == SOUTHERN_ISLANDS;
290 }
291
292 /// A read of an SGPR by a VMEM instruction requires 5 wait states when the
293 /// SGPR was written by a VALU Instruction.
294 bool hasVMEMReadSGPRVALUDefHazard() const {
295 return getGeneration() >= VOLCANIC_ISLANDS;
296 }
297
298 bool hasRFEHazards() const { return getGeneration() >= VOLCANIC_ISLANDS; }
299
300 /// Number of hazard wait states for s_setreg_b32/s_setreg_imm32_b32.
301 unsigned getSetRegWaitStates() const {
302 return getGeneration() <= SEA_ISLANDS ? 1 : 2;
303 }
304
305 bool supportsMinMaxDenormModes() const {
306 return getGeneration() >= AMDGPUSubtarget::GFX9;
307 }
308
309 /// \returns If target supports S_DENORM_MODE.
310 bool hasDenormModeInst() const {
311 return getGeneration() >= AMDGPUSubtarget::GFX10;
312 }
313
314 /// \returns If target supports ds_read/write_b128 and user enables generation
315 /// of ds_read/write_b128.
316 bool useDS128() const { return HasCIInsts && EnableDS128; }
317
318 /// \return If target supports ds_read/write_b96/128.
319 bool hasDS96AndDS128() const { return HasCIInsts; }
320
321 /// Have v_trunc_f64, v_ceil_f64, v_rndne_f64
322 bool haveRoundOpsF64() const { return HasCIInsts; }
323
324 /// \returns If MUBUF instructions always perform range checking, even for
325 /// buffer resources used for private memory access.
326 bool privateMemoryResourceIsRangeChecked() const {
327 return getGeneration() < AMDGPUSubtarget::GFX9;
328 }
329
330 /// \returns If target requires PRT Struct NULL support (zero result registers
331 /// for sparse texture support).
332 bool usePRTStrictNull() const { return EnablePRTStrictNull; }
333
334 bool hasUnalignedBufferAccessEnabled() const {
335 return HasUnalignedBufferAccess && HasUnalignedAccessMode;
336 }
337
338 bool hasUnalignedDSAccessEnabled() const {
339 return HasUnalignedDSAccess && HasUnalignedAccessMode;
340 }
341
342 bool hasUnalignedScratchAccessEnabled() const {
343 return HasUnalignedScratchAccess && HasUnalignedAccessMode;
344 }
345
346 bool isXNACKEnabled() const { return TargetID.isXnackOnOrAny(); }
347
348 bool hasRelaxedBufferOOBMode() const { return BufferOOBRelaxed; }
349 bool hasRelaxedTBufferOOBMode() const { return TBufferOOBRelaxed; }
350
351 /// Return the width, in bits, of the num_records field of a buffer resource
352 /// (V#) on this subtarget, or std::nullopt if not yet known.
353 std::optional<unsigned> getBufferResourceNumRecordsWidth() const {
354 if (BufferResourceNumRecordsWidth == 0)
355 return std::nullopt;
356 return BufferResourceNumRecordsWidth;
357 }
358
359 bool isCuModeEnabled() const { return EnableCuMode; }
360
361 /// \returns Whether a work-group runs on all of the block's SIMDs.
362 bool isFullSIMDMode() const {
363 return (HasGFX1250Insts && getGeneration() < GFX13) || !EnableCuMode;
364 }
365
366 bool isPreciseMemoryEnabled() const { return EnablePreciseMemory; }
367
368 bool hasFlatScrRegister() const { return hasFlatAddressSpace(); }
369
370 // Check if target supports ST addressing mode with FLAT scratch instructions.
371 // The ST addressing mode means no registers are used, either VGPR or SGPR,
372 // but only immediate offset is swizzled and added to the FLAT scratch base.
373 bool hasFlatScratchSTMode() const {
374 return hasFlatScratchInsts() && (hasGFX10_3Insts() || hasGFX940Insts());
375 }
376
377 bool hasFlatScratchSVSMode() const { return HasGFX940Insts || HasGFX11Insts; }
378
379 bool hasFlatScratchEnabled() const {
380 return hasArchitectedFlatScratch() ||
381 (EnableFlatScratch && hasFlatScratchInsts());
382 }
383
384 bool hasExportInsts() const {
385 return !hasGFX940Insts() && !hasGFX1250Insts();
386 }
387
388 bool hasVINTERPEncoding() const {
389 return HasGFX11Insts && !hasGFX1250Insts();
390 }
391
392 bool hasMultiDwordFlatScratchAddressing() const {
393 return getGeneration() >= GFX9;
394 }
395
396 bool hasFlatLgkmVMemCountInOrder() const { return getGeneration() > GFX9; }
397
398 bool hasD16LoadStore() const { return getGeneration() >= GFX9; }
399
400 bool d16PreservesUnusedBits() const {
401 return hasD16LoadStore() && !TargetID.isSramEccOnOrAny();
402 }
403
404 bool hasD16Images() const { return getGeneration() >= VOLCANIC_ISLANDS; }
405
406 /// Return if most LDS instructions have an m0 use that require m0 to be
407 /// initialized.
408 bool ldsRequiresM0Init() const { return getGeneration() < GFX9; }
409
410 // True if the hardware rewinds and replays GWS operations if a wave is
411 // preempted.
412 //
413 // If this is false, a GWS operation requires testing if a nack set the
414 // MEM_VIOL bit, and repeating if so.
415 bool hasGWSAutoReplay() const { return getGeneration() >= GFX9; }
416
417 /// \returns if target has ds_gws_sema_release_all instruction.
418 bool hasGWSSemaReleaseAll() const { return HasCIInsts; }
419
420 bool hasScalarAddSub64() const { return getGeneration() >= GFX12; }
421
422 bool hasScalarSMulU64() const { return getGeneration() >= GFX12; }
423
424 // Covers VS/PS/CS graphics shaders
425 bool isMesaGfxShader(const Function &F) const {
426 return isMesa3DOS() && AMDGPU::isShader(CC: F.getCallingConv());
427 }
428
429 bool hasMad64_32() const { return getGeneration() >= SEA_ISLANDS; }
430
431 bool hasAtomicFaddInsts() const {
432 return HasAtomicFaddRtnInsts || HasAtomicFaddNoRtnInsts;
433 }
434
435 bool vmemWriteNeedsExpWaitcnt() const {
436 return getGeneration() < SEA_ISLANDS;
437 }
438
439 bool hasInstPrefetch() const {
440 return getGeneration() == GFX10 || getGeneration() == GFX11;
441 }
442
443 bool hasPrefetch() const { return HasGFX12Insts; }
444
445 bool hasInstPrefSize() const { return isGFX11Plus(); }
446
447 void getInstPrefSizeArgs(uint32_t &Mask, uint32_t &Shift, uint32_t &Width,
448 uint32_t &CacheLineSize) const {
449 assert(isGFX11Plus());
450 CacheLineSize = getInstCacheLineSize();
451 if (getGeneration() == GFX11) {
452 Mask = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE;
453 Shift = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_SHIFT;
454 Width = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_WIDTH;
455 } else {
456 Mask = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE;
457 Shift = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_SHIFT;
458 Width = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_WIDTH;
459 }
460 }
461
462 // Has s_cmpk_* instructions.
463 bool hasSCmpK() const { return getGeneration() < GFX12; }
464
465 // Scratch is allocated in 256 dword per wave blocks for the entire
466 // wavefront. When viewed from the perspective of an arbitrary workitem, this
467 // is 4-byte aligned.
468 //
469 // Only 4-byte alignment is really needed to access anything. Transformations
470 // on the pointer value itself may rely on the alignment / known low bits of
471 // the pointer. Set this to something above the minimum to avoid needing
472 // dynamic realignment in common cases.
473 Align getStackAlignment() const { return Align(16); }
474
475 bool enableMachineScheduler() const override { return true; }
476
477 bool enableSSAMachineScheduler() const override { return true; }
478
479 bool useAA() const override;
480
481 bool enableSubRegLiveness() const override { return true; }
482
483 // XXX - Why is this here if it isn't in the default pass set?
484 bool enableEarlyIfConversion() const override { return true; }
485
486 void overrideSchedPolicy(MachineSchedPolicy &Policy,
487 const SchedRegion &Region) const override;
488
489 void overridePostRASchedPolicy(MachineSchedPolicy &Policy,
490 const SchedRegion &Region) const override;
491
492 void overridePipelinerPolicy(MachinePipelinerPolicy &Policy) const override;
493
494 void mirFileLoaded(MachineFunction &MF) const override;
495
496 unsigned getMaxNumUserSGPRs() const {
497 return AMDGPU::getMaxNumUserSGPRs(STI: *this);
498 }
499
500 bool useVGPRIndexMode() const;
501
502 bool hasScalarCompareEq64() const {
503 return getGeneration() >= VOLCANIC_ISLANDS;
504 }
505
506 bool hasLDSFPAtomicAddF32() const { return HasGFX8Insts; }
507 bool hasLDSFPAtomicAddF64() const {
508 return HasGFX90AInsts || HasGFX1250Insts;
509 }
510
511 /// \returns true if the subtarget has the v_permlane64_b32 instruction.
512 bool hasPermLane64() const { return getGeneration() >= GFX11; }
513
514 /// \returns true if the subtarget supports the ds_swizzle rotate and FFT
515 /// swizzle modes (GFX9+).
516 bool hasDsSwizzleRotateMode() const { return getGeneration() >= GFX9; }
517
518 bool hasDPPRowShare() const {
519 return HasDPP && (HasGFX90AInsts || getGeneration() >= GFX10);
520 }
521
522 // Has V_PK_MOV_B32 opcode
523 bool hasPkMovB32() const { return HasGFX90AInsts; }
524
525 bool hasBufferTFEFormatD16() const { return !HasGFX90AInsts; }
526
527 bool hasFmaakFmamkF32Insts() const {
528 return getGeneration() >= GFX10 || hasGFX940Insts();
529 }
530
531 bool hasFmaakFmamkF64Insts() const { return hasGFX1250Insts(); }
532
533 bool hasNonNSAEncoding() const { return getGeneration() < GFX12; }
534
535 unsigned getNSAMaxSize(bool HasSampler = false) const {
536 return AMDGPU::getNSAMaxSize(STI: *this, HasSampler);
537 }
538
539 bool hasMadF16() const;
540
541 // Scalar and global loads support scale_offset bit.
542 bool hasScaleOffset() const { return HasGFX1250Insts; }
543
544 // FLAT GLOBAL VOffset is signed
545 bool hasSignedGVSOffset() const { return HasGFX1250Insts; }
546
547 bool loadStoreOptEnabled() const { return EnableLoadStoreOpt; }
548
549 bool hasUserSGPRInit16BugInWave32() const {
550 return HasUserSGPRInit16Bug && isWave32();
551 }
552
553 bool has12DWordStoreHazard() const {
554 return getGeneration() != AMDGPUSubtarget::SOUTHERN_ISLANDS;
555 }
556
557 // \returns true if the subtarget supports DWORDX3 load/store instructions.
558 bool hasDwordx3LoadStores() const { return HasCIInsts; }
559
560 bool hasReadM0MovRelInterpHazard() const {
561 return getGeneration() == AMDGPUSubtarget::GFX9;
562 }
563
564 bool hasReadM0SendMsgHazard() const {
565 return getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS &&
566 getGeneration() <= AMDGPUSubtarget::GFX9;
567 }
568
569 bool hasReadM0LdsDmaHazard() const {
570 return getGeneration() == AMDGPUSubtarget::GFX9;
571 }
572
573 bool hasReadM0LdsDirectHazard() const {
574 return getGeneration() == AMDGPUSubtarget::GFX9;
575 }
576
577 bool hasLDSMisalignedBugInWGPMode() const {
578 return HasLDSMisalignedBug && !EnableCuMode;
579 }
580
581 // Shift amount of a 64 bit shift cannot be a highest allocated register
582 // if also at the end of the allocation block.
583 bool hasShift64HighRegBug() const { return HasGFX90AInsts; }
584
585 // v_dot2c_f32_f16 unconditionally flushes f16 subnormal inputs to zero
586 // regardless of the MODE register, unlike v_fma_mix_f32 which respects it.
587 bool dot2UnconditionalFlush() const {
588 return HasGFX90AInsts && !HasGFX940Insts;
589 }
590
591 // Has one cycle hazard on transcendental instruction feeding a
592 // non transcendental VALU.
593 bool hasTransForwardingHazard() const { return HasGFX940Insts; }
594
595 // Has one cycle hazard on a VALU instruction partially writing dst with
596 // a shift of result bits feeding another VALU instruction.
597 bool hasDstSelForwardingHazard() const { return HasGFX940Insts; }
598
599 // Cannot use op_sel with v_dot instructions.
600 bool hasDOTOpSelHazard() const { return HasGFX940Insts || HasGFX11Insts; }
601
602 // Does not have HW interlocs for VALU writing and then reading SGPRs.
603 bool hasVDecCoExecHazard() const { return HasGFX940Insts; }
604
605 bool hasHardClauses() const { return MaxHardClauseLength > 0; }
606
607 bool hasFPAtomicToDenormModeHazard() const {
608 return getGeneration() == GFX10;
609 }
610
611 bool hasVOP3DPP() const { return getGeneration() >= GFX11; }
612
613 bool hasLdsDirect() const { return getGeneration() >= GFX11; }
614
615 bool hasLdsWaitVMSRC() const { return getGeneration() >= GFX12; }
616
617 bool hasVALUPartialForwardingHazard() const {
618 return getGeneration() == GFX11;
619 }
620
621 bool hasCvtScaleForwardingHazard() const { return HasGFX950Insts; }
622
623 bool hasPermlaneForwardingHazard() const { return HasGFX950Insts; }
624
625 // All GFX9 targets experience a fetch delay when an instruction at the start
626 // of a loop header is split by a 32-byte fetch window boundary, but GFX950
627 // is uniquely sensitive to this: the delay triggers further performance
628 // degradation beyond the fetch latency itself.
629 bool hasLoopHeadInstSplitSensitivity() const { return HasGFX950Insts; }
630
631 bool requiresCodeObjectV6() const { return RequiresCOV6; }
632
633 bool useVGPRBlockOpsForCSR() const { return UseBlockVGPROpsForCSR; }
634
635 bool hasVALUMaskWriteHazard() const { return getGeneration() == GFX11; }
636
637 bool hasVALUReadSGPRHazard() const {
638 return HasGFX12Insts && !HasGFX1250Insts;
639 }
640
641 bool setRegModeNeedsVNOPs() const {
642 return HasGFX1250Insts && getGeneration() == GFX12;
643 }
644
645 /// Return if operations acting on VGPR tuples require even alignment.
646 bool needsAlignedVGPRs() const { return RequiresAlignVGPR; }
647
648 /// Return true if the target has the S_PACK_HL_B32_B16 instruction.
649 bool hasSPackHL() const { return HasGFX11Insts; }
650
651 /// Return true if the target has the V_CVT_PK_I16_F32/V_CVT_PK_U16_F32
652 /// instructions.
653 bool hasVCvtPkIU16F32() const { return HasGFX11Insts; }
654
655 /// Return true if the target's EXP instruction supports the NULL export
656 /// target.
657 bool hasNullExportTarget() const { return !HasGFX11Insts; }
658
659 bool hasFlatScratchSVSSwizzleBug() const { return getGeneration() == GFX11; }
660
661 /// Return true if the target has the S_DELAY_ALU instruction.
662 bool hasDelayAlu() const { return HasGFX11Insts; }
663
664 /// Returns true if the target supports
665 /// global_load_lds_dwordx3/global_load_lds_dwordx4 or
666 /// buffer_load_dwordx3/buffer_load_dwordx4 with the lds bit.
667 bool hasLDSLoadB96_B128() const { return hasGFX950Insts(); }
668
669 /// \returns true if the target uses LOADcnt/SAMPLEcnt/BVHcnt, DScnt/KMcnt
670 /// and STOREcnt rather than VMcnt, LGKMcnt and VScnt respectively.
671 bool hasExtendedWaitCounts() const { return getGeneration() >= GFX12; }
672
673 /// \returns true if the target has packed f32 instructions that only read 32
674 /// bits from a scalar operand (SGPR or literal) and replicates the bits to
675 /// both channels.
676 bool hasPKF32InstsReplicatingLower32BitsOfScalarInput() const {
677 return getGeneration() == GFX12 && HasGFX1250Insts;
678 }
679
680 bool hasAddPC64Inst() const { return HasGFX1250Insts; }
681
682 /// \returns true if the target supports expert scheduling mode 2 which relies
683 /// on the compiler to insert waits to avoid hazards between VMEM and VALU
684 /// instructions in some instances.
685 bool hasExpertSchedulingMode() const { return getGeneration() >= GFX12; }
686
687 /// \returns The maximum number of instructions that can be enclosed in an
688 /// S_CLAUSE on the given subtarget, or 0 for targets that do not support that
689 /// instruction.
690 unsigned maxHardClauseLength() const { return MaxHardClauseLength; }
691
692 /// Return the maximum number of waves per SIMD for kernels using \p SGPRs
693 /// SGPRs
694 unsigned getOccupancyWithNumSGPRs(unsigned SGPRs) const;
695
696 /// Return the maximum number of waves per SIMD for kernels using \p VGPRs
697 /// VGPRs
698 unsigned getOccupancyWithNumVGPRs(unsigned VGPRs,
699 unsigned DynamicVGPRBlockSize) const;
700
701 /// Subtarget's minimum/maximum occupancy, in number of waves per EU, that can
702 /// be achieved when the only function running on a CU is \p F, each workgroup
703 /// uses \p LDSSize bytes of LDS, and each wave uses \p NumSGPRs SGPRs and \p
704 /// NumVGPRs VGPRs. The flat workgroup sizes associated to the function are a
705 /// range, so this returns a range as well.
706 ///
707 /// Note that occupancy can be affected by the scratch allocation as well, but
708 /// we do not have enough information to compute it.
709 std::pair<unsigned, unsigned> computeOccupancy(const Function &F,
710 unsigned LDSSize = 0,
711 unsigned NumSGPRs = 0,
712 unsigned NumVGPRs = 0) const;
713
714 /// \returns true if the flat_scratch register should be initialized with the
715 /// pointer to the wave's scratch memory rather than a size and offset.
716 bool flatScratchIsPointer() const {
717 return getGeneration() >= AMDGPUSubtarget::GFX9;
718 }
719
720 /// \returns true if the machine has merged shaders in which s0-s7 are
721 /// reserved by the hardware and user SGPRs start at s8
722 bool hasMergedShaders() const { return getGeneration() >= GFX9; }
723
724 // \returns true if the target supports the pre-NGG legacy geometry path.
725 bool hasLegacyGeometry() const { return getGeneration() < GFX11; }
726
727 // \returns true if the target has split barriers feature
728 bool hasSplitBarriers() const { return getGeneration() >= GFX12; }
729
730 // \returns true if the target has WG_RR_MODE kernel descriptor mode bit
731 bool hasRrWGMode() const { return getGeneration() >= GFX12; }
732
733 /// \returns true if VADDR and SADDR fields in VSCRATCH can use negative
734 /// values.
735 bool hasSignedScratchOffsets() const { return getGeneration() >= GFX12; }
736
737 bool hasINVWBL2WaitCntRequirement() const { return HasGFX1250Insts; }
738
739 bool hasVOPD3() const { return HasGFX1250Insts; }
740
741 // \returns true if the target has V_PK_{MIN|MAX}3_{I|U}16 instructions.
742 bool hasPkMinMax3Insts() const { return HasGFX1250Insts; }
743
744 // \returns ture if target has S_GET_SHADER_CYCLES_U64 instruction.
745 bool hasSGetShaderCyclesInst() const { return HasGFX1250Insts; }
746
747 // \returns true if S_GETPC_B64 zero-extends the result from 48 bits instead
748 // of sign-extending. Note that GFX1250 has not only fixed the bug but also
749 // extended VA to 57 bits.
750 bool hasGetPCZeroExtension() const {
751 return HasGFX12Insts && !HasGFX1250Insts;
752 }
753
754 // \returns true if the target needs to create a prolog for backward
755 // compatibility when preloading kernel arguments.
756 bool needsKernArgPreloadProlog() const {
757 return hasKernargPreload() && !HasGFX1250Insts;
758 }
759
760 bool hasCondSubInsts() const { return HasGFX12Insts; }
761
762 bool hasSubClampInsts() const { return hasGFX10_3Insts(); }
763
764 bool hasAnyPackedFP32Ops() const {
765 return hasPackedFP32Ops() || hasPackedFP32SingleSGPROps();
766 };
767
768 bool hasAnyPackedFP64Ops() const { return hasPackedFP64SingleSGPROps(); };
769
770 bool hasAnyPackedU64Ops() const { return hasPackedU64SingleSGPROps(); };
771
772 /// \returns SGPR allocation granularity supported by the subtarget.
773 unsigned getSGPRAllocGranule() const {
774 return AMDGPU::getSGPRAllocGranule(AK: getTargetID().getGPUKind());
775 }
776
777 /// \returns SGPR encoding granularity supported by the subtarget.
778 unsigned getSGPREncodingGranule() const {
779 return AMDGPU::IsaInfo::getSGPREncodingGranule(STI: *this);
780 }
781
782 /// \returns Total number of SGPRs supported by the subtarget.
783 unsigned getTotalNumSGPRs() const {
784 return AMDGPU::getTotalNumSGPRs(AK: getTargetID().getGPUKind());
785 }
786
787 /// \returns Addressable number of SGPRs supported by the subtarget.
788 unsigned getAddressableNumSGPRs() const {
789 return AMDGPU::getAddressableNumSGPRs(AK: getTargetID().getGPUKind());
790 }
791
792 /// \returns Minimum number of SGPRs that meets the given number of waves per
793 /// execution unit requirement supported by the subtarget.
794 unsigned getMinNumSGPRs(unsigned WavesPerEU) const {
795 return AMDGPU::IsaInfo::getMinNumSGPRs(STI: *this, WavesPerEU);
796 }
797
798 /// \returns Maximum number of SGPRs that meets the given number of waves per
799 /// execution unit requirement supported by the subtarget.
800 unsigned getMaxNumSGPRs(unsigned WavesPerEU, bool Addressable) const {
801 return AMDGPU::IsaInfo::getMaxNumSGPRs(STI: *this, WavesPerEU, Addressable);
802 }
803
804 /// \returns Reserved number of SGPRs. This is common
805 /// utility function called by MachineFunction and
806 /// Function variants of getReservedNumSGPRs.
807 unsigned getBaseReservedNumSGPRs(const bool HasFlatScratch) const;
808 /// \returns Reserved number of SGPRs for given machine function \p MF.
809 unsigned getReservedNumSGPRs(const MachineFunction &MF) const;
810
811 /// \returns Reserved number of SGPRs for given function \p F.
812 unsigned getReservedNumSGPRs(const Function &F) const;
813
814 /// \returns Maximum number of preloaded SGPRs for the subtarget.
815 unsigned getMaxNumPreloadedSGPRs() const;
816
817 /// \returns max num SGPRs. This is the common utility
818 /// function called by MachineFunction and Function
819 /// variants of getMaxNumSGPRs.
820 unsigned getBaseMaxNumSGPRs(const Function &F,
821 std::pair<unsigned, unsigned> WavesPerEU,
822 unsigned PreloadedSGPRs,
823 unsigned ReservedNumSGPRs) const;
824
825 /// \returns Maximum number of SGPRs that meets number of waves per execution
826 /// unit requirement for function \p MF, or number of SGPRs explicitly
827 /// requested using "amdgpu-num-sgpr" attribute attached to function \p MF.
828 ///
829 /// \returns Value that meets number of waves per execution unit requirement
830 /// if explicitly requested value cannot be converted to integer, violates
831 /// subtarget's specifications, or does not meet number of waves per execution
832 /// unit requirement.
833 unsigned getMaxNumSGPRs(const MachineFunction &MF) const;
834
835 /// \returns Maximum number of SGPRs that meets number of waves per execution
836 /// unit requirement for function \p F, or number of SGPRs explicitly
837 /// requested using "amdgpu-num-sgpr" attribute attached to function \p F.
838 ///
839 /// \returns Value that meets number of waves per execution unit requirement
840 /// if explicitly requested value cannot be converted to integer, violates
841 /// subtarget's specifications, or does not meet number of waves per execution
842 /// unit requirement.
843 unsigned getMaxNumSGPRs(const Function &F) const;
844
845 /// \returns VGPR allocation granularity supported by the subtarget.
846 unsigned getVGPRAllocGranule(unsigned DynamicVGPRBlockSize) const {
847 return AMDGPU::IsaInfo::getVGPRAllocGranule(STI: *this, DynamicVGPRBlockSize);
848 }
849
850 /// \returns VGPR encoding granularity supported by the subtarget.
851 unsigned getVGPREncodingGranule() const {
852 return AMDGPU::IsaInfo::getVGPREncodingGranule(STI: *this);
853 }
854
855 /// \returns Total number of VGPRs supported by the subtarget.
856 unsigned getTotalNumVGPRs() const {
857 return AMDGPU::getTotalNumVGPRs(AK: getTargetID().getGPUKind(), IsWave32: isWave32());
858 }
859
860 /// \returns Addressable number of architectural VGPRs supported by the
861 /// subtarget.
862 unsigned getAddressableNumArchVGPRs() const {
863 return AMDGPU::IsaInfo::getAddressableNumArchVGPRs(STI: *this);
864 }
865
866 /// \returns Addressable number of VGPRs supported by the subtarget.
867 unsigned getAddressableNumVGPRs(unsigned DynamicVGPRBlockSize) const {
868 // Dynamic VGPR mode is a per-kernel mode, so it is not covered by the
869 // TargetParser query.
870 if (DynamicVGPRBlockSize != 0) {
871 return AMDGPU::IsaInfo::getAddressableNumVGPRs(STI: *this,
872 DynamicVGPRBlockSize);
873 }
874 return AMDGPU::getAddressableNumVGPRs(AK: getTargetID().getGPUKind(),
875 IsWave32: isWave32());
876 }
877
878 /// \returns the minimum number of VGPRs that will prevent achieving more than
879 /// the specified number of waves \p WavesPerEU.
880 unsigned getMinNumVGPRs(unsigned WavesPerEU,
881 unsigned DynamicVGPRBlockSize) const {
882 return AMDGPU::IsaInfo::getMinNumVGPRs(STI: *this, WavesPerEU,
883 DynamicVGPRBlockSize);
884 }
885
886 /// \returns the maximum number of VGPRs that can be used and still achieved
887 /// at least the specified number of waves \p WavesPerEU.
888 unsigned getMaxNumVGPRs(unsigned WavesPerEU,
889 unsigned DynamicVGPRBlockSize) const {
890 return AMDGPU::IsaInfo::getMaxNumVGPRs(STI: *this, WavesPerEU,
891 DynamicVGPRBlockSize);
892 }
893
894 /// \returns max num VGPRs. This is the common utility function
895 /// called by MachineFunction and Function variants of getMaxNumVGPRs.
896 unsigned
897 getBaseMaxNumVGPRs(const Function &F,
898 std::pair<unsigned, unsigned> NumVGPRBounds) const;
899
900 /// \returns Maximum number of VGPRs that meets number of waves per execution
901 /// unit requirement for function \p F, or number of VGPRs explicitly
902 /// requested using "amdgpu-num-vgpr" attribute attached to function \p F.
903 ///
904 /// \returns Value that meets number of waves per execution unit requirement
905 /// if explicitly requested value cannot be converted to integer, violates
906 /// subtarget's specifications, or does not meet number of waves per execution
907 /// unit requirement.
908 unsigned getMaxNumVGPRs(const Function &F) const;
909
910 /// Return a pair of maximum numbers of VGPRs and AGPRs that meet the number
911 /// of waves per execution unit required for the function \p MF.
912 std::pair<unsigned, unsigned> getMaxNumVectorRegs(const Function &F) const;
913
914 /// \returns Maximum number of VGPRs that meets number of waves per execution
915 /// unit requirement for function \p MF, or number of VGPRs explicitly
916 /// requested using "amdgpu-num-vgpr" attribute attached to function \p MF.
917 ///
918 /// \returns Value that meets number of waves per execution unit requirement
919 /// if explicitly requested value cannot be converted to integer, violates
920 /// subtarget's specifications, or does not meet number of waves per execution
921 /// unit requirement.
922 unsigned getMaxNumVGPRs(const MachineFunction &MF) const;
923
924 bool isWave32() const { return getWavefrontSize() == 32; }
925
926 bool isWave64() const { return getWavefrontSize() == 64; }
927
928 /// Returns if the wavesize of this subtarget is known reliable. This is false
929 /// only for the a default target-cpu that does not have an explicit
930 /// +wavefrontsize target feature.
931 bool isWaveSizeKnown() const {
932 return hasFeature(Feature: AMDGPU::FeatureWavefrontSize32) ||
933 hasFeature(Feature: AMDGPU::FeatureWavefrontSize64);
934 }
935
936 const TargetRegisterClass *getBoolRC() const {
937 return getRegisterInfo()->getBoolRC();
938 }
939
940 /// \returns Maximum number of work groups per compute unit supported by the
941 /// subtarget and limited by given \p FlatWorkGroupSize.
942 unsigned getMaxWorkGroupsPerCU(unsigned FlatWorkGroupSize) const override {
943 return AMDGPU::IsaInfo::getMaxWorkGroupsPerCU(STI: *this, FlatWorkGroupSize);
944 }
945
946 /// \returns Minimum flat work group size supported by the subtarget.
947 unsigned getMinFlatWorkGroupSize() const override {
948 return AMDGPU::getMinFlatWorkGroupSize();
949 }
950
951 /// \returns Maximum flat work group size supported by the subtarget.
952 unsigned getMaxFlatWorkGroupSize() const override {
953 return AMDGPU::getMaxFlatWorkGroupSize();
954 }
955
956 /// \returns Number of waves per execution unit required to support the given
957 /// \p FlatWorkGroupSize.
958 unsigned
959 getWavesPerEUForWorkGroup(unsigned FlatWorkGroupSize) const override {
960 return AMDGPU::IsaInfo::getWavesPerEUForWorkGroup(STI: *this, FlatWorkGroupSize);
961 }
962
963 /// \returns Minimum number of waves per execution unit supported by the
964 /// subtarget.
965 unsigned getMinWavesPerEU() const override {
966 return AMDGPU::getMinWavesPerEU();
967 }
968
969 void adjustSchedDependency(SUnit *Def, int DefOpIdx, SUnit *Use, int UseOpIdx,
970 SDep &Dep,
971 const TargetSchedModel *SchedModel) const override;
972
973 // \returns true if it's beneficial on this subtarget for the scheduler to
974 // cluster stores as well as loads.
975 bool shouldClusterStores() const { return getGeneration() >= GFX11; }
976
977 // \returns the number of address arguments from which to enable MIMG NSA
978 // on supported architectures.
979 unsigned getNSAThreshold(const MachineFunction &MF) const;
980
981 // \returns true if the subtarget has a hazard requiring an "s_nop 0"
982 // instruction before "s_sendmsg sendmsg(MSG_DEALLOC_VGPRS)".
983 bool requiresNopBeforeDeallocVGPRs() const { return !HasGFX1250Insts; }
984
985 // \returns true if the subtarget needs S_WAIT_ALU 0 before S_GETREG_B32 on
986 // STATUS, STATE_PRIV, EXCP_FLAG_PRIV, or EXCP_FLAG_USER.
987 bool requiresWaitIdleBeforeGetReg() const { return HasGFX1250Insts; }
988
989 bool requiresDisjointEarlyClobberAndUndef() const override {
990 // AMDGPU doesn't care if early-clobber and undef operands are allocated
991 // to the same register.
992 return false;
993 }
994
995 // DS_ATOMIC_ASYNC_BARRIER_ARRIVE_B64 shall not be claused with anything
996 // and surronded by S_WAIT_ALU(0xFFE3).
997 bool hasDsAtomicAsyncBarrierArriveB64PipeBug() const {
998 return getGeneration() == GFX12;
999 }
1000
1001 // Requires s_wait_alu(0) after s102/s103 write and src_flat_scratch_base
1002 // read.
1003 bool hasScratchBaseForwardingHazard() const {
1004 return HasGFX1250Insts && getGeneration() == GFX12;
1005 }
1006
1007 // src_flat_scratch_hi cannot be used as a source in SALU producing a 64-bit
1008 // result.
1009 bool hasFlatScratchHiInB64InstHazard() const {
1010 return HasGFX1250Insts && getGeneration() == GFX12;
1011 }
1012
1013 /// \returns true if the subtarget requires a wait for xcnt before VMEM
1014 /// accesses that must never be repeated in the event of a page fault/re-try.
1015 /// Atomic stores/rmw and all volatile accesses fall under this criteria.
1016 bool requiresWaitXCntForSingleAccessInstructions() const {
1017 return HasGFX1250Insts;
1018 }
1019
1020 /// \returns the number of significant bits in the immediate field of the
1021 /// S_NOP instruction.
1022 unsigned getSNopBits() const {
1023 if (getGeneration() >= AMDGPUSubtarget::GFX12)
1024 return 7;
1025 if (getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS)
1026 return 4;
1027 return 3;
1028 }
1029
1030 bool supportsBPermute() const {
1031 return getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS;
1032 }
1033
1034 bool supportsWaveWideBPermute() const {
1035 return (getGeneration() <= AMDGPUSubtarget::GFX9 ||
1036 getGeneration() == AMDGPUSubtarget::GFX12) ||
1037 isWave32();
1038 }
1039
1040 /// Return true if real (non-fake) variants of True16 instructions using
1041 /// 16-bit registers should be code-generated. Fake True16 instructions are
1042 /// identical to non-fake ones except that they take 32-bit registers as
1043 /// operands and always use their low halves.
1044 // TODO: Remove and use hasTrue16BitInsts() instead once True16 is fully
1045 // supported and the support for fake True16 instructions is removed.
1046 bool useRealTrue16Insts() const {
1047 return hasTrue16BitInsts() && EnableRealTrue16Insts;
1048 }
1049
1050 bool requiresWaitOnWorkgroupReleaseFence(bool TgSplit) const {
1051 return getGeneration() >= GFX10 || TgSplit;
1052 }
1053
1054 bool useDFAforSMS() const override { return false; }
1055
1056 bool enableWindowScheduler() const override { return false; }
1057
1058 // \returns true if ISel should select the native i64 min/max instructions
1059 // (V_MIN/MAX_{I|U}64).
1060 bool useMinMaxI64Insts() const {
1061 return hasMinMaxI64Insts() && !hasSlowMaxMinMulI64Insts();
1062 }
1063
1064 // \returns true if ISel should select the native i64 mul instruction
1065 // V_MUL_U64.
1066 bool useVMulU64Inst() const {
1067 return hasVMulU64Inst() && !hasSlowMaxMinMulI64Insts();
1068 }
1069};
1070
1071class GCNUserSGPRUsageInfo {
1072public:
1073 bool hasImplicitBufferPtr() const { return ImplicitBufferPtr; }
1074
1075 bool hasPrivateSegmentBuffer() const { return PrivateSegmentBuffer; }
1076
1077 bool hasDispatchPtr() const { return DispatchPtr; }
1078
1079 bool hasQueuePtr() const { return QueuePtr; }
1080
1081 bool hasKernargSegmentPtr() const { return KernargSegmentPtr; }
1082
1083 bool hasDispatchID() const { return DispatchID; }
1084
1085 bool hasFlatScratchInit() const { return FlatScratchInit; }
1086
1087 bool hasPrivateSegmentSize() const { return PrivateSegmentSize; }
1088
1089 unsigned getNumKernargPreloadSGPRs() const { return NumKernargPreloadSGPRs; }
1090
1091 unsigned getNumFreeUserSGPRs();
1092
1093 void allocKernargPreloadSGPRs(unsigned NumSGPRs);
1094
1095 enum UserSGPRID : unsigned {
1096 ImplicitBufferPtrID = 0,
1097 PrivateSegmentBufferID = 1,
1098 DispatchPtrID = 2,
1099 QueuePtrID = 3,
1100 KernargSegmentPtrID = 4,
1101 DispatchIdID = 5,
1102 FlatScratchInitID = 6,
1103 PrivateSegmentSizeID = 7
1104 };
1105
1106 // Returns the size in number of SGPRs for preload user SGPR field.
1107 static unsigned getNumUserSGPRForField(UserSGPRID ID) {
1108 switch (ID) {
1109 case ImplicitBufferPtrID:
1110 return 2;
1111 case PrivateSegmentBufferID:
1112 return 4;
1113 case DispatchPtrID:
1114 return 2;
1115 case QueuePtrID:
1116 return 2;
1117 case KernargSegmentPtrID:
1118 return 2;
1119 case DispatchIdID:
1120 return 2;
1121 case FlatScratchInitID:
1122 return 2;
1123 case PrivateSegmentSizeID:
1124 return 1;
1125 }
1126 llvm_unreachable("Unknown UserSGPRID.");
1127 }
1128
1129 GCNUserSGPRUsageInfo(const Function &F, const GCNSubtarget &ST);
1130
1131private:
1132 const GCNSubtarget &ST;
1133
1134 // Private memory buffer
1135 // Compute directly in sgpr[0:1]
1136 // Other shaders indirect 64-bits at sgpr[0:1]
1137 bool ImplicitBufferPtr = false;
1138
1139 bool PrivateSegmentBuffer = false;
1140
1141 bool DispatchPtr = false;
1142
1143 bool QueuePtr = false;
1144
1145 bool KernargSegmentPtr = false;
1146
1147 bool DispatchID = false;
1148
1149 bool FlatScratchInit = false;
1150
1151 bool PrivateSegmentSize = false;
1152
1153 unsigned NumKernargPreloadSGPRs = 0;
1154
1155 unsigned NumUsedUserSGPRs = 0;
1156};
1157
1158} // end namespace llvm
1159
1160#endif // LLVM_LIB_TARGET_AMDGPU_GCNSUBTARGET_H
1161