1//=====-- GCNSubtarget.h - Define GCN Subtarget for AMDGPU ------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//==-----------------------------------------------------------------------===//
8//
9/// \file
10/// AMD GCN specific subclass of TargetSubtarget.
11//
12//===----------------------------------------------------------------------===//
13
14#ifndef LLVM_LIB_TARGET_AMDGPU_GCNSUBTARGET_H
15#define LLVM_LIB_TARGET_AMDGPU_GCNSUBTARGET_H
16
17#include "AMDGPUCallLowering.h"
18#include "AMDGPURegisterBankInfo.h"
19#include "AMDGPUSubtarget.h"
20#include "SIFrameLowering.h"
21#include "SIISelLowering.h"
22#include "SIInstrInfo.h"
23#include "Utils/AMDGPUBaseInfo.h"
24#include "llvm/Support/AMDHSAKernelDescriptor.h"
25#include "llvm/Support/ErrorHandling.h"
26
27#define GET_SUBTARGETINFO_HEADER
28#include "AMDGPUGenSubtargetInfo.inc"
29
30namespace llvm {
31
32class GCNTargetMachine;
33
34/// Module flag names controlling out-of-bounds buffer access semantics.
35/// Each flag is an i32 with Module::Max merge behaviour and tri-state values:
36/// 0 = any (absent/default - backend currently treats as strict)
37/// 1 = relaxed
38/// 2 = strict
39namespace AMDGPUOOBMode {
40inline constexpr StringLiteral BufferFlag("amdgpu.buffer.oob.mode");
41inline constexpr StringLiteral TBufferFlag("amdgpu.tbuffer.oob.mode");
42} // namespace AMDGPUOOBMode
43
44class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
45 public AMDGPUSubtarget {
46public:
47 using AMDGPUSubtarget::getMaxWavesPerEU;
48
49 // Following 2 enums are documented at:
50 // - https://llvm.org/docs/AMDGPUUsage.html#trap-handler-abi
51 enum class TrapHandlerAbi {
52 NONE = 0x00,
53 AMDHSA = 0x01,
54 };
55
56 enum class TrapID {
57 LLVMAMDHSATrap = 0x02,
58 LLVMAMDHSADebugTrap = 0x03,
59 };
60
61private:
62 /// SelectionDAGISel related APIs.
63 std::unique_ptr<const SelectionDAGTargetInfo> TSInfo;
64
65 /// GlobalISel related APIs.
66 std::unique_ptr<AMDGPUCallLowering> CallLoweringInfo;
67 std::unique_ptr<InlineAsmLowering> InlineAsmLoweringInfo;
68 std::unique_ptr<InstructionSelector> InstSelector;
69 std::unique_ptr<LegalizerInfo> Legalizer;
70 std::unique_ptr<AMDGPURegisterBankInfo> RegBankInfo;
71
72protected:
73 // Basic subtarget description.
74 AMDGPU::TargetID TargetID;
75 unsigned Gen = INVALID;
76 InstrItineraryData InstrItins;
77 int LDSBankCount = 0;
78 unsigned MaxPrivateElementSize = 0;
79
80 // Instruction cache line size in bytes; set from TableGen subtarget features.
81 unsigned InstCacheLineSize = 0;
82
83 // Dynamically set bits that enable features.
84 bool DynamicVGPR = false;
85 bool DynamicVGPRBlockSize32 = false;
86 bool ScalarizeGlobal = false;
87 const bool BufferOOBRelaxed;
88 const bool TBufferOOBRelaxed;
89
90 /// The maximum number of instructions that may be placed within an S_CLAUSE,
91 /// which is one greater than the maximum argument to S_CLAUSE. A value of 0
92 /// indicates a lack of S_CLAUSE support.
93 unsigned MaxHardClauseLength = 0;
94
95#define GET_SUBTARGETINFO_MACRO(ATTRIBUTE, DEFAULT, GETTER) \
96 bool ATTRIBUTE = DEFAULT;
97#include "AMDGPUGenSubtargetInfo.inc"
98
99private:
100 SIInstrInfo InstrInfo;
101 SITargetLowering TLInfo;
102 SIFrameLowering FrameLowering;
103
104 /// Get the register that represents the actual dependency between the
105 /// definition and the use. The definition might only affect a subregister
106 /// that is not actually used. Works for both virtual and physical registers.
107 /// Note: Currently supports VOP3P instructions (without WMMA an SWMMAC).
108 /// Returns the definition register if there is a real dependency and no
109 /// better match is found.
110 Register getRealSchedDependency(const MachineInstr &DefI, int DefOpIdx,
111 const MachineInstr &UseI, int UseOpIdx) const;
112
113public:
114 GCNSubtarget(const Triple &TT, StringRef GPU, StringRef FS,
115 const GCNTargetMachine &TM, bool BufferOOBRelaxed = false,
116 bool TBufferOOBRelaxed = false);
117 ~GCNSubtarget() override;
118
119 GCNSubtarget &initializeSubtargetDependencies(const Triple &TT, StringRef GPU,
120 StringRef FS);
121
122 /// Diagnose inconsistent subtarget features before attempting to codegen
123 /// function \p F.
124 void checkSubtargetFeatures(const Function &F) const;
125
126 const SIInstrInfo *getInstrInfo() const override { return &InstrInfo; }
127
128 const SIFrameLowering *getFrameLowering() const override {
129 return &FrameLowering;
130 }
131
132 const SITargetLowering *getTargetLowering() const override { return &TLInfo; }
133
134 const SIRegisterInfo *getRegisterInfo() const override {
135 return &InstrInfo.getRegisterInfo();
136 }
137
138 const SelectionDAGTargetInfo *getSelectionDAGInfo() const override;
139
140 const CallLowering *getCallLowering() const override {
141 return CallLoweringInfo.get();
142 }
143
144 const InlineAsmLowering *getInlineAsmLowering() const override {
145 return InlineAsmLoweringInfo.get();
146 }
147
148 InstructionSelector *getInstructionSelector() const override {
149 return InstSelector.get();
150 }
151
152 const LegalizerInfo *getLegalizerInfo() const override {
153 return Legalizer.get();
154 }
155
156 const AMDGPURegisterBankInfo *getRegBankInfo() const override {
157 return RegBankInfo.get();
158 }
159
160 const AMDGPU::TargetID &getTargetID() const { return TargetID; }
161
162 const InstrItineraryData *getInstrItineraryData() const override {
163 return &InstrItins;
164 }
165
166 void ParseSubtargetFeatures(StringRef CPU, StringRef TuneCPU, StringRef FS);
167
168 Generation getGeneration() const { return (Generation)Gen; }
169
170 bool isGFX11Plus() const { return getGeneration() >= GFX11; }
171
172#define GET_SUBTARGETINFO_MACRO(ATTRIBUTE, DEFAULT, GETTER) \
173 bool GETTER() const override { return ATTRIBUTE; }
174#include "AMDGPUGenSubtargetInfo.inc"
175
176 unsigned getMaxWaveScratchSize() const {
177 // See COMPUTE_TMPRING_SIZE.WAVESIZE.
178 if (getGeneration() >= GFX12) {
179 // 18-bit field in units of 64-dword.
180 return (64 * 4) * ((1 << 18) - 1);
181 }
182 if (getGeneration() == GFX11) {
183 // 15-bit field in units of 64-dword.
184 return (64 * 4) * ((1 << 15) - 1);
185 }
186 // 13-bit field in units of 256-dword.
187 return (256 * 4) * ((1 << 13) - 1);
188 }
189
190 /// Return the number of high bits known to be zero for a frame index.
191 unsigned getKnownHighZeroBitsForFrameIndex() const {
192 return llvm::countl_zero(Val: getMaxWaveScratchSize()) + getWavefrontSizeLog2();
193 }
194
195 int getLDSBankCount() const { return LDSBankCount; }
196
197 /// Instruction cache line size in bytes (64 for pre-GFX11, 128 for GFX11+).
198 unsigned getInstCacheLineSize() const { return InstCacheLineSize; }
199
200 unsigned getMaxPrivateElementSize(bool ForBufferRSrc = false) const {
201 return (ForBufferRSrc || !hasFlatScratchEnabled()) ? MaxPrivateElementSize
202 : 16;
203 }
204
205 unsigned getConstantBusLimit(unsigned Opcode) const;
206
207 /// Returns if the result of this instruction with a 16-bit result returned in
208 /// a 32-bit register implicitly zeroes the high 16-bits, rather than preserve
209 /// the original value.
210 bool zeroesHigh16BitsOfDest(unsigned Opcode) const;
211
212 bool supportsWGP() const {
213 if (HasGFX1250Insts)
214 return false;
215 return getGeneration() >= GFX10;
216 }
217
218 bool hasHWFP64() const { return HasFP64; }
219
220 bool hasAddr64() const {
221 return (getGeneration() < AMDGPUSubtarget::VOLCANIC_ISLANDS);
222 }
223
224 bool hasFlat() const {
225 return (getGeneration() > AMDGPUSubtarget::SOUTHERN_ISLANDS);
226 }
227
228 // Return true if the target only has the reverse operand versions of VALU
229 // shift instructions (e.g. v_lshrrev_b32, and no v_lshr_b32).
230 bool hasOnlyRevVALUShifts() const {
231 return getGeneration() >= VOLCANIC_ISLANDS;
232 }
233
234 bool hasFractBug() const { return getGeneration() == SOUTHERN_ISLANDS; }
235
236 bool hasMed3_16() const { return getGeneration() >= AMDGPUSubtarget::GFX9; }
237
238 bool hasMin3Max3_16() const {
239 return getGeneration() >= AMDGPUSubtarget::GFX9;
240 }
241
242 bool hasSwap() const { return HasGFX9Insts; }
243
244 bool hasScalarPackInsts() const { return HasGFX9Insts; }
245
246 bool hasScalarMulHiInsts() const { return HasGFX9Insts; }
247
248 bool hasScalarSubwordLoads() const { return getGeneration() >= GFX12; }
249
250 bool hasAsyncMark() const { return hasVMemToLDSLoad() || HasAsynccnt; }
251
252 TrapHandlerAbi getTrapHandlerAbi() const {
253 return isAmdHsaOS() ? TrapHandlerAbi::AMDHSA : TrapHandlerAbi::NONE;
254 }
255
256 bool supportsGetDoorbellID() const {
257 // The S_GETREG DOORBELL_ID is supported by all GFX9 onward targets.
258 return getGeneration() >= GFX9;
259 }
260
261 /// True if the offset field of DS instructions works as expected. On SI, the
262 /// offset uses a 16-bit adder and does not always wrap properly.
263 bool hasUsableDSOffset() const { return getGeneration() >= SEA_ISLANDS; }
264
265 bool unsafeDSOffsetFoldingEnabled() const {
266 return EnableUnsafeDSOffsetFolding;
267 }
268
269 /// Condition output from div_scale is usable.
270 bool hasUsableDivScaleConditionOutput() const {
271 return getGeneration() != SOUTHERN_ISLANDS;
272 }
273
274 /// Extra wait hazard is needed in some cases before
275 /// s_cbranch_vccnz/s_cbranch_vccz.
276 bool hasReadVCCZBug() const { return getGeneration() <= SEA_ISLANDS; }
277
278 /// Writes to VCC_LO/VCC_HI update the VCCZ flag.
279 bool partialVCCWritesUpdateVCCZ() const { return getGeneration() >= GFX10; }
280
281 /// A read of an SGPR by SMRD instruction requires 4 wait states when the SGPR
282 /// was written by a VALU instruction.
283 bool hasSMRDReadVALUDefHazard() const {
284 return getGeneration() == SOUTHERN_ISLANDS;
285 }
286
287 /// A read of an SGPR by a VMEM instruction requires 5 wait states when the
288 /// SGPR was written by a VALU Instruction.
289 bool hasVMEMReadSGPRVALUDefHazard() const {
290 return getGeneration() >= VOLCANIC_ISLANDS;
291 }
292
293 bool hasRFEHazards() const { return getGeneration() >= VOLCANIC_ISLANDS; }
294
295 /// Number of hazard wait states for s_setreg_b32/s_setreg_imm32_b32.
296 unsigned getSetRegWaitStates() const {
297 return getGeneration() <= SEA_ISLANDS ? 1 : 2;
298 }
299
300 /// Return the amount of LDS that can be used that will not restrict the
301 /// occupancy lower than WaveCount.
302 unsigned getMaxLocalMemSizeWithWaveCount(unsigned WaveCount,
303 const Function &) const;
304
305 bool supportsMinMaxDenormModes() const {
306 return getGeneration() >= AMDGPUSubtarget::GFX9;
307 }
308
309 /// \returns If target supports S_DENORM_MODE.
310 bool hasDenormModeInst() const {
311 return getGeneration() >= AMDGPUSubtarget::GFX10;
312 }
313
314 /// \returns If target supports ds_read/write_b128 and user enables generation
315 /// of ds_read/write_b128.
316 bool useDS128() const { return HasCIInsts && EnableDS128; }
317
318 /// \return If target supports ds_read/write_b96/128.
319 bool hasDS96AndDS128() const { return HasCIInsts; }
320
321 /// Have v_trunc_f64, v_ceil_f64, v_rndne_f64
322 bool haveRoundOpsF64() const { return HasCIInsts; }
323
324 /// \returns If MUBUF instructions always perform range checking, even for
325 /// buffer resources used for private memory access.
326 bool privateMemoryResourceIsRangeChecked() const {
327 return getGeneration() < AMDGPUSubtarget::GFX9;
328 }
329
330 /// \returns If target requires PRT Struct NULL support (zero result registers
331 /// for sparse texture support).
332 bool usePRTStrictNull() const { return EnablePRTStrictNull; }
333
334 bool hasUnalignedBufferAccessEnabled() const {
335 return HasUnalignedBufferAccess && HasUnalignedAccessMode;
336 }
337
338 bool hasUnalignedDSAccessEnabled() const {
339 return HasUnalignedDSAccess && HasUnalignedAccessMode;
340 }
341
342 bool hasUnalignedScratchAccessEnabled() const {
343 return HasUnalignedScratchAccess && HasUnalignedAccessMode;
344 }
345
346 bool isXNACKEnabled() const { return TargetID.isXnackOnOrAny(); }
347
348 bool hasRelaxedBufferOOBMode() const { return BufferOOBRelaxed; }
349 bool hasRelaxedTBufferOOBMode() const { return TBufferOOBRelaxed; }
350
351 bool isCuModeEnabled() const { return EnableCuMode; }
352
353 bool isPreciseMemoryEnabled() const { return EnablePreciseMemory; }
354
355 bool hasFlatScrRegister() const { return hasFlatAddressSpace(); }
356
357 // Check if target supports ST addressing mode with FLAT scratch instructions.
358 // The ST addressing mode means no registers are used, either VGPR or SGPR,
359 // but only immediate offset is swizzled and added to the FLAT scratch base.
360 bool hasFlatScratchSTMode() const {
361 return hasFlatScratchInsts() && (hasGFX10_3Insts() || hasGFX940Insts());
362 }
363
364 bool hasFlatScratchSVSMode() const { return HasGFX940Insts || HasGFX11Insts; }
365
366 bool hasFlatScratchEnabled() const {
367 return hasArchitectedFlatScratch() ||
368 (EnableFlatScratch && hasFlatScratchInsts());
369 }
370
371 bool hasGlobalAddTidInsts() const { return HasGFX10_BEncoding; }
372
373 bool hasAtomicCSub() const { return HasGFX10_BEncoding; }
374
375 bool hasExportInsts() const {
376 return !hasGFX940Insts() && !hasGFX1250Insts();
377 }
378
379 bool hasVINTERPEncoding() const {
380 return HasGFX11Insts && !hasGFX1250Insts();
381 }
382
383 // DS_ADD_F64/DS_ADD_RTN_F64
384 bool hasLdsAtomicAddF64() const {
385 return hasGFX90AInsts() || hasGFX1250Insts();
386 }
387
388 bool hasMultiDwordFlatScratchAddressing() const {
389 return getGeneration() >= GFX9;
390 }
391
392 bool hasFlatLgkmVMemCountInOrder() const { return getGeneration() > GFX9; }
393
394 bool hasD16LoadStore() const { return getGeneration() >= GFX9; }
395
396 bool d16PreservesUnusedBits() const {
397 return hasD16LoadStore() && !TargetID.isSramEccOnOrAny();
398 }
399
400 bool hasD16Images() const { return getGeneration() >= VOLCANIC_ISLANDS; }
401
402 /// Return if most LDS instructions have an m0 use that require m0 to be
403 /// initialized.
404 bool ldsRequiresM0Init() const { return getGeneration() < GFX9; }
405
406 // True if the hardware rewinds and replays GWS operations if a wave is
407 // preempted.
408 //
409 // If this is false, a GWS operation requires testing if a nack set the
410 // MEM_VIOL bit, and repeating if so.
411 bool hasGWSAutoReplay() const { return getGeneration() >= GFX9; }
412
413 /// \returns if target has ds_gws_sema_release_all instruction.
414 bool hasGWSSemaReleaseAll() const { return HasCIInsts; }
415
416 bool hasScalarAddSub64() const { return getGeneration() >= GFX12; }
417
418 bool hasScalarSMulU64() const { return getGeneration() >= GFX12; }
419
420 // Covers VS/PS/CS graphics shaders
421 bool isMesaGfxShader(const Function &F) const {
422 return isMesa3DOS() && AMDGPU::isShader(CC: F.getCallingConv());
423 }
424
425 bool hasMad64_32() const { return getGeneration() >= SEA_ISLANDS; }
426
427 bool hasAtomicFaddInsts() const {
428 return HasAtomicFaddRtnInsts || HasAtomicFaddNoRtnInsts;
429 }
430
431 bool vmemWriteNeedsExpWaitcnt() const {
432 return getGeneration() < SEA_ISLANDS;
433 }
434
435 bool hasInstPrefetch() const {
436 return getGeneration() == GFX10 || getGeneration() == GFX11;
437 }
438
439 bool hasPrefetch() const { return HasGFX12Insts; }
440
441 bool hasInstPrefSize() const { return isGFX11Plus(); }
442
443 void getInstPrefSizeArgs(uint32_t &Mask, uint32_t &Shift, uint32_t &Width,
444 uint32_t &CacheLineSize) const {
445 assert(isGFX11Plus());
446 CacheLineSize = getInstCacheLineSize();
447 if (getGeneration() == GFX11) {
448 Mask = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE;
449 Shift = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_SHIFT;
450 Width = amdhsa::COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE_WIDTH;
451 } else {
452 Mask = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE;
453 Shift = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_SHIFT;
454 Width = amdhsa::COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE_WIDTH;
455 }
456 }
457
458 // Has s_cmpk_* instructions.
459 bool hasSCmpK() const { return getGeneration() < GFX12; }
460
461 // Scratch is allocated in 256 dword per wave blocks for the entire
462 // wavefront. When viewed from the perspective of an arbitrary workitem, this
463 // is 4-byte aligned.
464 //
465 // Only 4-byte alignment is really needed to access anything. Transformations
466 // on the pointer value itself may rely on the alignment / known low bits of
467 // the pointer. Set this to something above the minimum to avoid needing
468 // dynamic realignment in common cases.
469 Align getStackAlignment() const { return Align(16); }
470
471 bool enableMachineScheduler() const override { return true; }
472
473 bool useAA() const override;
474
475 bool enableSubRegLiveness() const override { return true; }
476
477 void setScalarizeGlobalBehavior(bool b) { ScalarizeGlobal = b; }
478 bool getScalarizeGlobalBehavior() const { return ScalarizeGlobal; }
479
480 // static wrappers
481 static bool hasHalfRate64Ops(const TargetSubtargetInfo &STI);
482
483 // XXX - Why is this here if it isn't in the default pass set?
484 bool enableEarlyIfConversion() const override { return true; }
485
486 void overrideSchedPolicy(MachineSchedPolicy &Policy,
487 const SchedRegion &Region) const override;
488
489 void overridePostRASchedPolicy(MachineSchedPolicy &Policy,
490 const SchedRegion &Region) const override;
491
492 void mirFileLoaded(MachineFunction &MF) const override;
493
494 unsigned getMaxNumUserSGPRs() const {
495 return AMDGPU::getMaxNumUserSGPRs(STI: *this);
496 }
497
498 bool useVGPRIndexMode() const;
499
500 bool hasScalarCompareEq64() const {
501 return getGeneration() >= VOLCANIC_ISLANDS;
502 }
503
504 bool hasLDSFPAtomicAddF32() const { return HasGFX8Insts; }
505 bool hasLDSFPAtomicAddF64() const {
506 return HasGFX90AInsts || HasGFX1250Insts;
507 }
508
509 /// \returns true if the subtarget has the v_permlane64_b32 instruction.
510 bool hasPermLane64() const { return getGeneration() >= GFX11; }
511
512 /// \returns true if the subtarget supports the ds_swizzle rotate and FFT
513 /// swizzle modes (GFX9+).
514 bool hasDsSwizzleRotateMode() const { return getGeneration() >= GFX9; }
515
516 bool hasDPPRowShare() const {
517 return HasDPP && (HasGFX90AInsts || getGeneration() >= GFX10);
518 }
519
520 // Has V_PK_MOV_B32 opcode
521 bool hasPkMovB32() const { return HasGFX90AInsts; }
522
523 bool hasFmaakFmamkF32Insts() const {
524 return getGeneration() >= GFX10 || hasGFX940Insts();
525 }
526
527 bool hasFmaakFmamkF64Insts() const { return hasGFX1250Insts(); }
528
529 bool hasNonNSAEncoding() const { return getGeneration() < GFX12; }
530
531 unsigned getNSAMaxSize(bool HasSampler = false) const {
532 return AMDGPU::getNSAMaxSize(STI: *this, HasSampler);
533 }
534
535 bool hasMadF16() const;
536
537 // Scalar and global loads support scale_offset bit.
538 bool hasScaleOffset() const { return HasGFX1250Insts; }
539
540 // FLAT GLOBAL VOffset is signed
541 bool hasSignedGVSOffset() const { return HasGFX1250Insts; }
542
543 bool loadStoreOptEnabled() const { return EnableLoadStoreOpt; }
544
545 bool hasUserSGPRInit16BugInWave32() const {
546 return HasUserSGPRInit16Bug && isWave32();
547 }
548
549 bool has12DWordStoreHazard() const {
550 return getGeneration() != AMDGPUSubtarget::SOUTHERN_ISLANDS;
551 }
552
553 // \returns true if the subtarget supports DWORDX3 load/store instructions.
554 bool hasDwordx3LoadStores() const { return HasCIInsts; }
555
556 bool hasReadM0MovRelInterpHazard() const {
557 return getGeneration() == AMDGPUSubtarget::GFX9;
558 }
559
560 bool hasReadM0SendMsgHazard() const {
561 return getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS &&
562 getGeneration() <= AMDGPUSubtarget::GFX9;
563 }
564
565 bool hasReadM0LdsDmaHazard() const {
566 return getGeneration() == AMDGPUSubtarget::GFX9;
567 }
568
569 bool hasReadM0LdsDirectHazard() const {
570 return getGeneration() == AMDGPUSubtarget::GFX9;
571 }
572
573 bool hasLDSMisalignedBugInWGPMode() const {
574 return HasLDSMisalignedBug && !EnableCuMode;
575 }
576
577 // Shift amount of a 64 bit shift cannot be a highest allocated register
578 // if also at the end of the allocation block.
579 bool hasShift64HighRegBug() const { return HasGFX90AInsts; }
580
581 // Has one cycle hazard on transcendental instruction feeding a
582 // non transcendental VALU.
583 bool hasTransForwardingHazard() const { return HasGFX940Insts; }
584
585 // Has one cycle hazard on a VALU instruction partially writing dst with
586 // a shift of result bits feeding another VALU instruction.
587 bool hasDstSelForwardingHazard() const { return HasGFX940Insts; }
588
589 // Cannot use op_sel with v_dot instructions.
590 bool hasDOTOpSelHazard() const { return HasGFX940Insts || HasGFX11Insts; }
591
592 // Does not have HW interlocs for VALU writing and then reading SGPRs.
593 bool hasVDecCoExecHazard() const { return HasGFX940Insts; }
594
595 bool hasHardClauses() const { return MaxHardClauseLength > 0; }
596
597 bool hasFPAtomicToDenormModeHazard() const {
598 return getGeneration() == GFX10;
599 }
600
601 bool hasVOP3DPP() const { return getGeneration() >= GFX11; }
602
603 bool hasLdsDirect() const { return getGeneration() >= GFX11; }
604
605 bool hasLdsWaitVMSRC() const { return getGeneration() >= GFX12; }
606
607 bool hasVALUPartialForwardingHazard() const {
608 return getGeneration() == GFX11;
609 }
610
611 bool hasCvtScaleForwardingHazard() const { return HasGFX950Insts; }
612
613 // All GFX9 targets experience a fetch delay when an instruction at the start
614 // of a loop header is split by a 32-byte fetch window boundary, but GFX950
615 // is uniquely sensitive to this: the delay triggers further performance
616 // degradation beyond the fetch latency itself.
617 bool hasLoopHeadInstSplitSensitivity() const { return HasGFX950Insts; }
618
619 bool requiresCodeObjectV6() const { return RequiresCOV6; }
620
621 bool useVGPRBlockOpsForCSR() const { return UseBlockVGPROpsForCSR; }
622
623 bool hasVALUMaskWriteHazard() const { return getGeneration() == GFX11; }
624
625 bool hasVALUReadSGPRHazard() const {
626 return HasGFX12Insts && !HasGFX1250Insts;
627 }
628
629 bool setRegModeNeedsVNOPs() const {
630 return HasGFX1250Insts && getGeneration() == GFX12;
631 }
632
633 /// Return if operations acting on VGPR tuples require even alignment.
634 bool needsAlignedVGPRs() const { return RequiresAlignVGPR; }
635
636 /// Return true if the target has the S_PACK_HL_B32_B16 instruction.
637 bool hasSPackHL() const { return HasGFX11Insts; }
638
639 /// Return true if the target has the V_CVT_PK_I16_F32/V_CVT_PK_U16_F32
640 /// instructions.
641 bool hasVCvtPkIU16F32() const { return HasGFX11Insts; }
642
643 /// Return true if the target's EXP instruction supports the NULL export
644 /// target.
645 bool hasNullExportTarget() const { return !HasGFX11Insts; }
646
647 bool hasFlatScratchSVSSwizzleBug() const { return getGeneration() == GFX11; }
648
649 /// Return true if the target has the S_DELAY_ALU instruction.
650 bool hasDelayAlu() const { return HasGFX11Insts; }
651
652 /// Returns true if the target supports
653 /// global_load_lds_dwordx3/global_load_lds_dwordx4 or
654 /// buffer_load_dwordx3/buffer_load_dwordx4 with the lds bit.
655 bool hasLDSLoadB96_B128() const { return hasGFX950Insts(); }
656
657 /// \returns true if the target uses LOADcnt/SAMPLEcnt/BVHcnt, DScnt/KMcnt
658 /// and STOREcnt rather than VMcnt, LGKMcnt and VScnt respectively.
659 bool hasExtendedWaitCounts() const { return getGeneration() >= GFX12; }
660
661 /// \returns true if the target has packed f32 instructions that only read 32
662 /// bits from a scalar operand (SGPR or literal) and replicates the bits to
663 /// both channels.
664 bool hasPKF32InstsReplicatingLower32BitsOfScalarInput() const {
665 return getGeneration() == GFX12 && HasGFX1250Insts;
666 }
667
668 bool hasAddPC64Inst() const { return HasGFX1250Insts; }
669
670 /// \returns true if the target supports expert scheduling mode 2 which relies
671 /// on the compiler to insert waits to avoid hazards between VMEM and VALU
672 /// instructions in some instances.
673 bool hasExpertSchedulingMode() const { return getGeneration() >= GFX12; }
674
675 /// \returns The maximum number of instructions that can be enclosed in an
676 /// S_CLAUSE on the given subtarget, or 0 for targets that do not support that
677 /// instruction.
678 unsigned maxHardClauseLength() const { return MaxHardClauseLength; }
679
680 /// Return the maximum number of waves per SIMD for kernels using \p SGPRs
681 /// SGPRs
682 unsigned getOccupancyWithNumSGPRs(unsigned SGPRs) const;
683
684 /// Return the maximum number of waves per SIMD for kernels using \p VGPRs
685 /// VGPRs
686 unsigned getOccupancyWithNumVGPRs(unsigned VGPRs,
687 unsigned DynamicVGPRBlockSize) const;
688
689 /// Subtarget's minimum/maximum occupancy, in number of waves per EU, that can
690 /// be achieved when the only function running on a CU is \p F, each workgroup
691 /// uses \p LDSSize bytes of LDS, and each wave uses \p NumSGPRs SGPRs and \p
692 /// NumVGPRs VGPRs. The flat workgroup sizes associated to the function are a
693 /// range, so this returns a range as well.
694 ///
695 /// Note that occupancy can be affected by the scratch allocation as well, but
696 /// we do not have enough information to compute it.
697 std::pair<unsigned, unsigned> computeOccupancy(const Function &F,
698 unsigned LDSSize = 0,
699 unsigned NumSGPRs = 0,
700 unsigned NumVGPRs = 0) const;
701
702 /// \returns true if the flat_scratch register should be initialized with the
703 /// pointer to the wave's scratch memory rather than a size and offset.
704 bool flatScratchIsPointer() const {
705 return getGeneration() >= AMDGPUSubtarget::GFX9;
706 }
707
708 /// \returns true if the machine has merged shaders in which s0-s7 are
709 /// reserved by the hardware and user SGPRs start at s8
710 bool hasMergedShaders() const { return getGeneration() >= GFX9; }
711
712 // \returns true if the target supports the pre-NGG legacy geometry path.
713 bool hasLegacyGeometry() const { return getGeneration() < GFX11; }
714
715 // \returns true if the target has split barriers feature
716 bool hasSplitBarriers() const { return getGeneration() >= GFX12; }
717
718 // \returns true if the target has WG_RR_MODE kernel descriptor mode bit
719 bool hasRrWGMode() const { return getGeneration() >= GFX12; }
720
721 /// \returns true if VADDR and SADDR fields in VSCRATCH can use negative
722 /// values.
723 bool hasSignedScratchOffsets() const { return getGeneration() >= GFX12; }
724
725 bool hasINVWBL2WaitCntRequirement() const { return HasGFX1250Insts; }
726
727 bool hasVOPD3() const { return HasGFX1250Insts; }
728
729 // \returns true if the target has V_PK_{MIN|MAX}3_{I|U}16 instructions.
730 bool hasPkMinMax3Insts() const { return HasGFX1250Insts; }
731
732 // \returns ture if target has S_GET_SHADER_CYCLES_U64 instruction.
733 bool hasSGetShaderCyclesInst() const { return HasGFX1250Insts; }
734
735 // \returns true if S_GETPC_B64 zero-extends the result from 48 bits instead
736 // of sign-extending. Note that GFX1250 has not only fixed the bug but also
737 // extended VA to 57 bits.
738 bool hasGetPCZeroExtension() const {
739 return HasGFX12Insts && !HasGFX1250Insts;
740 }
741
742 // \returns true if the target needs to create a prolog for backward
743 // compatibility when preloading kernel arguments.
744 bool needsKernArgPreloadProlog() const {
745 return hasKernargPreload() && !HasGFX1250Insts;
746 }
747
748 bool hasCondSubInsts() const { return HasGFX12Insts; }
749
750 bool hasSubClampInsts() const { return hasGFX10_3Insts(); }
751
752 /// \returns SGPR allocation granularity supported by the subtarget.
753 unsigned getSGPRAllocGranule() const {
754 return AMDGPU::IsaInfo::getSGPRAllocGranule(STI: *this);
755 }
756
757 /// \returns SGPR encoding granularity supported by the subtarget.
758 unsigned getSGPREncodingGranule() const {
759 return AMDGPU::IsaInfo::getSGPREncodingGranule(STI: *this);
760 }
761
762 /// \returns Total number of SGPRs supported by the subtarget.
763 unsigned getTotalNumSGPRs() const {
764 return AMDGPU::IsaInfo::getTotalNumSGPRs(STI: *this);
765 }
766
767 /// \returns Addressable number of SGPRs supported by the subtarget.
768 unsigned getAddressableNumSGPRs() const {
769 return AMDGPU::IsaInfo::getAddressableNumSGPRs(STI: *this);
770 }
771
772 /// \returns Minimum number of SGPRs that meets the given number of waves per
773 /// execution unit requirement supported by the subtarget.
774 unsigned getMinNumSGPRs(unsigned WavesPerEU) const {
775 return AMDGPU::IsaInfo::getMinNumSGPRs(STI: *this, WavesPerEU);
776 }
777
778 /// \returns Maximum number of SGPRs that meets the given number of waves per
779 /// execution unit requirement supported by the subtarget.
780 unsigned getMaxNumSGPRs(unsigned WavesPerEU, bool Addressable) const {
781 return AMDGPU::IsaInfo::getMaxNumSGPRs(STI: *this, WavesPerEU, Addressable);
782 }
783
784 /// \returns Reserved number of SGPRs. This is common
785 /// utility function called by MachineFunction and
786 /// Function variants of getReservedNumSGPRs.
787 unsigned getBaseReservedNumSGPRs(const bool HasFlatScratch) const;
788 /// \returns Reserved number of SGPRs for given machine function \p MF.
789 unsigned getReservedNumSGPRs(const MachineFunction &MF) const;
790
791 /// \returns Reserved number of SGPRs for given function \p F.
792 unsigned getReservedNumSGPRs(const Function &F) const;
793
794 /// \returns Maximum number of preloaded SGPRs for the subtarget.
795 unsigned getMaxNumPreloadedSGPRs() const;
796
797 /// \returns max num SGPRs. This is the common utility
798 /// function called by MachineFunction and Function
799 /// variants of getMaxNumSGPRs.
800 unsigned getBaseMaxNumSGPRs(const Function &F,
801 std::pair<unsigned, unsigned> WavesPerEU,
802 unsigned PreloadedSGPRs,
803 unsigned ReservedNumSGPRs) const;
804
805 /// \returns Maximum number of SGPRs that meets number of waves per execution
806 /// unit requirement for function \p MF, or number of SGPRs explicitly
807 /// requested using "amdgpu-num-sgpr" attribute attached to function \p MF.
808 ///
809 /// \returns Value that meets number of waves per execution unit requirement
810 /// if explicitly requested value cannot be converted to integer, violates
811 /// subtarget's specifications, or does not meet number of waves per execution
812 /// unit requirement.
813 unsigned getMaxNumSGPRs(const MachineFunction &MF) const;
814
815 /// \returns Maximum number of SGPRs that meets number of waves per execution
816 /// unit requirement for function \p F, or number of SGPRs explicitly
817 /// requested using "amdgpu-num-sgpr" attribute attached to function \p F.
818 ///
819 /// \returns Value that meets number of waves per execution unit requirement
820 /// if explicitly requested value cannot be converted to integer, violates
821 /// subtarget's specifications, or does not meet number of waves per execution
822 /// unit requirement.
823 unsigned getMaxNumSGPRs(const Function &F) const;
824
825 /// \returns VGPR allocation granularity supported by the subtarget.
826 unsigned getVGPRAllocGranule(unsigned DynamicVGPRBlockSize) const {
827 return AMDGPU::IsaInfo::getVGPRAllocGranule(STI: *this, DynamicVGPRBlockSize);
828 }
829
830 /// \returns VGPR encoding granularity supported by the subtarget.
831 unsigned getVGPREncodingGranule() const {
832 return AMDGPU::IsaInfo::getVGPREncodingGranule(STI: *this);
833 }
834
835 /// \returns Total number of VGPRs supported by the subtarget.
836 unsigned getTotalNumVGPRs() const {
837 return AMDGPU::IsaInfo::getTotalNumVGPRs(STI: *this);
838 }
839
840 /// \returns Addressable number of architectural VGPRs supported by the
841 /// subtarget.
842 unsigned getAddressableNumArchVGPRs() const {
843 return AMDGPU::IsaInfo::getAddressableNumArchVGPRs(STI: *this);
844 }
845
846 /// \returns Addressable number of VGPRs supported by the subtarget.
847 unsigned getAddressableNumVGPRs(unsigned DynamicVGPRBlockSize) const {
848 return AMDGPU::IsaInfo::getAddressableNumVGPRs(STI: *this, DynamicVGPRBlockSize);
849 }
850
851 /// \returns the minimum number of VGPRs that will prevent achieving more than
852 /// the specified number of waves \p WavesPerEU.
853 unsigned getMinNumVGPRs(unsigned WavesPerEU,
854 unsigned DynamicVGPRBlockSize) const {
855 return AMDGPU::IsaInfo::getMinNumVGPRs(STI: *this, WavesPerEU,
856 DynamicVGPRBlockSize);
857 }
858
859 /// \returns the maximum number of VGPRs that can be used and still achieved
860 /// at least the specified number of waves \p WavesPerEU.
861 unsigned getMaxNumVGPRs(unsigned WavesPerEU,
862 unsigned DynamicVGPRBlockSize) const {
863 return AMDGPU::IsaInfo::getMaxNumVGPRs(STI: *this, WavesPerEU,
864 DynamicVGPRBlockSize);
865 }
866
867 /// \returns max num VGPRs. This is the common utility function
868 /// called by MachineFunction and Function variants of getMaxNumVGPRs.
869 unsigned
870 getBaseMaxNumVGPRs(const Function &F,
871 std::pair<unsigned, unsigned> NumVGPRBounds) const;
872
873 /// \returns Maximum number of VGPRs that meets number of waves per execution
874 /// unit requirement for function \p F, or number of VGPRs explicitly
875 /// requested using "amdgpu-num-vgpr" attribute attached to function \p F.
876 ///
877 /// \returns Value that meets number of waves per execution unit requirement
878 /// if explicitly requested value cannot be converted to integer, violates
879 /// subtarget's specifications, or does not meet number of waves per execution
880 /// unit requirement.
881 unsigned getMaxNumVGPRs(const Function &F) const;
882
883 unsigned getMaxNumAGPRs(const Function &F) const { return getMaxNumVGPRs(F); }
884
885 /// Return a pair of maximum numbers of VGPRs and AGPRs that meet the number
886 /// of waves per execution unit required for the function \p MF.
887 std::pair<unsigned, unsigned> getMaxNumVectorRegs(const Function &F) const;
888
889 /// \returns Maximum number of VGPRs that meets number of waves per execution
890 /// unit requirement for function \p MF, or number of VGPRs explicitly
891 /// requested using "amdgpu-num-vgpr" attribute attached to function \p MF.
892 ///
893 /// \returns Value that meets number of waves per execution unit requirement
894 /// if explicitly requested value cannot be converted to integer, violates
895 /// subtarget's specifications, or does not meet number of waves per execution
896 /// unit requirement.
897 unsigned getMaxNumVGPRs(const MachineFunction &MF) const;
898
899 bool supportsWave32() const { return getGeneration() >= GFX10; }
900
901 bool supportsWave64() const { return !hasGFX1250Insts() || HasGFX13Insts; }
902
903 bool isWave32() const { return getWavefrontSize() == 32; }
904
905 bool isWave64() const { return getWavefrontSize() == 64; }
906
907 /// Returns if the wavesize of this subtarget is known reliable. This is false
908 /// only for the a default target-cpu that does not have an explicit
909 /// +wavefrontsize target feature.
910 bool isWaveSizeKnown() const {
911 return hasFeature(Feature: AMDGPU::FeatureWavefrontSize32) ||
912 hasFeature(Feature: AMDGPU::FeatureWavefrontSize64);
913 }
914
915 const TargetRegisterClass *getBoolRC() const {
916 return getRegisterInfo()->getBoolRC();
917 }
918
919 /// \returns Maximum number of work groups per compute unit supported by the
920 /// subtarget and limited by given \p FlatWorkGroupSize.
921 unsigned getMaxWorkGroupsPerCU(unsigned FlatWorkGroupSize) const override {
922 return AMDGPU::IsaInfo::getMaxWorkGroupsPerCU(STI: *this, FlatWorkGroupSize);
923 }
924
925 /// \returns Minimum flat work group size supported by the subtarget.
926 unsigned getMinFlatWorkGroupSize() const override {
927 return AMDGPU::IsaInfo::getMinFlatWorkGroupSize(STI: *this);
928 }
929
930 /// \returns Maximum flat work group size supported by the subtarget.
931 unsigned getMaxFlatWorkGroupSize() const override {
932 return AMDGPU::IsaInfo::getMaxFlatWorkGroupSize();
933 }
934
935 /// \returns Number of waves per execution unit required to support the given
936 /// \p FlatWorkGroupSize.
937 unsigned
938 getWavesPerEUForWorkGroup(unsigned FlatWorkGroupSize) const override {
939 return AMDGPU::IsaInfo::getWavesPerEUForWorkGroup(STI: *this, FlatWorkGroupSize);
940 }
941
942 /// \returns Minimum number of waves per execution unit supported by the
943 /// subtarget.
944 unsigned getMinWavesPerEU() const override {
945 return AMDGPU::IsaInfo::getMinWavesPerEU(STI: *this);
946 }
947
948 void adjustSchedDependency(SUnit *Def, int DefOpIdx, SUnit *Use, int UseOpIdx,
949 SDep &Dep,
950 const TargetSchedModel *SchedModel) const override;
951
952 // \returns true if it's beneficial on this subtarget for the scheduler to
953 // cluster stores as well as loads.
954 bool shouldClusterStores() const { return getGeneration() >= GFX11; }
955
956 // \returns the number of address arguments from which to enable MIMG NSA
957 // on supported architectures.
958 unsigned getNSAThreshold(const MachineFunction &MF) const;
959
960 // \returns true if the subtarget has a hazard requiring an "s_nop 0"
961 // instruction before "s_sendmsg sendmsg(MSG_DEALLOC_VGPRS)".
962 bool requiresNopBeforeDeallocVGPRs() const { return !HasGFX1250Insts; }
963
964 // \returns true if the subtarget needs S_WAIT_ALU 0 before S_GETREG_B32 on
965 // STATUS, STATE_PRIV, EXCP_FLAG_PRIV, or EXCP_FLAG_USER.
966 bool requiresWaitIdleBeforeGetReg() const { return HasGFX1250Insts; }
967
968 bool isDynamicVGPREnabled() const { return DynamicVGPR; }
969 unsigned getDynamicVGPRBlockSize() const {
970 return DynamicVGPRBlockSize32 ? 32 : 16;
971 }
972
973 bool requiresDisjointEarlyClobberAndUndef() const override {
974 // AMDGPU doesn't care if early-clobber and undef operands are allocated
975 // to the same register.
976 return false;
977 }
978
979 // DS_ATOMIC_ASYNC_BARRIER_ARRIVE_B64 shall not be claused with anything
980 // and surronded by S_WAIT_ALU(0xFFE3).
981 bool hasDsAtomicAsyncBarrierArriveB64PipeBug() const {
982 return getGeneration() == GFX12;
983 }
984
985 // Requires s_wait_alu(0) after s102/s103 write and src_flat_scratch_base
986 // read.
987 bool hasScratchBaseForwardingHazard() const {
988 return HasGFX1250Insts && getGeneration() == GFX12;
989 }
990
991 // src_flat_scratch_hi cannot be used as a source in SALU producing a 64-bit
992 // result.
993 bool hasFlatScratchHiInB64InstHazard() const {
994 return HasGFX1250Insts && getGeneration() == GFX12;
995 }
996
997 /// \returns true if the subtarget requires a wait for xcnt before VMEM
998 /// accesses that must never be repeated in the event of a page fault/re-try.
999 /// Atomic stores/rmw and all volatile accesses fall under this criteria.
1000 bool requiresWaitXCntForSingleAccessInstructions() const {
1001 return HasGFX1250Insts;
1002 }
1003
1004 /// \returns the number of significant bits in the immediate field of the
1005 /// S_NOP instruction.
1006 unsigned getSNopBits() const {
1007 if (getGeneration() >= AMDGPUSubtarget::GFX12)
1008 return 7;
1009 if (getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS)
1010 return 4;
1011 return 3;
1012 }
1013
1014 bool supportsBPermute() const {
1015 return getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS;
1016 }
1017
1018 bool supportsWaveWideBPermute() const {
1019 return (getGeneration() <= AMDGPUSubtarget::GFX9 ||
1020 getGeneration() == AMDGPUSubtarget::GFX12) ||
1021 isWave32();
1022 }
1023
1024 /// Return true if real (non-fake) variants of True16 instructions using
1025 /// 16-bit registers should be code-generated. Fake True16 instructions are
1026 /// identical to non-fake ones except that they take 32-bit registers as
1027 /// operands and always use their low halves.
1028 // TODO: Remove and use hasTrue16BitInsts() instead once True16 is fully
1029 // supported and the support for fake True16 instructions is removed.
1030 bool useRealTrue16Insts() const {
1031 return hasTrue16BitInsts() && EnableRealTrue16Insts;
1032 }
1033
1034 bool requiresWaitOnWorkgroupReleaseFence(bool TgSplit) const {
1035 return getGeneration() >= GFX10 || TgSplit;
1036 }
1037};
1038
1039class GCNUserSGPRUsageInfo {
1040public:
1041 bool hasImplicitBufferPtr() const { return ImplicitBufferPtr; }
1042
1043 bool hasPrivateSegmentBuffer() const { return PrivateSegmentBuffer; }
1044
1045 bool hasDispatchPtr() const { return DispatchPtr; }
1046
1047 bool hasQueuePtr() const { return QueuePtr; }
1048
1049 bool hasKernargSegmentPtr() const { return KernargSegmentPtr; }
1050
1051 bool hasDispatchID() const { return DispatchID; }
1052
1053 bool hasFlatScratchInit() const { return FlatScratchInit; }
1054
1055 bool hasPrivateSegmentSize() const { return PrivateSegmentSize; }
1056
1057 unsigned getNumKernargPreloadSGPRs() const { return NumKernargPreloadSGPRs; }
1058
1059 unsigned getNumUsedUserSGPRs() const { return NumUsedUserSGPRs; }
1060
1061 unsigned getNumFreeUserSGPRs();
1062
1063 void allocKernargPreloadSGPRs(unsigned NumSGPRs);
1064
1065 enum UserSGPRID : unsigned {
1066 ImplicitBufferPtrID = 0,
1067 PrivateSegmentBufferID = 1,
1068 DispatchPtrID = 2,
1069 QueuePtrID = 3,
1070 KernargSegmentPtrID = 4,
1071 DispatchIdID = 5,
1072 FlatScratchInitID = 6,
1073 PrivateSegmentSizeID = 7
1074 };
1075
1076 // Returns the size in number of SGPRs for preload user SGPR field.
1077 static unsigned getNumUserSGPRForField(UserSGPRID ID) {
1078 switch (ID) {
1079 case ImplicitBufferPtrID:
1080 return 2;
1081 case PrivateSegmentBufferID:
1082 return 4;
1083 case DispatchPtrID:
1084 return 2;
1085 case QueuePtrID:
1086 return 2;
1087 case KernargSegmentPtrID:
1088 return 2;
1089 case DispatchIdID:
1090 return 2;
1091 case FlatScratchInitID:
1092 return 2;
1093 case PrivateSegmentSizeID:
1094 return 1;
1095 }
1096 llvm_unreachable("Unknown UserSGPRID.");
1097 }
1098
1099 GCNUserSGPRUsageInfo(const Function &F, const GCNSubtarget &ST);
1100
1101private:
1102 const GCNSubtarget &ST;
1103
1104 // Private memory buffer
1105 // Compute directly in sgpr[0:1]
1106 // Other shaders indirect 64-bits at sgpr[0:1]
1107 bool ImplicitBufferPtr = false;
1108
1109 bool PrivateSegmentBuffer = false;
1110
1111 bool DispatchPtr = false;
1112
1113 bool QueuePtr = false;
1114
1115 bool KernargSegmentPtr = false;
1116
1117 bool DispatchID = false;
1118
1119 bool FlatScratchInit = false;
1120
1121 bool PrivateSegmentSize = false;
1122
1123 unsigned NumKernargPreloadSGPRs = 0;
1124
1125 unsigned NumUsedUserSGPRs = 0;
1126};
1127
1128} // end namespace llvm
1129
1130#endif // LLVM_LIB_TARGET_AMDGPU_GCNSUBTARGET_H
1131