1//===- AMDGPUBaseInfo.h - Top level definitions for AMDGPU ------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
9#ifndef LLVM_LIB_TARGET_AMDGPU_UTILS_AMDGPUBASEINFO_H
10#define LLVM_LIB_TARGET_AMDGPU_UTILS_AMDGPUBASEINFO_H
11
12#include "AMDGPUSubtarget.h"
13#include "SIDefines.h"
14#include "llvm/ADT/APFloat.h"
15#include "llvm/ADT/StringExtras.h"
16#include "llvm/ADT/StringTable.h"
17#include "llvm/IR/CallingConv.h"
18#include "llvm/IR/InstrTypes.h"
19#include "llvm/IR/Module.h"
20#include "llvm/Support/Alignment.h"
21#include "llvm/TargetParser/AMDGPUTargetParser.h"
22#include <array>
23#include <functional>
24#include <optional>
25#include <utility>
26
27// Pull in OpName enum definition and getNamedOperandIdx() declaration.
28#define GET_INSTRINFO_OPERAND_ENUM
29#include "AMDGPUGenInstrInfo.inc"
30
31struct amd_kernel_code_t;
32
33namespace llvm {
34
35struct Align;
36class Argument;
37class Function;
38class GlobalValue;
39class MachineInstr;
40class MCInstrInfo;
41class MCRegisterClass;
42class MCRegisterInfo;
43class MCSubtargetInfo;
44class MDNode;
45class StringRef;
46class Triple;
47class raw_ostream;
48
49namespace AMDGPU {
50
51struct AMDGPUMCKernelCodeT;
52struct IsaVersion;
53
54/// Generic target versions emitted by this version of LLVM.
55///
56/// These numbers are incremented every time a codegen breaking change occurs
57/// within a generic family.
58namespace GenericVersion {
59static constexpr unsigned GFX9 = 1;
60static constexpr unsigned GFX9_4 = 1;
61static constexpr unsigned GFX10_1 = 1;
62static constexpr unsigned GFX10_3 = 1;
63static constexpr unsigned GFX11 = 1;
64static constexpr unsigned GFX11_7 = 1;
65static constexpr unsigned GFX12 = 1;
66static constexpr unsigned GFX12_5 = 1;
67static constexpr unsigned GFX13 = 1;
68} // namespace GenericVersion
69
70enum { AMDHSA_COV4 = 4, AMDHSA_COV5 = 5, AMDHSA_COV6 = 6 };
71
72enum class FPType { None, FP4, FP8 };
73
74/// \returns True if \p STI is AMDHSA.
75bool isHsaAbi(const MCSubtargetInfo &STI);
76
77/// \returns Code object version from the IR module flag.
78unsigned getAMDHSACodeObjectVersion(const Module &M);
79
80/// \returns Code object version from ELF's e_ident[EI_ABIVERSION].
81unsigned getAMDHSACodeObjectVersion(unsigned ABIVersion);
82
83/// \returns The default HSA code object version. This should only be used when
84/// we lack a more accurate CodeObjectVersion value (e.g. from the IR module
85/// flag or a .amdhsa_code_object_version directive)
86unsigned getDefaultAMDHSACodeObjectVersion();
87
88/// \returns ABIVersion suitable for use in ELF's e_ident[EI_ABIVERSION]. \param
89/// CodeObjectVersion is a value returned by getAMDHSACodeObjectVersion().
90uint8_t getELFABIVersion(const Triple &OS, unsigned CodeObjectVersion);
91
92/// \returns The offset of the multigrid_sync_arg argument from implicitarg_ptr
93unsigned getMultigridSyncArgImplicitArgPosition(unsigned COV);
94
95/// \returns The offset of the hostcall pointer argument from implicitarg_ptr
96unsigned getHostcallImplicitArgPosition(unsigned COV);
97
98unsigned getDefaultQueueImplicitArgPosition(unsigned COV);
99unsigned getCompletionActionImplicitArgPosition(unsigned COV);
100
101struct GcnBufferFormatInfo {
102 unsigned Format;
103 unsigned BitsPerComp;
104 unsigned NumComponents;
105 unsigned NumFormat;
106 unsigned DataFormat;
107};
108
109struct MAIInstInfo {
110 uint32_t Opcode;
111 bool is_dgemm;
112 bool is_gfx940_xdl;
113};
114
115struct MFMA_F8F6F4_Info {
116 unsigned Opcode;
117 unsigned F8F8Opcode;
118 uint8_t NumRegsSrcA;
119 uint8_t NumRegsSrcB;
120};
121
122/// Normalized WMMA or SWMMAC family used to select co-execution rules.
123enum class WMMAVariant {
124 Unknown = 0,
125 IU8_16x16x64,
126 F8F6F4_16x16x128,
127 F8F6F4_16x16x128_BothF4,
128 FP8BF8_16x16x64,
129 F16BF16_16x16x32,
130 FP8BF8_16x16x128,
131 F4_32x16x128,
132};
133
134struct True16D16Info {
135 unsigned T16Op;
136 unsigned HiOp;
137 unsigned LoOp;
138};
139
140struct WMMAInstInfo {
141 uint32_t Opcode;
142 bool is_wmma_xdl;
143 bool HasMatrixScale;
144 WMMAVariant CoExecVariant;
145};
146
147#define GET_MIMGBaseOpcode_DECL
148#define GET_MIMGDim_DECL
149#define GET_MIMGEncoding_DECL
150#define GET_MIMGLZMapping_DECL
151#define GET_MIMGMIPMapping_DECL
152#define GET_MIMGBiASMapping_DECL
153#define GET_MAIInstInfoTable_DECL
154#define GET_isMFMA_F8F6F4Table_DECL
155#define GET_isCvtScaleF32_F32F16ToF8F4Table_DECL
156#define GET_True16D16Table_DECL
157#define GET_WMMAInstInfoTable_DECL
158#include "AMDGPUGenSearchableTables.inc"
159
160using TargetIDSetting = AMDGPU::TargetIDSetting;
161using TargetID = AMDGPU::TargetID;
162
163/// Construct TargetID from MCSubtargetInfo. \p FeatureString is used to
164/// determine explicitly requested xnack/sramecc settings.
165TargetID createAMDGPUTargetID(const MCSubtargetInfo &STI,
166 StringRef FeatureString);
167
168namespace IsaInfo {
169
170enum {
171 FIXED_NUM_SGPRS_FOR_INIT_BUG = AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG,
172 TRAP_NUM_SGPRS = 16
173};
174
175/// Returns true if \p Lhs and \p Rhs are incompatible (both specific but
176/// different).
177inline bool targetIDSettingsConflict(TargetIDSetting Lhs, TargetIDSetting Rhs) {
178 return Lhs != TargetIDSetting::Any && Rhs != TargetIDSetting::Any &&
179 Lhs != Rhs;
180}
181
182/// \returns Instruction cache line size in bytes for given subtarget \p STI.
183unsigned getInstCacheLineSize(const MCSubtargetInfo &STI);
184
185/// \returns Wavefront size for given subtarget \p STI.
186unsigned getWavefrontSize(const MCSubtargetInfo &STI);
187
188/// \returns Local memory size in bytes for given subtarget \p STI.
189unsigned getLocalMemorySize(const MCSubtargetInfo &STI);
190
191/// \returns Maximum addressable local memory size in bytes for given subtarget
192/// \p STI.
193unsigned getAddressableLocalMemorySize(const MCSubtargetInfo &STI);
194
195/// \returns Maximum number of work groups per compute unit for given subtarget
196/// \p STI and limited by given \p FlatWorkGroupSize.
197unsigned getMaxWorkGroupsPerCU(const MCSubtargetInfo &STI,
198 unsigned FlatWorkGroupSize);
199
200/// \returns Number of waves per execution unit required to support the given \p
201/// FlatWorkGroupSize.
202unsigned getWavesPerEUForWorkGroup(const MCSubtargetInfo &STI,
203 unsigned FlatWorkGroupSize);
204
205/// \returns Number of waves per work group for given subtarget \p STI and
206/// \p FlatWorkGroupSize.
207unsigned getWavesPerWorkGroup(const MCSubtargetInfo &STI,
208 unsigned FlatWorkGroupSize);
209
210/// \returns SGPR encoding granularity for given subtarget \p STI.
211unsigned getSGPREncodingGranule(const MCSubtargetInfo &STI);
212
213/// \returns Minimum number of SGPRs that meets the given number of waves per
214/// execution unit requirement for given subtarget \p STI.
215unsigned getMinNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU);
216
217/// \returns Maximum number of SGPRs that meets the given number of waves per
218/// execution unit requirement for given subtarget \p STI.
219unsigned getMaxNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
220 bool Addressable);
221
222/// \returns Number of extra SGPRs implicitly required by given subtarget \p
223/// STI when the given special registers are used.
224unsigned getNumExtraSGPRs(const MCSubtargetInfo &STI, bool VCCUsed,
225 bool FlatScrUsed, bool XNACKUsed);
226
227/// \returns Number of SGPR blocks needed for given subtarget \p STI when
228/// \p NumSGPRs are used. \p NumSGPRs should already include any special
229/// register counts.
230unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs);
231
232/// \returns VGPR allocation granularity for given subtarget \p STI.
233///
234/// For subtargets which support it, \p EnableWavefrontSize32 should match
235/// the ENABLE_WAVEFRONT_SIZE32 kernel descriptor field.
236unsigned
237getVGPRAllocGranule(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize,
238 std::optional<bool> EnableWavefrontSize32 = std::nullopt);
239
240/// \returns VGPR encoding granularity for given subtarget \p STI.
241///
242/// For subtargets which support it, \p EnableWavefrontSize32 should match
243/// the ENABLE_WAVEFRONT_SIZE32 kernel descriptor field.
244unsigned getVGPREncodingGranule(
245 const MCSubtargetInfo &STI,
246 std::optional<bool> EnableWavefrontSize32 = std::nullopt);
247
248/// For subtargets with a unified VGPR file and mixed ArchVGPR/AGPR usage,
249/// returns the allocation granule for ArchVGPRs.
250unsigned getArchVGPRAllocGranule();
251
252/// Maximum number of VGPR blocks that can be allocated in dynamic VGPR mode.
253static constexpr unsigned MaxDynamicVGPRBlocks = 8;
254
255/// \returns Addressable number of architectural VGPRs for a given subtarget \p
256/// STI.
257unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI);
258
259/// \returns Addressable number of VGPRs for given subtarget \p STI.
260unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
261 unsigned DynamicVGPRBlockSize);
262
263/// \returns Minimum number of VGPRs that meets given number of waves per
264/// execution unit requirement for given subtarget \p STI.
265unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
266 unsigned DynamicVGPRBlockSize);
267
268/// \returns Maximum number of VGPRs that meets given number of waves per
269/// execution unit requirement for given subtarget \p STI.
270unsigned getMaxNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
271 unsigned DynamicVGPRBlockSize);
272
273/// \returns Number of waves reachable for a given \p NumVGPRs usage for given
274/// subtarget \p STI.
275unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI,
276 unsigned NumVGPRs,
277 unsigned DynamicVGPRBlockSize);
278
279/// \returns Number of waves reachable for a given \p NumVGPRs usage, \p Granule
280/// size, \p MaxWaves possible, and \p TotalNumVGPRs available.
281unsigned getNumWavesPerEUWithNumVGPRs(unsigned NumVGPRs, unsigned Granule,
282 unsigned MaxWaves,
283 unsigned TotalNumVGPRs);
284
285/// \returns Whether allocated SGPRs can reduce occupancy on subtarget \p STI
286/// (true pre-GFX10). One named capability so callers don't test the version.
287bool isSGPROccupancyLimited(const MCSubtargetInfo &STI);
288
289/// \returns SGPR-limited occupancy (waves per EU) for subtarget \p STI: the
290/// inverse of getMaxNumSGPRs(). Unlike getMaxNumSGPRs() the budget is not
291/// clamped to the addressable count, since the allocated count callers pass in
292/// can exceed it.
293unsigned getOccupancyWithNumSGPRs(const MCSubtargetInfo &STI, unsigned SGPRs);
294
295/// \returns SGPR-limited occupancy computed from explicit budget parameters
296/// (\p MaxWaves, \p TotalNumSGPRs, \p Granule, \p TrapReserve). Subtarget-free
297/// core shared by the overload above and the occupancy MCExpr. Callers must
298/// check isSGPROccupancyLimited() first.
299unsigned getOccupancyWithNumSGPRs(unsigned SGPRs, unsigned MaxWaves,
300 unsigned TotalNumSGPRs, unsigned Granule,
301 unsigned TrapReserve);
302
303/// \returns Number of VGPR blocks that need to be allocated for the given
304/// subtarget \p STI when \p NumVGPRs are used.
305unsigned getAllocatedNumVGPRBlocks(
306 const MCSubtargetInfo &STI, unsigned NumVGPRs,
307 unsigned DynamicVGPRBlockSize,
308 std::optional<bool> EnableWavefrontSize32 = std::nullopt);
309
310} // end namespace IsaInfo
311
312// Represents a field in an encoded value.
313template <unsigned HighBit, unsigned LowBit, unsigned D = 0>
314struct EncodingField {
315 static_assert(HighBit >= LowBit, "Invalid bit range!");
316 static constexpr unsigned Offset = LowBit;
317 static constexpr unsigned Width = HighBit - LowBit + 1;
318
319 using ValueType = unsigned;
320 static_assert(Width <= sizeof(ValueType) * 8);
321 static constexpr ValueType Default = D;
322
323 ValueType Value;
324 constexpr EncodingField(ValueType Value) : Value(Value) {}
325
326 constexpr uint64_t encode() const { return Value; }
327 static ValueType decode(uint64_t Encoded) {
328 return static_cast<ValueType>(Encoded);
329 }
330};
331
332// Represents a single bit in an encoded value.
333template <unsigned Bit, unsigned D = 0>
334using EncodingBit = EncodingField<Bit, Bit, D>;
335
336// A helper for encoding and decoding multiple fields.
337template <typename... Fields> struct EncodingFields {
338 static constexpr uint64_t encode(Fields... Values) {
339 return ((Values.encode() << Values.Offset) | ...);
340 }
341
342 static std::tuple<typename Fields::ValueType...> decode(uint64_t Encoded) {
343 return {Fields::decode((Encoded >> Fields::Offset) &
344 maxUIntN(Fields::Width))...};
345 }
346};
347
348LLVM_READONLY
349inline bool hasNamedOperand(uint32_t Opcode, OpName NamedIdx) {
350 return getNamedOperandIdx(Opcode, Name: NamedIdx) != -1;
351}
352
353LLVM_READONLY
354int32_t getSOPPWithRelaxation(uint32_t Opcode);
355
356struct MIMGBaseOpcodeInfo {
357 MIMGBaseOpcode BaseOpcode;
358 bool Store;
359 bool Atomic;
360 bool AtomicX2;
361 bool Sampler;
362 bool Gather4;
363
364 uint8_t NumExtraArgs;
365 bool Gradients;
366 bool G16;
367 bool Coordinates;
368 bool LodOrClampOrMip;
369 bool HasD16;
370 bool MSAA;
371 bool BVH;
372 bool A16;
373 bool NoReturn;
374 bool PointSampleAccel;
375};
376
377LLVM_READONLY
378const MIMGBaseOpcodeInfo *getMIMGBaseOpcode(unsigned Opc);
379
380LLVM_READONLY
381const MIMGBaseOpcodeInfo *getMIMGBaseOpcodeInfo(unsigned BaseOpcode);
382
383struct MIMGDimInfo {
384 MIMGDim Dim;
385 MIMGDim NonArrayDim;
386 uint8_t NumCoords;
387 uint8_t NumGradients;
388 bool MSAA;
389 bool DA;
390 uint8_t Encoding;
391 StringTable::Offset AsmSuffix;
392};
393
394LLVM_READONLY
395const MIMGDimInfo *getMIMGDimInfo(unsigned DimEnum);
396
397LLVM_READONLY StringRef getMIMGDimInfoStr(StringTable::Offset);
398
399LLVM_READONLY
400const MIMGDimInfo *getMIMGDimInfoByEncoding(uint8_t DimEnc);
401
402LLVM_READONLY
403const MIMGDimInfo *getMIMGDimInfoByAsmSuffix(StringRef AsmSuffix);
404
405struct MIMGLZMappingInfo {
406 MIMGBaseOpcode L;
407 MIMGBaseOpcode LZ;
408};
409
410struct MIMGMIPMappingInfo {
411 MIMGBaseOpcode MIP;
412 MIMGBaseOpcode NONMIP;
413};
414
415struct MIMGBiasMappingInfo {
416 MIMGBaseOpcode Bias;
417 MIMGBaseOpcode NoBias;
418};
419
420struct MIMGOffsetMappingInfo {
421 MIMGBaseOpcode Offset;
422 MIMGBaseOpcode NoOffset;
423};
424
425struct MIMGG16MappingInfo {
426 MIMGBaseOpcode G;
427 MIMGBaseOpcode G16;
428};
429
430LLVM_READONLY
431const MIMGLZMappingInfo *getMIMGLZMappingInfo(unsigned L);
432
433struct WMMAOpcodeMappingInfo {
434 unsigned Opcode2Addr;
435 unsigned Opcode3Addr;
436};
437
438LLVM_READONLY
439const MIMGMIPMappingInfo *getMIMGMIPMappingInfo(unsigned MIP);
440
441LLVM_READONLY
442const MIMGBiasMappingInfo *getMIMGBiasMappingInfo(unsigned Bias);
443
444LLVM_READONLY
445const MIMGOffsetMappingInfo *getMIMGOffsetMappingInfo(unsigned Offset);
446
447LLVM_READONLY
448const MIMGG16MappingInfo *getMIMGG16MappingInfo(unsigned G);
449
450LLVM_READONLY
451int getMIMGOpcode(unsigned BaseOpcode, unsigned MIMGEncoding,
452 unsigned VDataDwords, unsigned VAddrDwords,
453 bool IndexedRsrc = false, bool IndexedSamp = false);
454
455LLVM_READONLY
456int getMaskedMIMGOp(unsigned Opc, unsigned NewChannels);
457
458LLVM_READONLY
459unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode,
460 const MIMGDimInfo *Dim, bool IsA16,
461 bool IsG16Supported);
462
463struct MIMGInfo {
464 uint32_t Opcode;
465 uint32_t BaseOpcode;
466 uint8_t MIMGEncoding;
467 uint8_t VDataDwords;
468 uint8_t VAddrDwords;
469 uint8_t VAddrOperands;
470 bool IndexedRsrc;
471 bool IndexedSamp;
472};
473
474LLVM_READONLY
475const MIMGInfo *getMIMGInfo(unsigned Opc);
476
477LLVM_READONLY
478int getMTBUFBaseOpcode(unsigned Opc);
479
480LLVM_READONLY
481int getMTBUFOpcode(unsigned BaseOpc, unsigned Elements);
482
483LLVM_READONLY
484int getMTBUFElements(unsigned Opc);
485
486LLVM_READONLY
487bool getMTBUFHasVAddr(unsigned Opc);
488
489LLVM_READONLY
490bool getMTBUFHasSrsrc(unsigned Opc);
491
492LLVM_READONLY
493bool getMTBUFHasSoffset(unsigned Opc);
494
495LLVM_READONLY
496int getMUBUFBaseOpcode(unsigned Opc);
497
498LLVM_READONLY
499int getMUBUFOpcode(unsigned BaseOpc, unsigned Elements);
500
501LLVM_READONLY
502int getMUBUFElements(unsigned Opc);
503
504LLVM_READONLY
505bool getMUBUFHasVAddr(unsigned Opc);
506
507LLVM_READONLY
508bool getMUBUFHasSrsrc(unsigned Opc);
509
510LLVM_READONLY
511bool getMUBUFHasSoffset(unsigned Opc);
512
513LLVM_READONLY
514bool getMUBUFIsBufferInv(unsigned Opc);
515
516LLVM_READONLY
517bool getMUBUFTfe(unsigned Opc);
518
519LLVM_READONLY
520bool getSMEMIsBuffer(unsigned Opc);
521
522LLVM_READONLY
523bool getVOP1IsSingle(unsigned Opc);
524
525LLVM_READONLY
526bool getVOP2IsSingle(unsigned Opc);
527
528LLVM_READONLY
529bool getVOP3IsSingle(unsigned Opc);
530
531LLVM_READONLY
532bool isVOPC64DPP(unsigned Opc);
533
534LLVM_READONLY
535bool isVOPCAsmOnly(unsigned Opc);
536
537/// Returns true if MAI operation is a double precision GEMM.
538LLVM_READONLY
539bool getMAIIsDGEMM(unsigned Opc);
540
541LLVM_READONLY
542bool getMAIIsGFX940XDL(unsigned Opc);
543
544LLVM_READONLY
545bool getWMMAIsXDL(unsigned Opc);
546
547LLVM_READONLY
548bool getHasMatrixScale(unsigned Opc);
549
550// Get an equivalent BitOp3 for a binary logical \p Opc.
551// \returns BitOp3 modifier for the logical operation or zero.
552// Used in VOPD3 conversion.
553unsigned getBitOp2(unsigned Opc);
554
555struct CanBeVOPD {
556 bool X;
557 bool Y;
558};
559
560/// \returns SIEncodingFamily used for VOPD encoding on a \p ST.
561LLVM_READONLY
562unsigned getVOPDEncodingFamily(const MCSubtargetInfo &ST);
563
564LLVM_READONLY
565CanBeVOPD getCanBeVOPD(unsigned Opc, unsigned EncodingFamily, bool VOPD3);
566
567LLVM_READNONE
568uint8_t mfmaScaleF8F6F4FormatToNumRegs(unsigned EncodingVal);
569
570LLVM_READONLY
571const MFMA_F8F6F4_Info *getMFMA_F8F6F4_WithFormatArgs(unsigned CBSZ,
572 unsigned BLGP,
573 unsigned F8F8Opcode);
574
575LLVM_READNONE
576uint8_t wmmaScaleF8F6F4FormatToNumRegs(unsigned Fmt);
577
578LLVM_READONLY
579const MFMA_F8F6F4_Info *getWMMA_F8F6F4_WithFormatArgs(unsigned FmtA,
580 unsigned FmtB,
581 unsigned F8F8Opcode);
582
583/// \return true if this combination is listed as valid.
584LLVM_READONLY
585bool isValidWMMAScaleFmtCombination(unsigned AFmt, unsigned AScale,
586 unsigned BFmt, unsigned BScale);
587
588/// \returns the matrix_a_fmt and matrix_b_fmt operands of an f8f6f4 WMMA.
589/// Works on both MCInst and MachineInstr.
590template <typename InstT>
591std::pair<unsigned, unsigned> getWMMAMatrixFmts(const InstT &MI) {
592 unsigned Opc = MI.getOpcode();
593 int AIdx = getNamedOperandIdx(Opcode: Opc, Name: OpName::matrix_a_fmt);
594 int BIdx = getNamedOperandIdx(Opcode: Opc, Name: OpName::matrix_b_fmt);
595 assert(AIdx != -1 && BIdx != -1 && "expected an f8f6f4 WMMA");
596 return {static_cast<unsigned>(MI.getOperand(AIdx).getImm()),
597 static_cast<unsigned>(MI.getOperand(BIdx).getImm())};
598}
599
600/// \returns true if either matrix input of an f8f6f4 WMMA is f8.
601template <typename InstT> bool isWMMAAnyF8(const InstT &MI) {
602 auto [AFmt, BFmt] = getWMMAMatrixFmts(MI);
603 return AFmt <= WMMA::MATRIX_FMT_BF8 || BFmt <= WMMA::MATRIX_FMT_BF8;
604}
605
606/// \returns true if both matrix inputs of an f8f6f4 WMMA are f4.
607template <typename InstT> bool isWMMABothF4(const InstT &MI) {
608 auto [AFmt, BFmt] = getWMMAMatrixFmts(MI);
609 return AFmt == WMMA::MATRIX_FMT_FP4 && BFmt == WMMA::MATRIX_FMT_FP4;
610}
611
612LLVM_READONLY
613const GcnBufferFormatInfo *getGcnBufferFormatInfo(uint8_t BitsPerComp,
614 uint8_t NumComponents,
615 uint8_t NumFormat,
616 const MCSubtargetInfo &STI);
617LLVM_READONLY
618const GcnBufferFormatInfo *getGcnBufferFormatInfo(uint8_t Format,
619 const MCSubtargetInfo &STI);
620
621LLVM_READONLY
622int32_t getMCOpcode(uint32_t Opcode, unsigned Gen);
623
624LLVM_READONLY
625unsigned getVOPDOpcode(unsigned Opc, bool VOPD3);
626
627LLVM_READONLY
628int getVOPDFull(unsigned OpX, unsigned OpY, unsigned EncodingFamily,
629 bool VOPD3);
630
631LLVM_READONLY
632bool isVOPD(unsigned Opc);
633
634LLVM_READNONE
635bool isMAC(unsigned Opc);
636
637LLVM_READNONE
638bool isPermlane16(unsigned Opc);
639
640LLVM_READNONE
641bool isGenericAtomic(unsigned Opc);
642
643LLVM_READNONE
644bool isCvt_F32_Fp8_Bf8_e64(unsigned Opc);
645
646namespace VOPD {
647
648enum Component : unsigned {
649 DST = 0,
650 SRC0,
651 SRC1,
652 SRC2,
653
654 DST_NUM = 1,
655 MAX_SRC_NUM = 3,
656 MAX_OPR_NUM = DST_NUM + MAX_SRC_NUM
657};
658
659// LSB mask for VGPR banks per VOPD component operand.
660// 4 banks result in a mask 3, setting 2 lower bits.
661constexpr unsigned VOPD_VGPR_BANK_MASKS[] = {1, 3, 3, 1};
662constexpr unsigned VOPD3_VGPR_BANK_MASKS[] = {1, 3, 3, 3};
663// GFX11 VOPD interlock hazard requires SRC0/SRC1 to have
664// different parities, not just on different banks. Else,
665// non-deterministic forwarding error may occur.
666constexpr unsigned VOPD_GFX11_VGPR_BANK_MASKS[] = {1, 1, 1, 1};
667
668enum ComponentIndex : unsigned { X = 0, Y = 1 };
669constexpr unsigned COMPONENTS[] = {ComponentIndex::X, ComponentIndex::Y};
670constexpr unsigned COMPONENTS_NUM = 2;
671
672// Properties of VOPD components.
673class ComponentProps {
674private:
675 unsigned SrcOperandsNum = 0;
676 unsigned MandatoryLiteralIdx = ~0u;
677 bool HasSrc2Acc = false;
678 unsigned NumVOPD3Mods = 0;
679 unsigned Opcode = 0;
680 bool IsVOP3 = false;
681
682public:
683 ComponentProps() = default;
684 ComponentProps(const MCInstrDesc &OpDesc, bool VOP3Layout = false);
685
686 // Return the total number of src operands this component has.
687 unsigned getCompSrcOperandsNum() const { return SrcOperandsNum; }
688
689 // Return the number of src operands of this component visible to the parser.
690 unsigned getCompParsedSrcOperandsNum() const {
691 return SrcOperandsNum - HasSrc2Acc;
692 }
693
694 // Return true iif this component has a mandatory literal.
695 bool hasMandatoryLiteral() const { return MandatoryLiteralIdx != ~0u; }
696
697 // If this component has a mandatory literal, return component operand
698 // index of this literal (i.e. either Component::SRC1 or Component::SRC2).
699 unsigned getMandatoryLiteralCompOperandIndex() const {
700 assert(hasMandatoryLiteral());
701 return MandatoryLiteralIdx;
702 }
703
704 // Return true iif this component has operand
705 // with component index CompSrcIdx and this operand may be a register.
706 bool hasRegSrcOperand(unsigned CompSrcIdx) const {
707 assert(CompSrcIdx < Component::MAX_SRC_NUM);
708 return SrcOperandsNum > CompSrcIdx && !hasMandatoryLiteralAt(CompSrcIdx);
709 }
710
711 // Return true iif this component has tied src2.
712 bool hasSrc2Acc() const { return HasSrc2Acc; }
713
714 // Return a number of source modifiers if instruction is used in VOPD3.
715 unsigned getCompVOPD3ModsNum() const { return NumVOPD3Mods; }
716
717 // Return opcode of the component.
718 unsigned getOpcode() const { return Opcode; }
719
720 // Returns if component opcode is in VOP3 encoding.
721 unsigned isVOP3() const { return IsVOP3; }
722
723 // Return index of BitOp3 operand or -1.
724 int getBitOp3OperandIdx() const;
725
726private:
727 bool hasMandatoryLiteralAt(unsigned CompSrcIdx) const {
728 assert(CompSrcIdx < Component::MAX_SRC_NUM);
729 return MandatoryLiteralIdx == Component::DST_NUM + CompSrcIdx;
730 }
731};
732
733enum ComponentKind : unsigned {
734 SINGLE = 0, // A single VOP1 or VOP2 instruction which may be used in VOPD.
735 COMPONENT_X, // A VOPD instruction, X component.
736 COMPONENT_Y, // A VOPD instruction, Y component.
737 MAX = COMPONENT_Y
738};
739
740// Interface functions of this class map VOPD component operand indices
741// to indices of operands in MachineInstr/MCInst or parsed operands array.
742//
743// Note that this class operates with 3 kinds of indices:
744// - VOPD component operand indices (Component::DST, Component::SRC0, etc.);
745// - MC operand indices (they refer operands in a MachineInstr/MCInst);
746// - parsed operand indices (they refer operands in parsed operands array).
747//
748// For SINGLE components mapping between these indices is trivial.
749// But things get more complicated for COMPONENT_X and
750// COMPONENT_Y because these components share the same
751// MachineInstr/MCInst and the same parsed operands array.
752// Below is an example of component operand to parsed operand
753// mapping for the following instruction:
754//
755// v_dual_add_f32 v255, v4, v5 :: v_dual_mov_b32 v6, v1
756//
757// PARSED COMPONENT PARSED
758// COMPONENT OPERANDS OPERAND INDEX OPERAND INDEX
759// -------------------------------------------------------------------
760// "v_dual_add_f32" 0
761// v_dual_add_f32 v255 0 (DST) --> 1
762// v4 1 (SRC0) --> 2
763// v5 2 (SRC1) --> 3
764// "::" 4
765// "v_dual_mov_b32" 5
766// v_dual_mov_b32 v6 0 (DST) --> 6
767// v1 1 (SRC0) --> 7
768// -------------------------------------------------------------------
769//
770class ComponentLayout {
771private:
772 // Regular MachineInstr/MCInst operands are ordered as follows:
773 // dst, src0 [, other src operands]
774 // VOPD MachineInstr/MCInst operands are ordered as follows:
775 // dstX, dstY, src0X [, other OpX operands], src0Y [, other OpY operands]
776 // Each ComponentKind has operand indices defined below.
777 static constexpr unsigned MC_DST_IDX[] = {0, 0, 1};
778
779 // VOPD3 instructions may have 2 or 3 source modifiers, src2 modifier is not
780 // used if there is tied accumulator. Indexing of this array:
781 // MC_SRC_IDX[VOPD3ModsNum][SrcNo]. This returns an index for a SINGLE
782 // instruction layout, add 1 for COMPONENT_X or COMPONENT_Y. For the second
783 // component add OpX.MCSrcNum + OpX.VOPD3ModsNum.
784 // For VOPD1/VOPD2 use column with zero modifiers.
785 static constexpr unsigned SINGLE_MC_SRC_IDX[4][3] = {
786 {1, 2, 3}, {2, 3, 4}, {2, 4, 5}, {2, 4, 6}};
787
788 // Parsed operands of regular instructions are ordered as follows:
789 // Mnemo dst src0 [vsrc1 ...]
790 // Parsed VOPD operands are ordered as follows:
791 // OpXMnemo dstX src0X [vsrc1X|imm vsrc1X|vsrc1X imm] '::'
792 // OpYMnemo dstY src0Y [vsrc1Y|imm vsrc1Y|vsrc1Y imm]
793 // Each ComponentKind has operand indices defined below.
794 static constexpr unsigned PARSED_DST_IDX[] = {1, 1,
795 4 /* + OpX.ParsedSrcNum */};
796 static constexpr unsigned FIRST_PARSED_SRC_IDX[] = {
797 2, 2, 5 /* + OpX.ParsedSrcNum */};
798
799private:
800 const ComponentKind Kind;
801 const ComponentProps PrevComp;
802 const unsigned VOPD3ModsNum;
803 const int BitOp3Idx; // Index of bitop3 operand or -1
804
805public:
806 // Create layout for COMPONENT_X or SINGLE component.
807 ComponentLayout(ComponentKind Kind, unsigned VOPD3ModsNum, int BitOp3Idx)
808 : Kind(Kind), VOPD3ModsNum(VOPD3ModsNum), BitOp3Idx(BitOp3Idx) {
809 assert(Kind == ComponentKind::SINGLE || Kind == ComponentKind::COMPONENT_X);
810 }
811
812 // Create layout for COMPONENT_Y which depends on COMPONENT_X layout.
813 ComponentLayout(const ComponentProps &OpXProps, unsigned VOPD3ModsNum,
814 int BitOp3Idx)
815 : Kind(ComponentKind::COMPONENT_Y), PrevComp(OpXProps),
816 VOPD3ModsNum(VOPD3ModsNum), BitOp3Idx(BitOp3Idx) {}
817
818public:
819 // Return the index of dst operand in MCInst operands.
820 unsigned getIndexOfDstInMCOperands() const { return MC_DST_IDX[Kind]; }
821
822 // Return the index of the specified src operand in MCInst operands.
823 unsigned getIndexOfSrcInMCOperands(unsigned CompSrcIdx, bool VOPD3) const {
824 assert(CompSrcIdx < Component::MAX_SRC_NUM);
825
826 if (Kind == SINGLE && CompSrcIdx == 2 && BitOp3Idx != -1)
827 return BitOp3Idx;
828
829 if (VOPD3) {
830 return SINGLE_MC_SRC_IDX[VOPD3ModsNum][CompSrcIdx] + getPrevCompSrcNum() +
831 getPrevCompVOPD3ModsNum() + (Kind != SINGLE ? 1 : 0);
832 }
833
834 return SINGLE_MC_SRC_IDX[0][CompSrcIdx] + getPrevCompSrcNum() +
835 (Kind != SINGLE ? 1 : 0);
836 }
837
838 // Return the index of dst operand in the parsed operands array.
839 unsigned getIndexOfDstInParsedOperands() const {
840 return PARSED_DST_IDX[Kind] + getPrevCompParsedSrcNum();
841 }
842
843 // Return the index of the specified src operand in the parsed operands array.
844 unsigned getIndexOfSrcInParsedOperands(unsigned CompSrcIdx) const {
845 assert(CompSrcIdx < Component::MAX_SRC_NUM);
846 return FIRST_PARSED_SRC_IDX[Kind] + getPrevCompParsedSrcNum() + CompSrcIdx;
847 }
848
849private:
850 unsigned getPrevCompSrcNum() const {
851 return PrevComp.getCompSrcOperandsNum();
852 }
853 unsigned getPrevCompParsedSrcNum() const {
854 return PrevComp.getCompParsedSrcOperandsNum();
855 }
856 unsigned getPrevCompVOPD3ModsNum() const {
857 return PrevComp.getCompVOPD3ModsNum();
858 }
859};
860
861// Layout and properties of VOPD components.
862class ComponentInfo : public ComponentProps, public ComponentLayout {
863public:
864 // Create ComponentInfo for COMPONENT_X or SINGLE component.
865 ComponentInfo(const MCInstrDesc &OpDesc,
866 ComponentKind Kind = ComponentKind::SINGLE,
867 bool VOP3Layout = false)
868 : ComponentProps(OpDesc, VOP3Layout),
869 ComponentLayout(Kind, getCompVOPD3ModsNum(), getBitOp3OperandIdx()) {}
870
871 // Create ComponentInfo for COMPONENT_Y which depends on COMPONENT_X layout.
872 ComponentInfo(const MCInstrDesc &OpDesc, const ComponentProps &OpXProps,
873 bool VOP3Layout = false)
874 : ComponentProps(OpDesc, VOP3Layout),
875 ComponentLayout(OpXProps, getCompVOPD3ModsNum(),
876 getBitOp3OperandIdx()) {}
877
878 // Map component operand index to parsed operand index.
879 // Return 0 if the specified operand does not exist.
880 unsigned getIndexInParsedOperands(unsigned CompOprIdx) const;
881};
882
883// Properties of VOPD instructions.
884class InstInfo {
885private:
886 const ComponentInfo CompInfo[COMPONENTS_NUM];
887
888public:
889 using RegIndices = std::array<MCRegister, Component::MAX_OPR_NUM>;
890
891 InstInfo(const MCInstrDesc &OpX, const MCInstrDesc &OpY)
892 : CompInfo{OpX, OpY} {}
893
894 InstInfo(const ComponentInfo &OprInfoX, const ComponentInfo &OprInfoY)
895 : CompInfo{OprInfoX, OprInfoY} {}
896
897 const ComponentInfo &operator[](size_t ComponentIdx) const {
898 assert(ComponentIdx < COMPONENTS_NUM);
899 return CompInfo[ComponentIdx];
900 }
901
902 // Check VOPD operands constraints.
903 // GetRegIdx(Component, MCOperandIdx) must return a VGPR register index
904 // for the specified component and MC operand. The callback must return 0
905 // if the operand is not a register or not a VGPR.
906 // If \p SkipSrc is set to true then constraints for source operands are not
907 // checked.
908 // If \p AllowSameVGPR is set then same VGPRs are allowed for X and Y sources
909 // even though it violates requirement to be from different banks.
910 // If \p VOPD3 is set to true both dst registers allowed to be either odd
911 // or even and instruction may have real src2 as opposed to tied accumulator.
912 // If \p HasGFX11InterlockHazard is set then X/Y SRC0 and SRC1 VGPRs
913 // must have different register-number parity.
914 bool
915 hasInvalidOperand(std::function<MCRegister(unsigned, unsigned)> GetRegIdx,
916 const MCRegisterInfo &MRI, bool SkipSrc = false,
917 bool AllowSameVGPR = false, bool VOPD3 = false,
918 bool HasGFX11InterlockHazard = false) const {
919 return getInvalidCompOperandIndex(GetRegIdx, MRI, SkipSrc, AllowSameVGPR,
920 VOPD3, HasGFX11InterlockHazard)
921 .has_value();
922 }
923
924 // Check VOPD operands constraints.
925 // Return the index of an invalid component operand, if any.
926 // If \p SkipSrc is set to true then constraints for source operands are not
927 // checked except for being from the same halves of VGPR file on gfx1250.
928 // If \p AllowSameVGPR is set then same VGPRs are allowed for X and Y sources
929 // even though it violates requirement to be from different banks.
930 // If \p VOPD3 is set to true both dst registers allowed to be either odd
931 // or even and instruction may have real src2 as opposed to tied accumulator.
932 // If \p HasGFX11InterlockHazard is set then X/Y SRC0 and SRC1 VGPRs
933 // must have different register-number parity.
934 std::optional<unsigned> getInvalidCompOperandIndex(
935 std::function<MCRegister(unsigned, unsigned)> GetRegIdx,
936 const MCRegisterInfo &MRI, bool SkipSrc = false,
937 bool AllowSameVGPR = false, bool VOPD3 = false,
938 bool HasGFX11InterlockHazard = false) const;
939
940private:
941 RegIndices
942 getRegIndices(unsigned ComponentIdx,
943 std::function<MCRegister(unsigned, unsigned)> GetRegIdx,
944 bool VOPD3) const;
945};
946
947} // namespace VOPD
948
949LLVM_READONLY
950std::pair<unsigned, unsigned> getVOPDComponents(unsigned VOPDOpcode);
951
952LLVM_READONLY
953// Get properties of 2 single VOP1/VOP2 instructions
954// used as components to create a VOPD instruction.
955VOPD::InstInfo getVOPDInstInfo(const MCInstrDesc &OpX, const MCInstrDesc &OpY);
956
957LLVM_READONLY
958// Get properties of VOPD X and Y components.
959VOPD::InstInfo getVOPDInstInfo(unsigned VOPDOpcode,
960 const MCInstrInfo *InstrInfo);
961
962LLVM_READONLY
963bool isAsyncStore(unsigned Opc);
964LLVM_READONLY
965bool isTensorStore(unsigned Opc);
966LLVM_READONLY
967unsigned getTemporalHintType(const MCInstrDesc TID);
968
969LLVM_READONLY
970bool isTrue16Inst(unsigned Opc);
971
972LLVM_READONLY
973FPType getFPDstSelType(unsigned Opc);
974
975bool isDPMACCInstruction(unsigned Opc);
976
977LLVM_READONLY
978unsigned mapWMMA2AddrTo3AddrOpcode(unsigned Opc);
979
980void initDefaultAMDKernelCodeT(AMDGPUMCKernelCodeT &Header,
981 const MCSubtargetInfo &STI);
982
983bool isReadOnlySegment(const GlobalValue *GV);
984
985/// \returns True if constants should be emitted to .text section for given
986/// target triple \p TT, false otherwise.
987bool shouldEmitConstantsToTextSection(const Triple &TT);
988
989/// Returns a valid charcode or 0 in the first entry if this is a valid physical
990/// register name. Followed by the start register number, and the register
991/// width. Does not validate the number of registers exists in the class. Unlike
992/// parseAsmConstraintPhysReg, this does not expect the name to be wrapped in
993/// "{}".
994std::tuple<char, unsigned, unsigned> parseAsmPhysRegName(StringRef TupleString);
995
996/// Returns a valid charcode or 0 in the first entry if this is a valid physical
997/// register constraint. Followed by the start register number, and the register
998/// width. Does not validate the number of registers exists in the class.
999std::tuple<char, unsigned, unsigned>
1000parseAsmConstraintPhysReg(StringRef Constraint);
1001
1002/// \returns A pair of integer values requested using \p F's \p Name attribute
1003/// in "first[,second]" format ("second" is optional unless \p OnlyFirstRequired
1004/// is false).
1005///
1006/// \returns \p Default if attribute is not present.
1007///
1008/// \returns \p Default and emits error if one of the requested values cannot be
1009/// converted to integer, or \p OnlyFirstRequired is false and "second" value is
1010/// not present.
1011std::pair<unsigned, unsigned>
1012getIntegerPairAttribute(const Function &F, StringRef Name,
1013 std::pair<unsigned, unsigned> Default,
1014 bool OnlyFirstRequired = false);
1015
1016/// \returns A pair of integer values requested using \p F's \p Name attribute
1017/// in "first[,second]" format ("second" is optional unless \p OnlyFirstRequired
1018/// is false).
1019///
1020/// \returns \p std::nullopt if attribute is not present.
1021///
1022/// \returns \p std::nullopt and emits error if one of the requested values
1023/// cannot be converted to integer, or \p OnlyFirstRequired is false and
1024/// "second" value is not present.
1025std::optional<std::pair<unsigned, std::optional<unsigned>>>
1026getIntegerPairAttribute(const Function &F, StringRef Name,
1027 bool OnlyFirstRequired = false);
1028
1029/// \returns Generate a vector of integer values requested using \p F's \p Name
1030/// attribute.
1031/// \returns A vector of size \p Size, with all elements set to \p DefaultVal,
1032/// if any error occurs. The corresponding error will also be emitted.
1033SmallVector<unsigned> getIntegerVecAttribute(const Function &F, StringRef Name,
1034 unsigned Size,
1035 unsigned DefaultVal);
1036/// Similar to the function above, but returns std::nullopt if any error occurs.
1037std::optional<SmallVector<unsigned>>
1038getIntegerVecAttribute(const Function &F, StringRef Name, unsigned Size);
1039
1040/// \returns The maximum number of workgroups for the function.
1041SmallVector<unsigned> getMaxNumWorkGroups(const Function &F);
1042
1043inline bool isTgSplitEnabled(const Function &F) {
1044 return F.hasFnAttribute(Kind: "amdgpu-tg-split");
1045}
1046
1047/// Checks if \p Val is inside \p MD, a !range-like metadata.
1048bool hasValueInRangeLikeMetadata(const MDNode &MD, int64_t Val);
1049
1050// The following methods are only meaningful on targets that support
1051// S_WAITCNT.
1052
1053/// \returns Vmcnt bit mask for given isa \p Version.
1054unsigned getVmcntBitMask(const IsaVersion &Version);
1055
1056/// \returns Expcnt bit mask for given isa \p Version.
1057unsigned getExpcntBitMask(const IsaVersion &Version);
1058
1059/// \returns Lgkmcnt bit mask for given isa \p Version.
1060unsigned getLgkmcntBitMask(const IsaVersion &Version);
1061
1062/// \returns Waitcnt bit mask for given isa \p Version.
1063unsigned getWaitcntBitMask(const IsaVersion &Version);
1064
1065/// \returns Decoded Vmcnt from given \p Waitcnt for given isa \p Version.
1066unsigned decodeVmcnt(const IsaVersion &Version, unsigned Waitcnt);
1067
1068/// \returns Decoded Expcnt from given \p Waitcnt for given isa \p Version.
1069unsigned decodeExpcnt(const IsaVersion &Version, unsigned Waitcnt);
1070
1071/// \returns Decoded Lgkmcnt from given \p Waitcnt for given isa \p Version.
1072unsigned decodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt);
1073
1074/// \returns Decoded Loadcnt from given \p Waitcnt for given isa \p Version.
1075unsigned decodeLoadcnt(const IsaVersion &Version, unsigned Waitcnt);
1076
1077/// \returns Decoded Storecnt from given \p Waitcnt for given isa \p Version.
1078unsigned decodeStorecnt(const IsaVersion &Version, unsigned Waitcnt);
1079
1080/// \returns Decoded Dscnt from given \p Waitcnt for given isa \p Version.
1081unsigned decodeDscnt(const IsaVersion &Version, unsigned Waitcnt);
1082
1083/// Decodes Vmcnt, Expcnt and Lgkmcnt from given \p Waitcnt for given isa
1084/// \p Version, and writes decoded values into \p Vmcnt, \p Expcnt and
1085/// \p Lgkmcnt respectively. Should not be used on gfx12+, the instruction
1086/// which needs it is deprecated
1087///
1088/// \details \p Vmcnt, \p Expcnt and \p Lgkmcnt are decoded as follows:
1089/// \p Vmcnt = \p Waitcnt[3:0] (pre-gfx9)
1090/// \p Vmcnt = \p Waitcnt[15:14,3:0] (gfx9,10)
1091/// \p Vmcnt = \p Waitcnt[15:10] (gfx11)
1092/// \p Expcnt = \p Waitcnt[6:4] (pre-gfx11)
1093/// \p Expcnt = \p Waitcnt[2:0] (gfx11)
1094/// \p Lgkmcnt = \p Waitcnt[11:8] (pre-gfx10)
1095/// \p Lgkmcnt = \p Waitcnt[13:8] (gfx10)
1096/// \p Lgkmcnt = \p Waitcnt[9:4] (gfx11)
1097///
1098void decodeWaitcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned &Vmcnt,
1099 unsigned &Expcnt, unsigned &Lgkmcnt);
1100
1101/// \returns \p Waitcnt with encoded \p Vmcnt for given isa \p Version.
1102unsigned encodeVmcnt(const IsaVersion &Version, unsigned Waitcnt,
1103 unsigned Vmcnt);
1104
1105/// \returns \p Waitcnt with encoded \p Expcnt for given isa \p Version.
1106unsigned encodeExpcnt(const IsaVersion &Version, unsigned Waitcnt,
1107 unsigned Expcnt);
1108
1109/// \returns \p Waitcnt with encoded \p Lgkmcnt for given isa \p Version.
1110unsigned encodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt,
1111 unsigned Lgkmcnt);
1112
1113/// Encodes \p Vmcnt, \p Expcnt and \p Lgkmcnt into Waitcnt for given isa
1114/// \p Version. Should not be used on gfx12+, the instruction which needs
1115/// it is deprecated
1116///
1117/// \details \p Vmcnt, \p Expcnt and \p Lgkmcnt are encoded as follows:
1118/// Waitcnt[2:0] = \p Expcnt (gfx11+)
1119/// Waitcnt[3:0] = \p Vmcnt (pre-gfx9)
1120/// Waitcnt[3:0] = \p Vmcnt[3:0] (gfx9,10)
1121/// Waitcnt[6:4] = \p Expcnt (pre-gfx11)
1122/// Waitcnt[9:4] = \p Lgkmcnt (gfx11)
1123/// Waitcnt[11:8] = \p Lgkmcnt (pre-gfx10)
1124/// Waitcnt[13:8] = \p Lgkmcnt (gfx10)
1125/// Waitcnt[15:10] = \p Vmcnt (gfx11)
1126/// Waitcnt[15:14] = \p Vmcnt[5:4] (gfx9,10)
1127///
1128/// \returns Waitcnt with encoded \p Vmcnt, \p Expcnt and \p Lgkmcnt for given
1129/// isa \p Version.
1130///
1131unsigned encodeWaitcnt(const IsaVersion &Version, unsigned Vmcnt,
1132 unsigned Expcnt, unsigned Lgkmcnt);
1133
1134/// \returns Waitcnt with encoded \p Loadcnt and \p Dscnt for given isa \p
1135/// Version.
1136unsigned encodeLoadcntDscnt(const IsaVersion &Version, unsigned Loadcnt,
1137 unsigned Dscnt);
1138
1139/// \returns Waitcnt with encoded \p Storecnt and \p Dscnt for given isa \p
1140/// Version.
1141unsigned encodeStorecntDscnt(const IsaVersion &Version, unsigned Storecnt,
1142 unsigned Dscnt);
1143
1144// The following methods are only meaningful on targets that support
1145// S_WAIT_*CNT, introduced with gfx12.
1146
1147/// \returns Loadcnt bit mask for given isa \p Version.
1148/// Returns 0 for versions that do not support LOADcnt
1149unsigned getLoadcntBitMask(const IsaVersion &Version);
1150
1151/// \returns Samplecnt bit mask for given isa \p Version.
1152/// Returns 0 for versions that do not support SAMPLEcnt
1153unsigned getSamplecntBitMask(const IsaVersion &Version);
1154
1155/// \returns Bvhcnt bit mask for given isa \p Version.
1156/// Returns 0 for versions that do not support BVHcnt
1157unsigned getBvhcntBitMask(const IsaVersion &Version);
1158
1159/// \returns Asynccnt bit mask for given isa \p Version.
1160/// Returns 0 for versions that do not support Asynccnt
1161unsigned getAsynccntBitMask(const IsaVersion &Version);
1162
1163/// \returns Dscnt bit mask for given isa \p Version.
1164/// Returns 0 for versions that do not support DScnt
1165unsigned getDscntBitMask(const IsaVersion &Version);
1166
1167/// \returns Dscnt bit mask for given isa \p Version.
1168/// Returns 0 for versions that do not support KMcnt
1169unsigned getKmcntBitMask(const IsaVersion &Version);
1170
1171/// \returns Xcnt bit mask for given isa \p Version.
1172/// Returns 0 for versions that do not support Xcnt.
1173unsigned getXcntBitMask(const IsaVersion &Version);
1174
1175/// \return STOREcnt or VScnt bit mask for given isa \p Version.
1176/// returns 0 for versions that do not support STOREcnt or VScnt.
1177/// STOREcnt and VScnt are the same counter, the name used
1178/// depends on the ISA version.
1179unsigned getStorecntBitMask(const IsaVersion &Version);
1180
1181namespace Hwreg {
1182
1183using HwregId = EncodingField<5, 0>;
1184using HwregOffset = EncodingField<10, 6>;
1185
1186struct HwregSize : EncodingField<15, 11, 32> {
1187 using EncodingField::EncodingField;
1188 constexpr uint64_t encode() const { return Value - 1; }
1189 static ValueType decode(uint64_t Encoded) {
1190 return static_cast<ValueType>(Encoded + 1);
1191 }
1192};
1193
1194using HwregEncoding = EncodingFields<HwregId, HwregOffset, HwregSize>;
1195
1196} // namespace Hwreg
1197
1198namespace DepCtr {
1199
1200int getDefaultDepCtrEncoding(const MCSubtargetInfo &STI);
1201int encodeDepCtr(const StringRef Name, int64_t Val, unsigned &UsedOprMask,
1202 const MCSubtargetInfo &STI);
1203bool isSymbolicDepCtrEncoding(unsigned Code, bool &HasNonDefaultVal,
1204 const MCSubtargetInfo &STI);
1205bool decodeDepCtr(unsigned Code, int &Id, StringRef &Name, unsigned &Val,
1206 bool &IsDefault, const MCSubtargetInfo &STI);
1207
1208/// \returns Maximum VaVdst value that can be encoded.
1209unsigned getVaVdstBitMask();
1210
1211/// \returns Maximum VaSdst value that can be encoded.
1212unsigned getVaSdstBitMask();
1213
1214/// \returns Maximum VaSsrc value that can be encoded.
1215unsigned getVaSsrcBitMask();
1216
1217/// \returns Maximum HoldCnt value that can be encoded.
1218unsigned getHoldCntBitMask(const IsaVersion &Version);
1219
1220/// \returns Maximum VmVsrc value that can be encoded.
1221unsigned getVmVsrcBitMask();
1222
1223/// \returns Maximum VaVcc value that can be encoded.
1224unsigned getVaVccBitMask();
1225
1226/// \returns Maximum SaSdst value that can be encoded.
1227unsigned getSaSdstBitMask();
1228
1229/// \returns Decoded VaVdst from given immediate \p Encoded.
1230unsigned decodeFieldVaVdst(unsigned Encoded);
1231
1232/// \returns Decoded VmVsrc from given immediate \p Encoded.
1233unsigned decodeFieldVmVsrc(unsigned Encoded);
1234
1235/// \returns Decoded SaSdst from given immediate \p Encoded.
1236unsigned decodeFieldSaSdst(unsigned Encoded);
1237
1238/// \returns Decoded VaSdst from given immediate \p Encoded.
1239unsigned decodeFieldVaSdst(unsigned Encoded);
1240
1241/// \returns Decoded VaVcc from given immediate \p Encoded.
1242unsigned decodeFieldVaVcc(unsigned Encoded);
1243
1244/// \returns Decoded SaSrc from given immediate \p Encoded.
1245unsigned decodeFieldVaSsrc(unsigned Encoded);
1246
1247/// \returns Decoded HoldCnt from given immediate \p Encoded.
1248unsigned decodeFieldHoldCnt(unsigned Encoded, const IsaVersion &Version);
1249
1250/// \returns \p VmVsrc as an encoded Depctr immediate.
1251unsigned encodeFieldVmVsrc(unsigned VmVsrc, const MCSubtargetInfo &STI);
1252
1253/// \returns \p Encoded combined with encoded \p VmVsrc.
1254unsigned encodeFieldVmVsrc(unsigned Encoded, unsigned VmVsrc);
1255
1256/// \returns \p VaVdst as an encoded Depctr immediate.
1257unsigned encodeFieldVaVdst(unsigned VaVdst, const MCSubtargetInfo &STI);
1258
1259/// \returns \p Encoded combined with encoded \p VaVdst.
1260unsigned encodeFieldVaVdst(unsigned Encoded, unsigned VaVdst);
1261
1262/// \returns \p SaSdst as an encoded Depctr immediate.
1263unsigned encodeFieldSaSdst(unsigned SaSdst, const MCSubtargetInfo &STI);
1264
1265/// \returns \p Encoded combined with encoded \p SaSdst.
1266unsigned encodeFieldSaSdst(unsigned Encoded, unsigned SaSdst);
1267
1268/// \returns \p VaSdst as an encoded Depctr immediate.
1269unsigned encodeFieldVaSdst(unsigned VaSdst, const MCSubtargetInfo &STI);
1270
1271/// \returns \p Encoded combined with encoded \p VaSdst.
1272unsigned encodeFieldVaSdst(unsigned Encoded, unsigned VaSdst);
1273
1274/// \returns \p VaVcc as an encoded Depctr immediate.
1275unsigned encodeFieldVaVcc(unsigned VaVcc, const MCSubtargetInfo &STI);
1276
1277/// \returns \p Encoded combined with encoded \p VaVcc.
1278unsigned encodeFieldVaVcc(unsigned Encoded, unsigned VaVcc);
1279
1280/// \returns \p HoldCnt as an encoded Depctr immediate.
1281unsigned encodeFieldHoldCnt(unsigned HoldCnt, const MCSubtargetInfo &STI);
1282
1283/// \returns \p Encoded combined with encoded \p HoldCnt.
1284unsigned encodeFieldHoldCnt(unsigned Encoded, unsigned HoldCnt,
1285 const IsaVersion &Version);
1286
1287/// \returns \p VaSsrc as an encoded Depctr immediate.
1288unsigned encodeFieldVaSsrc(unsigned VaSsrc, const MCSubtargetInfo &STI);
1289
1290/// \returns \p Encoded combined with encoded \p VaSsrc.
1291unsigned encodeFieldVaSsrc(unsigned Encoded, unsigned VaSsrc);
1292
1293} // namespace DepCtr
1294
1295namespace Exp {
1296
1297bool getTgtName(unsigned Id, StringRef &Name, int &Index);
1298
1299LLVM_READONLY
1300unsigned getTgtId(const StringRef Name);
1301
1302LLVM_READNONE
1303bool isSupportedTgtId(unsigned Id, const MCSubtargetInfo &STI);
1304
1305} // namespace Exp
1306
1307namespace MTBUFFormat {
1308
1309LLVM_READNONE
1310int64_t encodeDfmtNfmt(unsigned Dfmt, unsigned Nfmt);
1311
1312void decodeDfmtNfmt(unsigned Format, unsigned &Dfmt, unsigned &Nfmt);
1313
1314int64_t getDfmt(const StringRef Name);
1315
1316StringRef getDfmtName(unsigned Id);
1317
1318int64_t getNfmt(const StringRef Name, const MCSubtargetInfo &STI);
1319
1320StringRef getNfmtName(unsigned Id, const MCSubtargetInfo &STI);
1321
1322bool isValidDfmtNfmt(unsigned Val, const MCSubtargetInfo &STI);
1323
1324bool isValidNfmt(unsigned Val, const MCSubtargetInfo &STI);
1325
1326int64_t getUnifiedFormat(const StringRef Name, const MCSubtargetInfo &STI);
1327
1328StringRef getUnifiedFormatName(unsigned Id, const MCSubtargetInfo &STI);
1329
1330bool isValidUnifiedFormat(unsigned Val, const MCSubtargetInfo &STI);
1331
1332int64_t convertDfmtNfmt2Ufmt(unsigned Dfmt, unsigned Nfmt,
1333 const MCSubtargetInfo &STI);
1334
1335bool isValidFormatEncoding(unsigned Val, const MCSubtargetInfo &STI);
1336
1337unsigned getDefaultFormatEncoding(const MCSubtargetInfo &STI);
1338
1339} // namespace MTBUFFormat
1340
1341namespace SendMsg {
1342
1343LLVM_READNONE
1344bool isValidMsgId(int64_t MsgId, const MCSubtargetInfo &STI);
1345
1346LLVM_READNONE
1347bool isValidMsgOp(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI,
1348 bool Strict = true);
1349
1350LLVM_READNONE
1351bool isValidMsgStream(int64_t MsgId, int64_t OpId, int64_t StreamId,
1352 const MCSubtargetInfo &STI, bool Strict = true);
1353
1354LLVM_READNONE
1355bool msgRequiresOp(int64_t MsgId, const MCSubtargetInfo &STI);
1356
1357LLVM_READNONE
1358bool msgSupportsStream(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI);
1359
1360void decodeMsg(unsigned Val, uint16_t &MsgId, uint16_t &OpId,
1361 uint16_t &StreamId, const MCSubtargetInfo &STI);
1362
1363LLVM_READNONE
1364uint64_t encodeMsg(uint64_t MsgId, uint64_t OpId, uint64_t StreamId);
1365
1366/// Returns true if the message does not use the m0 operand.
1367bool msgDoesNotUseM0(int64_t MsgId, const MCSubtargetInfo &STI);
1368
1369} // namespace SendMsg
1370
1371unsigned getInitialPSInputAddr(const Function &F);
1372
1373bool getHasColorExport(const Function &F);
1374
1375bool getHasDepthExport(const Function &F);
1376
1377// Returns the value of the "amdgpu-dynamic-vgpr-block-size" attribute, or 0 if
1378// the attribute is missing or its value is invalid.
1379unsigned getDynamicVGPRBlockSize(const Function &F);
1380
1381LLVM_READNONE
1382constexpr bool isShader(CallingConv::ID CC) {
1383 switch (CC) {
1384 case CallingConv::AMDGPU_VS:
1385 case CallingConv::AMDGPU_LS:
1386 case CallingConv::AMDGPU_HS:
1387 case CallingConv::AMDGPU_ES:
1388 case CallingConv::AMDGPU_GS:
1389 case CallingConv::AMDGPU_PS:
1390 case CallingConv::AMDGPU_CS_Chain:
1391 case CallingConv::AMDGPU_CS_ChainPreserve:
1392 case CallingConv::AMDGPU_CS:
1393 return true;
1394 default:
1395 return false;
1396 }
1397}
1398
1399LLVM_READNONE
1400constexpr bool isGraphics(CallingConv::ID CC) {
1401 return isShader(CC) || CC == CallingConv::AMDGPU_Gfx ||
1402 CC == CallingConv::AMDGPU_Gfx_WholeWave;
1403}
1404
1405LLVM_READNONE
1406constexpr bool isCompute(CallingConv::ID CC) {
1407 return !isGraphics(CC) || CC == CallingConv::AMDGPU_CS;
1408}
1409
1410LLVM_READNONE
1411constexpr bool isEntryFunctionCC(CallingConv::ID CC) {
1412 switch (CC) {
1413 case CallingConv::AMDGPU_KERNEL:
1414 case CallingConv::SPIR_KERNEL:
1415 case CallingConv::AMDGPU_VS:
1416 case CallingConv::AMDGPU_GS:
1417 case CallingConv::AMDGPU_PS:
1418 case CallingConv::AMDGPU_CS:
1419 case CallingConv::AMDGPU_ES:
1420 case CallingConv::AMDGPU_HS:
1421 case CallingConv::AMDGPU_LS:
1422 return true;
1423 default:
1424 return false;
1425 }
1426}
1427
1428LLVM_READNONE
1429constexpr bool isChainCC(CallingConv::ID CC) {
1430 switch (CC) {
1431 case CallingConv::AMDGPU_CS_Chain:
1432 case CallingConv::AMDGPU_CS_ChainPreserve:
1433 return true;
1434 default:
1435 return false;
1436 }
1437}
1438
1439// These functions are considered entrypoints into the current module, i.e. they
1440// are allowed to be called from outside the current module. This is different
1441// from isEntryFunctionCC, which is only true for functions that are entered by
1442// the hardware. Module entry points include all entry functions but also
1443// include functions that can be called from other functions inside or outside
1444// the current module. Module entry functions are allowed to allocate LDS.
1445//
1446// AMDGPU_CS_Chain is intended for externally callable chain functions, so it is
1447// treated as a module entrypoint. AMDGPU_CS_ChainPreserve is used for internal
1448// helper functions (e.g. retry helpers), so it is not a module entrypoint.
1449LLVM_READNONE
1450constexpr bool isModuleEntryFunctionCC(CallingConv::ID CC) {
1451 switch (CC) {
1452 case CallingConv::AMDGPU_Gfx:
1453 case CallingConv::AMDGPU_CS_Chain:
1454 return true;
1455 default:
1456 return isEntryFunctionCC(CC);
1457 }
1458}
1459
1460LLVM_READNONE
1461constexpr inline bool isKernel(CallingConv::ID CC) {
1462 switch (CC) {
1463 case CallingConv::AMDGPU_KERNEL:
1464 case CallingConv::SPIR_KERNEL:
1465 return true;
1466 default:
1467 return false;
1468 }
1469}
1470
1471inline bool isKernel(const Function &F) { return isKernel(CC: F.getCallingConv()); }
1472
1473LLVM_READNONE
1474constexpr bool canGuaranteeTCO(CallingConv::ID CC) {
1475 return CC == CallingConv::Fast;
1476}
1477
1478/// Return true if we might ever do TCO for calls with this calling convention.
1479LLVM_READNONE
1480constexpr bool mayTailCallThisCC(CallingConv::ID CC) {
1481 switch (CC) {
1482 case CallingConv::C:
1483 case CallingConv::AMDGPU_Gfx:
1484 case CallingConv::AMDGPU_Gfx_WholeWave:
1485 return true;
1486 default:
1487 return canGuaranteeTCO(CC);
1488 }
1489}
1490
1491bool hasMIMG_R128(const MCSubtargetInfo &STI);
1492bool hasA16(const MCSubtargetInfo &STI);
1493bool hasG16(const MCSubtargetInfo &STI);
1494bool hasPackedD16(const MCSubtargetInfo &STI);
1495bool hasGDS(const MCSubtargetInfo &STI);
1496unsigned getNSAMaxSize(const MCSubtargetInfo &STI, bool HasSampler = false);
1497unsigned getMaxNumUserSGPRs(const MCSubtargetInfo &STI);
1498
1499bool isSI(const MCSubtargetInfo &STI);
1500bool isCI(const MCSubtargetInfo &STI);
1501bool isVI(const MCSubtargetInfo &STI);
1502bool isGFX9(const MCSubtargetInfo &STI);
1503bool isGFX9_GFX10(const MCSubtargetInfo &STI);
1504bool isGFX9_GFX10_GFX11(const MCSubtargetInfo &STI);
1505bool isGFX8_GFX9_GFX10(const MCSubtargetInfo &STI);
1506bool isGFX8Plus(const MCSubtargetInfo &STI);
1507bool isGFX9Plus(const MCSubtargetInfo &STI);
1508bool isNotGFX9Plus(const MCSubtargetInfo &STI);
1509bool isGFX10(const MCSubtargetInfo &STI);
1510bool isGFX10_GFX11(const MCSubtargetInfo &STI);
1511bool isGFX10Plus(const MCSubtargetInfo &STI);
1512bool isNotGFX10Plus(const MCSubtargetInfo &STI);
1513bool isGFX10Before1030(const MCSubtargetInfo &STI);
1514bool isGFX11(const MCSubtargetInfo &STI);
1515bool isGFX11Plus(const MCSubtargetInfo &STI);
1516bool isGFX12(const MCSubtargetInfo &STI);
1517bool isGFX12Plus(const MCSubtargetInfo &STI);
1518bool isGFX1250(const MCSubtargetInfo &STI);
1519bool isGFX1250Plus(const MCSubtargetInfo &STI);
1520bool isGFX13(const MCSubtargetInfo &STI);
1521bool isGFX13Plus(const MCSubtargetInfo &STI);
1522
1523/// \returns true if a work-group's waves run on all four SIMD32s (one
1524/// contiguous LDS) and not just on two.
1525bool isFullSIMDMode(const MCSubtargetInfo &STI);
1526
1527bool supportsWGP(const MCSubtargetInfo &STI);
1528bool isNotGFX12Plus(const MCSubtargetInfo &STI);
1529bool isNotGFX11Plus(const MCSubtargetInfo &STI);
1530bool isGCN3Encoding(const MCSubtargetInfo &STI);
1531bool isGFX10_BEncoding(const MCSubtargetInfo &STI);
1532bool hasGFX10_3Insts(const MCSubtargetInfo &STI);
1533bool isGFX10_3_GFX11(const MCSubtargetInfo &STI);
1534bool isGFX90A(const MCSubtargetInfo &STI);
1535bool isGFX940(const MCSubtargetInfo &STI);
1536bool hasArchitectedFlatScratch(const MCSubtargetInfo &STI);
1537bool hasMAIInsts(const MCSubtargetInfo &STI);
1538bool hasPopsExitingWaveID(const MCSubtargetInfo &STI);
1539
1540/// \returns true if the src_private_base and src_private_limit aperture
1541/// registers are available on \p STI. Targets with globally addressable
1542/// scratch have no private aperture and expose src_flat_scratch_base instead.
1543bool hasPrivateApertureRegs(const MCSubtargetInfo &STI);
1544
1545bool hasVOPD(const MCSubtargetInfo &STI);
1546bool hasDPPSrc1SGPR(const MCSubtargetInfo &STI);
1547
1548int getTotalNumVGPRs(bool has90AInsts, int32_t ArgNumAGPR, int32_t ArgNumVGPR);
1549unsigned hasKernargPreload(const MCSubtargetInfo &STI);
1550bool hasSMRDSignedImmOffset(const MCSubtargetInfo &ST);
1551
1552/// Is Reg - scalar register
1553bool isSGPR(MCRegister Reg, const MCRegisterInfo *TRI);
1554
1555/// \returns true if \p Reg is an indexed-resource index register, i.e. either a
1556/// 32-bit SGPR (uniform-indexed form) or a Lo256 VGPR (per-lane indexed form).
1557bool isRsrcIndexReg(MCRegister Reg, const MCRegisterInfo &MRI);
1558
1559/// \returns if \p Reg occupies the high 16-bits of a 32-bit register.
1560bool isHi16Reg(MCRegister Reg, const MCRegisterInfo &MRI);
1561
1562/// If \p Reg is a pseudo reg, return the correct hardware register given
1563/// \p STI otherwise return \p Reg.
1564MCRegister getMCReg(MCRegister Reg, const MCSubtargetInfo &STI);
1565
1566/// Convert hardware register \p Reg to a pseudo register
1567LLVM_READNONE
1568MCRegister mc2PseudoReg(MCRegister Reg);
1569
1570LLVM_READNONE
1571bool isInlineValue(MCRegister Reg);
1572
1573/// Is this an AMDGPU specific source operand? These include registers,
1574/// inline constants, literals and mandatory literals (KImm).
1575constexpr bool isSISrcOperand(const MCOperandInfo &OpInfo) {
1576 return OpInfo.OperandType >= AMDGPU::OPERAND_SRC_FIRST &&
1577 OpInfo.OperandType <= AMDGPU::OPERAND_SRC_LAST;
1578}
1579
1580inline bool isSISrcOperand(const MCInstrDesc &Desc, unsigned OpNo) {
1581 return isSISrcOperand(OpInfo: Desc.operands()[OpNo]);
1582}
1583
1584/// Is this a scalar (i.e. not packed) bf16 source operand?
1585constexpr bool isBF16SrcOperand(const MCOperandInfo &OpInfo) {
1586 return OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_BF16 ||
1587 OpInfo.OperandType == AMDGPU::OPERAND_REG_INLINE_C_BF16;
1588}
1589
1590/// Is this a KImm operand?
1591bool isKImmOperand(const MCInstrDesc &Desc, unsigned OpNo);
1592
1593/// Is this floating-point operand?
1594bool isSISrcFPOperand(const MCInstrDesc &Desc, unsigned OpNo);
1595
1596/// Get the size in bits of a register from the register class \p RC.
1597unsigned getRegBitWidth(unsigned RCID);
1598
1599/// Get the size in bits of a register from the register class \p RC.
1600unsigned getRegBitWidth(const MCRegisterClass &RC);
1601
1602LLVM_READNONE
1603inline unsigned getOperandSize(const MCOperandInfo &OpInfo) {
1604 switch (OpInfo.OperandType) {
1605 case AMDGPU::OPERAND_REG_IMM_INT32:
1606 case AMDGPU::OPERAND_REG_IMM_FP32:
1607 case AMDGPU::OPERAND_REG_INLINE_C_INT32:
1608 case AMDGPU::OPERAND_REG_INLINE_C_FP32:
1609 case AMDGPU::OPERAND_REG_INLINE_AC_INT32:
1610 case AMDGPU::OPERAND_REG_INLINE_AC_FP32:
1611 case AMDGPU::OPERAND_REG_IMM_V2INT32:
1612 case AMDGPU::OPERAND_REG_IMM_V2FP32:
1613 case AMDGPU::OPERAND_KIMM32:
1614 case AMDGPU::OPERAND_KIMM16: // mandatory literal is always size 4
1615 case AMDGPU::OPERAND_INLINE_SPLIT_BARRIER_INT32:
1616 return 4;
1617
1618 case AMDGPU::OPERAND_REG_IMM_INT64:
1619 case AMDGPU::OPERAND_REG_IMM_FP64:
1620 case AMDGPU::OPERAND_REG_INLINE_C_INT64:
1621 case AMDGPU::OPERAND_REG_INLINE_C_FP64:
1622 case AMDGPU::OPERAND_REG_INLINE_AC_FP64:
1623 case AMDGPU::OPERAND_REG_IMM_V2FP64:
1624 case AMDGPU::OPERAND_REG_IMM_V2INT64:
1625 case AMDGPU::OPERAND_KIMM64:
1626 return 8;
1627
1628 case AMDGPU::OPERAND_REG_IMM_INT16:
1629 case AMDGPU::OPERAND_REG_IMM_BF16:
1630 case AMDGPU::OPERAND_REG_IMM_FP16:
1631 case AMDGPU::OPERAND_REG_IMM_NOINLINE_FP16:
1632 case AMDGPU::OPERAND_REG_INLINE_C_INT16:
1633 case AMDGPU::OPERAND_REG_INLINE_C_BF16:
1634 case AMDGPU::OPERAND_REG_INLINE_C_FP16:
1635 case AMDGPU::OPERAND_REG_INLINE_C_V2INT16:
1636 case AMDGPU::OPERAND_REG_INLINE_C_V2BF16:
1637 case AMDGPU::OPERAND_REG_INLINE_C_V2FP16:
1638 case AMDGPU::OPERAND_REG_IMM_V2INT16:
1639 case AMDGPU::OPERAND_REG_IMM_V2BF16:
1640 case AMDGPU::OPERAND_REG_IMM_V2FP16:
1641 case AMDGPU::OPERAND_REG_IMM_V2FP16_SPLAT:
1642 case AMDGPU::OPERAND_REG_IMM_NOINLINE_V2FP16:
1643 return 2;
1644
1645 default:
1646 llvm_unreachable("unhandled operand type");
1647 }
1648}
1649
1650LLVM_READNONE
1651inline unsigned getOperandSize(const MCInstrDesc &Desc, unsigned OpNo) {
1652 return getOperandSize(OpInfo: Desc.operands()[OpNo]);
1653}
1654
1655/// Is this literal inlinable, and not one of the values intended for floating
1656/// point values.
1657LLVM_READNONE
1658inline bool isInlinableIntLiteral(int64_t Literal) {
1659 return Literal >= -16 && Literal <= 64;
1660}
1661
1662/// Is this literal inlinable
1663LLVM_READNONE
1664bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi);
1665
1666LLVM_READNONE
1667bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi);
1668
1669LLVM_READNONE
1670bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi);
1671
1672LLVM_READNONE
1673bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi);
1674
1675LLVM_READNONE
1676bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi);
1677
1678LLVM_READNONE
1679std::optional<unsigned> getInlineEncodingV2I16(uint32_t Literal);
1680
1681LLVM_READNONE
1682std::optional<unsigned> getInlineEncodingV2BF16(uint32_t Literal);
1683
1684LLVM_READNONE
1685std::optional<unsigned> getInlineEncodingV2F16(uint32_t Literal);
1686
1687LLVM_READNONE
1688std::optional<unsigned> getPKFMACF16InlineEncoding(uint32_t Literal,
1689 bool IsGFX11Plus);
1690
1691LLVM_READNONE
1692bool isInlinableLiteralV216(uint32_t Literal, uint8_t OpType);
1693
1694LLVM_READNONE
1695bool isInlinableLiteralV2I16(uint32_t Literal);
1696
1697LLVM_READNONE
1698bool isInlinableLiteralV2BF16(uint32_t Literal);
1699
1700LLVM_READNONE
1701bool isInlinableLiteralV2F16(uint32_t Literal);
1702
1703LLVM_READNONE
1704bool isPKFMACF16InlineConstant(uint32_t Literal, bool IsGFX11Plus);
1705
1706LLVM_READNONE
1707bool isValid32BitLiteral(uint64_t Val, bool IsFP64);
1708
1709LLVM_READNONE
1710int64_t encode32BitLiteral(int64_t Imm, OperandType Type, bool IsLit);
1711
1712bool isArgPassedInSGPR(const Argument *Arg);
1713
1714bool isArgPassedInSGPR(const CallBase *CB, unsigned ArgNo);
1715
1716/// The opcode is a packed fp32 instruction which only reads low 32 bits of
1717/// a scalar operand and propagates it to high channel.
1718LLVM_READONLY bool isPackedSingleSGPRFP32Inst(unsigned Opc);
1719
1720/// The opcode is a packed 64-bit instruction which only reads low 64 bits of
1721/// a scalar operand and propagates it to high channel.
1722LLVM_READONLY bool isPackedSingleSGPR64BitInst(unsigned Opc);
1723
1724/// Packed instructions that read a single SGPR for SGPR operands, except for
1725/// 64-bit elements which read two SGPRs.
1726LLVM_READONLY bool isSingleSGPRReadInst(unsigned Opc);
1727
1728LLVM_READONLY
1729bool isLegalSMRDEncodedUnsignedOffset(const MCSubtargetInfo &ST,
1730 int64_t EncodedOffset);
1731
1732LLVM_READONLY
1733bool isLegalSMRDEncodedSignedOffset(const MCSubtargetInfo &ST,
1734 int64_t EncodedOffset, bool IsBuffer);
1735
1736/// Convert \p ByteOffset to dwords if the subtarget uses dword SMRD immediate
1737/// offsets.
1738uint64_t convertSMRDOffsetUnits(const MCSubtargetInfo &ST, uint64_t ByteOffset);
1739
1740/// \returns The encoding that will be used for \p ByteOffset in the
1741/// SMRD offset field, or std::nullopt if it won't fit. On GFX9 and GFX10
1742/// S_LOAD instructions have a signed offset, on other subtargets it is
1743/// unsigned. S_BUFFER has an unsigned offset for all subtargets.
1744std::optional<int64_t> getSMRDEncodedOffset(const MCSubtargetInfo &ST,
1745 int64_t ByteOffset, bool IsBuffer,
1746 bool HasSOffset = false);
1747
1748/// \return The encoding that can be used for a 32-bit literal offset in an SMRD
1749/// instruction. This is only useful on CI.s
1750std::optional<int64_t> getSMRDEncodedLiteralOffset32(const MCSubtargetInfo &ST,
1751 int64_t ByteOffset);
1752
1753/// For pre-GFX12 FLAT instructions the offset must be positive;
1754/// MSB is ignored and forced to zero.
1755///
1756/// \return The number of bits available for the signed offset field in flat
1757/// instructions. Note that some forms of the instruction disallow negative
1758/// offsets.
1759unsigned getNumFlatOffsetBits(const MCSubtargetInfo &ST);
1760
1761LLVM_READNONE
1762inline bool isLegalDPALU_DPPControl(const MCSubtargetInfo &ST, unsigned DC) {
1763 if (isGFX12(STI: ST))
1764 return DC >= DPP::ROW_SHARE_FIRST && DC <= DPP::ROW_SHARE_LAST;
1765 if (isGFX90A(STI: ST))
1766 return DC >= DPP::ROW_NEWBCAST_FIRST && DC <= DPP::ROW_NEWBCAST_LAST;
1767 return false;
1768}
1769
1770/// \returns true if an instruction is a DP ALU DPP without any 64-bit operands.
1771bool isDPALU_DPP32BitOpc(unsigned Opc);
1772
1773/// \returns true if an instruction is a DP ALU DPP.
1774bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII,
1775 const MCSubtargetInfo &ST);
1776
1777/// \returns true if the intrinsic is divergent
1778bool isIntrinsicSourceOfDivergence(unsigned IntrID);
1779
1780/// \returns true if the intrinsic is uniform
1781bool isIntrinsicAlwaysUniform(unsigned IntrID);
1782
1783/// \returns a register class for the physical register \p Reg if it is a VGPR
1784/// or nullptr otherwise.
1785const MCRegisterClass *getVGPRPhysRegClass(MCRegister Reg,
1786 const MCRegisterInfo &MRI);
1787
1788/// \returns the MODE bits which have to be set by the S_SET_VGPR_MSB for the
1789/// physical register \p Reg.
1790unsigned getVGPREncodingMSBs(MCRegister Reg, const MCRegisterInfo &MRI);
1791
1792/// If \p Reg is a low VGPR return a corresponding high VGPR with \p MSBs set.
1793MCRegister getVGPRWithMSBs(MCRegister Reg, unsigned MSBs,
1794 const MCRegisterInfo &MRI);
1795
1796/// \returns VGPR MSBs encoded in a S_SETREG_IMM32_B32 \p MI if it sets
1797/// it. If \p HasSetregVGPRMSBFixup is true then size of the ID_MODE mask is
1798/// ignored.
1799std::optional<unsigned> convertSetRegImmToVgprMSBs(const MachineInstr &MI,
1800 bool HasSetregVGPRMSBFixup);
1801
1802/// \returns VGPR MSBs encoded in a S_SETREG_IMM32_B32 \p MI if it sets
1803/// it. If \p HasSetregVGPRMSBFixup is true then size of the ID_MODE mask is
1804/// ignored.
1805std::optional<unsigned> convertSetRegImmToVgprMSBs(const MCInst &MI,
1806 bool HasSetregVGPRMSBFixup);
1807
1808// Returns a table for the opcode with a given \p Desc to map the VGPR MSB
1809// set by the S_SET_VGPR_MSB to one of 4 sources. In case of VOPD returns 2
1810// maps, one for X and one for Y component.
1811std::pair<const AMDGPU::OpName *, const AMDGPU::OpName *>
1812getVGPRLoweringOperandTables(const MCInstrDesc &Desc);
1813
1814/// \returns true if a memory instruction supports scale_offset modifier.
1815bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode);
1816
1817class ClusterDimsAttr {
1818public:
1819 enum class Kind { Unknown, NoCluster, VariableDims, FixedDims };
1820
1821 ClusterDimsAttr() = default;
1822
1823 Kind getKind() const { return AttrKind; }
1824
1825 bool isUnknown() const { return getKind() == Kind::Unknown; }
1826
1827 bool isNoCluster() const { return getKind() == Kind::NoCluster; }
1828
1829 bool isFixedDims() const { return getKind() == Kind::FixedDims; }
1830
1831 bool isVariableDims() const { return getKind() == Kind::VariableDims; }
1832
1833 void setUnknown() { *this = ClusterDimsAttr(Kind::Unknown); }
1834
1835 void setVariableDims() { *this = ClusterDimsAttr(Kind::VariableDims); }
1836
1837 /// \returns the dims stored. Note that this function can only be called if
1838 /// the kind is \p Fixed.
1839 const std::array<unsigned, 3> &getDims() const;
1840
1841 bool operator==(const ClusterDimsAttr &RHS) const {
1842 return AttrKind == RHS.AttrKind && Dims == RHS.Dims;
1843 }
1844
1845 std::string to_string() const;
1846
1847 static ClusterDimsAttr get(const Function &F);
1848
1849private:
1850 enum Encoding { EncoNoCluster = 0, EncoVariableDims = 1024 };
1851
1852 ClusterDimsAttr(Kind AttrKind) : AttrKind(AttrKind) {}
1853
1854 std::array<unsigned, 3> Dims = {0, 0, 0};
1855
1856 Kind AttrKind = Kind::Unknown;
1857};
1858
1859/// Evaluate the constant-folded result of v_rcp for \p Val, accounting for
1860/// the hardware's denormal flushing on f32/f64 and its approximate rounding.
1861/// Returns std::nullopt if the hardware result is not guaranteed to match the
1862/// exact reciprocal.
1863std::optional<APFloat> evaluateRcp(const APFloat &Val);
1864
1865} // namespace AMDGPU
1866
1867raw_ostream &operator<<(raw_ostream &OS, const AMDGPU::TargetIDSetting S);
1868
1869} // end namespace llvm
1870
1871#endif // LLVM_LIB_TARGET_AMDGPU_UTILS_AMDGPUBASEINFO_H
1872