1//===-- SIRegisterInfo.h - SI Register Info Interface ----------*- C++ -*--===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// Interface definition for SIRegisterInfo
11//
12//===----------------------------------------------------------------------===//
13
14#ifndef LLVM_LIB_TARGET_AMDGPU_SIREGISTERINFO_H
15#define LLVM_LIB_TARGET_AMDGPU_SIREGISTERINFO_H
16
17#include "llvm/ADT/BitVector.h"
18
19#define GET_REGINFO_HEADER
20#include "AMDGPUGenRegisterInfo.inc"
21
22#include "SIDefines.h"
23
24namespace llvm {
25
26class GCNSubtarget;
27class LiveIntervals;
28class LiveRegUnits;
29class MachineInstrBuilder;
30class RegisterBank;
31struct SGPRSpillBuilder;
32
33/// Register allocation hint types. Helps eliminate unneeded COPY with True16
34namespace AMDGPURI {
35
36enum { Size16 = 1, Size32 = 2 };
37
38} // end namespace AMDGPURI
39
40class SIRegisterInfo final : public AMDGPUGenRegisterInfo {
41private:
42 const GCNSubtarget &ST;
43 bool SpillSGPRToVGPR;
44 bool isWave32;
45 BitVector RegPressureIgnoredUnits;
46
47 /// Sub reg indexes for getRegSplitParts.
48 /// First index represents subreg size from 1 to 32 Half DWORDS.
49 /// The inner vector is sorted by bit offset.
50 /// Provided a register can be fully split with given subregs,
51 /// all elements of the inner vector combined give a full lane mask.
52 static std::array<std::vector<int16_t>, 32> RegSplitParts;
53
54 // Table representing sub reg of given width and offset.
55 // First index is subreg size: 32, 64, 96, 128, 160, 192, 224, 256, 512.
56 // Second index is 32 different dword offsets.
57 static std::array<std::array<uint16_t, 32>, 9> SubRegFromChannelTable;
58
59 void reserveRegisterTuples(BitVector &, MCRegister Reg) const;
60
61public:
62 SIRegisterInfo(const GCNSubtarget &ST);
63
64 struct SpilledReg {
65 Register VGPR;
66 int Lane = -1;
67
68 SpilledReg() = default;
69 SpilledReg(Register R, int L) : VGPR(R), Lane(L) {}
70
71 bool hasLane() { return Lane != -1; }
72 bool hasReg() { return VGPR != 0; }
73 };
74
75 /// \returns the sub reg enum value for the given \p Channel
76 /// (e.g. getSubRegFromChannel(0) -> AMDGPU::sub0)
77 static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs = 1);
78
79 bool spillSGPRToVGPR() const {
80 return SpillSGPRToVGPR;
81 }
82
83 bool isCFISavedRegsSpillEnabled() const;
84
85 /// Return the largest available SGPR aligned to \p Align for the register
86 /// class \p RC.
87 MCRegister getAlignedHighSGPRForRC(const MachineFunction &MF,
88 const unsigned Align,
89 const TargetRegisterClass *RC) const;
90
91 /// Return the end register initially reserved for the scratch buffer in case
92 /// spilling is needed.
93 MCRegister reservedPrivateSegmentBufferReg(const MachineFunction &MF) const;
94
95 BitVector getReservedRegs(const MachineFunction &MF) const override;
96 bool isAsmClobberable(const MachineFunction &MF,
97 MCRegister PhysReg) const override;
98
99 const MCPhysReg *getCalleeSavedRegs(const MachineFunction *MF) const override;
100 const MCPhysReg *getCalleeSavedRegsViaCopy(const MachineFunction *MF) const;
101 const uint32_t *getCallPreservedMask(const MachineFunction &MF,
102 CallingConv::ID) const override;
103 const uint32_t *getNoPreservedMask() const override;
104
105 // Functions with the amdgpu_cs_chain or amdgpu_cs_chain_preserve calling
106 // conventions are free to use certain VGPRs without saving and restoring any
107 // lanes (not even inactive ones).
108 static bool isChainScratchRegister(Register VGPR);
109
110 unsigned getCSRFirstUseCost(const MachineFunction &) const override {
111 // The cost of 27 balances multiple factors that influence CSR cost:
112 // - Saving a SGPR CSR to VGPR lanes is relatively cheap.
113 // - In cases where this is not possible, stack access is very expensive.
114 // - CSRs are also the high registers, and we want to minimize the number of
115 // used registers as it impacts occupancy.
116 // Note: Register allocation only applies these cost to callee-save
117 // registers according to getCalleeSavedRegs, so handling of calling
118 // conventions with no CSR is handled there.
119 return 27;
120 }
121
122 // When building a block VGPR load, we only really transfer a subset of the
123 // registers in the block, based on a mask. Liveness analysis is not aware of
124 // the mask, so it might consider that any register in the block is available
125 // before the load and may therefore be scavenged. This is not ok for CSRs
126 // that are not clobbered, since the caller will expect them to be preserved.
127 // This method will add artificial implicit uses for those registers on the
128 // load instruction, so liveness analysis knows they're unavailable.
129 void addImplicitUsesForBlockCSRLoad(MachineInstrBuilder &MIB,
130 Register BlockReg) const;
131
132 // Iterate over all VGPRs in the given BlockReg and emit CFI for each VGPR
133 // as-needed depending on the (statically known) mask, relative to the given
134 // base Offset.
135 void buildCFIForBlockCSRStore(MachineBasicBlock &MBB,
136 MachineBasicBlock::iterator MBBI,
137 Register BlockReg, int64_t Offset) const;
138
139 const TargetRegisterClass *
140 getLargestLegalSuperClass(const TargetRegisterClass *RC,
141 const MachineFunction &MF) const override;
142
143 Register getFrameRegister(const MachineFunction &MF) const override;
144
145 bool hasBasePointer(const MachineFunction &MF) const;
146 Register getBaseRegister() const;
147
148 bool shouldRealignStack(const MachineFunction &MF) const override;
149 bool requiresRegisterScavenging(const MachineFunction &Fn) const override;
150
151 bool requiresFrameIndexScavenging(const MachineFunction &MF) const override;
152 bool requiresFrameIndexReplacementScavenging(
153 const MachineFunction &MF) const override;
154 bool requiresVirtualBaseRegisters(const MachineFunction &Fn) const override;
155
156 int64_t getScratchInstrOffset(const MachineInstr *MI) const;
157
158 int64_t getFrameIndexInstrOffset(const MachineInstr *MI,
159 int Idx) const override;
160
161 bool needsFrameBaseReg(MachineInstr *MI, int64_t Offset) const override;
162
163 Register materializeFrameBaseRegister(MachineBasicBlock *MBB, int FrameIdx,
164 int64_t Offset) const override;
165
166 void resolveFrameIndex(MachineInstr &MI, Register BaseReg,
167 int64_t Offset) const override;
168
169 bool isFrameOffsetLegal(const MachineInstr *MI, Register BaseReg,
170 int64_t Offset) const override;
171
172 /// Returns a legal register class to copy a register in the specified class
173 /// to or from. If it is possible to copy the register directly without using
174 /// a cross register class copy, return the specified RC. Returns NULL if it
175 /// is not possible to copy between two registers of the specified class.
176 const TargetRegisterClass *
177 getCrossCopyRegClass(const TargetRegisterClass *RC) const override;
178
179 const TargetRegisterClass *
180 getRegClassForBlockOp(const MachineFunction &MF) const {
181 return &AMDGPU::VReg_1024RegClass;
182 }
183
184 void buildVGPRSpillLoadStore(SGPRSpillBuilder &SB, int Index, int Offset,
185 bool IsLoad, bool IsKill = true) const;
186
187 /// If \p OnlyToVGPR is true, this will only succeed if this manages to find a
188 /// free VGPR lane to spill.
189 bool spillSGPR(MachineBasicBlock::iterator MI, int FI, RegScavenger *RS,
190 SlotIndexes *Indexes = nullptr, LiveIntervals *LIS = nullptr,
191 bool OnlyToVGPR = false, bool SpillToPhysVGPRLane = false,
192 bool NeedsCFI = false) const;
193
194 bool restoreSGPR(MachineBasicBlock::iterator MI, int FI, RegScavenger *RS,
195 SlotIndexes *Indexes = nullptr, LiveIntervals *LIS = nullptr,
196 bool OnlyToVGPR = false,
197 bool SpillToPhysVGPRLane = false) const;
198
199 bool spillEmergencySGPR(MachineBasicBlock::iterator MI,
200 MachineBasicBlock &RestoreMBB, Register SGPR,
201 RegScavenger *RS) const;
202
203 bool eliminateFrameIndex(MachineBasicBlock::iterator MI, int SPAdj,
204 unsigned FIOperandNum,
205 RegScavenger *RS) const override;
206
207 bool eliminateSGPRToVGPRSpillFrameIndex(
208 MachineBasicBlock::iterator MI, int FI, RegScavenger *RS,
209 SlotIndexes *Indexes = nullptr, LiveIntervals *LIS = nullptr,
210 bool SpillToPhysVGPRLane = false) const;
211
212 StringRef getRegAsmName(MCRegister Reg) const override;
213
214 // Pseudo regs are not allowed
215 unsigned getHWRegIndex(MCRegister Reg) const;
216
217 LLVM_READONLY
218 const TargetRegisterClass *getVGPRClassForBitWidth(unsigned BitWidth) const;
219
220 LLVM_READONLY const TargetRegisterClass *
221 getAlignedLo256VGPRClassForBitWidth(unsigned BitWidth) const;
222
223 LLVM_READONLY
224 const TargetRegisterClass *getAGPRClassForBitWidth(unsigned BitWidth) const;
225
226 LLVM_READONLY
227 const TargetRegisterClass *
228 getVectorSuperClassForBitWidth(unsigned BitWidth) const;
229
230 LLVM_READONLY
231 const TargetRegisterClass *
232 getDefaultVectorSuperClassForBitWidth(unsigned BitWidth) const;
233
234 LLVM_READONLY
235 static const TargetRegisterClass *getSGPRClassForBitWidth(unsigned BitWidth);
236
237 /// \returns true if this class contains only SGPR registers
238 static bool isSGPRClass(const TargetRegisterClass *RC) {
239 return hasSGPRs(RC) && !hasVGPRs(RC) && !hasAGPRs(RC);
240 }
241
242 /// \returns true if this class ID contains only SGPR registers
243 bool isSGPRClassID(unsigned RCID) const {
244 return isSGPRClass(RC: getRegClass(i: RCID));
245 }
246
247 bool isSGPRReg(const MachineRegisterInfo &MRI, Register Reg) const;
248 bool isSGPRPhysReg(Register Reg) const {
249 return isSGPRClass(RC: getPhysRegBaseClass(Reg));
250 }
251
252 bool isVGPRPhysReg(Register Reg) const {
253 return isVGPRClass(RC: getPhysRegBaseClass(Reg));
254 }
255
256 /// \returns true if this class contains only VGPR registers
257 static bool isVGPRClass(const TargetRegisterClass *RC) {
258 return hasVGPRs(RC) && !hasAGPRs(RC) && !hasSGPRs(RC);
259 }
260
261 /// \returns true if this class contains only AGPR registers
262 static bool isAGPRClass(const TargetRegisterClass *RC) {
263 return hasAGPRs(RC) && !hasVGPRs(RC) && !hasSGPRs(RC);
264 }
265
266 /// \returns true only if this class contains both VGPR and AGPR registers
267 bool isVectorSuperClass(const TargetRegisterClass *RC) const {
268 return hasVGPRs(RC) && hasAGPRs(RC) && !hasSGPRs(RC);
269 }
270
271 /// \returns true only if this class contains both VGPR and SGPR registers
272 bool isVSSuperClass(const TargetRegisterClass *RC) const {
273 return hasVGPRs(RC) && hasSGPRs(RC) && !hasAGPRs(RC);
274 }
275
276 /// \returns true if this class contains VGPR registers.
277 static bool hasVGPRs(const TargetRegisterClass *RC) {
278 return RC->TSFlags & SIRCFlags::HasVGPR;
279 }
280
281 /// \returns true if this class contains AGPR registers.
282 static bool hasAGPRs(const TargetRegisterClass *RC) {
283 return RC->TSFlags & SIRCFlags::HasAGPR;
284 }
285
286 /// \returns true if this class contains SGPR registers.
287 static bool hasSGPRs(const TargetRegisterClass *RC) {
288 return RC->TSFlags & SIRCFlags::HasSGPR;
289 }
290
291 /// \returns true if this class contains any vector registers.
292 static bool hasVectorRegisters(const TargetRegisterClass *RC) {
293 return hasVGPRs(RC) || hasAGPRs(RC);
294 }
295
296 /// \returns A VGPR reg class with the same width as \p SRC
297 const TargetRegisterClass *
298 getEquivalentVGPRClass(const TargetRegisterClass *SRC) const;
299
300 /// \returns An AGPR reg class with the same width as \p SRC
301 const TargetRegisterClass *
302 getEquivalentAGPRClass(const TargetRegisterClass *SRC) const;
303
304 /// \returns An AGPR+VGPR super reg class with the same width as \p SRC
305 const TargetRegisterClass *
306 getEquivalentAVClass(const TargetRegisterClass *SRC) const;
307
308 /// \returns A SGPR reg class with the same width as \p SRC
309 const TargetRegisterClass *
310 getEquivalentSGPRClass(const TargetRegisterClass *VRC) const;
311
312 /// Returns a register class which is compatible with \p SuperRC, such that a
313 /// subregister exists with class \p SubRC with subregister index \p
314 /// SubIdx. If this is impossible (e.g., an unaligned subregister index within
315 /// a register tuple), return null.
316 const TargetRegisterClass *
317 getCompatibleSubRegClass(const TargetRegisterClass *SuperRC,
318 const TargetRegisterClass *SubRC,
319 unsigned SubIdx) const;
320
321 /// \returns True if operands defined with this operand type can accept
322 /// a literal constant (i.e. any 32-bit immediate).
323 bool opCanUseLiteralConstant(unsigned OpType) const;
324
325 /// \returns True if operands defined with this operand type can accept
326 /// an inline constant. i.e. An integer value in the range (-16, 64) or
327 /// -4.0f, -2.0f, -1.0f, -0.5f, 0.0f, 0.5f, 1.0f, 2.0f, 4.0f.
328 bool opCanUseInlineConstant(unsigned OpType) const;
329
330 MCRegister findUnusedRegister(const MachineRegisterInfo &MRI,
331 const TargetRegisterClass *RC,
332 const MachineFunction &MF,
333 bool ReserveHighestVGPR = false) const;
334
335 const TargetRegisterClass *getRegClassForReg(const MachineRegisterInfo &MRI,
336 Register Reg) const;
337 const TargetRegisterClass *
338 getRegClassForOperandReg(const MachineRegisterInfo &MRI,
339 const MachineOperand &MO) const;
340
341 bool isVGPR(const MachineRegisterInfo &MRI, Register Reg) const;
342 bool isAGPR(const MachineRegisterInfo &MRI, Register Reg) const;
343 bool isVectorRegister(const MachineRegisterInfo &MRI, Register Reg) const {
344 return isVGPR(MRI, Reg) || isAGPR(MRI, Reg);
345 }
346
347 // FIXME: SGPRs are assumed to be uniform, but this is not true for i1 SGPRs
348 // (such as VCC) which hold a wave-wide vector of boolean values. Examining
349 // just the register class is not suffcient; it needs to be combined with a
350 // value type. The next predicate isUniformReg() does this correctly.
351 bool isDivergentRegClass(const TargetRegisterClass *RC) const override {
352 return !isSGPRClass(RC);
353 }
354
355 bool isUniformReg(const MachineRegisterInfo &MRI, const RegisterBankInfo &RBI,
356 Register Reg) const override;
357
358 ArrayRef<int16_t> getRegSplitParts(const TargetRegisterClass *RC,
359 unsigned EltSize) const;
360
361 unsigned getRegPressureLimit(const TargetRegisterClass *RC,
362 MachineFunction &MF) const override;
363
364 unsigned getRegPressureSetLimit(const MachineFunction &MF,
365 unsigned Idx) const override;
366
367 bool getRegAllocationHints(Register VirtReg, ArrayRef<MCPhysReg> Order,
368 SmallVectorImpl<MCPhysReg> &Hints,
369 const MachineFunction &MF, const VirtRegMap *VRM,
370 const LiveRegMatrix *Matrix) const override;
371
372 const int *getRegUnitPressureSets(MCRegUnit RegUnit) const override;
373
374 MCRegister getReturnAddressReg(const MachineFunction &MF) const;
375
376 const TargetRegisterClass *
377 getRegClassForSizeOnBank(unsigned Size, const RegisterBank &Bank) const;
378
379 const TargetRegisterClass *
380 getRegClassForTypeOnBank(LLT Ty, const RegisterBank &Bank) const {
381 return getRegClassForSizeOnBank(Size: Ty.getSizeInBits(), Bank);
382 }
383
384 const TargetRegisterClass *
385 getConstrainedRegClassForReg(Register Reg,
386 const MachineRegisterInfo &MRI) const override;
387
388 const TargetRegisterClass *getBoolRC() const {
389 return isWave32 ? &AMDGPU::SReg_32RegClass
390 : &AMDGPU::SReg_64RegClass;
391 }
392
393 const TargetRegisterClass *getWaveMaskRegClass() const {
394 return isWave32 ? &AMDGPU::SReg_32_XM0_XEXECRegClass
395 : &AMDGPU::SReg_64_XEXECRegClass;
396 }
397
398 // Return the appropriate register class to use for 64-bit VGPRs for the
399 // subtarget.
400 const TargetRegisterClass *getVGPR64Class() const;
401
402 MCRegister getVCC() const;
403
404 MCRegister getExec() const;
405
406 // Find reaching register definition
407 MachineInstr *findReachingDef(Register Reg, unsigned SubReg,
408 MachineInstr &Use,
409 MachineRegisterInfo &MRI,
410 LiveIntervals *LIS) const;
411
412 const uint32_t *getAllVGPRRegMask() const;
413 const uint32_t *getAllAGPRRegMask() const;
414 const uint32_t *getAllVectorRegMask() const;
415 const uint32_t *getAllAllocatableSRegMask() const;
416
417 // \returns number of 32 bit registers covered by a \p LM
418 static unsigned getNumCoveredRegs(LaneBitmask LM) {
419 // The assumption is that every lo16 subreg is an even bit and every hi16
420 // is an adjacent odd bit or vice versa.
421 uint64_t Mask = LM.getAsInteger();
422 uint64_t Even = Mask & 0xAAAAAAAAAAAAAAAAULL;
423 Mask = (Even >> 1) | Mask;
424 uint64_t Odd = Mask & 0x5555555555555555ULL;
425 return llvm::popcount(Value: Odd);
426 }
427
428 // \returns a DWORD offset of a \p SubReg
429 unsigned getChannelFromSubReg(unsigned SubReg) const {
430 return SubReg ? (getSubRegIdxOffset(Idx: SubReg) + 31) / 32 : 0;
431 }
432
433 // \returns a DWORD size of a \p SubReg
434 unsigned getNumChannelsFromSubReg(unsigned SubReg) const {
435 return getNumCoveredRegs(LM: getSubRegIndexLaneMask(SubIdx: SubReg));
436 }
437
438 // For a given 16 bit \p Reg \returns a 32 bit register holding it.
439 // \returns \p Reg otherwise.
440 MCPhysReg get32BitRegister(MCPhysReg Reg) const;
441
442 // Returns true if a given register class is properly aligned for
443 // the subtarget.
444 bool isProperlyAlignedRC(const TargetRegisterClass &RC) const;
445
446 /// Return all SGPR128 which satisfy the waves per execution unit requirement
447 /// of the subtarget.
448 ArrayRef<MCPhysReg> getAllSGPR128(const MachineFunction &MF) const;
449
450 /// Return all SGPR64 which satisfy the waves per execution unit requirement
451 /// of the subtarget.
452 ArrayRef<MCPhysReg> getAllSGPR64(const MachineFunction &MF) const;
453
454 /// Return all SGPR32 which satisfy the waves per execution unit requirement
455 /// of the subtarget.
456 ArrayRef<MCPhysReg> getAllSGPR32(const MachineFunction &MF) const;
457
458 // Insert spill or restore instructions.
459 // When lowering spill pseudos, the RegScavenger should be set.
460 // For creating spill instructions during frame lowering, where no scavenger
461 // is available, LiveUnits can be used.
462 void buildSpillLoadStore(MachineBasicBlock &MBB,
463 MachineBasicBlock::iterator MI, const DebugLoc &DL,
464 unsigned LoadStoreOp, int Index, Register ValueReg,
465 bool ValueIsKill, MCRegister ScratchOffsetReg,
466 int64_t InstrOffset, MachineMemOperand *MMO,
467 RegScavenger *RS, LiveRegUnits *LiveUnits = nullptr,
468 bool NeedsCFI = false) const;
469
470 // Return alignment in register file of first register in a register tuple.
471 unsigned getRegClassAlignmentNumBits(const TargetRegisterClass *RC) const {
472 return (RC->TSFlags & SIRCFlags::RegTupleAlignUnitsMask) * 32;
473 }
474
475 // Check if register class RC has required alignment.
476 bool isRegClassAligned(const TargetRegisterClass *RC,
477 unsigned AlignNumBits) const {
478 assert(AlignNumBits != 0);
479 unsigned RCAlign = getRegClassAlignmentNumBits(RC);
480 return RCAlign == AlignNumBits ||
481 (RCAlign > AlignNumBits && (RCAlign % AlignNumBits) == 0);
482 }
483
484 // Return alignment of a SubReg relative to start of a register in RC class.
485 // No check if the subreg is supported by the current RC is made.
486 unsigned getSubRegAlignmentNumBits(const TargetRegisterClass *RC,
487 unsigned SubReg) const;
488
489 // \returns a number of registers of a given \p RC used in a function.
490 // Does not go inside function calls. If \p IncludeCalls is true, it will
491 // include registers that may be clobbered by calls.
492 unsigned getNumUsedPhysRegs(const MachineRegisterInfo &MRI,
493 const TargetRegisterClass &RC,
494 bool IncludeCalls = true) const;
495
496 std::optional<uint8_t> getVRegFlagValue(StringRef Name) const override {
497 return Name == "WWM_REG" ? AMDGPU::VirtRegFlag::WWM_REG
498 : std::optional<uint8_t>{};
499 }
500
501 SmallVector<StringLiteral>
502 getVRegFlagsOfReg(Register Reg, const MachineFunction &MF) const override;
503
504 float
505 getSpillWeightScaleFactor(const TargetRegisterClass *RC) const override {
506 // Prioritize VGPR_32_Lo256 over other classes which may occupy registers
507 // beyond v256.
508 return AMDGPUGenRegisterInfo::getSpillWeightScaleFactor(RC) *
509 ((RC == &AMDGPU::VGPR_32_Lo256RegClass ||
510 RC == &AMDGPU::VReg_64_Lo256_Align2RegClass)
511 ? 2.0
512 : 1.0);
513 }
514};
515
516} // End namespace llvm
517
518#endif
519