1//===-- SIRegisterInfo.h - SI Register Info Interface ----------*- C++ -*--===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// Interface definition for SIRegisterInfo
11//
12//===----------------------------------------------------------------------===//
13
14#ifndef LLVM_LIB_TARGET_AMDGPU_SIREGISTERINFO_H
15#define LLVM_LIB_TARGET_AMDGPU_SIREGISTERINFO_H
16
17#include "llvm/ADT/BitVector.h"
18#include "llvm/ADT/SmallVector.h"
19#include "llvm/CodeGen/LiveRegMatrix.h"
20#include "llvm/CodeGen/Register.h"
21#include "llvm/CodeGen/VirtRegMap.h"
22#include "llvm/MC/MCRegister.h"
23
24#define GET_REGINFO_HEADER
25#include "AMDGPUGenRegisterInfo.inc"
26
27#include "SIDefines.h"
28
29namespace llvm {
30
31class GCNSubtarget;
32class LiveIntervals;
33class LiveRegUnits;
34class MachineInstrBuilder;
35class RegisterBank;
36struct SGPRSpillBuilder;
37
38/// Register allocation hint types. Helps eliminate unneeded COPY with True16
39namespace AMDGPURI {
40
41enum { Size16 = 1, Size32 = 2 };
42
43} // end namespace AMDGPURI
44
45class SIRegisterInfo final : public AMDGPUGenRegisterInfo {
46private:
47 const GCNSubtarget &ST;
48 bool SpillSGPRToVGPR;
49 bool isWave32;
50 BitVector RegPressureIgnoredUnits;
51
52 /// Sub reg indexes for getRegSplitParts.
53 /// First index represents subreg size from 1 to 32 Half DWORDS.
54 /// The inner vector is sorted by bit offset.
55 /// Provided a register can be fully split with given subregs,
56 /// all elements of the inner vector combined give a full lane mask.
57 static std::array<std::vector<int16_t>, 32> RegSplitParts;
58
59 // Table representing sub reg of given width and offset.
60 // First index is subreg size: 32, 64, 96, 128, 160, 192, 224, 256, 512.
61 // Second index is 32 different dword offsets.
62 static std::array<std::array<uint16_t, 32>, 9> SubRegFromChannelTable;
63
64 void reserveRegisterTuples(BitVector &, MCRegister Reg) const;
65
66 /// True if assigning Reg would fit in the current occupancy VGPR budget.
67 bool isRegWithinOccupancyBudget(MCPhysReg Reg, unsigned NumVGPRs,
68 unsigned NumAGPRs,
69 unsigned MaxVGPRsForCurrentOccupancy) const;
70
71public:
72 SIRegisterInfo(const GCNSubtarget &ST);
73
74 struct SpilledReg {
75 Register VGPR;
76 int Lane = -1;
77
78 SpilledReg() = default;
79 SpilledReg(Register R, int L) : VGPR(R), Lane(L) {}
80
81 bool hasLane() { return Lane != -1; }
82 };
83
84 /// \returns the sub reg enum value for the given \p Channel
85 /// (e.g. getSubRegFromChannel(0) -> AMDGPU::sub0)
86 static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs = 1);
87
88 bool spillSGPRToVGPR() const {
89 return SpillSGPRToVGPR;
90 }
91
92 bool isCFISavedRegsSpillEnabled() const;
93
94 /// Return the largest available SGPR aligned to \p Align for the register
95 /// class \p RC.
96 MCRegister getAlignedHighSGPRForRC(const MachineFunction &MF,
97 const unsigned Align,
98 const TargetRegisterClass *RC) const;
99
100 /// Return the end register initially reserved for the scratch buffer in case
101 /// spilling is needed.
102 MCRegister reservedPrivateSegmentBufferReg(const MachineFunction &MF) const;
103
104 BitVector getReservedRegs(const MachineFunction &MF) const override;
105 bool isAsmClobberable(const MachineFunction &MF,
106 MCRegister PhysReg) const override;
107
108 const MCPhysReg *getCalleeSavedRegs(const MachineFunction *MF) const override;
109 const MCPhysReg *getCalleeSavedRegsViaCopy(const MachineFunction *MF) const;
110 const uint32_t *getCallPreservedMask(const MachineFunction &MF,
111 CallingConv::ID) const override;
112 const uint32_t *getNoPreservedMask() const override;
113
114 // Functions with the amdgpu_cs_chain or amdgpu_cs_chain_preserve calling
115 // conventions are free to use certain VGPRs without saving and restoring any
116 // lanes (not even inactive ones).
117 static bool isChainScratchRegister(Register VGPR);
118
119 unsigned getCSRFirstUseCost(const MachineFunction &) const override {
120 // The cost of 27 balances multiple factors that influence CSR cost:
121 // - Saving a SGPR CSR to VGPR lanes is relatively cheap.
122 // - In cases where this is not possible, stack access is very expensive.
123 // - CSRs are also the high registers, and we want to minimize the number of
124 // used registers as it impacts occupancy.
125 // Note: Register allocation only applies these cost to callee-save
126 // registers according to getCalleeSavedRegs, so handling of calling
127 // conventions with no CSR is handled there.
128 return 27;
129 }
130
131 // When building a block VGPR load, we only really transfer a subset of the
132 // registers in the block, based on a mask. Liveness analysis is not aware of
133 // the mask, so it might consider that any register in the block is available
134 // before the load and may therefore be scavenged. This is not ok for CSRs
135 // that are not clobbered, since the caller will expect them to be preserved.
136 // This method will add artificial implicit uses for those registers on the
137 // load instruction, so liveness analysis knows they're unavailable.
138 void addImplicitUsesForBlockCSRLoad(MachineInstrBuilder &MIB,
139 Register BlockReg) const;
140
141 // Iterate over all VGPRs in the given BlockReg and emit CFI for each VGPR
142 // as-needed depending on the (statically known) mask, relative to the given
143 // base Offset.
144 void buildCFIForBlockCSRStore(MachineBasicBlock &MBB,
145 MachineBasicBlock::iterator MBBI,
146 Register BlockReg, int64_t Offset) const;
147
148 const TargetRegisterClass *
149 getLargestLegalSuperClass(const TargetRegisterClass *RC,
150 const MachineFunction &MF) const override;
151
152 Register getFrameRegister(const MachineFunction &MF) const override;
153
154 bool hasBasePointer(const MachineFunction &MF) const;
155 Register getBaseRegister() const;
156
157 bool shouldRealignStack(const MachineFunction &MF) const override;
158 bool requiresRegisterScavenging(const MachineFunction &Fn) const override;
159
160 bool requiresFrameIndexScavenging(const MachineFunction &MF) const override;
161 bool requiresFrameIndexReplacementScavenging(
162 const MachineFunction &MF) const override;
163 bool requiresVirtualBaseRegisters(const MachineFunction &Fn) const override;
164
165 int64_t getScratchInstrOffset(const MachineInstr *MI) const;
166
167 int64_t getFrameIndexInstrOffset(const MachineInstr *MI,
168 int Idx) const override;
169
170 bool needsFrameBaseReg(MachineInstr *MI, int64_t Offset) const override;
171
172 Register materializeFrameBaseRegister(MachineBasicBlock *MBB, int FrameIdx,
173 int64_t Offset) const override;
174
175 void resolveFrameIndex(MachineInstr &MI, Register BaseReg,
176 int64_t Offset) const override;
177
178 bool isFrameOffsetLegal(const MachineInstr *MI, Register BaseReg,
179 int64_t Offset) const override;
180
181 /// Returns a legal register class to copy a register in the specified class
182 /// to or from. If it is possible to copy the register directly without using
183 /// a cross register class copy, return the specified RC. Returns NULL if it
184 /// is not possible to copy between two registers of the specified class.
185 const TargetRegisterClass *
186 getCrossCopyRegClass(const TargetRegisterClass *RC) const override;
187
188 const TargetRegisterClass *
189 getRegClassForBlockOp(const MachineFunction &MF) const {
190 return &AMDGPU::VReg_1024RegClass;
191 }
192
193 void buildVGPRSpillLoadStore(SGPRSpillBuilder &SB, int Index, int Offset,
194 bool IsLoad, bool IsKill = true) const;
195
196 /// If \p OnlyToVGPR is true, this will only succeed if this manages to find a
197 /// free VGPR lane to spill.
198 bool spillSGPR(MachineBasicBlock::iterator MI, int FI, RegScavenger *RS,
199 SlotIndexes *Indexes = nullptr, LiveIntervals *LIS = nullptr,
200 bool OnlyToVGPR = false, bool SpillToPhysVGPRLane = false,
201 bool NeedsCFI = false) const;
202
203 bool restoreSGPR(MachineBasicBlock::iterator MI, int FI, RegScavenger *RS,
204 SlotIndexes *Indexes = nullptr, LiveIntervals *LIS = nullptr,
205 bool OnlyToVGPR = false,
206 bool SpillToPhysVGPRLane = false) const;
207
208 bool spillEmergencySGPR(MachineBasicBlock::iterator MI,
209 MachineBasicBlock &RestoreMBB, Register SGPR,
210 RegScavenger *RS) const;
211
212 bool eliminateFrameIndex(MachineBasicBlock::iterator MI, int SPAdj,
213 unsigned FIOperandNum,
214 RegScavenger *RS) const override;
215
216 bool eliminateSGPRToVGPRSpillFrameIndex(
217 MachineBasicBlock::iterator MI, int FI, RegScavenger *RS,
218 SlotIndexes *Indexes = nullptr, LiveIntervals *LIS = nullptr,
219 bool SpillToPhysVGPRLane = false) const;
220
221 StringRef getRegAsmName(MCRegister Reg) const override;
222
223 // Pseudo regs are not allowed
224 unsigned getHWRegIndex(MCRegister Reg) const;
225
226 LLVM_READONLY
227 const TargetRegisterClass *getVGPRClassForBitWidth(unsigned BitWidth) const;
228
229 LLVM_READONLY const TargetRegisterClass *
230 getAlignedLo256VGPRClassForBitWidth(unsigned BitWidth) const;
231
232 LLVM_READONLY
233 const TargetRegisterClass *getAGPRClassForBitWidth(unsigned BitWidth) const;
234
235 LLVM_READONLY
236 const TargetRegisterClass *
237 getVectorSuperClassForBitWidth(unsigned BitWidth) const;
238
239 LLVM_READONLY
240 const TargetRegisterClass *
241 getDefaultVectorSuperClassForBitWidth(unsigned BitWidth) const;
242
243 LLVM_READONLY
244 static const TargetRegisterClass *getSGPRClassForBitWidth(unsigned BitWidth);
245
246 /// \returns true if this class contains only SGPR registers
247 static bool isSGPRClass(const TargetRegisterClass *RC) {
248 return hasSGPRs(RC) && !hasVGPRs(RC) && !hasAGPRs(RC);
249 }
250
251 bool isSGPRReg(const MachineRegisterInfo &MRI, Register Reg) const;
252 bool isSGPRPhysReg(Register Reg) const {
253 return isSGPRClass(RC: getPhysRegBaseClass(Reg));
254 }
255
256 /// \returns true if this class contains only VGPR registers
257 static bool isVGPRClass(const TargetRegisterClass *RC) {
258 return hasVGPRs(RC) && !hasAGPRs(RC) && !hasSGPRs(RC);
259 }
260
261 /// \returns true if this class contains only AGPR registers
262 static bool isAGPRClass(const TargetRegisterClass *RC) {
263 return hasAGPRs(RC) && !hasVGPRs(RC) && !hasSGPRs(RC);
264 }
265
266 /// \returns true only if this class contains both VGPR and AGPR registers
267 bool isVectorSuperClass(const TargetRegisterClass *RC) const {
268 return hasVGPRs(RC) && hasAGPRs(RC) && !hasSGPRs(RC);
269 }
270
271 /// \returns true only if this class contains both VGPR and SGPR registers
272 bool isVSSuperClass(const TargetRegisterClass *RC) const {
273 return hasVGPRs(RC) && hasSGPRs(RC) && !hasAGPRs(RC);
274 }
275
276 /// \returns true if this class contains VGPR registers.
277 static bool hasVGPRs(const TargetRegisterClass *RC) {
278 return RC->TSFlags & SIRCFlags::HasVGPR;
279 }
280
281 /// \returns true if this class contains AGPR registers.
282 static bool hasAGPRs(const TargetRegisterClass *RC) {
283 return RC->TSFlags & SIRCFlags::HasAGPR;
284 }
285
286 /// \returns true if this class contains SGPR registers.
287 static bool hasSGPRs(const TargetRegisterClass *RC) {
288 return RC->TSFlags & SIRCFlags::HasSGPR;
289 }
290
291 /// \returns true if this class contains any vector registers.
292 static bool hasVectorRegisters(const TargetRegisterClass *RC) {
293 return hasVGPRs(RC) || hasAGPRs(RC);
294 }
295
296 /// \returns A VGPR reg class with the same width as \p SRC
297 const TargetRegisterClass *
298 getEquivalentVGPRClass(const TargetRegisterClass *SRC) const;
299
300 /// \returns An AGPR reg class with the same width as \p SRC
301 const TargetRegisterClass *
302 getEquivalentAGPRClass(const TargetRegisterClass *SRC) const;
303
304 /// \returns An AGPR+VGPR super reg class with the same width as \p SRC
305 const TargetRegisterClass *
306 getEquivalentAVClass(const TargetRegisterClass *SRC) const;
307
308 /// \returns A SGPR reg class with the same width as \p SRC
309 const TargetRegisterClass *
310 getEquivalentSGPRClass(const TargetRegisterClass *VRC) const;
311
312 /// Returns a register class which is compatible with \p SuperRC, such that a
313 /// subregister exists with class \p SubRC with subregister index \p
314 /// SubIdx. If this is impossible (e.g., an unaligned subregister index within
315 /// a register tuple), return null.
316 const TargetRegisterClass *
317 getCompatibleSubRegClass(const TargetRegisterClass *SuperRC,
318 const TargetRegisterClass *SubRC,
319 unsigned SubIdx) const;
320
321 /// \returns True if operands defined with this operand type can accept
322 /// a literal constant (i.e. any 32-bit immediate).
323 bool opCanUseLiteralConstant(unsigned OpType) const;
324
325 /// \returns True if operands defined with this operand type can accept
326 /// an inline constant. i.e. An integer value in the range (-16, 64) or
327 /// -4.0f, -2.0f, -1.0f, -0.5f, 0.0f, 0.5f, 1.0f, 2.0f, 4.0f.
328 bool opCanUseInlineConstant(unsigned OpType) const;
329
330 MCRegister findUnusedRegister(const MachineRegisterInfo &MRI,
331 const TargetRegisterClass *RC,
332 const MachineFunction &MF,
333 bool ReserveHighestVGPR = false) const;
334
335 const TargetRegisterClass *getRegClassForReg(const MachineRegisterInfo &MRI,
336 Register Reg) const;
337 const TargetRegisterClass *
338 getRegClassForOperandReg(const MachineRegisterInfo &MRI,
339 const MachineOperand &MO) const;
340
341 bool isVGPR(const MachineRegisterInfo &MRI, Register Reg) const;
342 bool isAGPR(const MachineRegisterInfo &MRI, Register Reg) const;
343 bool isVectorRegister(const MachineRegisterInfo &MRI, Register Reg) const {
344 return isVGPR(MRI, Reg) || isAGPR(MRI, Reg);
345 }
346
347 // FIXME: SGPRs are assumed to be uniform, but this is not true for i1 SGPRs
348 // (such as VCC) which hold a wave-wide vector of boolean values. Examining
349 // just the register class is not suffcient; it needs to be combined with a
350 // value type. The next predicate isUniformReg() does this correctly.
351 bool isDivergentRegClass(const TargetRegisterClass *RC) const override {
352 return !isSGPRClass(RC);
353 }
354
355 bool isUniformReg(const MachineRegisterInfo &MRI, const RegisterBankInfo &RBI,
356 Register Reg) const override;
357
358 ArrayRef<int16_t> getRegSplitParts(const TargetRegisterClass *RC,
359 unsigned EltSize) const;
360
361 unsigned getRegPressureLimit(const TargetRegisterClass *RC,
362 MachineFunction &MF) const override;
363
364 unsigned getRegPressureSetLimit(const MachineFunction &MF,
365 unsigned Idx) const override;
366
367 bool getRegAllocationHints(Register VirtReg, ArrayRef<MCPhysReg> Order,
368 SmallSetVector<MCPhysReg, 16> &Hints,
369 const MachineFunction &MF, const VirtRegMap *VRM,
370 const LiveRegMatrix *Matrix) const override;
371
372 bool shouldApplyAntiHints(const MachineFunction &MF,
373 unsigned NumAllocatedVGPRs,
374 unsigned &MaxVGPRsForCurrentOccupancy) const;
375
376 void filterAndSortForAntiHintedRegs(
377 Register VirtReg, MutableArrayRef<MCPhysReg> CustomOrder,
378 const BitVector &AntiHintedRegUnits, const MachineFunction &MF,
379 const LiveRegMatrix *Matrix = nullptr,
380 const RegisterClassInfo *RegClassInfo = nullptr) const override;
381
382 const int *getRegUnitPressureSets(MCRegUnit RegUnit) const override;
383
384 MCRegister getReturnAddressReg(const MachineFunction &MF) const;
385
386 const TargetRegisterClass *
387 getRegClassForSizeOnBank(unsigned Size, const RegisterBank &Bank) const;
388
389 const TargetRegisterClass *
390 getRegClassForTypeOnBank(LLT Ty, const RegisterBank &Bank) const {
391 return getRegClassForSizeOnBank(Size: Ty.getSizeInBits(), Bank);
392 }
393
394 const TargetRegisterClass *
395 getConstrainedRegClassForReg(Register Reg,
396 const MachineRegisterInfo &MRI) const override;
397
398 const TargetRegisterClass *getBoolRC() const {
399 return isWave32 ? &AMDGPU::SReg_32RegClass
400 : &AMDGPU::SReg_64RegClass;
401 }
402
403 const TargetRegisterClass *getWaveMaskRegClass() const {
404 return isWave32 ? &AMDGPU::SReg_32_XM0_XEXECRegClass
405 : &AMDGPU::SReg_64_XEXECRegClass;
406 }
407
408 // Return the appropriate register class to use for 64-bit VGPRs for the
409 // subtarget.
410 const TargetRegisterClass *getVGPR64Class() const;
411
412 MCRegister getVCC() const;
413
414 MCRegister getExec() const;
415
416 // Find reaching register definition
417 MachineInstr *findReachingDef(Register Reg, unsigned SubReg,
418 MachineInstr &Use,
419 MachineRegisterInfo &MRI,
420 LiveIntervals *LIS) const;
421
422 const uint32_t *getAllVGPRRegMask() const;
423 const uint32_t *getAllAGPRRegMask() const;
424 const uint32_t *getAllVectorRegMask() const;
425
426 // \returns number of 32 bit registers covered by a \p LM
427 static unsigned getNumCoveredRegs(LaneBitmask LM) {
428 // The assumption is that every lo16 subreg is an even bit and every hi16
429 // is an adjacent odd bit or vice versa.
430 uint64_t Mask = LM.getAsInteger();
431 uint64_t Even = Mask & 0xAAAAAAAAAAAAAAAAULL;
432 Mask = (Even >> 1) | Mask;
433 uint64_t Odd = Mask & 0x5555555555555555ULL;
434 return llvm::popcount(Value: Odd);
435 }
436
437 // \returns a DWORD offset of a \p SubReg
438 unsigned getChannelFromSubReg(unsigned SubReg) const {
439 return SubReg ? (getSubRegIdxOffset(Idx: SubReg) + 31) / 32 : 0;
440 }
441
442 // For a given 16 bit \p Reg \returns a 32 bit register holding it.
443 // \returns \p Reg otherwise.
444 MCPhysReg get32BitRegister(MCPhysReg Reg) const;
445
446 // Returns true if a given register class is properly aligned for
447 // the subtarget.
448 bool isProperlyAlignedRC(const TargetRegisterClass &RC) const;
449
450 /// Return all SGPR128 which satisfy the waves per execution unit requirement
451 /// of the subtarget.
452 ArrayRef<MCPhysReg> getAllSGPR128(const MachineFunction &MF) const;
453
454 /// Return all SGPR64 which satisfy the waves per execution unit requirement
455 /// of the subtarget.
456 ArrayRef<MCPhysReg> getAllSGPR64(const MachineFunction &MF) const;
457
458 /// Return all SGPR32 which satisfy the waves per execution unit requirement
459 /// of the subtarget.
460 ArrayRef<MCPhysReg> getAllSGPR32(const MachineFunction &MF) const;
461
462 // Insert spill or restore instructions.
463 // When lowering spill pseudos, the RegScavenger should be set.
464 // For creating spill instructions during frame lowering, where no scavenger
465 // is available, LiveUnits can be used.
466 void buildSpillLoadStore(MachineBasicBlock &MBB,
467 MachineBasicBlock::iterator MI, const DebugLoc &DL,
468 unsigned LoadStoreOp, int Index, Register ValueReg,
469 bool ValueIsKill, MCRegister ScratchOffsetReg,
470 int64_t InstrOffset, MachineMemOperand *MMO,
471 RegScavenger *RS, LiveRegUnits *LiveUnits = nullptr,
472 bool NeedsCFI = false) const;
473
474 // Return alignment in register file of first register in a register tuple.
475 unsigned getRegClassAlignmentNumBits(const TargetRegisterClass *RC) const {
476 return (RC->TSFlags & SIRCFlags::RegTupleAlignUnitsMask) * 32;
477 }
478
479 // Check if register class RC has required alignment.
480 bool isRegClassAligned(const TargetRegisterClass *RC,
481 unsigned AlignNumBits) const {
482 assert(AlignNumBits != 0);
483 unsigned RCAlign = getRegClassAlignmentNumBits(RC);
484 return RCAlign == AlignNumBits ||
485 (RCAlign > AlignNumBits && (RCAlign % AlignNumBits) == 0);
486 }
487
488 // Return alignment of a SubReg relative to start of a register in RC class.
489 // No check if the subreg is supported by the current RC is made.
490 unsigned getSubRegAlignmentNumBits(const TargetRegisterClass *RC,
491 unsigned SubReg) const;
492
493 // \returns a number of registers of a given \p RC used in a function.
494 // Does not go inside function calls. If \p IncludeCalls is true, it will
495 // include registers that may be clobbered by calls.
496 unsigned getNumUsedPhysRegs(const MachineRegisterInfo &MRI,
497 const TargetRegisterClass &RC,
498 bool IncludeCalls = true) const;
499
500 std::optional<uint8_t> getVRegFlagValue(StringRef Name) const override {
501 return Name == "WWM_REG" ? AMDGPU::VirtRegFlag::WWM_REG
502 : std::optional<uint8_t>{};
503 }
504
505 SmallVector<StringLiteral>
506 getVRegFlagsOfReg(Register Reg, const MachineFunction &MF) const override;
507
508 float
509 getSpillWeightScaleFactor(const TargetRegisterClass *RC) const override {
510 // Prioritize VGPR_32_Lo256 over other classes which may occupy registers
511 // beyond v256.
512 return AMDGPUGenRegisterInfo::getSpillWeightScaleFactor(RC) *
513 ((RC == &AMDGPU::VGPR_32_Lo256RegClass ||
514 RC == &AMDGPU::VReg_64_Lo256_Align2RegClass)
515 ? 2.0
516 : 1.0);
517 }
518};
519
520} // End namespace llvm
521
522#endif
523