1//===- SIPreAllocateWWMRegs.cpp - WWM Register Pre-allocation -------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// Pass to pre-allocated WWM registers
11//
12//===----------------------------------------------------------------------===//
13
14#include "SIPreAllocateWWMRegs.h"
15#include "AMDGPU.h"
16#include "GCNSubtarget.h"
17#include "SIMachineFunctionInfo.h"
18#include "llvm/ADT/PostOrderIterator.h"
19#include "llvm/CodeGen/LiveIntervals.h"
20#include "llvm/CodeGen/LiveRegMatrix.h"
21#include "llvm/CodeGen/MachineFrameInfo.h"
22#include "llvm/CodeGen/MachineFunctionPass.h"
23#include "llvm/CodeGen/RegisterClassInfo.h"
24#include "llvm/CodeGen/VirtRegMap.h"
25#include "llvm/InitializePasses.h"
26
27using namespace llvm;
28
29#define DEBUG_TYPE "si-pre-allocate-wwm-regs"
30
31static cl::opt<bool>
32 EnablePreallocateSGPRSpillVGPRs("amdgpu-prealloc-sgpr-spill-vgprs",
33 cl::init(Val: false), cl::Hidden);
34
35bool llvm::isPreallocateSGPRSpillVGPRsEnabled(const MachineFunction &MF) {
36 return EnablePreallocateSGPRSpillVGPRs ||
37 MF.getFunction().hasFnAttribute(Kind: "amdgpu-prealloc-sgpr-spill-vgprs");
38}
39
40namespace {
41
42class SIPreAllocateWWMRegs {
43private:
44 const SIInstrInfo *TII;
45 const SIRegisterInfo *TRI;
46 MachineRegisterInfo *MRI;
47 LiveIntervals *LIS;
48 LiveRegMatrix *Matrix;
49 VirtRegMap *VRM;
50 RegisterClassInfo &RCI;
51
52 std::vector<unsigned> RegsToRewrite;
53#ifndef NDEBUG
54 void printWWMInfo(const MachineInstr &MI);
55#endif
56 bool processDef(MachineOperand &MO);
57 void rewriteRegs(MachineFunction &MF);
58
59public:
60 SIPreAllocateWWMRegs(LiveIntervals *LIS, LiveRegMatrix *Matrix,
61 VirtRegMap *VRM, RegisterClassInfo &RCI)
62 : LIS(LIS), Matrix(Matrix), VRM(VRM), RCI(RCI) {}
63 bool run(MachineFunction &MF);
64};
65
66class SIPreAllocateWWMRegsLegacy : public MachineFunctionPass {
67public:
68 static char ID;
69
70 SIPreAllocateWWMRegsLegacy() : MachineFunctionPass(ID) {}
71
72 bool runOnMachineFunction(MachineFunction &MF) override;
73
74 void getAnalysisUsage(AnalysisUsage &AU) const override {
75 AU.addRequired<LiveIntervalsWrapperPass>();
76 AU.addRequired<VirtRegMapWrapperLegacy>();
77 AU.addRequired<LiveRegMatrixWrapperLegacy>();
78 AU.addRequired<MachineRegisterClassInfoWrapperPass>();
79 AU.setPreservesAll();
80 MachineFunctionPass::getAnalysisUsage(AU);
81 }
82};
83
84} // End anonymous namespace.
85
86INITIALIZE_PASS_BEGIN(SIPreAllocateWWMRegsLegacy, DEBUG_TYPE,
87 "SI Pre-allocate WWM Registers", false, false)
88INITIALIZE_PASS_DEPENDENCY(LiveIntervalsWrapperPass)
89INITIALIZE_PASS_DEPENDENCY(VirtRegMapWrapperLegacy)
90INITIALIZE_PASS_DEPENDENCY(LiveRegMatrixWrapperLegacy)
91INITIALIZE_PASS_DEPENDENCY(MachineRegisterClassInfoWrapperPass)
92INITIALIZE_PASS_END(SIPreAllocateWWMRegsLegacy, DEBUG_TYPE,
93 "SI Pre-allocate WWM Registers", false, false)
94
95char SIPreAllocateWWMRegsLegacy::ID = 0;
96
97char &llvm::SIPreAllocateWWMRegsLegacyID = SIPreAllocateWWMRegsLegacy::ID;
98
99bool SIPreAllocateWWMRegs::processDef(MachineOperand &MO) {
100 Register Reg = MO.getReg();
101 if (Reg.isPhysical())
102 return false;
103
104 if (!SIRegisterInfo::hasVGPRs(RC: MRI->getRegClass(Reg)))
105 return false;
106
107 if (VRM->hasPhys(virtReg: Reg))
108 return false;
109
110 LiveInterval &LI = LIS->getInterval(Reg);
111
112 for (MCRegister PhysReg : RCI.getOrder(RC: MRI->getRegClass(Reg))) {
113 if (!MRI->isPhysRegUsed(PhysReg, /*SkipRegMaskTest=*/true) &&
114 Matrix->checkInterference(VirtReg: LI, PhysReg) == LiveRegMatrix::IK_Free) {
115 Matrix->assign(VirtReg: LI, PhysReg);
116 assert(PhysReg != 0);
117 RegsToRewrite.push_back(x: Reg);
118 return true;
119 }
120 }
121
122 llvm_unreachable("physreg not found for WWM expression");
123}
124
125void SIPreAllocateWWMRegs::rewriteRegs(MachineFunction &MF) {
126 for (MachineBasicBlock &MBB : MF) {
127 for (MachineInstr &MI : MBB) {
128 for (MachineOperand &MO : MI.operands()) {
129 if (!MO.isReg())
130 continue;
131
132 const Register VirtReg = MO.getReg();
133 if (VirtReg.isPhysical())
134 continue;
135
136 if (!VirtReg.isValid())
137 continue;
138
139 if (!VRM->hasPhys(virtReg: VirtReg))
140 continue;
141
142 Register PhysReg = VRM->getPhys(virtReg: VirtReg);
143 const unsigned SubReg = MO.getSubReg();
144 if (SubReg != 0) {
145 PhysReg = TRI->getSubReg(Reg: PhysReg, Idx: SubReg);
146 MO.setSubReg(0);
147 }
148
149 MO.setReg(PhysReg);
150 MO.setIsRenamable(false);
151 }
152 }
153 }
154
155 SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
156
157 for (unsigned Reg : RegsToRewrite) {
158 const Register PhysReg = VRM->getPhys(virtReg: Reg);
159 assert(PhysReg != 0);
160
161 LiveInterval &LI = LIS->getInterval(Reg);
162 Matrix->unassign(VirtReg: LI, /*ClearAllReferencingSegments=*/true);
163 LIS->removeInterval(Reg);
164
165 MFI->reserveWWMRegister(Reg: PhysReg);
166 }
167
168 RegsToRewrite.clear();
169
170 // Update the set of reserved registers to include WWM ones
171 // without unnecessarily invalidating RegClassInfo.
172 MRI->freezeReservedRegs();
173 RCI.updateReservedRegs(ReservedInput: MRI->getReservedRegs());
174}
175
176#ifndef NDEBUG
177LLVM_DUMP_METHOD void
178SIPreAllocateWWMRegs::printWWMInfo(const MachineInstr &MI) {
179
180 unsigned Opc = MI.getOpcode();
181
182 if (Opc == AMDGPU::ENTER_STRICT_WWM || Opc == AMDGPU::ENTER_STRICT_WQM) {
183 dbgs() << "Entering ";
184 } else {
185 assert(Opc == AMDGPU::EXIT_STRICT_WWM || Opc == AMDGPU::EXIT_STRICT_WQM);
186 dbgs() << "Exiting ";
187 }
188
189 if (Opc == AMDGPU::ENTER_STRICT_WWM || Opc == AMDGPU::EXIT_STRICT_WWM) {
190 dbgs() << "Strict WWM ";
191 } else {
192 assert(Opc == AMDGPU::ENTER_STRICT_WQM || Opc == AMDGPU::EXIT_STRICT_WQM);
193 dbgs() << "Strict WQM ";
194 }
195
196 dbgs() << "region: " << MI;
197}
198
199#endif
200
201bool SIPreAllocateWWMRegsLegacy::runOnMachineFunction(MachineFunction &MF) {
202 auto *LIS = &getAnalysis<LiveIntervalsWrapperPass>().getLIS();
203 auto *Matrix = &getAnalysis<LiveRegMatrixWrapperLegacy>().getLRM();
204 auto *VRM = &getAnalysis<VirtRegMapWrapperLegacy>().getVRM();
205 auto &RCI = getAnalysis<MachineRegisterClassInfoWrapperPass>().getRCI();
206 return SIPreAllocateWWMRegs(LIS, Matrix, VRM, RCI).run(MF);
207}
208
209bool SIPreAllocateWWMRegs::run(MachineFunction &MF) {
210 LLVM_DEBUG(dbgs() << "SIPreAllocateWWMRegs: function " << MF.getName() << "\n");
211
212 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
213
214 TII = ST.getInstrInfo();
215 TRI = &TII->getRegisterInfo();
216 MRI = &MF.getRegInfo();
217
218 bool PreallocateSGPRSpillVGPRs = isPreallocateSGPRSpillVGPRsEnabled(MF);
219
220 bool RegsAssigned = false;
221
222 // We use a reverse post-order traversal of the control-flow graph to
223 // guarantee that we visit definitions in dominance order. Since WWM
224 // expressions are guaranteed to never involve phi nodes, and we can only
225 // escape WWM through the special WWM instruction, this means that this is a
226 // perfect elimination order, so we can never do any better.
227 ReversePostOrderTraversal<MachineFunction*> RPOT(&MF);
228
229 for (MachineBasicBlock *MBB : RPOT) {
230 bool InWWM = false;
231 for (MachineInstr &MI : *MBB) {
232 if (MI.getOpcode() == AMDGPU::SI_SPILL_S32_TO_VGPR) {
233 if (PreallocateSGPRSpillVGPRs)
234 RegsAssigned |= processDef(MO&: MI.getOperand(i: 0));
235 continue;
236 }
237
238 if (MI.getOpcode() == AMDGPU::ENTER_STRICT_WWM ||
239 MI.getOpcode() == AMDGPU::ENTER_STRICT_WQM) {
240 LLVM_DEBUG(printWWMInfo(MI));
241 InWWM = true;
242 continue;
243 }
244
245 if (MI.getOpcode() == AMDGPU::EXIT_STRICT_WWM ||
246 MI.getOpcode() == AMDGPU::EXIT_STRICT_WQM) {
247 LLVM_DEBUG(printWWMInfo(MI));
248 InWWM = false;
249 }
250
251 if (!InWWM)
252 continue;
253
254 LLVM_DEBUG(dbgs() << "Processing " << MI);
255
256 for (MachineOperand &DefOpnd : MI.defs()) {
257 RegsAssigned |= processDef(MO&: DefOpnd);
258 }
259 }
260 }
261
262 if (!RegsAssigned)
263 return false;
264
265 rewriteRegs(MF);
266 return true;
267}
268
269PreservedAnalyses
270SIPreAllocateWWMRegsPass::run(MachineFunction &MF,
271 MachineFunctionAnalysisManager &MFAM) {
272 auto *LIS = &MFAM.getResult<LiveIntervalsAnalysis>(IR&: MF);
273 auto *Matrix = &MFAM.getResult<LiveRegMatrixAnalysis>(IR&: MF);
274 auto *VRM = &MFAM.getResult<VirtRegMapAnalysis>(IR&: MF);
275 auto &RCI = MFAM.getResult<MachineRegisterClassAnalysis>(IR&: MF);
276 SIPreAllocateWWMRegs(LIS, Matrix, VRM, RCI).run(MF);
277 return PreservedAnalyses::all();
278}
279