1//===-- GCNPreRAOptimizations.cpp -----------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This pass combines split register tuple initialization into a single pseudo:
11///
12/// undef %0.sub1:sreg_64 = S_MOV_B32 1
13/// %0.sub0:sreg_64 = S_MOV_B32 2
14/// =>
15/// %0:sreg_64 = S_MOV_B64_IMM_PSEUDO 0x200000001
16///
17/// This is to allow rematerialization of a value instead of spilling. It is
18/// supposed to be done after register coalescer to allow it to do its job and
19/// before actual register allocation to allow rematerialization.
20///
21/// Right now the pass only handles 64 bit SGPRs with immediate initializers,
22/// although the same shall be possible with other register classes and
23/// instructions if necessary.
24///
25/// This pass also adds register allocation hints to COPY.
26/// The hints will be post-processed by SIRegisterInfo::getRegAllocationHints.
27/// When using True16, we often see COPY moving a 16-bit value between a VGPR_32
28/// and a VGPR_16. If we use the VGPR_16 that corresponds to the lo16 bits of
29/// the VGPR_32, the COPY can be completely eliminated.
30///
31//===----------------------------------------------------------------------===//
32
33#include "GCNPreRAOptimizations.h"
34#include "AMDGPU.h"
35#include "GCNPreRAAntiHints.h"
36#include "GCNSubtarget.h"
37#include "MCTargetDesc/AMDGPUMCTargetDesc.h"
38#include "SIInstrInfo.h"
39#include "SIRegisterInfo.h"
40#include "llvm/CodeGen/LiveIntervals.h"
41#include "llvm/CodeGen/MachineFunctionPass.h"
42#include "llvm/CodeGen/Register.h"
43#include "llvm/CodeGen/TargetSchedule.h"
44#include "llvm/InitializePasses.h"
45#include "llvm/Support/CommandLine.h"
46
47using namespace llvm;
48
49#define DEBUG_TYPE "amdgpu-pre-ra-optimizations"
50
51static cl::opt<bool>
52 EnableAntiHints("amdgpu-anti-hints", cl::Hidden,
53 cl::desc("Enable register allocation anti-hints."),
54 cl::init(Val: true));
55
56namespace {
57
58class GCNPreRAOptimizationsImpl {
59private:
60 const SIInstrInfo *TII;
61 const SIRegisterInfo *TRI;
62 MachineRegisterInfo *MRI;
63 LiveIntervals *LIS;
64 TargetSchedModel SchedModel;
65
66 bool processReg(Register Reg);
67 void hintTrue16Copy(const MachineInstr &MI);
68 bool optimizeBVHStack(MachineInstr &MI);
69
70public:
71 GCNPreRAOptimizationsImpl(LiveIntervals *LS) : LIS(LS) {}
72 bool run(MachineFunction &MF);
73};
74
75class GCNPreRAOptimizationsLegacy : public MachineFunctionPass {
76public:
77 static char ID;
78
79 GCNPreRAOptimizationsLegacy() : MachineFunctionPass(ID) {}
80
81 bool runOnMachineFunction(MachineFunction &MF) override;
82
83 StringRef getPassName() const override {
84 return "AMDGPU Pre-RA optimizations";
85 }
86
87 void getAnalysisUsage(AnalysisUsage &AU) const override {
88 AU.addRequired<LiveIntervalsWrapperPass>();
89 AU.setPreservesAll();
90 MachineFunctionPass::getAnalysisUsage(AU);
91 }
92};
93} // End anonymous namespace.
94
95INITIALIZE_PASS_BEGIN(GCNPreRAOptimizationsLegacy, DEBUG_TYPE,
96 "AMDGPU Pre-RA optimizations", false, false)
97INITIALIZE_PASS_DEPENDENCY(LiveIntervalsWrapperPass)
98INITIALIZE_PASS_END(GCNPreRAOptimizationsLegacy, DEBUG_TYPE,
99 "Pre-RA optimizations", false, false)
100
101char GCNPreRAOptimizationsLegacy::ID = 0;
102
103char &llvm::GCNPreRAOptimizationsID = GCNPreRAOptimizationsLegacy::ID;
104
105bool GCNPreRAOptimizationsImpl::processReg(Register Reg) {
106 MachineInstr *Def0 = nullptr;
107 MachineInstr *Def1 = nullptr;
108 uint64_t Init = 0;
109 bool Changed = false;
110 SmallSet<Register, 32> ModifiedRegs;
111 bool IsAGPRDst = TRI->isAGPRClass(RC: MRI->getRegClass(Reg));
112
113 for (MachineInstr &I : MRI->def_instructions(Reg)) {
114 switch (I.getOpcode()) {
115 default:
116 return false;
117 case AMDGPU::V_ACCVGPR_WRITE_B32_e64:
118 break;
119 case AMDGPU::COPY: {
120 // Some subtargets cannot do an AGPR to AGPR copy directly, and need an
121 // intermdiate temporary VGPR register. Try to find the defining
122 // accvgpr_write to avoid temporary registers.
123
124 if (!IsAGPRDst)
125 return false;
126
127 Register SrcReg = I.getOperand(i: 1).getReg();
128
129 if (!SrcReg.isVirtual())
130 break;
131
132 // Check if source of copy is from another AGPR.
133 bool IsAGPRSrc = TRI->isAGPRClass(RC: MRI->getRegClass(Reg: SrcReg));
134 if (!IsAGPRSrc)
135 break;
136
137 // def_instructions() does not look at subregs so it may give us a
138 // different instruction that defines the same vreg but different subreg
139 // so we have to manually check subreg.
140 Register SrcSubReg = I.getOperand(i: 1).getSubReg();
141 for (auto &Def : MRI->def_instructions(Reg: SrcReg)) {
142 if (SrcSubReg != Def.getOperand(i: 0).getSubReg())
143 continue;
144
145 if (Def.getOpcode() == AMDGPU::V_ACCVGPR_WRITE_B32_e64) {
146 const MachineOperand &DefSrcMO = Def.getOperand(i: 1);
147
148 // Immediates are not an issue and can be propagated in
149 // postrapseudos pass. Only handle cases where defining
150 // accvgpr_write source is a vreg.
151 if (DefSrcMO.isReg() && DefSrcMO.getReg().isVirtual()) {
152 // Propagate source reg of accvgpr write to this copy instruction
153 I.getOperand(i: 1).setReg(DefSrcMO.getReg());
154 I.getOperand(i: 1).setSubReg(DefSrcMO.getSubReg());
155
156 // Reg uses were changed, collect unique set of registers to update
157 // live intervals at the end.
158 ModifiedRegs.insert(V: DefSrcMO.getReg());
159 ModifiedRegs.insert(V: SrcReg);
160
161 Changed = true;
162 }
163
164 // Found the defining accvgpr_write, stop looking any further.
165 break;
166 }
167 }
168 break;
169 }
170 case AMDGPU::S_MOV_B32:
171 if (I.getOperand(i: 0).getReg() != Reg || !I.getOperand(i: 1).isImm() ||
172 I.getNumOperands() != 2)
173 return false;
174
175 switch (I.getOperand(i: 0).getSubReg()) {
176 default:
177 return false;
178 case AMDGPU::sub0:
179 if (Def0)
180 return false;
181 Def0 = &I;
182 Init |= Lo_32(Value: I.getOperand(i: 1).getImm());
183 break;
184 case AMDGPU::sub1:
185 if (Def1)
186 return false;
187 Def1 = &I;
188 Init |= static_cast<uint64_t>(I.getOperand(i: 1).getImm()) << 32;
189 break;
190 }
191 break;
192 }
193 }
194
195 // For AGPR reg, check if live intervals need to be updated.
196 if (IsAGPRDst) {
197 if (Changed) {
198 for (Register RegToUpdate : ModifiedRegs) {
199 LIS->removeInterval(Reg: RegToUpdate);
200 LIS->createAndComputeVirtRegInterval(Reg: RegToUpdate);
201 }
202 }
203
204 return Changed;
205 }
206
207 // For SGPR reg, check if we can combine instructions.
208 if (!Def0 || !Def1 || Def0->getParent() != Def1->getParent())
209 return Changed;
210
211 LLVM_DEBUG(dbgs() << "Combining:\n " << *Def0 << " " << *Def1
212 << " =>\n");
213
214 if (SlotIndex::isEarlierInstr(A: LIS->getInstructionIndex(Instr: *Def1),
215 B: LIS->getInstructionIndex(Instr: *Def0)))
216 std::swap(a&: Def0, b&: Def1);
217
218 LIS->RemoveMachineInstrFromMaps(MI&: *Def0);
219 LIS->RemoveMachineInstrFromMaps(MI&: *Def1);
220 auto NewI = BuildMI(BB&: *Def0->getParent(), I&: *Def0, MIMD: Def0->getDebugLoc(),
221 MCID: TII->get(Opcode: AMDGPU::S_MOV_B64_IMM_PSEUDO), DestReg: Reg)
222 .addImm(Val: Init);
223
224 Def0->eraseFromParent();
225 Def1->eraseFromParent();
226 LIS->InsertMachineInstrInMaps(MI&: *NewI);
227 LIS->removeInterval(Reg);
228 LIS->createAndComputeVirtRegInterval(Reg);
229
230 LLVM_DEBUG(dbgs() << " " << *NewI);
231
232 return true;
233}
234
235bool GCNPreRAOptimizationsLegacy::runOnMachineFunction(MachineFunction &MF) {
236 if (skipFunction(F: MF.getFunction()))
237 return false;
238 LiveIntervals *LIS = &getAnalysis<LiveIntervalsWrapperPass>().getLIS();
239 return GCNPreRAOptimizationsImpl(LIS).run(MF);
240}
241
242PreservedAnalyses
243GCNPreRAOptimizationsPass::run(MachineFunction &MF,
244 MachineFunctionAnalysisManager &MFAM) {
245 LiveIntervals *LIS = &MFAM.getResult<LiveIntervalsAnalysis>(IR&: MF);
246 GCNPreRAOptimizationsImpl(LIS).run(MF);
247 return PreservedAnalyses::all();
248}
249
250void GCNPreRAOptimizationsImpl::hintTrue16Copy(const MachineInstr &MI) {
251 Register Dst = MI.getOperand(i: 0).getReg();
252 Register Src = MI.getOperand(i: 1).getReg();
253 const TargetRegisterClass *DstRC = TRI->getRegClassForReg(MRI: *MRI, Reg: Dst);
254 bool IsDst16Bit = AMDGPU::VGPR_16RegClass.hasSubClassEq(RC: DstRC);
255 if (Dst.isVirtual() && IsDst16Bit && Src.isPhysical() &&
256 TRI->getRegClassForReg(MRI: *MRI, Reg: Src) == &AMDGPU::VGPR_32RegClass)
257 MRI->setRegAllocationHint(VReg: Dst, Type: 0, PrefReg: TRI->getSubReg(Reg: Src, Idx: AMDGPU::lo16));
258 if (Src.isVirtual() && MRI->getRegClass(Reg: Src) == &AMDGPU::VGPR_16RegClass &&
259 Dst.isPhysical() && DstRC == &AMDGPU::VGPR_32RegClass)
260 MRI->setRegAllocationHint(VReg: Src, Type: 0, PrefReg: TRI->getSubReg(Reg: Dst, Idx: AMDGPU::lo16));
261 if (!Dst.isVirtual() || !Src.isVirtual())
262 return;
263 if (MRI->getRegClass(Reg: Dst) == &AMDGPU::VGPR_32RegClass &&
264 MRI->getRegClass(Reg: Src) == &AMDGPU::VGPR_16RegClass) {
265 MRI->setRegAllocationHint(VReg: Dst, Type: AMDGPURI::Size32, PrefReg: Src);
266 MRI->setRegAllocationHint(VReg: Src, Type: AMDGPURI::Size16, PrefReg: Dst);
267 }
268 if (IsDst16Bit && MRI->getRegClass(Reg: Src) == &AMDGPU::VGPR_32RegClass)
269 MRI->setRegAllocationHint(VReg: Dst, Type: AMDGPURI::Size16, PrefReg: Src);
270}
271
272bool GCNPreRAOptimizationsImpl::optimizeBVHStack(MachineInstr &MI) {
273 SmallVector<Register, 2> UseRegs;
274
275 // Find BVH sources for this DS_BVH_STACK instruction.
276 auto CheckUse = [&](MachineOperand &Use) {
277 Register Reg = Use.getReg();
278 for (const MachineInstr &Src : MRI->def_instructions(Reg)) {
279 if (!SIInstrInfo::isImage(MI: Src))
280 continue;
281 const AMDGPU::MIMGInfo *Info = AMDGPU::getMIMGInfo(Opc: Src.getOpcode());
282 const AMDGPU::MIMGBaseOpcodeInfo *BaseInfo =
283 AMDGPU::getMIMGBaseOpcodeInfo(BaseOpcode: Info->BaseOpcode);
284 if (!BaseInfo->BVH)
285 continue;
286 UseRegs.push_back(Elt: Reg);
287 break;
288 }
289 };
290 CheckUse(*TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::data0));
291 CheckUse(*TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::data1));
292
293 if (UseRegs.empty())
294 return false;
295
296 // Add implicit uses for entire BVH source registers.
297 // This avoids partial reallocation of register which could
298 // introduce a premature s_wait_bvhcnt.
299 for (Register Reg : UseRegs) {
300 MI.addOperand(Op: MachineOperand::CreateReg(Reg, isDef: false, isImp: true));
301 LIS->removeInterval(Reg);
302 LIS->createAndComputeVirtRegInterval(Reg);
303 }
304 LLVM_DEBUG(dbgs() << "Added implicit uses to: " << MI);
305
306 return true;
307}
308
309bool GCNPreRAOptimizationsImpl::run(MachineFunction &MF) {
310 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
311 TII = ST.getInstrInfo();
312 MRI = &MF.getRegInfo();
313 TRI = ST.getRegisterInfo();
314
315 // Anti-hints only steer register allocation, so they do not count as a
316 // modification of the function.
317 if (EnableAntiHints) {
318 SchedModel.init(TSInfo: &ST);
319 AMDGPU::HazardContext HCtx{.TII: TII, .TRI: TRI, .MRI: MRI, .LIS: LIS, .ST: &ST, .SchedModel: &SchedModel};
320 AMDGPU::applyAntiHintRules(MF, Ctx: HCtx);
321 }
322
323 bool Changed = false;
324
325 for (unsigned I = 0, E = MRI->getNumVirtRegs(); I != E; ++I) {
326 Register Reg = Register::index2VirtReg(Index: I);
327 if (!LIS->hasInterval(Reg))
328 continue;
329 const TargetRegisterClass *RC = MRI->getRegClass(Reg);
330 if ((RC->getSizeInBits() != 64 || !TRI->isSGPRClass(RC)) &&
331 (ST.hasGFX90AInsts() || !TRI->isAGPRClass(RC)))
332 continue;
333
334 Changed |= processReg(Reg);
335 }
336
337 const bool HasBVHStack = ST.hasBVHDualAndBVH8Insts();
338 const bool HasRealTrue16 = ST.useRealTrue16Insts();
339
340 if (!HasRealTrue16 && !HasBVHStack)
341 return Changed;
342
343 for (MachineBasicBlock &MBB : MF) {
344 for (MachineInstr &MI : MBB) {
345 // Add RA hints to improve True16 COPY elimination.
346 if (HasRealTrue16 && MI.getOpcode() == AMDGPU::COPY) {
347 hintTrue16Copy(MI);
348 continue;
349 }
350 // Add implicit uses to avoid early wait on intersect ray instructions.
351 if (HasBVHStack &&
352 (MI.getOpcode() == AMDGPU::DS_BVH_STACK_RTN_B32 ||
353 MI.getOpcode() == AMDGPU::DS_BVH_STACK_PUSH8_POP1_RTN_B32 ||
354 MI.getOpcode() == AMDGPU::DS_BVH_STACK_PUSH8_POP2_RTN_B64)) {
355 Changed |= optimizeBVHStack(MI);
356 continue;
357 }
358 }
359 }
360
361 return Changed;
362}
363