1//===-- SIPostRABundler.cpp -----------------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This pass creates bundles of memory instructions to protect adjacent loads
11/// and stores from being rescheduled apart from each other post-RA.
12///
13//===----------------------------------------------------------------------===//
14
15#include "SIPostRABundler.h"
16#include "AMDGPU.h"
17#include "GCNSubtarget.h"
18#include "llvm/CodeGen/MachineFunctionPass.h"
19
20using namespace llvm;
21
22#define DEBUG_TYPE "si-post-ra-bundler"
23
24namespace {
25
26class SIPostRABundlerLegacy : public MachineFunctionPass {
27public:
28 static char ID;
29
30public:
31 SIPostRABundlerLegacy() : MachineFunctionPass(ID) {}
32
33 bool runOnMachineFunction(MachineFunction &MF) override;
34
35 StringRef getPassName() const override {
36 return "SI post-RA bundler";
37 }
38
39 void getAnalysisUsage(AnalysisUsage &AU) const override {
40 AU.setPreservesAll();
41 MachineFunctionPass::getAnalysisUsage(AU);
42 }
43};
44
45class SIPostRABundler {
46public:
47 bool run(MachineFunction &MF);
48
49private:
50 const SIRegisterInfo *TRI;
51
52 SmallSet<Register, 16> Defs;
53
54 void collectUsedRegUnits(const MachineInstr &MI,
55 BitVector &UsedRegUnits) const;
56
57 bool isBundleCandidate(const MachineInstr &MI) const;
58 bool isDependentLoad(const MachineInstr &MI) const;
59 bool canBundle(const MachineInstr &MI, const MachineInstr &NextMI) const;
60};
61
62} // End anonymous namespace.
63
64INITIALIZE_PASS(SIPostRABundlerLegacy, DEBUG_TYPE, "SI post-RA bundler", false,
65 false)
66
67char SIPostRABundlerLegacy::ID = 0;
68
69char &llvm::SIPostRABundlerLegacyID = SIPostRABundlerLegacy::ID;
70
71bool SIPostRABundler::isDependentLoad(const MachineInstr &MI) const {
72 if (!MI.mayLoad())
73 return false;
74
75 for (const MachineOperand &Op : MI.explicit_operands()) {
76 if (!Op.isReg())
77 continue;
78 Register Reg = Op.getReg();
79 for (Register Def : Defs)
80 if (TRI->regsOverlap(RegA: Reg, RegB: Def))
81 return true;
82 }
83
84 return false;
85}
86
87void SIPostRABundler::collectUsedRegUnits(const MachineInstr &MI,
88 BitVector &UsedRegUnits) const {
89 if (MI.isDebugInstr())
90 return;
91
92 for (const MachineOperand &Op : MI.operands()) {
93 if (!Op.isReg() || !Op.readsReg())
94 continue;
95
96 Register Reg = Op.getReg();
97 assert(!Op.getSubReg() &&
98 "subregister indexes should not be present after RA");
99
100 for (MCRegUnit Unit : TRI->regunits(Reg))
101 UsedRegUnits.set(static_cast<unsigned>(Unit));
102 }
103}
104
105static bool isMemoryInst(const MachineInstr &MI) {
106 return SIInstrFlags::isMUBUF(O: MI) || SIInstrFlags::isMTBUF(O: MI) ||
107 SIInstrFlags::isSMRD(O: MI) || SIInstrFlags::isDS(O: MI) ||
108 SIInstrFlags::isFLAT(O: MI) || SIInstrFlags::isMIMG(O: MI) ||
109 SIInstrFlags::isVIMAGE(O: MI) || SIInstrFlags::isVSAMPLE(O: MI);
110}
111
112static bool hasSameMemFormat(const MachineInstr &A, const MachineInstr &B) {
113 return SIInstrFlags::isMUBUF(O: A) == SIInstrFlags::isMUBUF(O: B) &&
114 SIInstrFlags::isMTBUF(O: A) == SIInstrFlags::isMTBUF(O: B) &&
115 SIInstrFlags::isSMRD(O: A) == SIInstrFlags::isSMRD(O: B) &&
116 SIInstrFlags::isDS(O: A) == SIInstrFlags::isDS(O: B) &&
117 SIInstrFlags::isFLAT(O: A) == SIInstrFlags::isFLAT(O: B) &&
118 SIInstrFlags::isMIMG(O: A) == SIInstrFlags::isMIMG(O: B) &&
119 SIInstrFlags::isVIMAGE(O: A) == SIInstrFlags::isVIMAGE(O: B) &&
120 SIInstrFlags::isVSAMPLE(O: A) == SIInstrFlags::isVSAMPLE(O: B);
121}
122
123bool SIPostRABundler::isBundleCandidate(const MachineInstr &MI) const {
124 return isMemoryInst(MI) && MI.mayLoadOrStore() && !MI.isBundled();
125}
126
127bool SIPostRABundler::canBundle(const MachineInstr &MI,
128 const MachineInstr &NextMI) const {
129 return isMemoryInst(MI) && MI.mayLoadOrStore() && !NextMI.isBundled() &&
130 NextMI.mayLoad() == MI.mayLoad() &&
131 NextMI.mayStore() == MI.mayStore() && hasSameMemFormat(A: MI, B: NextMI) &&
132 !isDependentLoad(MI: NextMI);
133}
134
135bool SIPostRABundlerLegacy::runOnMachineFunction(MachineFunction &MF) {
136 if (skipFunction(F: MF.getFunction()))
137 return false;
138 return SIPostRABundler().run(MF);
139}
140
141PreservedAnalyses SIPostRABundlerPass::run(MachineFunction &MF,
142 MachineFunctionAnalysisManager &) {
143 SIPostRABundler().run(MF);
144 return PreservedAnalyses::all();
145}
146
147bool SIPostRABundler::run(MachineFunction &MF) {
148
149 TRI = MF.getSubtarget<GCNSubtarget>().getRegisterInfo();
150 BitVector BundleUsedRegUnits(TRI->getNumRegUnits());
151 BitVector KillUsedRegUnits(TRI->getNumRegUnits());
152
153 bool Changed = false;
154 for (MachineBasicBlock &MBB : MF) {
155 bool HasIGLPInstrs = llvm::any_of(Range: MBB.instrs(), P: [](MachineInstr &MI) {
156 unsigned Opc = MI.getOpcode();
157 return Opc == AMDGPU::SCHED_GROUP_BARRIER || Opc == AMDGPU::IGLP_OPT;
158 });
159
160 // Don't cluster with IGLP instructions.
161 if (HasIGLPInstrs)
162 continue;
163
164 MachineBasicBlock::instr_iterator Next;
165 MachineBasicBlock::instr_iterator B = MBB.instr_begin();
166 MachineBasicBlock::instr_iterator E = MBB.instr_end();
167
168 for (auto I = B; I != E; I = Next) {
169 Next = std::next(x: I);
170 if (!isBundleCandidate(MI: *I))
171 continue;
172
173 assert(Defs.empty());
174
175 if (I->getNumExplicitDefs() != 0)
176 Defs.insert(V: I->defs().begin()->getReg());
177
178 MachineBasicBlock::instr_iterator BundleStart = I;
179 MachineBasicBlock::instr_iterator BundleEnd = I;
180 unsigned ClauseLength = 1;
181 for (I = Next; I != E; I = Next) {
182 Next = std::next(x: I);
183
184 assert(BundleEnd != I);
185 if (canBundle(MI: *BundleEnd, NextMI: *I)) {
186 BundleEnd = I;
187 if (I->getNumExplicitDefs() != 0)
188 Defs.insert(V: I->defs().begin()->getReg());
189 ++ClauseLength;
190 } else if (!I->isMetaInstruction() ||
191 I->getOpcode() == AMDGPU::SCHED_BARRIER) {
192 // SCHED_BARRIER is not bundled to be honored by scheduler later.
193 // Allow other meta instructions in between bundle candidates, but do
194 // not start or end a bundle on one.
195 //
196 // TODO: It may be better to move meta instructions like dbg_value
197 // after the bundle. We're relying on the memory legalizer to unbundle
198 // these.
199 break;
200 }
201 }
202
203 Next = std::next(x: BundleEnd);
204 if (ClauseLength > 1) {
205 Changed = true;
206
207 // Before register allocation, kills are inserted after potential soft
208 // clauses to hint register allocation. Look for kills that look like
209 // this, and erase them.
210 if (Next != E && Next->isKill()) {
211
212 // TODO: Should maybe back-propagate kill flags to the bundle.
213 for (const MachineInstr &BundleMI : make_range(x: BundleStart, y: Next))
214 collectUsedRegUnits(MI: BundleMI, UsedRegUnits&: BundleUsedRegUnits);
215
216 BundleUsedRegUnits.flip();
217
218 while (Next != E && Next->isKill()) {
219 MachineInstr &Kill = *Next;
220 collectUsedRegUnits(MI: Kill, UsedRegUnits&: KillUsedRegUnits);
221
222 KillUsedRegUnits &= BundleUsedRegUnits;
223
224 // Erase the kill if it's a subset of the used registers.
225 //
226 // TODO: Should we just remove all kills? Is there any real reason to
227 // keep them after RA?
228 if (KillUsedRegUnits.none()) {
229 ++Next;
230 Kill.eraseFromParent();
231 } else
232 break;
233
234 KillUsedRegUnits.reset();
235 }
236
237 BundleUsedRegUnits.reset();
238 }
239
240 finalizeBundle(MBB, FirstMI: BundleStart, LastMI: Next);
241 }
242
243 Defs.clear();
244 }
245 }
246
247 return Changed;
248}
249