1//===--------------------- SIOptimizeVGPRLiveRange.cpp -------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This pass tries to remove unnecessary VGPR live ranges in divergent if-else
11/// structures and waterfall loops.
12///
13/// When we do structurization, we usually transform an if-else into two
14/// successive if-then (with a flow block to do predicate inversion). Consider a
15/// simple case after structurization: A divergent value %a was defined before
16/// if-else and used in both THEN (use in THEN is optional) and ELSE part:
17/// bb.if:
18/// %a = ...
19/// ...
20/// bb.then:
21/// ... = op %a
22/// ... // %a can be dead here
23/// bb.flow:
24/// ...
25/// bb.else:
26/// ... = %a
27/// ...
28/// bb.endif
29///
30/// As register allocator has no idea of the thread-control-flow, it will just
31/// assume %a would be alive in the whole range of bb.then because of a later
32/// use in bb.else. On AMDGPU architecture, the VGPR is accessed with respect
33/// to exec mask. For this if-else case, the lanes active in bb.then will be
34/// inactive in bb.else, and vice-versa. So we are safe to say that %a was dead
35/// after the last use in bb.then until the end of the block. The reason is
36/// the instructions in bb.then will only overwrite lanes that will never be
37/// accessed in bb.else.
38///
39/// This pass aims to tell register allocator that %a is in-fact dead,
40/// through inserting a phi-node in bb.flow saying that %a is undef when coming
41/// from bb.then, and then replace the uses in the bb.else with the result of
42/// newly inserted phi.
43///
44/// Two key conditions must be met to ensure correctness:
45/// 1.) The def-point should be in the same loop-level as if-else-endif to make
46/// sure the second loop iteration still get correct data.
47/// 2.) There should be no further uses after the IF-ELSE region.
48///
49///
50/// Waterfall loops get inserted around instructions that use divergent values
51/// but can only be executed with a uniform value. For example an indirect call
52/// to a divergent address:
53/// bb.start:
54/// %a = ...
55/// %fun = ...
56/// ...
57/// bb.loop:
58/// call %fun (%a)
59/// ... // %a can be dead here
60/// loop %bb.loop
61///
62/// The loop block is executed multiple times, but it is run exactly once for
63/// each active lane. Similar to the if-else case, the register allocator
64/// assumes that %a is live throughout the loop as it is used again in the next
65/// iteration. If %a is a VGPR that is unused after the loop, it does not need
66/// to be live after its last use in the loop block. By inserting a phi-node at
67/// the start of bb.loop that is undef when coming from bb.loop, the register
68/// allocation knows that the value of %a does not need to be preserved through
69/// iterations of the loop.
70///
71//
72//===----------------------------------------------------------------------===//
73
74#include "SIOptimizeVGPRLiveRange.h"
75#include "AMDGPU.h"
76#include "GCNSubtarget.h"
77#include "SIMachineFunctionInfo.h"
78#include "llvm/CodeGen/LiveIntervals.h"
79#include "llvm/CodeGen/MachineDominators.h"
80#include "llvm/CodeGen/MachineLoopInfo.h"
81#include "llvm/CodeGen/TargetRegisterInfo.h"
82#include "llvm/IR/Dominators.h"
83#include "llvm/InitializePasses.h"
84
85using namespace llvm;
86
87#define DEBUG_TYPE "si-opt-vgpr-liverange"
88
89namespace {
90
91class SIOptimizeVGPRLiveRange {
92private:
93 const SIRegisterInfo *TRI = nullptr;
94 const SIInstrInfo *TII = nullptr;
95 LiveIntervals *LIS = nullptr;
96 MachineDominatorTree *MDT = nullptr;
97 const MachineLoopInfo *Loops = nullptr;
98 MachineRegisterInfo *MRI = nullptr;
99
100 // Is \p Reg alive completely through \p MBB (live-in and live-out with no
101 // intervening def/kill)?
102 bool isLiveThrough(Register Reg, const MachineBasicBlock *MBB) const;
103
104 // Is \p Reg live into \p MBB? This is true when it is live through MBB or
105 // killed in MBB. A register only used by PHIs in MBB is not considered live
106 // in.
107 bool isLiveIntoMBB(Register Reg, const MachineBasicBlock *MBB) const;
108
109public:
110 SIOptimizeVGPRLiveRange(LiveIntervals *LIS, MachineDominatorTree *MDT,
111 MachineLoopInfo *Loops)
112 : LIS(LIS), MDT(MDT), Loops(Loops) {}
113 bool run(MachineFunction &MF);
114
115 MachineBasicBlock *getElseTarget(MachineBasicBlock *MBB) const;
116
117 void collectElseRegionBlocks(MachineBasicBlock *Flow,
118 MachineBasicBlock *Endif,
119 SmallSetVector<MachineBasicBlock *, 16> &) const;
120
121 void
122 collectCandidateRegisters(MachineBasicBlock *If, MachineBasicBlock *Flow,
123 MachineBasicBlock *Endif,
124 SmallSetVector<MachineBasicBlock *, 16> &ElseBlocks,
125 SmallVectorImpl<Register> &CandidateRegs) const;
126
127 void collectWaterfallCandidateRegisters(
128 MachineBasicBlock *LoopHeader, MachineBasicBlock *LoopEnd,
129 SmallSetVector<Register, 16> &CandidateRegs,
130 SmallSetVector<MachineBasicBlock *, 2> &Blocks,
131 SmallVectorImpl<MachineInstr *> &Instructions) const;
132
133 void
134 optimizeLiveRange(Register Reg, MachineBasicBlock *If,
135 MachineBasicBlock *Flow, MachineBasicBlock *Endif,
136 SmallSetVector<MachineBasicBlock *, 16> &ElseBlocks) const;
137
138 void optimizeWaterfallLiveRange(
139 Register Reg, MachineBasicBlock *LoopHeader,
140 SmallSetVector<MachineBasicBlock *, 2> &LoopBlocks,
141 SmallVectorImpl<MachineInstr *> &Instructions) const;
142};
143
144class SIOptimizeVGPRLiveRangeLegacy : public MachineFunctionPass {
145public:
146 static char ID;
147
148 SIOptimizeVGPRLiveRangeLegacy() : MachineFunctionPass(ID) {}
149
150 bool runOnMachineFunction(MachineFunction &MF) override;
151
152 StringRef getPassName() const override {
153 return "SI Optimize VGPR LiveRange";
154 }
155
156 void getAnalysisUsage(AnalysisUsage &AU) const override {
157 AU.setPreservesCFG();
158 AU.addRequired<LiveIntervalsWrapperPass>();
159 AU.addPreserved<SlotIndexesWrapperPass>();
160 AU.addPreserved<LiveIntervalsWrapperPass>();
161 AU.addRequired<MachineDominatorTreeWrapperPass>();
162 AU.addRequired<MachineLoopInfoWrapperPass>();
163 MachineFunctionPass::getAnalysisUsage(AU);
164 }
165
166 MachineFunctionProperties getRequiredProperties() const override {
167 return MachineFunctionProperties().setIsSSA();
168 }
169
170 MachineFunctionProperties getClearedProperties() const override {
171 return MachineFunctionProperties().setNoPHIs();
172 }
173};
174
175} // end anonymous namespace
176
177// Check whether the MBB is a else flow block and get the branching target which
178// is the Endif block
179MachineBasicBlock *
180SIOptimizeVGPRLiveRange::getElseTarget(MachineBasicBlock *MBB) const {
181 for (auto &BR : MBB->terminators()) {
182 if (BR.getOpcode() == AMDGPU::SI_ELSE)
183 return BR.getOperand(i: 2).getMBB();
184 }
185 return nullptr;
186}
187
188bool SIOptimizeVGPRLiveRange::isLiveThrough(
189 Register Reg, const MachineBasicBlock *MBB) const {
190 const LiveInterval &LI = LIS->getInterval(Reg);
191 return LIS->isLiveInToMBB(LR: LI, mbb: MBB) && LIS->isLiveOutOfMBB(LR: LI, mbb: MBB);
192}
193
194bool SIOptimizeVGPRLiveRange::isLiveIntoMBB(
195 Register Reg, const MachineBasicBlock *MBB) const {
196 const LiveInterval &LI = LIS->getInterval(Reg);
197 return LIS->isLiveInToMBB(LR: LI, mbb: MBB);
198}
199
200void SIOptimizeVGPRLiveRange::collectElseRegionBlocks(
201 MachineBasicBlock *Flow, MachineBasicBlock *Endif,
202 SmallSetVector<MachineBasicBlock *, 16> &Blocks) const {
203 assert(Flow != Endif);
204
205 MachineBasicBlock *MBB = Endif;
206 unsigned Cur = 0;
207 while (MBB) {
208 for (auto *Pred : MBB->predecessors()) {
209 if (Pred != Flow)
210 Blocks.insert(X: Pred);
211 }
212
213 if (Cur < Blocks.size())
214 MBB = Blocks[Cur++];
215 else
216 MBB = nullptr;
217 }
218
219 LLVM_DEBUG({
220 dbgs() << "Found Else blocks: ";
221 for (auto *MBB : Blocks)
222 dbgs() << printMBBReference(*MBB) << ' ';
223 dbgs() << '\n';
224 });
225}
226
227/// Collect the killed registers in the ELSE region which are not alive through
228/// the whole THEN region.
229void SIOptimizeVGPRLiveRange::collectCandidateRegisters(
230 MachineBasicBlock *If, MachineBasicBlock *Flow, MachineBasicBlock *Endif,
231 SmallSetVector<MachineBasicBlock *, 16> &ElseBlocks,
232 SmallVectorImpl<Register> &CandidateRegs) const {
233
234 SmallSet<Register, 8> KillsInElse;
235
236 for (auto *Else : ElseBlocks) {
237 for (auto &MI : Else->instrs()) {
238 if (MI.isDebugInstr())
239 continue;
240
241 for (auto &MO : MI.operands()) {
242 if (!MO.isReg() || !MO.getReg() || MO.isDef())
243 continue;
244
245 Register MOReg = MO.getReg();
246 // We can only optimize VGPR/AGPR/AV virtual registers.
247 if (!MOReg.isVirtual() ||
248 !TRI->hasVectorRegisters(RC: MRI->getRegClass(Reg: MOReg)))
249 continue;
250
251 if (MO.readsReg()) {
252 const MachineBasicBlock *DefMBB = MRI->getDefBlock(Reg: MOReg);
253 // Make sure two conditions are met:
254 // a.) the value is defined before/in the IF block
255 // b.) should be defined in the same loop-level.
256 if ((isLiveThrough(Reg: MOReg, MBB: If) || DefMBB == If) &&
257 Loops->getLoopFor(BB: DefMBB) == Loops->getLoopFor(BB: If)) {
258 // Check if the register is live into the endif block. If not,
259 // consider it killed in the else region.
260 if (!isLiveIntoMBB(Reg: MOReg, MBB: Endif)) {
261 KillsInElse.insert(V: MOReg);
262 } else {
263 LLVM_DEBUG(dbgs() << "Excluding " << printReg(MOReg, TRI)
264 << " as Live in Endif\n");
265 }
266 }
267 }
268 }
269 }
270 }
271
272 // Check the phis in the Endif, looking for value coming from the ELSE
273 // region. Make sure the phi-use is the last use.
274 for (auto &MI : Endif->phis()) {
275 for (unsigned Idx = 1; Idx < MI.getNumOperands(); Idx += 2) {
276 auto &MO = MI.getOperand(i: Idx);
277 auto *Pred = MI.getOperand(i: Idx + 1).getMBB();
278 if (Pred == Flow)
279 continue;
280 assert(ElseBlocks.contains(Pred) && "Should be from Else region\n");
281
282 if (!MO.isReg() || !MO.getReg() || MO.isUndef())
283 continue;
284
285 Register Reg = MO.getReg();
286 if (!Reg.isVirtual() || !TRI->hasVectorRegisters(RC: MRI->getRegClass(Reg)))
287 continue;
288
289 if (isLiveIntoMBB(Reg, MBB: Endif)) {
290 LLVM_DEBUG(dbgs() << "Excluding " << printReg(Reg, TRI)
291 << " as Live in Endif\n");
292 continue;
293 }
294 // Make sure two conditions are met:
295 // a.) the value is defined before/in the IF block
296 // b.) should be defined in the same loop-level.
297 const MachineBasicBlock *DefMBB = MRI->getDefBlock(Reg);
298 if ((isLiveThrough(Reg, MBB: If) || DefMBB == If) &&
299 Loops->getLoopFor(BB: DefMBB) == Loops->getLoopFor(BB: If))
300 KillsInElse.insert(V: Reg);
301 }
302 }
303
304 auto IsLiveThroughThen = [&](Register Reg) {
305 for (auto I = MRI->use_nodbg_begin(RegNo: Reg), E = MRI->use_nodbg_end(); I != E;
306 ++I) {
307 if (!I->readsReg())
308 continue;
309 auto *UseMI = I->getParent();
310 auto *UseMBB = UseMI->getParent();
311 if (UseMBB == Flow || UseMBB == Endif) {
312 if (!UseMI->isPHI())
313 return true;
314
315 auto *IncomingMBB = UseMI->getOperand(i: I.getOperandNo() + 1).getMBB();
316 // The register is live through the path If->Flow or Flow->Endif.
317 // we should not optimize for such cases.
318 if ((UseMBB == Flow && IncomingMBB != If) ||
319 (UseMBB == Endif && IncomingMBB == Flow))
320 return true;
321 }
322 }
323 return false;
324 };
325
326 for (auto Reg : KillsInElse) {
327 if (!IsLiveThroughThen(Reg))
328 CandidateRegs.push_back(Elt: Reg);
329 }
330}
331
332/// Collect the registers used in the waterfall loop block that are defined
333/// before.
334void SIOptimizeVGPRLiveRange::collectWaterfallCandidateRegisters(
335 MachineBasicBlock *LoopHeader, MachineBasicBlock *LoopEnd,
336 SmallSetVector<Register, 16> &CandidateRegs,
337 SmallSetVector<MachineBasicBlock *, 2> &Blocks,
338 SmallVectorImpl<MachineInstr *> &Instructions) const {
339
340 // Collect loop instructions, potentially spanning multiple blocks
341 auto *MBB = LoopHeader;
342 for (;;) {
343 Blocks.insert(X: MBB);
344 for (auto &MI : *MBB) {
345 if (MI.isDebugInstr())
346 continue;
347 Instructions.push_back(Elt: &MI);
348 }
349 if (MBB == LoopEnd)
350 break;
351
352 if ((MBB != LoopHeader && MBB->pred_size() != 1) ||
353 (MBB == LoopHeader && MBB->pred_size() != 2) || MBB->succ_size() != 1) {
354 LLVM_DEBUG(dbgs() << "Unexpected edges in CFG, ignoring loop\n");
355 return;
356 }
357
358 MBB = *MBB->succ_begin();
359 }
360
361 for (auto *I : Instructions) {
362 auto &MI = *I;
363
364 for (auto &MO : MI.all_uses()) {
365 if (!MO.getReg())
366 continue;
367
368 Register MOReg = MO.getReg();
369 // We can only optimize VGPR/AGPR/AV virtual registers.
370 if (!MOReg.isVirtual() ||
371 !TRI->hasVectorRegisters(RC: MRI->getRegClass(Reg: MOReg)))
372 continue;
373
374 if (MO.readsReg()) {
375 MachineBasicBlock *DefMBB = MRI->getDefBlock(Reg: MOReg);
376 // Make sure the value is defined before the LOOP block
377 if (!Blocks.contains(key: DefMBB) && !CandidateRegs.contains(key: MOReg)) {
378 // If the variable is used after the loop, the register coalescer will
379 // merge the newly created register and remove the phi node again.
380 // Just do nothing in that case.
381 bool IsUsed = false;
382 for (auto *Succ : LoopEnd->successors()) {
383 if (!Blocks.contains(key: Succ) && isLiveIntoMBB(Reg: MOReg, MBB: Succ)) {
384 IsUsed = true;
385 break;
386 }
387 }
388 if (!IsUsed) {
389 LLVM_DEBUG(dbgs() << "Found candidate reg: "
390 << printReg(MOReg, TRI, 0, MRI) << '\n');
391 CandidateRegs.insert(X: MOReg);
392 } else {
393 LLVM_DEBUG(dbgs() << "Reg is used after loop, ignoring: "
394 << printReg(MOReg, TRI, 0, MRI) << '\n');
395 }
396 }
397 }
398 }
399 }
400}
401
402void SIOptimizeVGPRLiveRange::optimizeLiveRange(
403 Register Reg, MachineBasicBlock *If, MachineBasicBlock *Flow,
404 MachineBasicBlock *Endif,
405 SmallSetVector<MachineBasicBlock *, 16> &ElseBlocks) const {
406 // Insert a new PHI, marking the value from the THEN region being
407 // undef.
408 LLVM_DEBUG(dbgs() << "Optimizing " << printReg(Reg, TRI) << '\n');
409 const auto *RC = MRI->getRegClass(Reg);
410 Register NewReg = MRI->createVirtualRegister(RegClass: RC);
411 Register UndefReg = MRI->createVirtualRegister(RegClass: RC);
412 MachineInstrBuilder PHI = BuildMI(BB&: *Flow, I: Flow->getFirstNonPHI(), MIMD: DebugLoc(),
413 MCID: TII->get(Opcode: TargetOpcode::PHI), DestReg: NewReg);
414 for (auto *Pred : Flow->predecessors()) {
415 if (Pred == If)
416 PHI.addReg(RegNo: Reg).addMBB(MBB: Pred);
417 else
418 PHI.addReg(RegNo: UndefReg, Flags: RegState::Undef).addMBB(MBB: Pred);
419 }
420
421 // Replace all uses in the ELSE region or the PHIs in ENDIF block
422 // Use early increment range because setReg() will update the linked list.
423 for (auto &O : make_early_inc_range(Range: MRI->use_operands(Reg))) {
424 auto *UseMI = O.getParent();
425 auto *UseBlock = UseMI->getParent();
426 // Replace uses in Endif block
427 if (UseBlock == Endif) {
428 if (UseMI->isPHI())
429 O.setReg(NewReg);
430 else if (UseMI->isDebugInstr())
431 continue;
432 else {
433 // DetectDeadLanes may mark register uses as undef without removing
434 // them, in which case a non-phi instruction using the original register
435 // may exist in the Endif block even though the register is not live
436 // into it.
437 assert(!O.readsReg());
438 }
439 continue;
440 }
441
442 // Replace uses in Else region
443 if (ElseBlocks.contains(key: UseBlock))
444 O.setReg(NewReg);
445 }
446
447 // The new PHI is a def of NewReg and a use of Reg and UndefReg; the uses of
448 // Reg in the Else/Endif region were rewritten to NewReg. Kill flags moved
449 // with the rewritten operands may no longer mark the last use, so drop them
450 // and let the recomputed intervals be the source of truth.
451 MRI->clearKillFlags(Reg);
452 MRI->clearKillFlags(Reg: NewReg);
453 LIS->InsertMachineInstrInMaps(MI&: *PHI);
454 LIS->removeInterval(Reg);
455 LIS->createAndComputeVirtRegInterval(Reg);
456 LIS->createAndComputeVirtRegInterval(Reg: NewReg);
457 LIS->createAndComputeVirtRegInterval(Reg: UndefReg);
458}
459
460void SIOptimizeVGPRLiveRange::optimizeWaterfallLiveRange(
461 Register Reg, MachineBasicBlock *LoopHeader,
462 SmallSetVector<MachineBasicBlock *, 2> &Blocks,
463 SmallVectorImpl<MachineInstr *> &Instructions) const {
464 // Insert a new PHI, marking the value from the last loop iteration undef.
465 LLVM_DEBUG(dbgs() << "Optimizing " << printReg(Reg, TRI) << '\n');
466 const auto *RC = MRI->getRegClass(Reg);
467 Register NewReg = MRI->createVirtualRegister(RegClass: RC);
468 Register UndefReg = MRI->createVirtualRegister(RegClass: RC);
469
470 // Replace all uses in the LOOP region
471 // Use early increment range because setReg() will update the linked list.
472 for (auto &O : make_early_inc_range(Range: MRI->use_operands(Reg))) {
473 auto *UseMI = O.getParent();
474 auto *UseBlock = UseMI->getParent();
475 // Replace uses in Loop blocks
476 if (Blocks.contains(key: UseBlock))
477 O.setReg(NewReg);
478 }
479
480 MachineInstrBuilder PHI =
481 BuildMI(BB&: *LoopHeader, I: LoopHeader->getFirstNonPHI(), MIMD: DebugLoc(),
482 MCID: TII->get(Opcode: TargetOpcode::PHI), DestReg: NewReg);
483 for (auto *Pred : LoopHeader->predecessors()) {
484 if (Blocks.contains(key: Pred))
485 PHI.addReg(RegNo: UndefReg, Flags: RegState::Undef).addMBB(MBB: Pred);
486 else
487 PHI.addReg(RegNo: Reg).addMBB(MBB: Pred);
488 }
489
490 LIS->InsertMachineInstrInMaps(MI&: *PHI);
491 LIS->removeInterval(Reg);
492 LIS->createAndComputeVirtRegInterval(Reg);
493 LIS->createAndComputeVirtRegInterval(Reg: NewReg);
494 LIS->createAndComputeVirtRegInterval(Reg: UndefReg);
495}
496
497char SIOptimizeVGPRLiveRangeLegacy::ID = 0;
498
499INITIALIZE_PASS_BEGIN(SIOptimizeVGPRLiveRangeLegacy, DEBUG_TYPE,
500 "SI Optimize VGPR LiveRange", false, false)
501INITIALIZE_PASS_DEPENDENCY(MachineDominatorTreeWrapperPass)
502INITIALIZE_PASS_DEPENDENCY(MachineLoopInfoWrapperPass)
503INITIALIZE_PASS_DEPENDENCY(LiveIntervalsWrapperPass)
504INITIALIZE_PASS_END(SIOptimizeVGPRLiveRangeLegacy, DEBUG_TYPE,
505 "SI Optimize VGPR LiveRange", false, false)
506
507char &llvm::SIOptimizeVGPRLiveRangeLegacyID = SIOptimizeVGPRLiveRangeLegacy::ID;
508
509bool SIOptimizeVGPRLiveRangeLegacy::runOnMachineFunction(MachineFunction &MF) {
510 if (skipFunction(F: MF.getFunction()))
511 return false;
512
513 LiveIntervals *LIS = &getAnalysis<LiveIntervalsWrapperPass>().getLIS();
514 MachineDominatorTree *MDT =
515 &getAnalysis<MachineDominatorTreeWrapperPass>().getDomTree();
516 MachineLoopInfo *Loops = &getAnalysis<MachineLoopInfoWrapperPass>().getLI();
517 return SIOptimizeVGPRLiveRange(LIS, MDT, Loops).run(MF);
518}
519
520PreservedAnalyses
521SIOptimizeVGPRLiveRangePass::run(MachineFunction &MF,
522 MachineFunctionAnalysisManager &MFAM) {
523 MFPropsModifier _(*this, MF);
524 LiveIntervals *LIS = &MFAM.getResult<LiveIntervalsAnalysis>(IR&: MF);
525 MachineDominatorTree *MDT = &MFAM.getResult<MachineDominatorTreeAnalysis>(IR&: MF);
526 MachineLoopInfo *Loops = &MFAM.getResult<MachineLoopAnalysis>(IR&: MF);
527
528 bool Changed = SIOptimizeVGPRLiveRange(LIS, MDT, Loops).run(MF);
529 if (!Changed)
530 return PreservedAnalyses::all();
531
532 auto PA = getMachineFunctionPassPreservedAnalyses();
533 PA.preserve<SlotIndexesAnalysis>();
534 PA.preserve<LiveIntervalsAnalysis>();
535 PA.preserveSet<CFGAnalyses>();
536 return PA;
537}
538
539bool SIOptimizeVGPRLiveRange::run(MachineFunction &MF) {
540 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
541 TII = ST.getInstrInfo();
542 TRI = &TII->getRegisterInfo();
543 MRI = &MF.getRegInfo();
544
545 bool MadeChange = false;
546
547 // TODO: we need to think about the order of visiting the blocks to get
548 // optimal result for nesting if-else cases.
549 for (MachineBasicBlock &MBB : MF) {
550 for (auto &MI : MBB.terminators()) {
551 // Detect the if-else blocks
552 if (MI.getOpcode() == AMDGPU::SI_IF) {
553 MachineBasicBlock *IfTarget = MI.getOperand(i: 2).getMBB();
554 auto *Endif = getElseTarget(MBB: IfTarget);
555 if (!Endif)
556 continue;
557
558 // Skip unexpected control flow.
559 if (!MDT->dominates(A: &MBB, B: IfTarget) || !MDT->dominates(A: IfTarget, B: Endif))
560 continue;
561
562 SmallSetVector<MachineBasicBlock *, 16> ElseBlocks;
563 SmallVector<Register> CandidateRegs;
564
565 LLVM_DEBUG(dbgs() << "Checking IF-ELSE-ENDIF: "
566 << printMBBReference(MBB) << ' '
567 << printMBBReference(*IfTarget) << ' '
568 << printMBBReference(*Endif) << '\n');
569
570 // Collect all the blocks in the ELSE region
571 collectElseRegionBlocks(Flow: IfTarget, Endif, Blocks&: ElseBlocks);
572
573 // Collect the registers can be optimized
574 collectCandidateRegisters(If: &MBB, Flow: IfTarget, Endif, ElseBlocks,
575 CandidateRegs);
576 MadeChange |= !CandidateRegs.empty();
577 // Now we are safe to optimize.
578 for (auto Reg : CandidateRegs)
579 optimizeLiveRange(Reg, If: &MBB, Flow: IfTarget, Endif, ElseBlocks);
580 } else if (MI.getOpcode() == AMDGPU::SI_WATERFALL_LOOP) {
581 auto *LoopHeader = MI.getOperand(i: 0).getMBB();
582 auto *LoopEnd = &MBB;
583
584 LLVM_DEBUG(dbgs() << "Checking Waterfall loop: "
585 << printMBBReference(*LoopHeader) << '\n');
586
587 SmallSetVector<Register, 16> CandidateRegs;
588 SmallVector<MachineInstr *, 16> Instructions;
589 SmallSetVector<MachineBasicBlock *, 2> Blocks;
590
591 collectWaterfallCandidateRegisters(LoopHeader, LoopEnd, CandidateRegs,
592 Blocks, Instructions);
593 MadeChange |= !CandidateRegs.empty();
594 // Now we are safe to optimize.
595 for (auto Reg : CandidateRegs)
596 optimizeWaterfallLiveRange(Reg, LoopHeader, Blocks, Instructions);
597 }
598 }
599 }
600
601 return MadeChange;
602}
603