1//===- SIPeepholeSDWA.cpp - Peephole optimization for SDWA instructions ---===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file This pass tries to apply several peephole SDWA patterns.
10///
11/// E.g. original:
12/// V_LSHRREV_B32_e32 %0, 16, %1
13/// V_ADD_CO_U32_e32 %2, %0, %3
14/// V_LSHLREV_B32_e32 %4, 16, %2
15///
16/// Replace:
17/// V_ADD_CO_U32_sdwa %4, %1, %3
18/// dst_sel:WORD_1 dst_unused:UNUSED_PAD src0_sel:WORD_1 src1_sel:DWORD
19///
20//===----------------------------------------------------------------------===//
21
22#include "SIPeepholeSDWA.h"
23#include "AMDGPU.h"
24#include "GCNSubtarget.h"
25#include "llvm/ADT/Statistic.h"
26#include "llvm/CodeGen/MachineFunctionPass.h"
27#include <optional>
28
29using namespace llvm;
30
31#define DEBUG_TYPE "si-peephole-sdwa"
32
33STATISTIC(NumSDWAPatternsFound, "Number of SDWA patterns found.");
34STATISTIC(NumSDWAInstructionsPeepholed,
35 "Number of instruction converted to SDWA.");
36
37namespace {
38
39bool isConvertibleToSDWA(MachineInstr &MI, const GCNSubtarget &ST,
40 const SIInstrInfo *TII);
41class SDWAOperand;
42class SDWADstOperand;
43
44using SDWAOperandsVector = SmallVector<SDWAOperand *, 4>;
45using SDWAOperandsMap = MapVector<MachineInstr *, SDWAOperandsVector>;
46
47class SIPeepholeSDWA {
48private:
49 MachineRegisterInfo *MRI;
50 const SIRegisterInfo *TRI;
51 const SIInstrInfo *TII;
52
53 MapVector<MachineInstr *, std::unique_ptr<SDWAOperand>> SDWAOperands;
54 SDWAOperandsMap PotentialMatches;
55 SmallVector<MachineInstr *, 8> ConvertedInstructions;
56
57 std::optional<int64_t> foldToImm(const MachineOperand &Op) const;
58
59 // If MI is a v_and_b32 with a 0xffff or 0xff immediate, return the masked
60 // value operand and the matching SDWA selector (WORD_0 / BYTE_0).
61 std::optional<std::pair<MachineOperand *, AMDGPU::SDWA::SdwaSel>>
62 matchAndMask(MachineInstr &MI) const;
63
64 // VOPC SDWA instructions carry the SDWA TSFlag but have no dst_sel operand.
65 bool isSDWAWithDstSel(const MachineInstr &Inst) const;
66
67 void matchSDWAOperands(MachineBasicBlock &MBB);
68 std::unique_ptr<SDWAOperand> matchSDWAOperand(MachineInstr &MI);
69 void pseudoOpConvertToVOP2(MachineInstr &MI,
70 const GCNSubtarget &ST) const;
71 void convertVcndmaskToVOP2(MachineInstr &MI, const GCNSubtarget &ST) const;
72 MachineInstr *createSDWAVersion(MachineInstr &MI);
73 bool convertToSDWA(MachineInstr &MI, const SDWAOperandsVector &SDWAOperands);
74 void legalizeScalarOperands(MachineInstr &MI, const GCNSubtarget &ST) const;
75 bool splitLshlOrForSDWA(MachineBasicBlock &MBB);
76
77public:
78 bool run(MachineFunction &MF);
79};
80
81class SIPeepholeSDWALegacy : public MachineFunctionPass {
82public:
83 static char ID;
84
85 SIPeepholeSDWALegacy() : MachineFunctionPass(ID) {}
86
87 StringRef getPassName() const override { return "SI Peephole SDWA"; }
88
89 bool runOnMachineFunction(MachineFunction &MF) override;
90
91 void getAnalysisUsage(AnalysisUsage &AU) const override {
92 AU.setPreservesCFG();
93 MachineFunctionPass::getAnalysisUsage(AU);
94 }
95};
96
97using namespace AMDGPU::SDWA;
98
99class SDWAOperand {
100private:
101 MachineOperand *Target; // Operand that would be used in converted instruction
102 MachineOperand *Replaced; // Operand that would be replace by Target
103
104 /// Returns true iff the SDWA selection of this SDWAOperand can be combined
105 /// with the SDWA selections of its uses in \p MI.
106 virtual bool canCombineSelections(const MachineInstr &MI,
107 const SIInstrInfo *TII) = 0;
108
109public:
110 SDWAOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp)
111 : Target(TargetOp), Replaced(ReplacedOp) {
112 assert(Target->isReg());
113 assert(Replaced->isReg());
114 }
115
116 virtual ~SDWAOperand() = default;
117
118 virtual MachineInstr *potentialToConvert(const SIInstrInfo *TII,
119 const GCNSubtarget &ST,
120 SDWAOperandsMap *PotentialMatches = nullptr) = 0;
121 virtual bool convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) = 0;
122
123 MachineOperand *getTargetOperand() const { return Target; }
124 MachineOperand *getReplacedOperand() const { return Replaced; }
125 MachineInstr *getParentInst() const { return Target->getParent(); }
126
127 MachineRegisterInfo *getMRI() const {
128 return &getParentInst()->getMF()->getRegInfo();
129 }
130
131#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
132 virtual void print(raw_ostream& OS) const = 0;
133 void dump() const { print(dbgs()); }
134#endif
135};
136
137class SDWASrcOperand : public SDWAOperand {
138private:
139 SdwaSel SrcSel;
140 bool Abs;
141 bool Neg;
142 bool Sext;
143
144public:
145 SDWASrcOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp,
146 SdwaSel SrcSel_ = DWORD, bool Abs_ = false, bool Neg_ = false,
147 bool Sext_ = false)
148 : SDWAOperand(TargetOp, ReplacedOp), SrcSel(SrcSel_), Abs(Abs_),
149 Neg(Neg_), Sext(Sext_) {}
150
151 MachineInstr *potentialToConvert(const SIInstrInfo *TII,
152 const GCNSubtarget &ST,
153 SDWAOperandsMap *PotentialMatches = nullptr) override;
154 bool convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) override;
155 bool canCombineSelections(const MachineInstr &MI,
156 const SIInstrInfo *TII) override;
157
158 SdwaSel getSrcSel() const { return SrcSel; }
159 bool getAbs() const { return Abs; }
160 bool getNeg() const { return Neg; }
161 bool getSext() const { return Sext; }
162
163 uint64_t getSrcMods(const SIInstrInfo *TII,
164 const MachineOperand *SrcOp) const;
165
166#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
167 void print(raw_ostream& OS) const override;
168#endif
169};
170
171class SDWADstOperand : public SDWAOperand {
172private:
173 SdwaSel DstSel;
174 DstUnused DstUn;
175
176public:
177 SDWADstOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp,
178 SdwaSel DstSel_ = DWORD, DstUnused DstUn_ = UNUSED_PAD)
179 : SDWAOperand(TargetOp, ReplacedOp), DstSel(DstSel_), DstUn(DstUn_) {}
180
181 MachineInstr *potentialToConvert(const SIInstrInfo *TII,
182 const GCNSubtarget &ST,
183 SDWAOperandsMap *PotentialMatches = nullptr) override;
184 bool convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) override;
185 bool canCombineSelections(const MachineInstr &MI,
186 const SIInstrInfo *TII) override;
187
188 SdwaSel getDstSel() const { return DstSel; }
189 DstUnused getDstUnused() const { return DstUn; }
190
191#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
192 void print(raw_ostream& OS) const override;
193#endif
194};
195
196class SDWADstPreserveOperand : public SDWADstOperand {
197private:
198 MachineOperand *Preserve;
199
200public:
201 SDWADstPreserveOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp,
202 MachineOperand *PreserveOp, SdwaSel DstSel_ = DWORD)
203 : SDWADstOperand(TargetOp, ReplacedOp, DstSel_, UNUSED_PRESERVE),
204 Preserve(PreserveOp) {}
205
206 bool convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) override;
207 bool canCombineSelections(const MachineInstr &MI,
208 const SIInstrInfo *TII) override;
209
210 MachineOperand *getPreservedOperand() const { return Preserve; }
211
212#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
213 void print(raw_ostream& OS) const override;
214#endif
215};
216
217} // end anonymous namespace
218
219INITIALIZE_PASS(SIPeepholeSDWALegacy, DEBUG_TYPE, "SI Peephole SDWA", false,
220 false)
221
222char SIPeepholeSDWALegacy::ID = 0;
223
224char &llvm::SIPeepholeSDWALegacyID = SIPeepholeSDWALegacy::ID;
225
226#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
227static raw_ostream& operator<<(raw_ostream &OS, SdwaSel Sel) {
228 switch(Sel) {
229 case BYTE_0: OS << "BYTE_0"; break;
230 case BYTE_1: OS << "BYTE_1"; break;
231 case BYTE_2: OS << "BYTE_2"; break;
232 case BYTE_3: OS << "BYTE_3"; break;
233 case WORD_0: OS << "WORD_0"; break;
234 case WORD_1: OS << "WORD_1"; break;
235 case DWORD: OS << "DWORD"; break;
236 }
237 return OS;
238}
239
240static raw_ostream& operator<<(raw_ostream &OS, const DstUnused &Un) {
241 switch(Un) {
242 case UNUSED_PAD: OS << "UNUSED_PAD"; break;
243 case UNUSED_SEXT: OS << "UNUSED_SEXT"; break;
244 case UNUSED_PRESERVE: OS << "UNUSED_PRESERVE"; break;
245 }
246 return OS;
247}
248
249LLVM_DUMP_METHOD
250void SDWASrcOperand::print(raw_ostream& OS) const {
251 OS << "SDWA src: " << *getTargetOperand()
252 << " src_sel:" << getSrcSel()
253 << " abs:" << getAbs() << " neg:" << getNeg()
254 << " sext:" << getSext() << '\n';
255}
256
257LLVM_DUMP_METHOD
258void SDWADstOperand::print(raw_ostream& OS) const {
259 OS << "SDWA dst: " << *getTargetOperand()
260 << " dst_sel:" << getDstSel()
261 << " dst_unused:" << getDstUnused() << '\n';
262}
263
264LLVM_DUMP_METHOD
265void SDWADstPreserveOperand::print(raw_ostream& OS) const {
266 OS << "SDWA preserve dst: " << *getTargetOperand()
267 << " dst_sel:" << getDstSel()
268 << " preserve:" << *getPreservedOperand() << '\n';
269}
270
271#endif
272
273static void copyRegOperand(MachineOperand &To, const MachineOperand &From) {
274 assert(To.isReg() && From.isReg());
275 To.setReg(From.getReg());
276 To.setSubReg(From.getSubReg());
277 To.setIsUndef(From.isUndef());
278 if (To.isUse()) {
279 To.setIsKill(From.isKill());
280 } else {
281 To.setIsDead(From.isDead());
282 }
283}
284
285static bool isSameReg(const MachineOperand &LHS, const MachineOperand &RHS) {
286 return LHS.isReg() &&
287 RHS.isReg() &&
288 LHS.getReg() == RHS.getReg() &&
289 LHS.getSubReg() == RHS.getSubReg();
290}
291
292static MachineOperand *findSingleRegUse(const MachineOperand *Reg,
293 const MachineRegisterInfo *MRI) {
294 if (!Reg->isReg() || !Reg->isDef())
295 return nullptr;
296
297 return MRI->getOneNonDBGUse(RegNo: Reg->getReg());
298}
299
300static MachineOperand *findSingleRegDef(const MachineOperand *Reg,
301 const MachineRegisterInfo *MRI) {
302 if (!Reg->isReg())
303 return nullptr;
304
305 return MRI->getOneDef(Reg: Reg->getReg());
306}
307
308/// Combine an SDWA instruction's existing SDWA selection \p Sel with
309/// the SDWA selection \p OperandSel of its operand. If the selections
310/// are compatible, return the combined selection, otherwise return a
311/// nullopt.
312/// For example, if we have Sel = BYTE_0 Sel and OperandSel = WORD_1:
313/// BYTE_0 Sel (WORD_1 Sel (%X)) -> BYTE_2 Sel (%X)
314static std::optional<SdwaSel> combineSdwaSel(SdwaSel Sel, SdwaSel OperandSel) {
315 if (Sel == SdwaSel::DWORD)
316 return OperandSel;
317
318 if (Sel == OperandSel || OperandSel == SdwaSel::DWORD)
319 return Sel;
320
321 if (Sel == SdwaSel::WORD_1 || Sel == SdwaSel::BYTE_2 ||
322 Sel == SdwaSel::BYTE_3)
323 return {};
324
325 if (OperandSel == SdwaSel::WORD_0)
326 return Sel;
327
328 if (OperandSel == SdwaSel::WORD_1) {
329 if (Sel == SdwaSel::BYTE_0)
330 return SdwaSel::BYTE_2;
331 if (Sel == SdwaSel::BYTE_1)
332 return SdwaSel::BYTE_3;
333 if (Sel == SdwaSel::WORD_0)
334 return SdwaSel::WORD_1;
335 }
336
337 return {};
338}
339
340uint64_t SDWASrcOperand::getSrcMods(const SIInstrInfo *TII,
341 const MachineOperand *SrcOp) const {
342 uint64_t Mods = 0;
343 const auto *MI = SrcOp->getParent();
344 if (TII->getNamedOperand(MI: *MI, OperandName: AMDGPU::OpName::src0) == SrcOp) {
345 if (auto *Mod = TII->getNamedOperand(MI: *MI, OperandName: AMDGPU::OpName::src0_modifiers)) {
346 Mods = Mod->getImm();
347 }
348 } else if (TII->getNamedOperand(MI: *MI, OperandName: AMDGPU::OpName::src1) == SrcOp) {
349 if (auto *Mod = TII->getNamedOperand(MI: *MI, OperandName: AMDGPU::OpName::src1_modifiers)) {
350 Mods = Mod->getImm();
351 }
352 }
353 if (Abs || Neg) {
354 assert(!Sext &&
355 "Float and integer src modifiers can't be set simultaneously");
356 Mods |= Abs ? SISrcMods::ABS : 0u;
357 Mods ^= Neg ? SISrcMods::NEG : 0u;
358 } else if (Sext) {
359 Mods |= SISrcMods::SEXT;
360 }
361
362 return Mods;
363}
364
365MachineInstr *SDWASrcOperand::potentialToConvert(const SIInstrInfo *TII,
366 const GCNSubtarget &ST,
367 SDWAOperandsMap *PotentialMatches) {
368 if (PotentialMatches != nullptr) {
369 // Fill out the map for all uses if all can be converted
370 MachineOperand *Reg = getReplacedOperand();
371 if (!Reg->isReg() || !Reg->isDef())
372 return nullptr;
373
374 for (MachineInstr &UseMI : getMRI()->use_nodbg_instructions(Reg: Reg->getReg()))
375 // Check that all instructions that use Reg can be converted
376 if (!isConvertibleToSDWA(MI&: UseMI, ST, TII) ||
377 !canCombineSelections(MI: UseMI, TII))
378 return nullptr;
379
380 // Now that it's guaranteed all uses are legal, iterate over the uses again
381 // to add them for later conversion.
382 for (MachineOperand &UseMO : getMRI()->use_nodbg_operands(Reg: Reg->getReg())) {
383 // Should not get a subregister here
384 assert(isSameReg(UseMO, *Reg));
385
386 SDWAOperandsMap &potentialMatchesMap = *PotentialMatches;
387 MachineInstr *UseMI = UseMO.getParent();
388 potentialMatchesMap[UseMI].push_back(Elt: this);
389 }
390 return nullptr;
391 }
392
393 // For SDWA src operand potential instruction is one that use register
394 // defined by parent instruction
395 MachineOperand *PotentialMO = findSingleRegUse(Reg: getReplacedOperand(), MRI: getMRI());
396 if (!PotentialMO)
397 return nullptr;
398
399 MachineInstr *Parent = PotentialMO->getParent();
400
401 return canCombineSelections(MI: *Parent, TII) ? Parent : nullptr;
402}
403
404bool SDWASrcOperand::convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) {
405 assert((!Sext || !TII->getSubtarget().zeroesHigh16BitsOfDest(
406 getParentInst()->getOpcode())) &&
407 "Cannot use sign-extension with instruction that zeroes high bits");
408 switch (MI.getOpcode()) {
409 case AMDGPU::V_CVT_F32_FP8_sdwa:
410 case AMDGPU::V_CVT_F32_BF8_sdwa:
411 case AMDGPU::V_CVT_PK_F32_FP8_sdwa:
412 case AMDGPU::V_CVT_PK_F32_BF8_sdwa:
413 // Does not support input modifiers: noabs, noneg, nosext.
414 return false;
415 case AMDGPU::V_CNDMASK_B32_sdwa:
416 // SISrcMods uses the same bitmask for SEXT and NEG modifiers and
417 // hence the compiler can only support one type of modifier for
418 // each SDWA instruction. For V_CNDMASK_B32_sdwa, this is NEG
419 // since its operands get printed using
420 // AMDGPUInstPrinter::printOperandAndFPInputMods which produces
421 // the output intended for NEG if SEXT is set.
422 //
423 // The ISA does actually support both modifiers on most SDWA
424 // instructions.
425 //
426 // FIXME Accept SEXT here after fixing this issue.
427 if (Sext)
428 return false;
429 break;
430 }
431
432 // Find operand in instruction that matches source operand and replace it with
433 // target operand. Set corresponding src_sel
434 bool IsPreserveSrc = false;
435 MachineOperand *Src = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0);
436 MachineOperand *SrcSel = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0_sel);
437 MachineOperand *SrcMods =
438 TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0_modifiers);
439 assert(Src && (Src->isReg() || Src->isImm()));
440 if (!isSameReg(LHS: *Src, RHS: *getReplacedOperand())) {
441 // If this is not src0 then it could be src1
442 Src = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1);
443 SrcSel = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1_sel);
444 SrcMods = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1_modifiers);
445
446 if (!Src ||
447 !isSameReg(LHS: *Src, RHS: *getReplacedOperand())) {
448 // It's possible this Src is a tied operand for
449 // UNUSED_PRESERVE, in which case we can either
450 // abandon the peephole attempt, or if legal we can
451 // copy the target operand into the tied slot
452 // if the preserve operation will effectively cause the same
453 // result by overwriting the rest of the dst.
454 MachineOperand *Dst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst);
455 MachineOperand *DstUnused =
456 TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::dst_unused);
457
458 if (Dst &&
459 DstUnused->getImm() == AMDGPU::SDWA::DstUnused::UNUSED_PRESERVE) {
460 // This will work if the tied src is accessing WORD_0, and the dst is
461 // writing WORD_1. Modifiers don't matter because all the bits that
462 // would be impacted are being overwritten by the dst.
463 // Any other case will not work.
464 SdwaSel DstSel = static_cast<SdwaSel>(
465 TII->getNamedImmOperand(MI, OperandName: AMDGPU::OpName::dst_sel));
466 if (DstSel == AMDGPU::SDWA::SdwaSel::WORD_1 &&
467 getSrcSel() == AMDGPU::SDWA::SdwaSel::WORD_0) {
468 IsPreserveSrc = true;
469 auto DstIdx = AMDGPU::getNamedOperandIdx(Opcode: MI.getOpcode(),
470 Name: AMDGPU::OpName::vdst);
471 auto TiedIdx = MI.findTiedOperandIdx(OpIdx: DstIdx);
472 Src = &MI.getOperand(i: TiedIdx);
473 SrcSel = nullptr;
474 SrcMods = nullptr;
475 } else {
476 // Not legal to convert this src
477 return false;
478 }
479 }
480 }
481 assert(Src && Src->isReg());
482
483 if ((MI.getOpcode() == AMDGPU::V_FMAC_F16_sdwa ||
484 MI.getOpcode() == AMDGPU::V_FMAC_F32_sdwa ||
485 MI.getOpcode() == AMDGPU::V_MAC_F16_sdwa ||
486 MI.getOpcode() == AMDGPU::V_MAC_F32_sdwa) &&
487 !isSameReg(LHS: *Src, RHS: *getReplacedOperand())) {
488 // In case of v_mac_f16/32_sdwa this pass can try to apply src operand to
489 // src2. This is not allowed.
490 return false;
491 }
492
493 assert(isSameReg(*Src, *getReplacedOperand()) &&
494 (IsPreserveSrc || (SrcSel && SrcMods)));
495 }
496 copyRegOperand(To&: *Src, From: *getTargetOperand());
497 if (!IsPreserveSrc) {
498 SdwaSel ExistingSel = static_cast<SdwaSel>(SrcSel->getImm());
499 SrcSel->setImm(*combineSdwaSel(Sel: ExistingSel, OperandSel: getSrcSel()));
500 SrcMods->setImm(getSrcMods(TII, SrcOp: Src));
501 }
502 getTargetOperand()->setIsKill(false);
503 return true;
504}
505
506/// Verify that the SDWA selection operand \p SrcSelOpName of the SDWA
507/// instruction \p MI can be combined with the selection \p OpSel.
508static bool canCombineOpSel(const MachineInstr &MI, const SIInstrInfo *TII,
509 AMDGPU::OpName SrcSelOpName, SdwaSel OpSel) {
510 assert(TII->isSDWA(MI.getOpcode()));
511
512 const MachineOperand *SrcSelOp = TII->getNamedOperand(MI, OperandName: SrcSelOpName);
513 SdwaSel SrcSel = static_cast<SdwaSel>(SrcSelOp->getImm());
514
515 return combineSdwaSel(Sel: SrcSel, OperandSel: OpSel).has_value();
516}
517
518/// Verify that \p Op is the same register as the operand of the SDWA
519/// instruction \p MI named by \p SrcOpName and that the SDWA
520/// selection \p SrcSelOpName can be combined with the \p OpSel.
521static bool canCombineOpSel(const MachineInstr &MI, const SIInstrInfo *TII,
522 AMDGPU::OpName SrcOpName,
523 AMDGPU::OpName SrcSelOpName, MachineOperand *Op,
524 SdwaSel OpSel) {
525 assert(TII->isSDWA(MI.getOpcode()));
526
527 const MachineOperand *Src = TII->getNamedOperand(MI, OperandName: SrcOpName);
528 if (!Src || !isSameReg(LHS: *Src, RHS: *Op))
529 return true;
530
531 return canCombineOpSel(MI, TII, SrcSelOpName, OpSel);
532}
533
534bool SDWASrcOperand::canCombineSelections(const MachineInstr &MI,
535 const SIInstrInfo *TII) {
536 if (!TII->isSDWA(Opcode: MI.getOpcode()))
537 return true;
538
539 using namespace AMDGPU;
540
541 return canCombineOpSel(MI, TII, SrcOpName: OpName::src0, SrcSelOpName: OpName::src0_sel,
542 Op: getReplacedOperand(), OpSel: getSrcSel()) &&
543 canCombineOpSel(MI, TII, SrcOpName: OpName::src1, SrcSelOpName: OpName::src1_sel,
544 Op: getReplacedOperand(), OpSel: getSrcSel());
545}
546
547MachineInstr *SDWADstOperand::potentialToConvert(const SIInstrInfo *TII,
548 const GCNSubtarget &ST,
549 SDWAOperandsMap *PotentialMatches) {
550 // For SDWA dst operand potential instruction is one that defines register
551 // that this operand uses
552 MachineRegisterInfo *MRI = getMRI();
553 MachineInstr *ParentMI = getParentInst();
554
555 MachineOperand *PotentialMO = findSingleRegDef(Reg: getReplacedOperand(), MRI);
556 if (!PotentialMO)
557 return nullptr;
558
559 // Check that ParentMI is the only instruction that uses replaced register
560 for (MachineInstr &UseInst : MRI->use_nodbg_instructions(Reg: PotentialMO->getReg())) {
561 if (&UseInst != ParentMI)
562 return nullptr;
563 }
564
565 MachineInstr *Parent = PotentialMO->getParent();
566 return canCombineSelections(MI: *Parent, TII) ? Parent : nullptr;
567}
568
569bool SDWADstOperand::convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) {
570 // Replace vdst operand in MI with target operand. Set dst_sel and dst_unused
571
572 if ((MI.getOpcode() == AMDGPU::V_FMAC_F16_sdwa ||
573 MI.getOpcode() == AMDGPU::V_FMAC_F32_sdwa ||
574 MI.getOpcode() == AMDGPU::V_MAC_F16_sdwa ||
575 MI.getOpcode() == AMDGPU::V_MAC_F32_sdwa) &&
576 getDstSel() != AMDGPU::SDWA::DWORD) {
577 // v_mac_f16/32_sdwa allow dst_sel to be equal only to DWORD
578 return false;
579 }
580
581 MachineOperand *Operand = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst);
582 assert(Operand &&
583 Operand->isReg() &&
584 isSameReg(*Operand, *getReplacedOperand()));
585 copyRegOperand(To&: *Operand, From: *getTargetOperand());
586 MachineOperand *DstSel= TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::dst_sel);
587 assert(DstSel);
588
589 SdwaSel ExistingSel = static_cast<SdwaSel>(DstSel->getImm());
590 DstSel->setImm(combineSdwaSel(Sel: ExistingSel, OperandSel: getDstSel()).value());
591
592 MachineOperand *DstUnused= TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::dst_unused);
593 assert(DstUnused);
594 DstUnused->setImm(getDstUnused());
595
596 // Remove original instruction because it would conflict with our new
597 // instruction by register definition
598 getParentInst()->eraseFromParent();
599 return true;
600}
601
602bool SDWADstOperand::canCombineSelections(const MachineInstr &MI,
603 const SIInstrInfo *TII) {
604 if (!TII->isSDWA(Opcode: MI.getOpcode()))
605 return true;
606
607 return canCombineOpSel(MI, TII, SrcSelOpName: AMDGPU::OpName::dst_sel, OpSel: getDstSel());
608}
609
610bool SDWADstPreserveOperand::convertToSDWA(MachineInstr &MI,
611 const SIInstrInfo *TII) {
612 // MI should be moved right before v_or_b32.
613 // For this we should clear all kill flags on uses of MI src-operands or else
614 // we can encounter problem with use of killed operand.
615 for (MachineOperand &MO : MI.uses()) {
616 if (!MO.isReg())
617 continue;
618 getMRI()->clearKillFlags(Reg: MO.getReg());
619 }
620
621 // Move MI before v_or_b32
622 MI.getParent()->remove(I: &MI);
623 getParentInst()->getParent()->insert(I: getParentInst(), MI: &MI);
624
625 // Add Implicit use of preserved register
626 MachineInstrBuilder MIB(*MI.getMF(), MI);
627 MIB.addReg(RegNo: getPreservedOperand()->getReg(),
628 Flags: RegState::ImplicitKill,
629 SubReg: getPreservedOperand()->getSubReg());
630
631 // Tie dst to implicit use
632 MI.tieOperands(DefIdx: AMDGPU::getNamedOperandIdx(Opcode: MI.getOpcode(), Name: AMDGPU::OpName::vdst),
633 UseIdx: MI.getNumOperands() - 1);
634
635 // Convert MI as any other SDWADstOperand and remove v_or_b32
636 return SDWADstOperand::convertToSDWA(MI, TII);
637}
638
639bool SDWADstPreserveOperand::canCombineSelections(const MachineInstr &MI,
640 const SIInstrInfo *TII) {
641 return SDWADstOperand::canCombineSelections(MI, TII);
642}
643
644std::optional<int64_t>
645SIPeepholeSDWA::foldToImm(const MachineOperand &Op) const {
646 if (Op.isImm()) {
647 return Op.getImm();
648 }
649
650 // If this is not immediate then it can be copy of immediate value, e.g.:
651 // %1 = S_MOV_B32 255;
652 if (Op.isReg()) {
653 for (const MachineOperand &Def : MRI->def_operands(Reg: Op.getReg())) {
654 if (!isSameReg(LHS: Op, RHS: Def))
655 continue;
656
657 const MachineInstr *DefInst = Def.getParent();
658 if (!TII->isFoldableCopy(MI: *DefInst))
659 return std::nullopt;
660
661 const MachineOperand &Copied = DefInst->getOperand(i: 1);
662 if (!Copied.isImm())
663 return std::nullopt;
664
665 return Copied.getImm();
666 }
667 }
668
669 return std::nullopt;
670}
671
672std::optional<std::pair<MachineOperand *, SdwaSel>>
673SIPeepholeSDWA::matchAndMask(MachineInstr &MI) const {
674 if (MI.getOpcode() != AMDGPU::V_AND_B32_e32 &&
675 MI.getOpcode() != AMDGPU::V_AND_B32_e64)
676 return std::nullopt;
677
678 MachineOperand *Src0 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0);
679 MachineOperand *Src1 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1);
680 MachineOperand *ValSrc = Src1;
681 std::optional<int64_t> Imm = foldToImm(Op: *Src0);
682 if (!Imm) {
683 Imm = foldToImm(Op: *Src1);
684 ValSrc = Src0;
685 }
686 if (!Imm || (*Imm != 0x0000ffff && *Imm != 0x000000ff))
687 return std::nullopt;
688
689 return std::make_pair(x&: ValSrc, y: *Imm == 0x0000ffff ? WORD_0 : BYTE_0);
690}
691
692bool SIPeepholeSDWA::isSDWAWithDstSel(const MachineInstr &Inst) const {
693 return TII->isSDWA(MI: Inst) &&
694 AMDGPU::hasNamedOperand(Opcode: Inst.getOpcode(), NamedIdx: AMDGPU::OpName::dst_sel);
695}
696
697std::unique_ptr<SDWAOperand>
698SIPeepholeSDWA::matchSDWAOperand(MachineInstr &MI) {
699 unsigned Opcode = MI.getOpcode();
700 switch (Opcode) {
701 case AMDGPU::V_LSHRREV_B32_e32:
702 case AMDGPU::V_ASHRREV_I32_e32:
703 case AMDGPU::V_LSHLREV_B32_e32:
704 case AMDGPU::V_LSHRREV_B32_e64:
705 case AMDGPU::V_ASHRREV_I32_e64:
706 case AMDGPU::V_LSHLREV_B32_e64: {
707 // from: v_lshrrev_b32_e32 v1, 16/24, v0
708 // to SDWA src:v0 src_sel:WORD_1/BYTE_3
709
710 // from: v_ashrrev_i32_e32 v1, 16/24, v0
711 // to SDWA src:v0 src_sel:WORD_1/BYTE_3 sext:1
712
713 // from: v_lshlrev_b32_e32 v1, 16/24, v0
714 // to SDWA dst:v1 dst_sel:WORD_1/BYTE_3 dst_unused:UNUSED_PAD
715 MachineOperand *Src0 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0);
716 auto Imm = foldToImm(Op: *Src0);
717 if (!Imm)
718 break;
719
720 if (*Imm != 16 && *Imm != 24)
721 break;
722
723 MachineOperand *Src1 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1);
724 MachineOperand *Dst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst);
725 if (!Src1->isReg() || Src1->getReg().isPhysical() ||
726 Dst->getReg().isPhysical())
727 break;
728
729 if (Opcode == AMDGPU::V_LSHLREV_B32_e32 ||
730 Opcode == AMDGPU::V_LSHLREV_B32_e64) {
731 return std::make_unique<SDWADstOperand>(
732 args&: Dst, args&: Src1, args: *Imm == 16 ? WORD_1 : BYTE_3, args: UNUSED_PAD);
733 }
734 return std::make_unique<SDWASrcOperand>(
735 args&: Src1, args&: Dst, args: *Imm == 16 ? WORD_1 : BYTE_3, args: false, args: false,
736 args: Opcode != AMDGPU::V_LSHRREV_B32_e32 &&
737 Opcode != AMDGPU::V_LSHRREV_B32_e64);
738 break;
739 }
740
741 case AMDGPU::V_LSHRREV_B16_e32:
742 case AMDGPU::V_LSHLREV_B16_e32:
743 case AMDGPU::V_LSHRREV_B16_e64:
744 case AMDGPU::V_LSHRREV_B16_opsel_e64:
745 case AMDGPU::V_LSHLREV_B16_opsel_e64:
746 case AMDGPU::V_LSHLREV_B16_e64: {
747 // V_ASHRREV_I16_e32 and V_ASHRREV_I16_e64 are
748 // not included here because they zero-fill the high 16-bits.
749
750 // from: v_lshrrev_b16_e32 v1, 8, v0
751 // to SDWA src:v0 src_sel:BYTE_1
752
753 // from: v_lshlrev_b16_e32 v1, 8, v0
754 // to SDWA dst:v1 dst_sel:BYTE_1 dst_unused:UNUSED_PAD
755 MachineOperand *Src0 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0);
756 auto Imm = foldToImm(Op: *Src0);
757 if (!Imm || *Imm != 8)
758 break;
759
760 MachineOperand *Src1 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1);
761 MachineOperand *Dst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst);
762
763 if (!Src1->isReg() || Src1->getReg().isPhysical() ||
764 Dst->getReg().isPhysical())
765 break;
766
767 if (Opcode == AMDGPU::V_LSHLREV_B16_e32 ||
768 Opcode == AMDGPU::V_LSHLREV_B16_opsel_e64 ||
769 Opcode == AMDGPU::V_LSHLREV_B16_e64)
770 return std::make_unique<SDWADstOperand>(args&: Dst, args&: Src1, args: BYTE_1, args: UNUSED_PAD);
771 return std::make_unique<SDWASrcOperand>(args&: Src1, args&: Dst, args: BYTE_1, args: false, args: false,
772 args: false);
773 break;
774 }
775
776 case AMDGPU::V_BFE_I32_e64:
777 case AMDGPU::V_BFE_U32_e64: {
778 // e.g.:
779 // from: v_bfe_u32 v1, v0, 8, 8
780 // to SDWA src:v0 src_sel:BYTE_1
781
782 // offset | width | src_sel
783 // ------------------------
784 // 0 | 8 | BYTE_0
785 // 0 | 16 | WORD_0
786 // 0 | 32 | DWORD ?
787 // 8 | 8 | BYTE_1
788 // 16 | 8 | BYTE_2
789 // 16 | 16 | WORD_1
790 // 24 | 8 | BYTE_3
791
792 MachineOperand *Src1 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1);
793 auto Offset = foldToImm(Op: *Src1);
794 if (!Offset)
795 break;
796
797 MachineOperand *Src2 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src2);
798 auto Width = foldToImm(Op: *Src2);
799 if (!Width)
800 break;
801
802 SdwaSel SrcSel = DWORD;
803
804 if (*Offset == 0 && *Width == 8)
805 SrcSel = BYTE_0;
806 else if (*Offset == 0 && *Width == 16)
807 SrcSel = WORD_0;
808 else if (*Offset == 0 && *Width == 32)
809 SrcSel = DWORD;
810 else if (*Offset == 8 && *Width == 8)
811 SrcSel = BYTE_1;
812 else if (*Offset == 16 && *Width == 8)
813 SrcSel = BYTE_2;
814 else if (*Offset == 16 && *Width == 16)
815 SrcSel = WORD_1;
816 else if (*Offset == 24 && *Width == 8)
817 SrcSel = BYTE_3;
818 else
819 break;
820
821 MachineOperand *Src0 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0);
822 MachineOperand *Dst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst);
823
824 if (!Src0->isReg() || Src0->getReg().isPhysical() ||
825 Dst->getReg().isPhysical())
826 break;
827
828 return std::make_unique<SDWASrcOperand>(
829 args&: Src0, args&: Dst, args&: SrcSel, args: false, args: false, args: Opcode != AMDGPU::V_BFE_U32_e64);
830 }
831
832 case AMDGPU::V_AND_B32_e32:
833 case AMDGPU::V_AND_B32_e64: {
834 // e.g.:
835 // from: v_and_b32_e32 v1, 0x0000ffff/0x000000ff, v0
836 // to SDWA src:v0 src_sel:WORD_0/BYTE_0
837 auto Mask = matchAndMask(MI);
838 if (!Mask)
839 break;
840 MachineOperand *ValSrc = Mask->first;
841
842 MachineOperand *Dst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst);
843
844 if (!ValSrc->isReg() || ValSrc->getReg().isPhysical() ||
845 Dst->getReg().isPhysical())
846 break;
847
848 return std::make_unique<SDWASrcOperand>(args&: ValSrc, args&: Dst, args&: Mask->second);
849 }
850
851 case AMDGPU::V_OR_B32_e32:
852 case AMDGPU::V_OR_B32_e64: {
853 // Patterns for dst_unused:UNUSED_PRESERVE.
854 // e.g., from:
855 // v_add_f16_sdwa v0, v1, v2 dst_sel:WORD_1 dst_unused:UNUSED_PAD
856 // src1_sel:WORD_1 src2_sel:WORD1
857 // v_add_f16_e32 v3, v1, v2
858 // v_or_b32_e32 v4, v0, v3
859 // to SDWA preserve dst:v4 dst_sel:WORD_1 dst_unused:UNUSED_PRESERVE preserve:v3
860
861 // Check if one of operands of v_or_b32 is SDWA instruction
862 using CheckRetType =
863 std::optional<std::pair<MachineOperand *, MachineOperand *>>;
864 auto CheckOROperandsForSDWA =
865 [&](const MachineOperand *Op1, const MachineOperand *Op2) -> CheckRetType {
866 if (!Op1 || !Op1->isReg() || !Op2 || !Op2->isReg())
867 return CheckRetType(std::nullopt);
868
869 MachineOperand *Op1Def = findSingleRegDef(Reg: Op1, MRI);
870 if (!Op1Def)
871 return CheckRetType(std::nullopt);
872
873 MachineInstr *Op1Inst = Op1Def->getParent();
874 if (!isSDWAWithDstSel(Inst: *Op1Inst))
875 return CheckRetType(std::nullopt);
876
877 MachineOperand *Op2Def = findSingleRegDef(Reg: Op2, MRI);
878 if (!Op2Def)
879 return CheckRetType(std::nullopt);
880
881 return CheckRetType(std::pair(Op1Def, Op2Def));
882 };
883
884 MachineOperand *OrSDWA = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0);
885 MachineOperand *OrOther = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1);
886 assert(OrSDWA && OrOther);
887 auto Res = CheckOROperandsForSDWA(OrSDWA, OrOther);
888 if (!Res) {
889 OrSDWA = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1);
890 OrOther = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0);
891 assert(OrSDWA && OrOther);
892 Res = CheckOROperandsForSDWA(OrSDWA, OrOther);
893 if (!Res)
894 break;
895 }
896
897 MachineOperand *OrSDWADef = Res->first;
898 MachineOperand *OrOtherDef = Res->second;
899 assert(OrSDWADef && OrOtherDef);
900
901 MachineInstr *SDWAInst = OrSDWADef->getParent();
902 MachineInstr *OtherInst = OrOtherDef->getParent();
903
904 // Check that OtherInstr is actually bitwise compatible with SDWAInst = their
905 // destination patterns don't overlap. Compatible instruction can be either
906 // regular instruction with compatible bitness or SDWA instruction with
907 // correct dst_sel
908 // SDWAInst | OtherInst bitness / OtherInst dst_sel
909 // -----------------------------------------------------
910 // DWORD | no / no
911 // WORD_0 | no / BYTE_2/3, WORD_1
912 // WORD_1 | 8/16-bit instructions / BYTE_0/1, WORD_0
913 // BYTE_0 | no / BYTE_1/2/3, WORD_1
914 // BYTE_1 | 8-bit / BYTE_0/2/3, WORD_1
915 // BYTE_2 | 8/16-bit / BYTE_0/1/3. WORD_0
916 // BYTE_3 | 8/16/24-bit / BYTE_0/1/2, WORD_0
917 // E.g. if SDWAInst is v_add_f16_sdwa dst_sel:WORD_1 then v_add_f16 is OK
918 // but v_add_f32 is not.
919
920 // TODO: add support for non-SDWA instructions as OtherInst.
921 // For now this only works with SDWA instructions. For regular instructions
922 // there is no way to determine if the instruction writes only 8/16/24-bit
923 // out of full register size and all registers are at min 32-bit wide.
924 if (!isSDWAWithDstSel(Inst: *OtherInst))
925 break;
926
927 SdwaSel DstSel = static_cast<SdwaSel>(
928 TII->getNamedImmOperand(MI: *SDWAInst, OperandName: AMDGPU::OpName::dst_sel));
929 SdwaSel OtherDstSel = static_cast<SdwaSel>(
930 TII->getNamedImmOperand(MI: *OtherInst, OperandName: AMDGPU::OpName::dst_sel));
931
932 bool DstSelAgree = false;
933 switch (DstSel) {
934 case WORD_0: DstSelAgree = ((OtherDstSel == BYTE_2) ||
935 (OtherDstSel == BYTE_3) ||
936 (OtherDstSel == WORD_1));
937 break;
938 case WORD_1: DstSelAgree = ((OtherDstSel == BYTE_0) ||
939 (OtherDstSel == BYTE_1) ||
940 (OtherDstSel == WORD_0));
941 break;
942 case BYTE_0: DstSelAgree = ((OtherDstSel == BYTE_1) ||
943 (OtherDstSel == BYTE_2) ||
944 (OtherDstSel == BYTE_3) ||
945 (OtherDstSel == WORD_1));
946 break;
947 case BYTE_1: DstSelAgree = ((OtherDstSel == BYTE_0) ||
948 (OtherDstSel == BYTE_2) ||
949 (OtherDstSel == BYTE_3) ||
950 (OtherDstSel == WORD_1));
951 break;
952 case BYTE_2: DstSelAgree = ((OtherDstSel == BYTE_0) ||
953 (OtherDstSel == BYTE_1) ||
954 (OtherDstSel == BYTE_3) ||
955 (OtherDstSel == WORD_0));
956 break;
957 case BYTE_3: DstSelAgree = ((OtherDstSel == BYTE_0) ||
958 (OtherDstSel == BYTE_1) ||
959 (OtherDstSel == BYTE_2) ||
960 (OtherDstSel == WORD_0));
961 break;
962 default: DstSelAgree = false;
963 }
964
965 if (!DstSelAgree)
966 break;
967
968 // Also OtherInst dst_unused should be UNUSED_PAD
969 DstUnused OtherDstUnused = static_cast<DstUnused>(
970 TII->getNamedImmOperand(MI: *OtherInst, OperandName: AMDGPU::OpName::dst_unused));
971 if (OtherDstUnused != DstUnused::UNUSED_PAD)
972 break;
973
974 // Create DstPreserveOperand
975 MachineOperand *OrDst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst);
976 assert(OrDst && OrDst->isReg());
977
978 return std::make_unique<SDWADstPreserveOperand>(
979 args&: OrDst, args&: OrSDWADef, args&: OrOtherDef, args&: DstSel);
980
981 }
982 }
983
984 return std::unique_ptr<SDWAOperand>(nullptr);
985}
986
987#if !defined(NDEBUG)
988static raw_ostream& operator<<(raw_ostream &OS, const SDWAOperand &Operand) {
989 Operand.print(OS);
990 return OS;
991}
992#endif
993
994void SIPeepholeSDWA::matchSDWAOperands(MachineBasicBlock &MBB) {
995 for (MachineInstr &MI : MBB) {
996 if (auto Operand = matchSDWAOperand(MI)) {
997 LLVM_DEBUG(dbgs() << "Match: " << MI << "To: " << *Operand << '\n');
998 SDWAOperands[&MI] = std::move(Operand);
999 ++NumSDWAPatternsFound;
1000 }
1001 }
1002}
1003
1004// Convert the V_ADD_CO_U32_e64 into V_ADD_CO_U32_e32. This allows
1005// isConvertibleToSDWA to perform its transformation on V_ADD_CO_U32_e32 into
1006// V_ADD_CO_U32_sdwa.
1007//
1008// We are transforming from a VOP3 into a VOP2 form of the instruction.
1009// %19:vgpr_32 = V_AND_B32_e32 255,
1010// killed %16:vgpr_32, implicit $exec
1011// %47:vgpr_32, %49:sreg_64_xexec = V_ADD_CO_U32_e64
1012// %26.sub0:vreg_64, %19:vgpr_32, implicit $exec
1013// %48:vgpr_32, dead %50:sreg_64_xexec = V_ADDC_U32_e64
1014// %26.sub1:vreg_64, %54:vgpr_32, killed %49:sreg_64_xexec, implicit $exec
1015//
1016// becomes
1017// %47:vgpr_32 = V_ADD_CO_U32_sdwa
1018// 0, %26.sub0:vreg_64, 0, killed %16:vgpr_32, 0, 6, 0, 6, 0,
1019// implicit-def $vcc, implicit $exec
1020// %48:vgpr_32, dead %50:sreg_64_xexec = V_ADDC_U32_e64
1021// %26.sub1:vreg_64, %54:vgpr_32, killed $vcc, implicit $exec
1022void SIPeepholeSDWA::pseudoOpConvertToVOP2(MachineInstr &MI,
1023 const GCNSubtarget &ST) const {
1024 int Opc = MI.getOpcode();
1025 assert((Opc == AMDGPU::V_ADD_CO_U32_e64 || Opc == AMDGPU::V_SUB_CO_U32_e64) &&
1026 "Currently only handles V_ADD_CO_U32_e64 or V_SUB_CO_U32_e64");
1027
1028 // Can the candidate MI be shrunk?
1029 if (!TII->canShrink(MI, MRI: *MRI))
1030 return;
1031 Opc = AMDGPU::getVOPe32(Opcode: Opc);
1032 // Find the related ADD instruction.
1033 const MachineOperand *Sdst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::sdst);
1034 if (!Sdst)
1035 return;
1036 MachineOperand *NextOp = findSingleRegUse(Reg: Sdst, MRI);
1037 if (!NextOp)
1038 return;
1039 MachineInstr &MISucc = *NextOp->getParent();
1040
1041 // Make sure the carry in/out are subsequently unused.
1042 MachineOperand *CarryIn = TII->getNamedOperand(MI&: MISucc, OperandName: AMDGPU::OpName::src2);
1043 if (!CarryIn)
1044 return;
1045 MachineOperand *CarryOut = TII->getNamedOperand(MI&: MISucc, OperandName: AMDGPU::OpName::sdst);
1046 if (!CarryOut)
1047 return;
1048 if (!MRI->hasOneNonDBGUse(RegNo: CarryIn->getReg()) ||
1049 !MRI->use_nodbg_empty(RegNo: CarryOut->getReg()))
1050 return;
1051 // Make sure VCC or its subregs are dead before MI.
1052 MachineBasicBlock &MBB = *MI.getParent();
1053 if (MISucc.getParent() != &MBB)
1054 return; // Loop depends on MI and MISucc in same MBB.
1055 MachineBasicBlock::LivenessQueryResult Liveness =
1056 MBB.computeRegisterLiveness(TRI, Reg: AMDGPU::VCC, Before: MI, Neighborhood: 25);
1057 if (Liveness != MachineBasicBlock::LQR_Dead)
1058 return;
1059 // Check if VCC is referenced in range of (MI,MISucc].
1060 for (auto I = std::next(x: MI.getIterator()), E = MISucc.getIterator();
1061 I != E; ++I) {
1062 if (I->modifiesRegister(Reg: AMDGPU::VCC, TRI))
1063 return;
1064 }
1065
1066 // Replace MI with V_{SUB|ADD}_I32_e32
1067 BuildMI(BB&: MBB, I&: MI, MIMD: MI.getDebugLoc(), MCID: TII->get(Opcode: Opc))
1068 .add(MO: *TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst))
1069 .add(MO: *TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0))
1070 .add(MO: *TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1))
1071 .setMIFlags(MI.getFlags());
1072
1073 MI.eraseFromParent();
1074
1075 // Since the carry output of MI is now VCC, update its use in MISucc.
1076
1077 MISucc.substituteRegister(FromReg: CarryIn->getReg(), ToReg: TRI->getVCC(), SubIdx: 0, RegInfo: *TRI);
1078}
1079
1080/// Try to convert an \p MI in VOP3 which takes an src2 carry-in
1081/// operand into the corresponding VOP2 form which expects the
1082/// argument in VCC. To this end, add an copy from the carry-in to
1083/// VCC. The conversion will only be applied if \p MI can be shrunk
1084/// to VOP2 and if VCC can be proven to be dead before \p MI.
1085void SIPeepholeSDWA::convertVcndmaskToVOP2(MachineInstr &MI,
1086 const GCNSubtarget &ST) const {
1087 assert(MI.getOpcode() == AMDGPU::V_CNDMASK_B32_e64);
1088
1089 LLVM_DEBUG(dbgs() << "Attempting VOP2 conversion: " << MI);
1090 if (!TII->canShrink(MI, MRI: *MRI)) {
1091 LLVM_DEBUG(dbgs() << "Cannot shrink instruction\n");
1092 return;
1093 }
1094
1095 const MachineOperand &CarryIn =
1096 *TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src2);
1097 Register CarryReg = CarryIn.getReg();
1098 MachineInstr *CarryDef = MRI->getVRegDef(Reg: CarryReg);
1099 if (!CarryDef) {
1100 LLVM_DEBUG(dbgs() << "Missing carry-in operand definition\n");
1101 return;
1102 }
1103
1104 // Make sure VCC or its subregs are dead before MI.
1105 MCRegister Vcc = TRI->getVCC();
1106 MachineBasicBlock &MBB = *MI.getParent();
1107 MachineBasicBlock::LivenessQueryResult Liveness =
1108 MBB.computeRegisterLiveness(TRI, Reg: Vcc, Before: MI);
1109 if (Liveness != MachineBasicBlock::LQR_Dead) {
1110 LLVM_DEBUG(dbgs() << "VCC not known to be dead before instruction\n");
1111 return;
1112 }
1113
1114 BuildMI(BB&: MBB, I&: MI, MIMD: MI.getDebugLoc(), MCID: TII->get(Opcode: AMDGPU::COPY), DestReg: Vcc).add(MO: CarryIn);
1115
1116 auto Converted = BuildMI(BB&: MBB, I&: MI, MIMD: MI.getDebugLoc(),
1117 MCID: TII->get(Opcode: AMDGPU::getVOPe32(Opcode: MI.getOpcode())))
1118 .add(MO: *TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst))
1119 .add(MO: *TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0))
1120 .add(MO: *TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1))
1121 .setMIFlags(MI.getFlags());
1122 TII->fixImplicitOperands(MI&: *Converted);
1123 LLVM_DEBUG(dbgs() << "Converted to VOP2: " << *Converted);
1124 (void)Converted;
1125 MI.eraseFromParent();
1126}
1127
1128namespace {
1129bool isConvertibleToSDWA(MachineInstr &MI,
1130 const GCNSubtarget &ST,
1131 const SIInstrInfo* TII) {
1132 // Check if this is already an SDWA instruction
1133 unsigned Opc = MI.getOpcode();
1134 if (TII->isSDWA(Opcode: Opc))
1135 return true;
1136
1137 // Can only be handled after ealier conversion to
1138 // AMDGPU::V_CNDMASK_B32_e32 which is not always possible.
1139 if (Opc == AMDGPU::V_CNDMASK_B32_e64)
1140 return false;
1141
1142 // Check if this instruction has opcode that supports SDWA
1143 if (AMDGPU::getSDWAOp(Opcode: Opc) == -1)
1144 Opc = AMDGPU::getVOPe32(Opcode: Opc);
1145
1146 if (AMDGPU::getSDWAOp(Opcode: Opc) == -1)
1147 return false;
1148
1149 if (!ST.hasSDWAOmod() && TII->hasModifiersSet(MI, OpName: AMDGPU::OpName::omod))
1150 return false;
1151
1152 if (TII->isVOPC(Opcode: Opc)) {
1153 if (!ST.hasSDWASdst()) {
1154 const MachineOperand *SDst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::sdst);
1155 if (SDst && (SDst->getReg() != AMDGPU::VCC &&
1156 SDst->getReg() != AMDGPU::VCC_LO))
1157 return false;
1158 }
1159
1160 if (!ST.hasSDWAOutModsVOPC() &&
1161 (TII->hasModifiersSet(MI, OpName: AMDGPU::OpName::clamp) ||
1162 TII->hasModifiersSet(MI, OpName: AMDGPU::OpName::omod)))
1163 return false;
1164
1165 } else if (TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::sdst) ||
1166 !TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst)) {
1167 return false;
1168 }
1169
1170 if (!ST.hasSDWAMac() && (Opc == AMDGPU::V_FMAC_F16_e32 ||
1171 Opc == AMDGPU::V_FMAC_F32_e32 ||
1172 Opc == AMDGPU::V_MAC_F16_e32 ||
1173 Opc == AMDGPU::V_MAC_F32_e32))
1174 return false;
1175
1176 // Check if target supports this SDWA opcode
1177 if (TII->pseudoToMCOpcode(Opcode: Opc) == -1 ||
1178 TII->pseudoToMCOpcode(Opcode: AMDGPU::getSDWAOp(Opcode: Opc)) == -1)
1179 return false;
1180
1181 if (MachineOperand *Src0 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0)) {
1182 if (!Src0->isReg() && !Src0->isImm())
1183 return false;
1184 }
1185
1186 if (MachineOperand *Src1 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1)) {
1187 if (!Src1->isReg() && !Src1->isImm())
1188 return false;
1189 }
1190
1191 return true;
1192}
1193} // namespace
1194
1195MachineInstr *SIPeepholeSDWA::createSDWAVersion(MachineInstr &MI) {
1196 unsigned Opcode = MI.getOpcode();
1197 assert(!TII->isSDWA(Opcode));
1198
1199 int SDWAOpcode = AMDGPU::getSDWAOp(Opcode);
1200 if (SDWAOpcode == -1)
1201 SDWAOpcode = AMDGPU::getSDWAOp(Opcode: AMDGPU::getVOPe32(Opcode));
1202 assert(SDWAOpcode != -1);
1203
1204 const MCInstrDesc &SDWADesc = TII->get(Opcode: SDWAOpcode);
1205
1206 // Create SDWA version of instruction MI and initialize its operands
1207 MachineInstrBuilder SDWAInst =
1208 BuildMI(BB&: *MI.getParent(), I&: MI, MIMD: MI.getDebugLoc(), MCID: SDWADesc)
1209 .setMIFlags(MI.getFlags());
1210
1211 // Copy dst, if it is present in original then should also be present in SDWA
1212 MachineOperand *Dst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst);
1213 if (Dst) {
1214 assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::vdst));
1215 SDWAInst.add(MO: *Dst);
1216 } else if ((Dst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::sdst))) {
1217 assert(Dst && AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::sdst));
1218 SDWAInst.add(MO: *Dst);
1219 } else {
1220 assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::sdst));
1221 SDWAInst.addReg(RegNo: TRI->getVCC(), Flags: RegState::Define);
1222 }
1223
1224 // Copy src0, initialize src0_modifiers. All sdwa instructions has src0 and
1225 // src0_modifiers (except for v_nop_sdwa, but it can't get here)
1226 MachineOperand *Src0 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0);
1227 assert(Src0 && AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src0) &&
1228 AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src0_modifiers));
1229 if (auto *Mod = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0_modifiers))
1230 SDWAInst.addImm(Val: Mod->getImm());
1231 else
1232 SDWAInst.addImm(Val: 0);
1233 SDWAInst.add(MO: *Src0);
1234
1235 // Copy src1 if present, initialize src1_modifiers.
1236 MachineOperand *Src1 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1);
1237 if (Src1) {
1238 assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src1) &&
1239 AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src1_modifiers));
1240 if (auto *Mod = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1_modifiers))
1241 SDWAInst.addImm(Val: Mod->getImm());
1242 else
1243 SDWAInst.addImm(Val: 0);
1244 SDWAInst.add(MO: *Src1);
1245 }
1246
1247 if (SDWAOpcode == AMDGPU::V_FMAC_F16_sdwa ||
1248 SDWAOpcode == AMDGPU::V_FMAC_F32_sdwa ||
1249 SDWAOpcode == AMDGPU::V_MAC_F16_sdwa ||
1250 SDWAOpcode == AMDGPU::V_MAC_F32_sdwa) {
1251 // v_mac_f16/32 has additional src2 operand tied to vdst
1252 MachineOperand *Src2 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src2);
1253 assert(Src2);
1254 SDWAInst.add(MO: *Src2);
1255 }
1256
1257 // Copy clamp if present, initialize otherwise
1258 assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::clamp));
1259 MachineOperand *Clamp = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::clamp);
1260 if (Clamp) {
1261 SDWAInst.add(MO: *Clamp);
1262 } else {
1263 SDWAInst.addImm(Val: 0);
1264 }
1265
1266 // Copy omod if present, initialize otherwise if needed
1267 if (AMDGPU::hasNamedOperand(Opcode: SDWAOpcode, NamedIdx: AMDGPU::OpName::omod)) {
1268 MachineOperand *OMod = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::omod);
1269 if (OMod) {
1270 SDWAInst.add(MO: *OMod);
1271 } else {
1272 SDWAInst.addImm(Val: 0);
1273 }
1274 }
1275
1276 // Initialize SDWA specific operands
1277 if (AMDGPU::hasNamedOperand(Opcode: SDWAOpcode, NamedIdx: AMDGPU::OpName::dst_sel))
1278 SDWAInst.addImm(Val: AMDGPU::SDWA::SdwaSel::DWORD);
1279
1280 if (AMDGPU::hasNamedOperand(Opcode: SDWAOpcode, NamedIdx: AMDGPU::OpName::dst_unused))
1281 SDWAInst.addImm(Val: AMDGPU::SDWA::DstUnused::UNUSED_PAD);
1282
1283 assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src0_sel));
1284 SDWAInst.addImm(Val: AMDGPU::SDWA::SdwaSel::DWORD);
1285
1286 if (Src1) {
1287 assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src1_sel));
1288 SDWAInst.addImm(Val: AMDGPU::SDWA::SdwaSel::DWORD);
1289 }
1290
1291 // Check for a preserved register that needs to be copied.
1292 MachineInstr *Ret = SDWAInst.getInstr();
1293 TII->fixImplicitOperands(MI&: *Ret);
1294 return Ret;
1295}
1296
1297bool SIPeepholeSDWA::convertToSDWA(MachineInstr &MI,
1298 const SDWAOperandsVector &SDWAOperands) {
1299 LLVM_DEBUG(dbgs() << "Convert instruction:" << MI);
1300
1301 MachineInstr *SDWAInst;
1302 if (TII->isSDWA(Opcode: MI.getOpcode())) {
1303 // Clone the instruction to allow revoking changes
1304 // made to MI during the processing of the operands
1305 // if the conversion fails.
1306 SDWAInst = MI.getMF()->CloneMachineInstr(Orig: &MI);
1307 MI.getParent()->insert(I: MI.getIterator(), M: SDWAInst);
1308 } else {
1309 SDWAInst = createSDWAVersion(MI);
1310 }
1311
1312 // Apply all sdwa operand patterns.
1313 bool Converted = false;
1314 for (auto &Operand : SDWAOperands) {
1315 LLVM_DEBUG(dbgs() << *SDWAInst << "\nOperand: " << *Operand);
1316 // There should be no intersection between SDWA operands and potential MIs
1317 // e.g.:
1318 // v_and_b32 v0, 0xff, v1 -> src:v1 sel:BYTE_0
1319 // v_and_b32 v2, 0xff, v0 -> src:v0 sel:BYTE_0
1320 // v_add_u32 v3, v4, v2
1321 //
1322 // In that example it is possible that we would fold 2nd instruction into
1323 // 3rd (v_add_u32_sdwa) and then try to fold 1st instruction into 2nd (that
1324 // was already destroyed). So if SDWAOperand is also a potential MI then do
1325 // not apply it.
1326 if (PotentialMatches.count(Key: Operand->getParentInst()) == 0)
1327 Converted |= Operand->convertToSDWA(MI&: *SDWAInst, TII);
1328 }
1329
1330 if (!Converted) {
1331 SDWAInst->eraseFromParent();
1332 return false;
1333 }
1334
1335 ConvertedInstructions.push_back(Elt: SDWAInst);
1336 for (MachineOperand &MO : SDWAInst->uses()) {
1337 if (!MO.isReg())
1338 continue;
1339
1340 MRI->clearKillFlags(Reg: MO.getReg());
1341 }
1342 LLVM_DEBUG(dbgs() << "\nInto:" << *SDWAInst << '\n');
1343 ++NumSDWAInstructionsPeepholed;
1344
1345 MI.eraseFromParent();
1346 return true;
1347}
1348
1349// If an instruction was converted to SDWA it should not have immediates or SGPR
1350// operands (allowed one SGPR on GFX9). Copy its scalar operands into VGPRs.
1351void SIPeepholeSDWA::legalizeScalarOperands(MachineInstr &MI,
1352 const GCNSubtarget &ST) const {
1353 const MCInstrDesc &Desc = TII->get(Opcode: MI.getOpcode());
1354 unsigned ConstantBusCount = 0;
1355 for (MachineOperand &Op : MI.explicit_uses()) {
1356 if (Op.isReg()) {
1357 if (TRI->isVGPR(MRI: *MRI, Reg: Op.getReg()))
1358 continue;
1359
1360 if (ST.hasSDWAScalar() && ConstantBusCount == 0) {
1361 ++ConstantBusCount;
1362 continue;
1363 }
1364 } else if (!Op.isImm())
1365 continue;
1366
1367 unsigned I = Op.getOperandNo();
1368 const TargetRegisterClass *OpRC = TII->getRegClass(MCID: Desc, OpNum: I);
1369 if (!OpRC || !TRI->isVSSuperClass(RC: OpRC))
1370 continue;
1371
1372 Register VGPR = MRI->createVirtualRegister(RegClass: &AMDGPU::VGPR_32RegClass);
1373 auto Copy = BuildMI(BB&: *MI.getParent(), I: MI.getIterator(), MIMD: MI.getDebugLoc(),
1374 MCID: TII->get(Opcode: AMDGPU::V_MOV_B32_e32), DestReg: VGPR);
1375 if (Op.isImm())
1376 Copy.addImm(Val: Op.getImm());
1377 else if (Op.isReg())
1378 Copy.addReg(RegNo: Op.getReg(), Flags: getKillRegState(B: Op.isKill()), SubReg: Op.getSubReg());
1379 Op.ChangeToRegister(Reg: VGPR, isDef: false);
1380 }
1381}
1382
1383// Re-fold the masked high-half pack (hi << 16) | (z & 0xffff) into a single
1384// v_or_b32_sdwa src1_sel:WORD_0, which ISel's fused v_lshl_or_b32 blocks.
1385bool SIPeepholeSDWA::splitLshlOrForSDWA(MachineBasicBlock &MBB) {
1386 struct Candidate {
1387 MachineInstr *LshlOr;
1388 MachineInstr *AndMI;
1389 MachineOperand *Hi;
1390 MachineOperand *ValSrc;
1391 };
1392 SmallVector<Candidate, 4> Candidates;
1393
1394 for (MachineInstr &MI : MBB) {
1395 if (MI.getOpcode() != AMDGPU::V_LSHL_OR_B32_e64)
1396 continue;
1397
1398 MachineOperand *Shift = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1);
1399 std::optional<int64_t> ShiftImm = foldToImm(Op: *Shift);
1400 if (!ShiftImm || *ShiftImm != 16)
1401 continue;
1402
1403 MachineOperand *Hi = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0);
1404 MachineOperand *Src2 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src2);
1405 // Src2 must be a virtual reg so getVRegDef below is valid.
1406 if (!Hi->isReg() || !Src2->isReg() || !Src2->getReg().isVirtual())
1407 continue;
1408
1409 // The 0xffff mask must come from a single-use v_and so it can be dropped.
1410 if (!MRI->hasOneNonDBGUse(RegNo: Src2->getReg()))
1411 continue;
1412 MachineInstr *AndMI = MRI->getVRegDef(Reg: Src2->getReg());
1413 if (!AndMI)
1414 continue;
1415 std::optional<std::pair<MachineOperand *, SdwaSel>> Mask =
1416 matchAndMask(MI&: *AndMI);
1417 if (!Mask || Mask->second != WORD_0)
1418 continue;
1419 MachineOperand *ValSrc = Mask->first;
1420 if (!ValSrc->isReg() || !TRI->isVGPR(MRI: *MRI, Reg: ValSrc->getReg()))
1421 continue;
1422
1423 Candidates.push_back(Elt: {.LshlOr: &MI, .AndMI: AndMI, .Hi: Hi, .ValSrc: ValSrc});
1424 }
1425
1426 for (const Candidate &C : Candidates) {
1427 MachineOperand *Dst = TII->getNamedOperand(MI&: *C.LshlOr, OperandName: AMDGPU::OpName::vdst);
1428
1429 Register ShiftReg = MRI->createVirtualRegister(RegClass: &AMDGPU::VGPR_32RegClass);
1430 BuildMI(BB&: *C.LshlOr->getParent(), I&: *C.LshlOr, MIMD: C.LshlOr->getDebugLoc(),
1431 MCID: TII->get(Opcode: AMDGPU::V_LSHLREV_B32_e64), DestReg: ShiftReg)
1432 .addImm(Val: 16)
1433 .add(MO: *C.Hi);
1434
1435 // vdst, src0_mods, src0, src1_mods, src1, clamp, dst_sel, dst_unused,
1436 // src0_sel, src1_sel.
1437 BuildMI(BB&: *C.LshlOr->getParent(), I&: *C.LshlOr, MIMD: C.LshlOr->getDebugLoc(),
1438 MCID: TII->get(Opcode: AMDGPU::V_OR_B32_sdwa))
1439 .add(MO: *Dst)
1440 .addImm(Val: 0)
1441 .addReg(RegNo: ShiftReg)
1442 .addImm(Val: 0)
1443 .add(MO: *C.ValSrc)
1444 .addImm(Val: 0)
1445 .addImm(Val: DWORD)
1446 .addImm(Val: UNUSED_PAD)
1447 .addImm(Val: DWORD)
1448 .addImm(Val: WORD_0);
1449
1450 MRI->clearKillFlags(Reg: C.ValSrc->getReg());
1451 C.LshlOr->eraseFromParent();
1452 C.AndMI->eraseFromParent();
1453 }
1454
1455 return !Candidates.empty();
1456}
1457
1458bool SIPeepholeSDWALegacy::runOnMachineFunction(MachineFunction &MF) {
1459 if (skipFunction(F: MF.getFunction()))
1460 return false;
1461
1462 return SIPeepholeSDWA().run(MF);
1463}
1464
1465bool SIPeepholeSDWA::run(MachineFunction &MF) {
1466 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
1467
1468 if (!ST.hasSDWA())
1469 return false;
1470
1471 MRI = &MF.getRegInfo();
1472 TRI = ST.getRegisterInfo();
1473 TII = ST.getInstrInfo();
1474
1475 // Find all SDWA operands in MF.
1476 bool Ret = false;
1477 for (MachineBasicBlock &MBB : MF) {
1478 bool Changed = false;
1479 do {
1480 Ret |= splitLshlOrForSDWA(MBB);
1481
1482 // Preprocess the ADD/SUB pairs so they could be SDWA'ed.
1483 // Look for a possible ADD or SUB that resulted from a previously lowered
1484 // V_{ADD|SUB}_U64_PSEUDO. The function pseudoOpConvertToVOP2
1485 // lowers the pair of instructions into e32 form.
1486 matchSDWAOperands(MBB);
1487 for (const auto &OperandPair : SDWAOperands) {
1488 const auto &Operand = OperandPair.second;
1489 MachineInstr *PotentialMI = Operand->potentialToConvert(TII, ST);
1490 if (!PotentialMI)
1491 continue;
1492
1493 switch (PotentialMI->getOpcode()) {
1494 case AMDGPU::V_ADD_CO_U32_e64:
1495 case AMDGPU::V_SUB_CO_U32_e64:
1496 pseudoOpConvertToVOP2(MI&: *PotentialMI, ST);
1497 break;
1498 case AMDGPU::V_CNDMASK_B32_e64:
1499 convertVcndmaskToVOP2(MI&: *PotentialMI, ST);
1500 break;
1501 };
1502 }
1503 SDWAOperands.clear();
1504
1505 // Generate potential match list.
1506 matchSDWAOperands(MBB);
1507
1508 for (const auto &OperandPair : SDWAOperands) {
1509 const auto &Operand = OperandPair.second;
1510 MachineInstr *PotentialMI =
1511 Operand->potentialToConvert(TII, ST, PotentialMatches: &PotentialMatches);
1512
1513 if (PotentialMI && isConvertibleToSDWA(MI&: *PotentialMI, ST, TII))
1514 PotentialMatches[PotentialMI].push_back(Elt: Operand.get());
1515 }
1516
1517 for (auto &PotentialPair : PotentialMatches) {
1518 MachineInstr &PotentialMI = *PotentialPair.first;
1519 convertToSDWA(MI&: PotentialMI, SDWAOperands: PotentialPair.second);
1520 }
1521
1522 PotentialMatches.clear();
1523 SDWAOperands.clear();
1524
1525 Changed = !ConvertedInstructions.empty();
1526
1527 if (Changed)
1528 Ret = true;
1529 while (!ConvertedInstructions.empty())
1530 legalizeScalarOperands(MI&: *ConvertedInstructions.pop_back_val(), ST);
1531 } while (Changed);
1532 }
1533
1534 return Ret;
1535}
1536
1537PreservedAnalyses SIPeepholeSDWAPass::run(MachineFunction &MF,
1538 MachineFunctionAnalysisManager &) {
1539 if (MF.getFunction().hasOptNone() || !SIPeepholeSDWA().run(MF))
1540 return PreservedAnalyses::all();
1541
1542 PreservedAnalyses PA = getMachineFunctionPassPreservedAnalyses();
1543 PA.preserveSet<CFGAnalyses>();
1544 return PA;
1545}
1546