| 1 | //===- SIPeepholeSDWA.cpp - Peephole optimization for SDWA instructions ---===// |
| 2 | // |
| 3 | // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. |
| 4 | // See https://llvm.org/LICENSE.txt for license information. |
| 5 | // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception |
| 6 | // |
| 7 | //===----------------------------------------------------------------------===// |
| 8 | // |
| 9 | /// \file This pass tries to apply several peephole SDWA patterns. |
| 10 | /// |
| 11 | /// E.g. original: |
| 12 | /// V_LSHRREV_B32_e32 %0, 16, %1 |
| 13 | /// V_ADD_CO_U32_e32 %2, %0, %3 |
| 14 | /// V_LSHLREV_B32_e32 %4, 16, %2 |
| 15 | /// |
| 16 | /// Replace: |
| 17 | /// V_ADD_CO_U32_sdwa %4, %1, %3 |
| 18 | /// dst_sel:WORD_1 dst_unused:UNUSED_PAD src0_sel:WORD_1 src1_sel:DWORD |
| 19 | /// |
| 20 | //===----------------------------------------------------------------------===// |
| 21 | |
| 22 | #include "SIPeepholeSDWA.h" |
| 23 | #include "AMDGPU.h" |
| 24 | #include "GCNSubtarget.h" |
| 25 | #include "llvm/ADT/Statistic.h" |
| 26 | #include "llvm/CodeGen/MachineFunctionPass.h" |
| 27 | #include <optional> |
| 28 | |
| 29 | using namespace llvm; |
| 30 | |
| 31 | #define DEBUG_TYPE "si-peephole-sdwa" |
| 32 | |
| 33 | STATISTIC(NumSDWAPatternsFound, "Number of SDWA patterns found." ); |
| 34 | STATISTIC(NumSDWAInstructionsPeepholed, |
| 35 | "Number of instruction converted to SDWA." ); |
| 36 | |
| 37 | namespace { |
| 38 | |
| 39 | bool isConvertibleToSDWA(MachineInstr &MI, const GCNSubtarget &ST, |
| 40 | const SIInstrInfo *TII); |
| 41 | class SDWAOperand; |
| 42 | class SDWADstOperand; |
| 43 | |
| 44 | using SDWAOperandsVector = SmallVector<SDWAOperand *, 4>; |
| 45 | using SDWAOperandsMap = MapVector<MachineInstr *, SDWAOperandsVector>; |
| 46 | |
| 47 | class SIPeepholeSDWA { |
| 48 | private: |
| 49 | MachineRegisterInfo *MRI; |
| 50 | const SIRegisterInfo *TRI; |
| 51 | const SIInstrInfo *TII; |
| 52 | |
| 53 | MapVector<MachineInstr *, std::unique_ptr<SDWAOperand>> SDWAOperands; |
| 54 | SDWAOperandsMap PotentialMatches; |
| 55 | SmallVector<MachineInstr *, 8> ConvertedInstructions; |
| 56 | |
| 57 | std::optional<int64_t> foldToImm(const MachineOperand &Op) const; |
| 58 | |
| 59 | // If MI is a v_and_b32 with a 0xffff or 0xff immediate, return the masked |
| 60 | // value operand and the matching SDWA selector (WORD_0 / BYTE_0). |
| 61 | std::optional<std::pair<MachineOperand *, AMDGPU::SDWA::SdwaSel>> |
| 62 | matchAndMask(MachineInstr &MI) const; |
| 63 | |
| 64 | // VOPC SDWA instructions carry the SDWA TSFlag but have no dst_sel operand. |
| 65 | bool isSDWAWithDstSel(const MachineInstr &Inst) const; |
| 66 | |
| 67 | void matchSDWAOperands(MachineBasicBlock &MBB); |
| 68 | std::unique_ptr<SDWAOperand> matchSDWAOperand(MachineInstr &MI); |
| 69 | void pseudoOpConvertToVOP2(MachineInstr &MI, |
| 70 | const GCNSubtarget &ST) const; |
| 71 | void convertVcndmaskToVOP2(MachineInstr &MI, const GCNSubtarget &ST) const; |
| 72 | MachineInstr *createSDWAVersion(MachineInstr &MI); |
| 73 | bool convertToSDWA(MachineInstr &MI, const SDWAOperandsVector &SDWAOperands); |
| 74 | void legalizeScalarOperands(MachineInstr &MI, const GCNSubtarget &ST) const; |
| 75 | bool splitLshlOrForSDWA(MachineBasicBlock &MBB); |
| 76 | |
| 77 | public: |
| 78 | bool run(MachineFunction &MF); |
| 79 | }; |
| 80 | |
| 81 | class SIPeepholeSDWALegacy : public MachineFunctionPass { |
| 82 | public: |
| 83 | static char ID; |
| 84 | |
| 85 | SIPeepholeSDWALegacy() : MachineFunctionPass(ID) {} |
| 86 | |
| 87 | StringRef getPassName() const override { return "SI Peephole SDWA" ; } |
| 88 | |
| 89 | bool runOnMachineFunction(MachineFunction &MF) override; |
| 90 | |
| 91 | void getAnalysisUsage(AnalysisUsage &AU) const override { |
| 92 | AU.setPreservesCFG(); |
| 93 | MachineFunctionPass::getAnalysisUsage(AU); |
| 94 | } |
| 95 | }; |
| 96 | |
| 97 | using namespace AMDGPU::SDWA; |
| 98 | |
| 99 | class SDWAOperand { |
| 100 | private: |
| 101 | MachineOperand *Target; // Operand that would be used in converted instruction |
| 102 | MachineOperand *Replaced; // Operand that would be replace by Target |
| 103 | |
| 104 | /// Returns true iff the SDWA selection of this SDWAOperand can be combined |
| 105 | /// with the SDWA selections of its uses in \p MI. |
| 106 | virtual bool canCombineSelections(const MachineInstr &MI, |
| 107 | const SIInstrInfo *TII) = 0; |
| 108 | |
| 109 | public: |
| 110 | SDWAOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp) |
| 111 | : Target(TargetOp), Replaced(ReplacedOp) { |
| 112 | assert(Target->isReg()); |
| 113 | assert(Replaced->isReg()); |
| 114 | } |
| 115 | |
| 116 | virtual ~SDWAOperand() = default; |
| 117 | |
| 118 | virtual MachineInstr *potentialToConvert(const SIInstrInfo *TII, |
| 119 | const GCNSubtarget &ST, |
| 120 | SDWAOperandsMap *PotentialMatches = nullptr) = 0; |
| 121 | virtual bool convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) = 0; |
| 122 | |
| 123 | MachineOperand *getTargetOperand() const { return Target; } |
| 124 | MachineOperand *getReplacedOperand() const { return Replaced; } |
| 125 | MachineInstr *getParentInst() const { return Target->getParent(); } |
| 126 | |
| 127 | MachineRegisterInfo *getMRI() const { |
| 128 | return &getParentInst()->getMF()->getRegInfo(); |
| 129 | } |
| 130 | |
| 131 | #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP) |
| 132 | virtual void print(raw_ostream& OS) const = 0; |
| 133 | void dump() const { print(dbgs()); } |
| 134 | #endif |
| 135 | }; |
| 136 | |
| 137 | class SDWASrcOperand : public SDWAOperand { |
| 138 | private: |
| 139 | SdwaSel SrcSel; |
| 140 | bool Abs; |
| 141 | bool Neg; |
| 142 | bool Sext; |
| 143 | |
| 144 | public: |
| 145 | SDWASrcOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp, |
| 146 | SdwaSel SrcSel_ = DWORD, bool Abs_ = false, bool Neg_ = false, |
| 147 | bool Sext_ = false) |
| 148 | : SDWAOperand(TargetOp, ReplacedOp), SrcSel(SrcSel_), Abs(Abs_), |
| 149 | Neg(Neg_), Sext(Sext_) {} |
| 150 | |
| 151 | MachineInstr *potentialToConvert(const SIInstrInfo *TII, |
| 152 | const GCNSubtarget &ST, |
| 153 | SDWAOperandsMap *PotentialMatches = nullptr) override; |
| 154 | bool convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) override; |
| 155 | bool canCombineSelections(const MachineInstr &MI, |
| 156 | const SIInstrInfo *TII) override; |
| 157 | |
| 158 | SdwaSel getSrcSel() const { return SrcSel; } |
| 159 | bool getAbs() const { return Abs; } |
| 160 | bool getNeg() const { return Neg; } |
| 161 | bool getSext() const { return Sext; } |
| 162 | |
| 163 | uint64_t getSrcMods(const SIInstrInfo *TII, |
| 164 | const MachineOperand *SrcOp) const; |
| 165 | |
| 166 | #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP) |
| 167 | void print(raw_ostream& OS) const override; |
| 168 | #endif |
| 169 | }; |
| 170 | |
| 171 | class SDWADstOperand : public SDWAOperand { |
| 172 | private: |
| 173 | SdwaSel DstSel; |
| 174 | DstUnused DstUn; |
| 175 | |
| 176 | public: |
| 177 | SDWADstOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp, |
| 178 | SdwaSel DstSel_ = DWORD, DstUnused DstUn_ = UNUSED_PAD) |
| 179 | : SDWAOperand(TargetOp, ReplacedOp), DstSel(DstSel_), DstUn(DstUn_) {} |
| 180 | |
| 181 | MachineInstr *potentialToConvert(const SIInstrInfo *TII, |
| 182 | const GCNSubtarget &ST, |
| 183 | SDWAOperandsMap *PotentialMatches = nullptr) override; |
| 184 | bool convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) override; |
| 185 | bool canCombineSelections(const MachineInstr &MI, |
| 186 | const SIInstrInfo *TII) override; |
| 187 | |
| 188 | SdwaSel getDstSel() const { return DstSel; } |
| 189 | DstUnused getDstUnused() const { return DstUn; } |
| 190 | |
| 191 | #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP) |
| 192 | void print(raw_ostream& OS) const override; |
| 193 | #endif |
| 194 | }; |
| 195 | |
| 196 | class SDWADstPreserveOperand : public SDWADstOperand { |
| 197 | private: |
| 198 | MachineOperand *Preserve; |
| 199 | |
| 200 | public: |
| 201 | SDWADstPreserveOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp, |
| 202 | MachineOperand *PreserveOp, SdwaSel DstSel_ = DWORD) |
| 203 | : SDWADstOperand(TargetOp, ReplacedOp, DstSel_, UNUSED_PRESERVE), |
| 204 | Preserve(PreserveOp) {} |
| 205 | |
| 206 | bool convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) override; |
| 207 | bool canCombineSelections(const MachineInstr &MI, |
| 208 | const SIInstrInfo *TII) override; |
| 209 | |
| 210 | MachineOperand *getPreservedOperand() const { return Preserve; } |
| 211 | |
| 212 | #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP) |
| 213 | void print(raw_ostream& OS) const override; |
| 214 | #endif |
| 215 | }; |
| 216 | |
| 217 | } // end anonymous namespace |
| 218 | |
| 219 | INITIALIZE_PASS(SIPeepholeSDWALegacy, DEBUG_TYPE, "SI Peephole SDWA" , false, |
| 220 | false) |
| 221 | |
| 222 | char SIPeepholeSDWALegacy::ID = 0; |
| 223 | |
| 224 | char &llvm::SIPeepholeSDWALegacyID = SIPeepholeSDWALegacy::ID; |
| 225 | |
| 226 | #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP) |
| 227 | static raw_ostream& operator<<(raw_ostream &OS, SdwaSel Sel) { |
| 228 | switch(Sel) { |
| 229 | case BYTE_0: OS << "BYTE_0" ; break; |
| 230 | case BYTE_1: OS << "BYTE_1" ; break; |
| 231 | case BYTE_2: OS << "BYTE_2" ; break; |
| 232 | case BYTE_3: OS << "BYTE_3" ; break; |
| 233 | case WORD_0: OS << "WORD_0" ; break; |
| 234 | case WORD_1: OS << "WORD_1" ; break; |
| 235 | case DWORD: OS << "DWORD" ; break; |
| 236 | } |
| 237 | return OS; |
| 238 | } |
| 239 | |
| 240 | static raw_ostream& operator<<(raw_ostream &OS, const DstUnused &Un) { |
| 241 | switch(Un) { |
| 242 | case UNUSED_PAD: OS << "UNUSED_PAD" ; break; |
| 243 | case UNUSED_SEXT: OS << "UNUSED_SEXT" ; break; |
| 244 | case UNUSED_PRESERVE: OS << "UNUSED_PRESERVE" ; break; |
| 245 | } |
| 246 | return OS; |
| 247 | } |
| 248 | |
| 249 | LLVM_DUMP_METHOD |
| 250 | void SDWASrcOperand::print(raw_ostream& OS) const { |
| 251 | OS << "SDWA src: " << *getTargetOperand() |
| 252 | << " src_sel:" << getSrcSel() |
| 253 | << " abs:" << getAbs() << " neg:" << getNeg() |
| 254 | << " sext:" << getSext() << '\n'; |
| 255 | } |
| 256 | |
| 257 | LLVM_DUMP_METHOD |
| 258 | void SDWADstOperand::print(raw_ostream& OS) const { |
| 259 | OS << "SDWA dst: " << *getTargetOperand() |
| 260 | << " dst_sel:" << getDstSel() |
| 261 | << " dst_unused:" << getDstUnused() << '\n'; |
| 262 | } |
| 263 | |
| 264 | LLVM_DUMP_METHOD |
| 265 | void SDWADstPreserveOperand::print(raw_ostream& OS) const { |
| 266 | OS << "SDWA preserve dst: " << *getTargetOperand() |
| 267 | << " dst_sel:" << getDstSel() |
| 268 | << " preserve:" << *getPreservedOperand() << '\n'; |
| 269 | } |
| 270 | |
| 271 | #endif |
| 272 | |
| 273 | static void copyRegOperand(MachineOperand &To, const MachineOperand &From) { |
| 274 | assert(To.isReg() && From.isReg()); |
| 275 | To.setReg(From.getReg()); |
| 276 | To.setSubReg(From.getSubReg()); |
| 277 | To.setIsUndef(From.isUndef()); |
| 278 | if (To.isUse()) { |
| 279 | To.setIsKill(From.isKill()); |
| 280 | } else { |
| 281 | To.setIsDead(From.isDead()); |
| 282 | } |
| 283 | } |
| 284 | |
| 285 | static bool isSameReg(const MachineOperand &LHS, const MachineOperand &RHS) { |
| 286 | return LHS.isReg() && |
| 287 | RHS.isReg() && |
| 288 | LHS.getReg() == RHS.getReg() && |
| 289 | LHS.getSubReg() == RHS.getSubReg(); |
| 290 | } |
| 291 | |
| 292 | static MachineOperand *findSingleRegUse(const MachineOperand *Reg, |
| 293 | const MachineRegisterInfo *MRI) { |
| 294 | if (!Reg->isReg() || !Reg->isDef()) |
| 295 | return nullptr; |
| 296 | |
| 297 | return MRI->getOneNonDBGUse(RegNo: Reg->getReg()); |
| 298 | } |
| 299 | |
| 300 | static MachineOperand *findSingleRegDef(const MachineOperand *Reg, |
| 301 | const MachineRegisterInfo *MRI) { |
| 302 | if (!Reg->isReg()) |
| 303 | return nullptr; |
| 304 | |
| 305 | return MRI->getOneDef(Reg: Reg->getReg()); |
| 306 | } |
| 307 | |
| 308 | /// Combine an SDWA instruction's existing SDWA selection \p Sel with |
| 309 | /// the SDWA selection \p OperandSel of its operand. If the selections |
| 310 | /// are compatible, return the combined selection, otherwise return a |
| 311 | /// nullopt. |
| 312 | /// For example, if we have Sel = BYTE_0 Sel and OperandSel = WORD_1: |
| 313 | /// BYTE_0 Sel (WORD_1 Sel (%X)) -> BYTE_2 Sel (%X) |
| 314 | static std::optional<SdwaSel> combineSdwaSel(SdwaSel Sel, SdwaSel OperandSel) { |
| 315 | if (Sel == SdwaSel::DWORD) |
| 316 | return OperandSel; |
| 317 | |
| 318 | if (Sel == OperandSel || OperandSel == SdwaSel::DWORD) |
| 319 | return Sel; |
| 320 | |
| 321 | if (Sel == SdwaSel::WORD_1 || Sel == SdwaSel::BYTE_2 || |
| 322 | Sel == SdwaSel::BYTE_3) |
| 323 | return {}; |
| 324 | |
| 325 | if (OperandSel == SdwaSel::WORD_0) |
| 326 | return Sel; |
| 327 | |
| 328 | if (OperandSel == SdwaSel::WORD_1) { |
| 329 | if (Sel == SdwaSel::BYTE_0) |
| 330 | return SdwaSel::BYTE_2; |
| 331 | if (Sel == SdwaSel::BYTE_1) |
| 332 | return SdwaSel::BYTE_3; |
| 333 | if (Sel == SdwaSel::WORD_0) |
| 334 | return SdwaSel::WORD_1; |
| 335 | } |
| 336 | |
| 337 | return {}; |
| 338 | } |
| 339 | |
| 340 | uint64_t SDWASrcOperand::getSrcMods(const SIInstrInfo *TII, |
| 341 | const MachineOperand *SrcOp) const { |
| 342 | uint64_t Mods = 0; |
| 343 | const auto *MI = SrcOp->getParent(); |
| 344 | if (TII->getNamedOperand(MI: *MI, OperandName: AMDGPU::OpName::src0) == SrcOp) { |
| 345 | if (auto *Mod = TII->getNamedOperand(MI: *MI, OperandName: AMDGPU::OpName::src0_modifiers)) { |
| 346 | Mods = Mod->getImm(); |
| 347 | } |
| 348 | } else if (TII->getNamedOperand(MI: *MI, OperandName: AMDGPU::OpName::src1) == SrcOp) { |
| 349 | if (auto *Mod = TII->getNamedOperand(MI: *MI, OperandName: AMDGPU::OpName::src1_modifiers)) { |
| 350 | Mods = Mod->getImm(); |
| 351 | } |
| 352 | } |
| 353 | if (Abs || Neg) { |
| 354 | assert(!Sext && |
| 355 | "Float and integer src modifiers can't be set simultaneously" ); |
| 356 | Mods |= Abs ? SISrcMods::ABS : 0u; |
| 357 | Mods ^= Neg ? SISrcMods::NEG : 0u; |
| 358 | } else if (Sext) { |
| 359 | Mods |= SISrcMods::SEXT; |
| 360 | } |
| 361 | |
| 362 | return Mods; |
| 363 | } |
| 364 | |
| 365 | MachineInstr *SDWASrcOperand::potentialToConvert(const SIInstrInfo *TII, |
| 366 | const GCNSubtarget &ST, |
| 367 | SDWAOperandsMap *PotentialMatches) { |
| 368 | if (PotentialMatches != nullptr) { |
| 369 | // Fill out the map for all uses if all can be converted |
| 370 | MachineOperand *Reg = getReplacedOperand(); |
| 371 | if (!Reg->isReg() || !Reg->isDef()) |
| 372 | return nullptr; |
| 373 | |
| 374 | for (MachineInstr &UseMI : getMRI()->use_nodbg_instructions(Reg: Reg->getReg())) |
| 375 | // Check that all instructions that use Reg can be converted |
| 376 | if (!isConvertibleToSDWA(MI&: UseMI, ST, TII) || |
| 377 | !canCombineSelections(MI: UseMI, TII)) |
| 378 | return nullptr; |
| 379 | |
| 380 | // Now that it's guaranteed all uses are legal, iterate over the uses again |
| 381 | // to add them for later conversion. |
| 382 | for (MachineOperand &UseMO : getMRI()->use_nodbg_operands(Reg: Reg->getReg())) { |
| 383 | // Should not get a subregister here |
| 384 | assert(isSameReg(UseMO, *Reg)); |
| 385 | |
| 386 | SDWAOperandsMap &potentialMatchesMap = *PotentialMatches; |
| 387 | MachineInstr *UseMI = UseMO.getParent(); |
| 388 | potentialMatchesMap[UseMI].push_back(Elt: this); |
| 389 | } |
| 390 | return nullptr; |
| 391 | } |
| 392 | |
| 393 | // For SDWA src operand potential instruction is one that use register |
| 394 | // defined by parent instruction |
| 395 | MachineOperand *PotentialMO = findSingleRegUse(Reg: getReplacedOperand(), MRI: getMRI()); |
| 396 | if (!PotentialMO) |
| 397 | return nullptr; |
| 398 | |
| 399 | MachineInstr *Parent = PotentialMO->getParent(); |
| 400 | |
| 401 | return canCombineSelections(MI: *Parent, TII) ? Parent : nullptr; |
| 402 | } |
| 403 | |
| 404 | bool SDWASrcOperand::convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) { |
| 405 | assert((!Sext || !TII->getSubtarget().zeroesHigh16BitsOfDest( |
| 406 | getParentInst()->getOpcode())) && |
| 407 | "Cannot use sign-extension with instruction that zeroes high bits" ); |
| 408 | switch (MI.getOpcode()) { |
| 409 | case AMDGPU::V_CVT_F32_FP8_sdwa: |
| 410 | case AMDGPU::V_CVT_F32_BF8_sdwa: |
| 411 | case AMDGPU::V_CVT_PK_F32_FP8_sdwa: |
| 412 | case AMDGPU::V_CVT_PK_F32_BF8_sdwa: |
| 413 | // Does not support input modifiers: noabs, noneg, nosext. |
| 414 | return false; |
| 415 | case AMDGPU::V_CNDMASK_B32_sdwa: |
| 416 | // SISrcMods uses the same bitmask for SEXT and NEG modifiers and |
| 417 | // hence the compiler can only support one type of modifier for |
| 418 | // each SDWA instruction. For V_CNDMASK_B32_sdwa, this is NEG |
| 419 | // since its operands get printed using |
| 420 | // AMDGPUInstPrinter::printOperandAndFPInputMods which produces |
| 421 | // the output intended for NEG if SEXT is set. |
| 422 | // |
| 423 | // The ISA does actually support both modifiers on most SDWA |
| 424 | // instructions. |
| 425 | // |
| 426 | // FIXME Accept SEXT here after fixing this issue. |
| 427 | if (Sext) |
| 428 | return false; |
| 429 | break; |
| 430 | } |
| 431 | |
| 432 | // Find operand in instruction that matches source operand and replace it with |
| 433 | // target operand. Set corresponding src_sel |
| 434 | bool IsPreserveSrc = false; |
| 435 | MachineOperand *Src = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0); |
| 436 | MachineOperand *SrcSel = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0_sel); |
| 437 | MachineOperand *SrcMods = |
| 438 | TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0_modifiers); |
| 439 | assert(Src && (Src->isReg() || Src->isImm())); |
| 440 | if (!isSameReg(LHS: *Src, RHS: *getReplacedOperand())) { |
| 441 | // If this is not src0 then it could be src1 |
| 442 | Src = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1); |
| 443 | SrcSel = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1_sel); |
| 444 | SrcMods = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1_modifiers); |
| 445 | |
| 446 | if (!Src || |
| 447 | !isSameReg(LHS: *Src, RHS: *getReplacedOperand())) { |
| 448 | // It's possible this Src is a tied operand for |
| 449 | // UNUSED_PRESERVE, in which case we can either |
| 450 | // abandon the peephole attempt, or if legal we can |
| 451 | // copy the target operand into the tied slot |
| 452 | // if the preserve operation will effectively cause the same |
| 453 | // result by overwriting the rest of the dst. |
| 454 | MachineOperand *Dst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst); |
| 455 | MachineOperand *DstUnused = |
| 456 | TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::dst_unused); |
| 457 | |
| 458 | if (Dst && |
| 459 | DstUnused->getImm() == AMDGPU::SDWA::DstUnused::UNUSED_PRESERVE) { |
| 460 | // This will work if the tied src is accessing WORD_0, and the dst is |
| 461 | // writing WORD_1. Modifiers don't matter because all the bits that |
| 462 | // would be impacted are being overwritten by the dst. |
| 463 | // Any other case will not work. |
| 464 | SdwaSel DstSel = static_cast<SdwaSel>( |
| 465 | TII->getNamedImmOperand(MI, OperandName: AMDGPU::OpName::dst_sel)); |
| 466 | if (DstSel == AMDGPU::SDWA::SdwaSel::WORD_1 && |
| 467 | getSrcSel() == AMDGPU::SDWA::SdwaSel::WORD_0) { |
| 468 | IsPreserveSrc = true; |
| 469 | auto DstIdx = AMDGPU::getNamedOperandIdx(Opcode: MI.getOpcode(), |
| 470 | Name: AMDGPU::OpName::vdst); |
| 471 | auto TiedIdx = MI.findTiedOperandIdx(OpIdx: DstIdx); |
| 472 | Src = &MI.getOperand(i: TiedIdx); |
| 473 | SrcSel = nullptr; |
| 474 | SrcMods = nullptr; |
| 475 | } else { |
| 476 | // Not legal to convert this src |
| 477 | return false; |
| 478 | } |
| 479 | } |
| 480 | } |
| 481 | assert(Src && Src->isReg()); |
| 482 | |
| 483 | if ((MI.getOpcode() == AMDGPU::V_FMAC_F16_sdwa || |
| 484 | MI.getOpcode() == AMDGPU::V_FMAC_F32_sdwa || |
| 485 | MI.getOpcode() == AMDGPU::V_MAC_F16_sdwa || |
| 486 | MI.getOpcode() == AMDGPU::V_MAC_F32_sdwa) && |
| 487 | !isSameReg(LHS: *Src, RHS: *getReplacedOperand())) { |
| 488 | // In case of v_mac_f16/32_sdwa this pass can try to apply src operand to |
| 489 | // src2. This is not allowed. |
| 490 | return false; |
| 491 | } |
| 492 | |
| 493 | assert(isSameReg(*Src, *getReplacedOperand()) && |
| 494 | (IsPreserveSrc || (SrcSel && SrcMods))); |
| 495 | } |
| 496 | copyRegOperand(To&: *Src, From: *getTargetOperand()); |
| 497 | if (!IsPreserveSrc) { |
| 498 | SdwaSel ExistingSel = static_cast<SdwaSel>(SrcSel->getImm()); |
| 499 | SrcSel->setImm(*combineSdwaSel(Sel: ExistingSel, OperandSel: getSrcSel())); |
| 500 | SrcMods->setImm(getSrcMods(TII, SrcOp: Src)); |
| 501 | } |
| 502 | getTargetOperand()->setIsKill(false); |
| 503 | return true; |
| 504 | } |
| 505 | |
| 506 | /// Verify that the SDWA selection operand \p SrcSelOpName of the SDWA |
| 507 | /// instruction \p MI can be combined with the selection \p OpSel. |
| 508 | static bool canCombineOpSel(const MachineInstr &MI, const SIInstrInfo *TII, |
| 509 | AMDGPU::OpName SrcSelOpName, SdwaSel OpSel) { |
| 510 | assert(TII->isSDWA(MI.getOpcode())); |
| 511 | |
| 512 | const MachineOperand *SrcSelOp = TII->getNamedOperand(MI, OperandName: SrcSelOpName); |
| 513 | SdwaSel SrcSel = static_cast<SdwaSel>(SrcSelOp->getImm()); |
| 514 | |
| 515 | return combineSdwaSel(Sel: SrcSel, OperandSel: OpSel).has_value(); |
| 516 | } |
| 517 | |
| 518 | /// Verify that \p Op is the same register as the operand of the SDWA |
| 519 | /// instruction \p MI named by \p SrcOpName and that the SDWA |
| 520 | /// selection \p SrcSelOpName can be combined with the \p OpSel. |
| 521 | static bool canCombineOpSel(const MachineInstr &MI, const SIInstrInfo *TII, |
| 522 | AMDGPU::OpName SrcOpName, |
| 523 | AMDGPU::OpName SrcSelOpName, MachineOperand *Op, |
| 524 | SdwaSel OpSel) { |
| 525 | assert(TII->isSDWA(MI.getOpcode())); |
| 526 | |
| 527 | const MachineOperand *Src = TII->getNamedOperand(MI, OperandName: SrcOpName); |
| 528 | if (!Src || !isSameReg(LHS: *Src, RHS: *Op)) |
| 529 | return true; |
| 530 | |
| 531 | return canCombineOpSel(MI, TII, SrcSelOpName, OpSel); |
| 532 | } |
| 533 | |
| 534 | bool SDWASrcOperand::canCombineSelections(const MachineInstr &MI, |
| 535 | const SIInstrInfo *TII) { |
| 536 | if (!TII->isSDWA(Opcode: MI.getOpcode())) |
| 537 | return true; |
| 538 | |
| 539 | using namespace AMDGPU; |
| 540 | |
| 541 | return canCombineOpSel(MI, TII, SrcOpName: OpName::src0, SrcSelOpName: OpName::src0_sel, |
| 542 | Op: getReplacedOperand(), OpSel: getSrcSel()) && |
| 543 | canCombineOpSel(MI, TII, SrcOpName: OpName::src1, SrcSelOpName: OpName::src1_sel, |
| 544 | Op: getReplacedOperand(), OpSel: getSrcSel()); |
| 545 | } |
| 546 | |
| 547 | MachineInstr *SDWADstOperand::potentialToConvert(const SIInstrInfo *TII, |
| 548 | const GCNSubtarget &ST, |
| 549 | SDWAOperandsMap *PotentialMatches) { |
| 550 | // For SDWA dst operand potential instruction is one that defines register |
| 551 | // that this operand uses |
| 552 | MachineRegisterInfo *MRI = getMRI(); |
| 553 | MachineInstr *ParentMI = getParentInst(); |
| 554 | |
| 555 | MachineOperand *PotentialMO = findSingleRegDef(Reg: getReplacedOperand(), MRI); |
| 556 | if (!PotentialMO) |
| 557 | return nullptr; |
| 558 | |
| 559 | // Check that ParentMI is the only instruction that uses replaced register |
| 560 | for (MachineInstr &UseInst : MRI->use_nodbg_instructions(Reg: PotentialMO->getReg())) { |
| 561 | if (&UseInst != ParentMI) |
| 562 | return nullptr; |
| 563 | } |
| 564 | |
| 565 | MachineInstr *Parent = PotentialMO->getParent(); |
| 566 | return canCombineSelections(MI: *Parent, TII) ? Parent : nullptr; |
| 567 | } |
| 568 | |
| 569 | bool SDWADstOperand::convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) { |
| 570 | // Replace vdst operand in MI with target operand. Set dst_sel and dst_unused |
| 571 | |
| 572 | if ((MI.getOpcode() == AMDGPU::V_FMAC_F16_sdwa || |
| 573 | MI.getOpcode() == AMDGPU::V_FMAC_F32_sdwa || |
| 574 | MI.getOpcode() == AMDGPU::V_MAC_F16_sdwa || |
| 575 | MI.getOpcode() == AMDGPU::V_MAC_F32_sdwa) && |
| 576 | getDstSel() != AMDGPU::SDWA::DWORD) { |
| 577 | // v_mac_f16/32_sdwa allow dst_sel to be equal only to DWORD |
| 578 | return false; |
| 579 | } |
| 580 | |
| 581 | MachineOperand *Operand = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst); |
| 582 | assert(Operand && |
| 583 | Operand->isReg() && |
| 584 | isSameReg(*Operand, *getReplacedOperand())); |
| 585 | copyRegOperand(To&: *Operand, From: *getTargetOperand()); |
| 586 | MachineOperand *DstSel= TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::dst_sel); |
| 587 | assert(DstSel); |
| 588 | |
| 589 | SdwaSel ExistingSel = static_cast<SdwaSel>(DstSel->getImm()); |
| 590 | DstSel->setImm(combineSdwaSel(Sel: ExistingSel, OperandSel: getDstSel()).value()); |
| 591 | |
| 592 | MachineOperand *DstUnused= TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::dst_unused); |
| 593 | assert(DstUnused); |
| 594 | DstUnused->setImm(getDstUnused()); |
| 595 | |
| 596 | // Remove original instruction because it would conflict with our new |
| 597 | // instruction by register definition |
| 598 | getParentInst()->eraseFromParent(); |
| 599 | return true; |
| 600 | } |
| 601 | |
| 602 | bool SDWADstOperand::canCombineSelections(const MachineInstr &MI, |
| 603 | const SIInstrInfo *TII) { |
| 604 | if (!TII->isSDWA(Opcode: MI.getOpcode())) |
| 605 | return true; |
| 606 | |
| 607 | return canCombineOpSel(MI, TII, SrcSelOpName: AMDGPU::OpName::dst_sel, OpSel: getDstSel()); |
| 608 | } |
| 609 | |
| 610 | bool SDWADstPreserveOperand::convertToSDWA(MachineInstr &MI, |
| 611 | const SIInstrInfo *TII) { |
| 612 | // MI should be moved right before v_or_b32. |
| 613 | // For this we should clear all kill flags on uses of MI src-operands or else |
| 614 | // we can encounter problem with use of killed operand. |
| 615 | for (MachineOperand &MO : MI.uses()) { |
| 616 | if (!MO.isReg()) |
| 617 | continue; |
| 618 | getMRI()->clearKillFlags(Reg: MO.getReg()); |
| 619 | } |
| 620 | |
| 621 | // Move MI before v_or_b32 |
| 622 | MI.getParent()->remove(I: &MI); |
| 623 | getParentInst()->getParent()->insert(I: getParentInst(), MI: &MI); |
| 624 | |
| 625 | // Add Implicit use of preserved register |
| 626 | MachineInstrBuilder MIB(*MI.getMF(), MI); |
| 627 | MIB.addReg(RegNo: getPreservedOperand()->getReg(), |
| 628 | Flags: RegState::ImplicitKill, |
| 629 | SubReg: getPreservedOperand()->getSubReg()); |
| 630 | |
| 631 | // Tie dst to implicit use |
| 632 | MI.tieOperands(DefIdx: AMDGPU::getNamedOperandIdx(Opcode: MI.getOpcode(), Name: AMDGPU::OpName::vdst), |
| 633 | UseIdx: MI.getNumOperands() - 1); |
| 634 | |
| 635 | // Convert MI as any other SDWADstOperand and remove v_or_b32 |
| 636 | return SDWADstOperand::convertToSDWA(MI, TII); |
| 637 | } |
| 638 | |
| 639 | bool SDWADstPreserveOperand::canCombineSelections(const MachineInstr &MI, |
| 640 | const SIInstrInfo *TII) { |
| 641 | return SDWADstOperand::canCombineSelections(MI, TII); |
| 642 | } |
| 643 | |
| 644 | std::optional<int64_t> |
| 645 | SIPeepholeSDWA::foldToImm(const MachineOperand &Op) const { |
| 646 | if (Op.isImm()) { |
| 647 | return Op.getImm(); |
| 648 | } |
| 649 | |
| 650 | // If this is not immediate then it can be copy of immediate value, e.g.: |
| 651 | // %1 = S_MOV_B32 255; |
| 652 | if (Op.isReg()) { |
| 653 | for (const MachineOperand &Def : MRI->def_operands(Reg: Op.getReg())) { |
| 654 | if (!isSameReg(LHS: Op, RHS: Def)) |
| 655 | continue; |
| 656 | |
| 657 | const MachineInstr *DefInst = Def.getParent(); |
| 658 | if (!TII->isFoldableCopy(MI: *DefInst)) |
| 659 | return std::nullopt; |
| 660 | |
| 661 | const MachineOperand &Copied = DefInst->getOperand(i: 1); |
| 662 | if (!Copied.isImm()) |
| 663 | return std::nullopt; |
| 664 | |
| 665 | return Copied.getImm(); |
| 666 | } |
| 667 | } |
| 668 | |
| 669 | return std::nullopt; |
| 670 | } |
| 671 | |
| 672 | std::optional<std::pair<MachineOperand *, SdwaSel>> |
| 673 | SIPeepholeSDWA::matchAndMask(MachineInstr &MI) const { |
| 674 | if (MI.getOpcode() != AMDGPU::V_AND_B32_e32 && |
| 675 | MI.getOpcode() != AMDGPU::V_AND_B32_e64) |
| 676 | return std::nullopt; |
| 677 | |
| 678 | MachineOperand *Src0 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0); |
| 679 | MachineOperand *Src1 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1); |
| 680 | MachineOperand *ValSrc = Src1; |
| 681 | std::optional<int64_t> Imm = foldToImm(Op: *Src0); |
| 682 | if (!Imm) { |
| 683 | Imm = foldToImm(Op: *Src1); |
| 684 | ValSrc = Src0; |
| 685 | } |
| 686 | if (!Imm || (*Imm != 0x0000ffff && *Imm != 0x000000ff)) |
| 687 | return std::nullopt; |
| 688 | |
| 689 | return std::make_pair(x&: ValSrc, y: *Imm == 0x0000ffff ? WORD_0 : BYTE_0); |
| 690 | } |
| 691 | |
| 692 | bool SIPeepholeSDWA::isSDWAWithDstSel(const MachineInstr &Inst) const { |
| 693 | return TII->isSDWA(MI: Inst) && |
| 694 | AMDGPU::hasNamedOperand(Opcode: Inst.getOpcode(), NamedIdx: AMDGPU::OpName::dst_sel); |
| 695 | } |
| 696 | |
| 697 | std::unique_ptr<SDWAOperand> |
| 698 | SIPeepholeSDWA::matchSDWAOperand(MachineInstr &MI) { |
| 699 | unsigned Opcode = MI.getOpcode(); |
| 700 | switch (Opcode) { |
| 701 | case AMDGPU::V_LSHRREV_B32_e32: |
| 702 | case AMDGPU::V_ASHRREV_I32_e32: |
| 703 | case AMDGPU::V_LSHLREV_B32_e32: |
| 704 | case AMDGPU::V_LSHRREV_B32_e64: |
| 705 | case AMDGPU::V_ASHRREV_I32_e64: |
| 706 | case AMDGPU::V_LSHLREV_B32_e64: { |
| 707 | // from: v_lshrrev_b32_e32 v1, 16/24, v0 |
| 708 | // to SDWA src:v0 src_sel:WORD_1/BYTE_3 |
| 709 | |
| 710 | // from: v_ashrrev_i32_e32 v1, 16/24, v0 |
| 711 | // to SDWA src:v0 src_sel:WORD_1/BYTE_3 sext:1 |
| 712 | |
| 713 | // from: v_lshlrev_b32_e32 v1, 16/24, v0 |
| 714 | // to SDWA dst:v1 dst_sel:WORD_1/BYTE_3 dst_unused:UNUSED_PAD |
| 715 | MachineOperand *Src0 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0); |
| 716 | auto Imm = foldToImm(Op: *Src0); |
| 717 | if (!Imm) |
| 718 | break; |
| 719 | |
| 720 | if (*Imm != 16 && *Imm != 24) |
| 721 | break; |
| 722 | |
| 723 | MachineOperand *Src1 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1); |
| 724 | MachineOperand *Dst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst); |
| 725 | if (!Src1->isReg() || Src1->getReg().isPhysical() || |
| 726 | Dst->getReg().isPhysical()) |
| 727 | break; |
| 728 | |
| 729 | if (Opcode == AMDGPU::V_LSHLREV_B32_e32 || |
| 730 | Opcode == AMDGPU::V_LSHLREV_B32_e64) { |
| 731 | return std::make_unique<SDWADstOperand>( |
| 732 | args&: Dst, args&: Src1, args: *Imm == 16 ? WORD_1 : BYTE_3, args: UNUSED_PAD); |
| 733 | } |
| 734 | return std::make_unique<SDWASrcOperand>( |
| 735 | args&: Src1, args&: Dst, args: *Imm == 16 ? WORD_1 : BYTE_3, args: false, args: false, |
| 736 | args: Opcode != AMDGPU::V_LSHRREV_B32_e32 && |
| 737 | Opcode != AMDGPU::V_LSHRREV_B32_e64); |
| 738 | break; |
| 739 | } |
| 740 | |
| 741 | case AMDGPU::V_LSHRREV_B16_e32: |
| 742 | case AMDGPU::V_LSHLREV_B16_e32: |
| 743 | case AMDGPU::V_LSHRREV_B16_e64: |
| 744 | case AMDGPU::V_LSHRREV_B16_opsel_e64: |
| 745 | case AMDGPU::V_LSHLREV_B16_opsel_e64: |
| 746 | case AMDGPU::V_LSHLREV_B16_e64: { |
| 747 | // V_ASHRREV_I16_e32 and V_ASHRREV_I16_e64 are |
| 748 | // not included here because they zero-fill the high 16-bits. |
| 749 | |
| 750 | // from: v_lshrrev_b16_e32 v1, 8, v0 |
| 751 | // to SDWA src:v0 src_sel:BYTE_1 |
| 752 | |
| 753 | // from: v_lshlrev_b16_e32 v1, 8, v0 |
| 754 | // to SDWA dst:v1 dst_sel:BYTE_1 dst_unused:UNUSED_PAD |
| 755 | MachineOperand *Src0 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0); |
| 756 | auto Imm = foldToImm(Op: *Src0); |
| 757 | if (!Imm || *Imm != 8) |
| 758 | break; |
| 759 | |
| 760 | MachineOperand *Src1 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1); |
| 761 | MachineOperand *Dst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst); |
| 762 | |
| 763 | if (!Src1->isReg() || Src1->getReg().isPhysical() || |
| 764 | Dst->getReg().isPhysical()) |
| 765 | break; |
| 766 | |
| 767 | if (Opcode == AMDGPU::V_LSHLREV_B16_e32 || |
| 768 | Opcode == AMDGPU::V_LSHLREV_B16_opsel_e64 || |
| 769 | Opcode == AMDGPU::V_LSHLREV_B16_e64) |
| 770 | return std::make_unique<SDWADstOperand>(args&: Dst, args&: Src1, args: BYTE_1, args: UNUSED_PAD); |
| 771 | return std::make_unique<SDWASrcOperand>(args&: Src1, args&: Dst, args: BYTE_1, args: false, args: false, |
| 772 | args: false); |
| 773 | break; |
| 774 | } |
| 775 | |
| 776 | case AMDGPU::V_BFE_I32_e64: |
| 777 | case AMDGPU::V_BFE_U32_e64: { |
| 778 | // e.g.: |
| 779 | // from: v_bfe_u32 v1, v0, 8, 8 |
| 780 | // to SDWA src:v0 src_sel:BYTE_1 |
| 781 | |
| 782 | // offset | width | src_sel |
| 783 | // ------------------------ |
| 784 | // 0 | 8 | BYTE_0 |
| 785 | // 0 | 16 | WORD_0 |
| 786 | // 0 | 32 | DWORD ? |
| 787 | // 8 | 8 | BYTE_1 |
| 788 | // 16 | 8 | BYTE_2 |
| 789 | // 16 | 16 | WORD_1 |
| 790 | // 24 | 8 | BYTE_3 |
| 791 | |
| 792 | MachineOperand *Src1 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1); |
| 793 | auto Offset = foldToImm(Op: *Src1); |
| 794 | if (!Offset) |
| 795 | break; |
| 796 | |
| 797 | MachineOperand *Src2 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src2); |
| 798 | auto Width = foldToImm(Op: *Src2); |
| 799 | if (!Width) |
| 800 | break; |
| 801 | |
| 802 | SdwaSel SrcSel = DWORD; |
| 803 | |
| 804 | if (*Offset == 0 && *Width == 8) |
| 805 | SrcSel = BYTE_0; |
| 806 | else if (*Offset == 0 && *Width == 16) |
| 807 | SrcSel = WORD_0; |
| 808 | else if (*Offset == 0 && *Width == 32) |
| 809 | SrcSel = DWORD; |
| 810 | else if (*Offset == 8 && *Width == 8) |
| 811 | SrcSel = BYTE_1; |
| 812 | else if (*Offset == 16 && *Width == 8) |
| 813 | SrcSel = BYTE_2; |
| 814 | else if (*Offset == 16 && *Width == 16) |
| 815 | SrcSel = WORD_1; |
| 816 | else if (*Offset == 24 && *Width == 8) |
| 817 | SrcSel = BYTE_3; |
| 818 | else |
| 819 | break; |
| 820 | |
| 821 | MachineOperand *Src0 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0); |
| 822 | MachineOperand *Dst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst); |
| 823 | |
| 824 | if (!Src0->isReg() || Src0->getReg().isPhysical() || |
| 825 | Dst->getReg().isPhysical()) |
| 826 | break; |
| 827 | |
| 828 | return std::make_unique<SDWASrcOperand>( |
| 829 | args&: Src0, args&: Dst, args&: SrcSel, args: false, args: false, args: Opcode != AMDGPU::V_BFE_U32_e64); |
| 830 | } |
| 831 | |
| 832 | case AMDGPU::V_AND_B32_e32: |
| 833 | case AMDGPU::V_AND_B32_e64: { |
| 834 | // e.g.: |
| 835 | // from: v_and_b32_e32 v1, 0x0000ffff/0x000000ff, v0 |
| 836 | // to SDWA src:v0 src_sel:WORD_0/BYTE_0 |
| 837 | auto Mask = matchAndMask(MI); |
| 838 | if (!Mask) |
| 839 | break; |
| 840 | MachineOperand *ValSrc = Mask->first; |
| 841 | |
| 842 | MachineOperand *Dst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst); |
| 843 | |
| 844 | if (!ValSrc->isReg() || ValSrc->getReg().isPhysical() || |
| 845 | Dst->getReg().isPhysical()) |
| 846 | break; |
| 847 | |
| 848 | return std::make_unique<SDWASrcOperand>(args&: ValSrc, args&: Dst, args&: Mask->second); |
| 849 | } |
| 850 | |
| 851 | case AMDGPU::V_OR_B32_e32: |
| 852 | case AMDGPU::V_OR_B32_e64: { |
| 853 | // Patterns for dst_unused:UNUSED_PRESERVE. |
| 854 | // e.g., from: |
| 855 | // v_add_f16_sdwa v0, v1, v2 dst_sel:WORD_1 dst_unused:UNUSED_PAD |
| 856 | // src1_sel:WORD_1 src2_sel:WORD1 |
| 857 | // v_add_f16_e32 v3, v1, v2 |
| 858 | // v_or_b32_e32 v4, v0, v3 |
| 859 | // to SDWA preserve dst:v4 dst_sel:WORD_1 dst_unused:UNUSED_PRESERVE preserve:v3 |
| 860 | |
| 861 | // Check if one of operands of v_or_b32 is SDWA instruction |
| 862 | using CheckRetType = |
| 863 | std::optional<std::pair<MachineOperand *, MachineOperand *>>; |
| 864 | auto CheckOROperandsForSDWA = |
| 865 | [&](const MachineOperand *Op1, const MachineOperand *Op2) -> CheckRetType { |
| 866 | if (!Op1 || !Op1->isReg() || !Op2 || !Op2->isReg()) |
| 867 | return CheckRetType(std::nullopt); |
| 868 | |
| 869 | MachineOperand *Op1Def = findSingleRegDef(Reg: Op1, MRI); |
| 870 | if (!Op1Def) |
| 871 | return CheckRetType(std::nullopt); |
| 872 | |
| 873 | MachineInstr *Op1Inst = Op1Def->getParent(); |
| 874 | if (!isSDWAWithDstSel(Inst: *Op1Inst)) |
| 875 | return CheckRetType(std::nullopt); |
| 876 | |
| 877 | MachineOperand *Op2Def = findSingleRegDef(Reg: Op2, MRI); |
| 878 | if (!Op2Def) |
| 879 | return CheckRetType(std::nullopt); |
| 880 | |
| 881 | return CheckRetType(std::pair(Op1Def, Op2Def)); |
| 882 | }; |
| 883 | |
| 884 | MachineOperand *OrSDWA = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0); |
| 885 | MachineOperand *OrOther = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1); |
| 886 | assert(OrSDWA && OrOther); |
| 887 | auto Res = CheckOROperandsForSDWA(OrSDWA, OrOther); |
| 888 | if (!Res) { |
| 889 | OrSDWA = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1); |
| 890 | OrOther = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0); |
| 891 | assert(OrSDWA && OrOther); |
| 892 | Res = CheckOROperandsForSDWA(OrSDWA, OrOther); |
| 893 | if (!Res) |
| 894 | break; |
| 895 | } |
| 896 | |
| 897 | MachineOperand *OrSDWADef = Res->first; |
| 898 | MachineOperand *OrOtherDef = Res->second; |
| 899 | assert(OrSDWADef && OrOtherDef); |
| 900 | |
| 901 | MachineInstr *SDWAInst = OrSDWADef->getParent(); |
| 902 | MachineInstr *OtherInst = OrOtherDef->getParent(); |
| 903 | |
| 904 | // Check that OtherInstr is actually bitwise compatible with SDWAInst = their |
| 905 | // destination patterns don't overlap. Compatible instruction can be either |
| 906 | // regular instruction with compatible bitness or SDWA instruction with |
| 907 | // correct dst_sel |
| 908 | // SDWAInst | OtherInst bitness / OtherInst dst_sel |
| 909 | // ----------------------------------------------------- |
| 910 | // DWORD | no / no |
| 911 | // WORD_0 | no / BYTE_2/3, WORD_1 |
| 912 | // WORD_1 | 8/16-bit instructions / BYTE_0/1, WORD_0 |
| 913 | // BYTE_0 | no / BYTE_1/2/3, WORD_1 |
| 914 | // BYTE_1 | 8-bit / BYTE_0/2/3, WORD_1 |
| 915 | // BYTE_2 | 8/16-bit / BYTE_0/1/3. WORD_0 |
| 916 | // BYTE_3 | 8/16/24-bit / BYTE_0/1/2, WORD_0 |
| 917 | // E.g. if SDWAInst is v_add_f16_sdwa dst_sel:WORD_1 then v_add_f16 is OK |
| 918 | // but v_add_f32 is not. |
| 919 | |
| 920 | // TODO: add support for non-SDWA instructions as OtherInst. |
| 921 | // For now this only works with SDWA instructions. For regular instructions |
| 922 | // there is no way to determine if the instruction writes only 8/16/24-bit |
| 923 | // out of full register size and all registers are at min 32-bit wide. |
| 924 | if (!isSDWAWithDstSel(Inst: *OtherInst)) |
| 925 | break; |
| 926 | |
| 927 | SdwaSel DstSel = static_cast<SdwaSel>( |
| 928 | TII->getNamedImmOperand(MI: *SDWAInst, OperandName: AMDGPU::OpName::dst_sel)); |
| 929 | SdwaSel OtherDstSel = static_cast<SdwaSel>( |
| 930 | TII->getNamedImmOperand(MI: *OtherInst, OperandName: AMDGPU::OpName::dst_sel)); |
| 931 | |
| 932 | bool DstSelAgree = false; |
| 933 | switch (DstSel) { |
| 934 | case WORD_0: DstSelAgree = ((OtherDstSel == BYTE_2) || |
| 935 | (OtherDstSel == BYTE_3) || |
| 936 | (OtherDstSel == WORD_1)); |
| 937 | break; |
| 938 | case WORD_1: DstSelAgree = ((OtherDstSel == BYTE_0) || |
| 939 | (OtherDstSel == BYTE_1) || |
| 940 | (OtherDstSel == WORD_0)); |
| 941 | break; |
| 942 | case BYTE_0: DstSelAgree = ((OtherDstSel == BYTE_1) || |
| 943 | (OtherDstSel == BYTE_2) || |
| 944 | (OtherDstSel == BYTE_3) || |
| 945 | (OtherDstSel == WORD_1)); |
| 946 | break; |
| 947 | case BYTE_1: DstSelAgree = ((OtherDstSel == BYTE_0) || |
| 948 | (OtherDstSel == BYTE_2) || |
| 949 | (OtherDstSel == BYTE_3) || |
| 950 | (OtherDstSel == WORD_1)); |
| 951 | break; |
| 952 | case BYTE_2: DstSelAgree = ((OtherDstSel == BYTE_0) || |
| 953 | (OtherDstSel == BYTE_1) || |
| 954 | (OtherDstSel == BYTE_3) || |
| 955 | (OtherDstSel == WORD_0)); |
| 956 | break; |
| 957 | case BYTE_3: DstSelAgree = ((OtherDstSel == BYTE_0) || |
| 958 | (OtherDstSel == BYTE_1) || |
| 959 | (OtherDstSel == BYTE_2) || |
| 960 | (OtherDstSel == WORD_0)); |
| 961 | break; |
| 962 | default: DstSelAgree = false; |
| 963 | } |
| 964 | |
| 965 | if (!DstSelAgree) |
| 966 | break; |
| 967 | |
| 968 | // Also OtherInst dst_unused should be UNUSED_PAD |
| 969 | DstUnused OtherDstUnused = static_cast<DstUnused>( |
| 970 | TII->getNamedImmOperand(MI: *OtherInst, OperandName: AMDGPU::OpName::dst_unused)); |
| 971 | if (OtherDstUnused != DstUnused::UNUSED_PAD) |
| 972 | break; |
| 973 | |
| 974 | // Create DstPreserveOperand |
| 975 | MachineOperand *OrDst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst); |
| 976 | assert(OrDst && OrDst->isReg()); |
| 977 | |
| 978 | return std::make_unique<SDWADstPreserveOperand>( |
| 979 | args&: OrDst, args&: OrSDWADef, args&: OrOtherDef, args&: DstSel); |
| 980 | |
| 981 | } |
| 982 | } |
| 983 | |
| 984 | return std::unique_ptr<SDWAOperand>(nullptr); |
| 985 | } |
| 986 | |
| 987 | #if !defined(NDEBUG) |
| 988 | static raw_ostream& operator<<(raw_ostream &OS, const SDWAOperand &Operand) { |
| 989 | Operand.print(OS); |
| 990 | return OS; |
| 991 | } |
| 992 | #endif |
| 993 | |
| 994 | void SIPeepholeSDWA::matchSDWAOperands(MachineBasicBlock &MBB) { |
| 995 | for (MachineInstr &MI : MBB) { |
| 996 | if (auto Operand = matchSDWAOperand(MI)) { |
| 997 | LLVM_DEBUG(dbgs() << "Match: " << MI << "To: " << *Operand << '\n'); |
| 998 | SDWAOperands[&MI] = std::move(Operand); |
| 999 | ++NumSDWAPatternsFound; |
| 1000 | } |
| 1001 | } |
| 1002 | } |
| 1003 | |
| 1004 | // Convert the V_ADD_CO_U32_e64 into V_ADD_CO_U32_e32. This allows |
| 1005 | // isConvertibleToSDWA to perform its transformation on V_ADD_CO_U32_e32 into |
| 1006 | // V_ADD_CO_U32_sdwa. |
| 1007 | // |
| 1008 | // We are transforming from a VOP3 into a VOP2 form of the instruction. |
| 1009 | // %19:vgpr_32 = V_AND_B32_e32 255, |
| 1010 | // killed %16:vgpr_32, implicit $exec |
| 1011 | // %47:vgpr_32, %49:sreg_64_xexec = V_ADD_CO_U32_e64 |
| 1012 | // %26.sub0:vreg_64, %19:vgpr_32, implicit $exec |
| 1013 | // %48:vgpr_32, dead %50:sreg_64_xexec = V_ADDC_U32_e64 |
| 1014 | // %26.sub1:vreg_64, %54:vgpr_32, killed %49:sreg_64_xexec, implicit $exec |
| 1015 | // |
| 1016 | // becomes |
| 1017 | // %47:vgpr_32 = V_ADD_CO_U32_sdwa |
| 1018 | // 0, %26.sub0:vreg_64, 0, killed %16:vgpr_32, 0, 6, 0, 6, 0, |
| 1019 | // implicit-def $vcc, implicit $exec |
| 1020 | // %48:vgpr_32, dead %50:sreg_64_xexec = V_ADDC_U32_e64 |
| 1021 | // %26.sub1:vreg_64, %54:vgpr_32, killed $vcc, implicit $exec |
| 1022 | void SIPeepholeSDWA::pseudoOpConvertToVOP2(MachineInstr &MI, |
| 1023 | const GCNSubtarget &ST) const { |
| 1024 | int Opc = MI.getOpcode(); |
| 1025 | assert((Opc == AMDGPU::V_ADD_CO_U32_e64 || Opc == AMDGPU::V_SUB_CO_U32_e64) && |
| 1026 | "Currently only handles V_ADD_CO_U32_e64 or V_SUB_CO_U32_e64" ); |
| 1027 | |
| 1028 | // Can the candidate MI be shrunk? |
| 1029 | if (!TII->canShrink(MI, MRI: *MRI)) |
| 1030 | return; |
| 1031 | Opc = AMDGPU::getVOPe32(Opcode: Opc); |
| 1032 | // Find the related ADD instruction. |
| 1033 | const MachineOperand *Sdst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::sdst); |
| 1034 | if (!Sdst) |
| 1035 | return; |
| 1036 | MachineOperand *NextOp = findSingleRegUse(Reg: Sdst, MRI); |
| 1037 | if (!NextOp) |
| 1038 | return; |
| 1039 | MachineInstr &MISucc = *NextOp->getParent(); |
| 1040 | |
| 1041 | // Make sure the carry in/out are subsequently unused. |
| 1042 | MachineOperand *CarryIn = TII->getNamedOperand(MI&: MISucc, OperandName: AMDGPU::OpName::src2); |
| 1043 | if (!CarryIn) |
| 1044 | return; |
| 1045 | MachineOperand *CarryOut = TII->getNamedOperand(MI&: MISucc, OperandName: AMDGPU::OpName::sdst); |
| 1046 | if (!CarryOut) |
| 1047 | return; |
| 1048 | if (!MRI->hasOneNonDBGUse(RegNo: CarryIn->getReg()) || |
| 1049 | !MRI->use_nodbg_empty(RegNo: CarryOut->getReg())) |
| 1050 | return; |
| 1051 | // Make sure VCC or its subregs are dead before MI. |
| 1052 | MachineBasicBlock &MBB = *MI.getParent(); |
| 1053 | if (MISucc.getParent() != &MBB) |
| 1054 | return; // Loop depends on MI and MISucc in same MBB. |
| 1055 | MachineBasicBlock::LivenessQueryResult Liveness = |
| 1056 | MBB.computeRegisterLiveness(TRI, Reg: AMDGPU::VCC, Before: MI, Neighborhood: 25); |
| 1057 | if (Liveness != MachineBasicBlock::LQR_Dead) |
| 1058 | return; |
| 1059 | // Check if VCC is referenced in range of (MI,MISucc]. |
| 1060 | for (auto I = std::next(x: MI.getIterator()), E = MISucc.getIterator(); |
| 1061 | I != E; ++I) { |
| 1062 | if (I->modifiesRegister(Reg: AMDGPU::VCC, TRI)) |
| 1063 | return; |
| 1064 | } |
| 1065 | |
| 1066 | // Replace MI with V_{SUB|ADD}_I32_e32 |
| 1067 | BuildMI(BB&: MBB, I&: MI, MIMD: MI.getDebugLoc(), MCID: TII->get(Opcode: Opc)) |
| 1068 | .add(MO: *TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst)) |
| 1069 | .add(MO: *TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0)) |
| 1070 | .add(MO: *TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1)) |
| 1071 | .setMIFlags(MI.getFlags()); |
| 1072 | |
| 1073 | MI.eraseFromParent(); |
| 1074 | |
| 1075 | // Since the carry output of MI is now VCC, update its use in MISucc. |
| 1076 | |
| 1077 | MISucc.substituteRegister(FromReg: CarryIn->getReg(), ToReg: TRI->getVCC(), SubIdx: 0, RegInfo: *TRI); |
| 1078 | } |
| 1079 | |
| 1080 | /// Try to convert an \p MI in VOP3 which takes an src2 carry-in |
| 1081 | /// operand into the corresponding VOP2 form which expects the |
| 1082 | /// argument in VCC. To this end, add an copy from the carry-in to |
| 1083 | /// VCC. The conversion will only be applied if \p MI can be shrunk |
| 1084 | /// to VOP2 and if VCC can be proven to be dead before \p MI. |
| 1085 | void SIPeepholeSDWA::convertVcndmaskToVOP2(MachineInstr &MI, |
| 1086 | const GCNSubtarget &ST) const { |
| 1087 | assert(MI.getOpcode() == AMDGPU::V_CNDMASK_B32_e64); |
| 1088 | |
| 1089 | LLVM_DEBUG(dbgs() << "Attempting VOP2 conversion: " << MI); |
| 1090 | if (!TII->canShrink(MI, MRI: *MRI)) { |
| 1091 | LLVM_DEBUG(dbgs() << "Cannot shrink instruction\n" ); |
| 1092 | return; |
| 1093 | } |
| 1094 | |
| 1095 | const MachineOperand &CarryIn = |
| 1096 | *TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src2); |
| 1097 | Register CarryReg = CarryIn.getReg(); |
| 1098 | MachineInstr *CarryDef = MRI->getVRegDef(Reg: CarryReg); |
| 1099 | if (!CarryDef) { |
| 1100 | LLVM_DEBUG(dbgs() << "Missing carry-in operand definition\n" ); |
| 1101 | return; |
| 1102 | } |
| 1103 | |
| 1104 | // Make sure VCC or its subregs are dead before MI. |
| 1105 | MCRegister Vcc = TRI->getVCC(); |
| 1106 | MachineBasicBlock &MBB = *MI.getParent(); |
| 1107 | MachineBasicBlock::LivenessQueryResult Liveness = |
| 1108 | MBB.computeRegisterLiveness(TRI, Reg: Vcc, Before: MI); |
| 1109 | if (Liveness != MachineBasicBlock::LQR_Dead) { |
| 1110 | LLVM_DEBUG(dbgs() << "VCC not known to be dead before instruction\n" ); |
| 1111 | return; |
| 1112 | } |
| 1113 | |
| 1114 | BuildMI(BB&: MBB, I&: MI, MIMD: MI.getDebugLoc(), MCID: TII->get(Opcode: AMDGPU::COPY), DestReg: Vcc).add(MO: CarryIn); |
| 1115 | |
| 1116 | auto Converted = BuildMI(BB&: MBB, I&: MI, MIMD: MI.getDebugLoc(), |
| 1117 | MCID: TII->get(Opcode: AMDGPU::getVOPe32(Opcode: MI.getOpcode()))) |
| 1118 | .add(MO: *TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst)) |
| 1119 | .add(MO: *TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0)) |
| 1120 | .add(MO: *TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1)) |
| 1121 | .setMIFlags(MI.getFlags()); |
| 1122 | TII->fixImplicitOperands(MI&: *Converted); |
| 1123 | LLVM_DEBUG(dbgs() << "Converted to VOP2: " << *Converted); |
| 1124 | (void)Converted; |
| 1125 | MI.eraseFromParent(); |
| 1126 | } |
| 1127 | |
| 1128 | namespace { |
| 1129 | bool isConvertibleToSDWA(MachineInstr &MI, |
| 1130 | const GCNSubtarget &ST, |
| 1131 | const SIInstrInfo* TII) { |
| 1132 | // Check if this is already an SDWA instruction |
| 1133 | unsigned Opc = MI.getOpcode(); |
| 1134 | if (TII->isSDWA(Opcode: Opc)) |
| 1135 | return true; |
| 1136 | |
| 1137 | // Can only be handled after ealier conversion to |
| 1138 | // AMDGPU::V_CNDMASK_B32_e32 which is not always possible. |
| 1139 | if (Opc == AMDGPU::V_CNDMASK_B32_e64) |
| 1140 | return false; |
| 1141 | |
| 1142 | // Check if this instruction has opcode that supports SDWA |
| 1143 | if (AMDGPU::getSDWAOp(Opcode: Opc) == -1) |
| 1144 | Opc = AMDGPU::getVOPe32(Opcode: Opc); |
| 1145 | |
| 1146 | if (AMDGPU::getSDWAOp(Opcode: Opc) == -1) |
| 1147 | return false; |
| 1148 | |
| 1149 | if (!ST.hasSDWAOmod() && TII->hasModifiersSet(MI, OpName: AMDGPU::OpName::omod)) |
| 1150 | return false; |
| 1151 | |
| 1152 | if (TII->isVOPC(Opcode: Opc)) { |
| 1153 | if (!ST.hasSDWASdst()) { |
| 1154 | const MachineOperand *SDst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::sdst); |
| 1155 | if (SDst && (SDst->getReg() != AMDGPU::VCC && |
| 1156 | SDst->getReg() != AMDGPU::VCC_LO)) |
| 1157 | return false; |
| 1158 | } |
| 1159 | |
| 1160 | if (!ST.hasSDWAOutModsVOPC() && |
| 1161 | (TII->hasModifiersSet(MI, OpName: AMDGPU::OpName::clamp) || |
| 1162 | TII->hasModifiersSet(MI, OpName: AMDGPU::OpName::omod))) |
| 1163 | return false; |
| 1164 | |
| 1165 | } else if (TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::sdst) || |
| 1166 | !TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst)) { |
| 1167 | return false; |
| 1168 | } |
| 1169 | |
| 1170 | if (!ST.hasSDWAMac() && (Opc == AMDGPU::V_FMAC_F16_e32 || |
| 1171 | Opc == AMDGPU::V_FMAC_F32_e32 || |
| 1172 | Opc == AMDGPU::V_MAC_F16_e32 || |
| 1173 | Opc == AMDGPU::V_MAC_F32_e32)) |
| 1174 | return false; |
| 1175 | |
| 1176 | // Check if target supports this SDWA opcode |
| 1177 | if (TII->pseudoToMCOpcode(Opcode: Opc) == -1 || |
| 1178 | TII->pseudoToMCOpcode(Opcode: AMDGPU::getSDWAOp(Opcode: Opc)) == -1) |
| 1179 | return false; |
| 1180 | |
| 1181 | if (MachineOperand *Src0 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0)) { |
| 1182 | if (!Src0->isReg() && !Src0->isImm()) |
| 1183 | return false; |
| 1184 | } |
| 1185 | |
| 1186 | if (MachineOperand *Src1 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1)) { |
| 1187 | if (!Src1->isReg() && !Src1->isImm()) |
| 1188 | return false; |
| 1189 | } |
| 1190 | |
| 1191 | return true; |
| 1192 | } |
| 1193 | } // namespace |
| 1194 | |
| 1195 | MachineInstr *SIPeepholeSDWA::createSDWAVersion(MachineInstr &MI) { |
| 1196 | unsigned Opcode = MI.getOpcode(); |
| 1197 | assert(!TII->isSDWA(Opcode)); |
| 1198 | |
| 1199 | int SDWAOpcode = AMDGPU::getSDWAOp(Opcode); |
| 1200 | if (SDWAOpcode == -1) |
| 1201 | SDWAOpcode = AMDGPU::getSDWAOp(Opcode: AMDGPU::getVOPe32(Opcode)); |
| 1202 | assert(SDWAOpcode != -1); |
| 1203 | |
| 1204 | const MCInstrDesc &SDWADesc = TII->get(Opcode: SDWAOpcode); |
| 1205 | |
| 1206 | // Create SDWA version of instruction MI and initialize its operands |
| 1207 | MachineInstrBuilder SDWAInst = |
| 1208 | BuildMI(BB&: *MI.getParent(), I&: MI, MIMD: MI.getDebugLoc(), MCID: SDWADesc) |
| 1209 | .setMIFlags(MI.getFlags()); |
| 1210 | |
| 1211 | // Copy dst, if it is present in original then should also be present in SDWA |
| 1212 | MachineOperand *Dst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::vdst); |
| 1213 | if (Dst) { |
| 1214 | assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::vdst)); |
| 1215 | SDWAInst.add(MO: *Dst); |
| 1216 | } else if ((Dst = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::sdst))) { |
| 1217 | assert(Dst && AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::sdst)); |
| 1218 | SDWAInst.add(MO: *Dst); |
| 1219 | } else { |
| 1220 | assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::sdst)); |
| 1221 | SDWAInst.addReg(RegNo: TRI->getVCC(), Flags: RegState::Define); |
| 1222 | } |
| 1223 | |
| 1224 | // Copy src0, initialize src0_modifiers. All sdwa instructions has src0 and |
| 1225 | // src0_modifiers (except for v_nop_sdwa, but it can't get here) |
| 1226 | MachineOperand *Src0 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0); |
| 1227 | assert(Src0 && AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src0) && |
| 1228 | AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src0_modifiers)); |
| 1229 | if (auto *Mod = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0_modifiers)) |
| 1230 | SDWAInst.addImm(Val: Mod->getImm()); |
| 1231 | else |
| 1232 | SDWAInst.addImm(Val: 0); |
| 1233 | SDWAInst.add(MO: *Src0); |
| 1234 | |
| 1235 | // Copy src1 if present, initialize src1_modifiers. |
| 1236 | MachineOperand *Src1 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1); |
| 1237 | if (Src1) { |
| 1238 | assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src1) && |
| 1239 | AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src1_modifiers)); |
| 1240 | if (auto *Mod = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1_modifiers)) |
| 1241 | SDWAInst.addImm(Val: Mod->getImm()); |
| 1242 | else |
| 1243 | SDWAInst.addImm(Val: 0); |
| 1244 | SDWAInst.add(MO: *Src1); |
| 1245 | } |
| 1246 | |
| 1247 | if (SDWAOpcode == AMDGPU::V_FMAC_F16_sdwa || |
| 1248 | SDWAOpcode == AMDGPU::V_FMAC_F32_sdwa || |
| 1249 | SDWAOpcode == AMDGPU::V_MAC_F16_sdwa || |
| 1250 | SDWAOpcode == AMDGPU::V_MAC_F32_sdwa) { |
| 1251 | // v_mac_f16/32 has additional src2 operand tied to vdst |
| 1252 | MachineOperand *Src2 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src2); |
| 1253 | assert(Src2); |
| 1254 | SDWAInst.add(MO: *Src2); |
| 1255 | } |
| 1256 | |
| 1257 | // Copy clamp if present, initialize otherwise |
| 1258 | assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::clamp)); |
| 1259 | MachineOperand *Clamp = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::clamp); |
| 1260 | if (Clamp) { |
| 1261 | SDWAInst.add(MO: *Clamp); |
| 1262 | } else { |
| 1263 | SDWAInst.addImm(Val: 0); |
| 1264 | } |
| 1265 | |
| 1266 | // Copy omod if present, initialize otherwise if needed |
| 1267 | if (AMDGPU::hasNamedOperand(Opcode: SDWAOpcode, NamedIdx: AMDGPU::OpName::omod)) { |
| 1268 | MachineOperand *OMod = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::omod); |
| 1269 | if (OMod) { |
| 1270 | SDWAInst.add(MO: *OMod); |
| 1271 | } else { |
| 1272 | SDWAInst.addImm(Val: 0); |
| 1273 | } |
| 1274 | } |
| 1275 | |
| 1276 | // Initialize SDWA specific operands |
| 1277 | if (AMDGPU::hasNamedOperand(Opcode: SDWAOpcode, NamedIdx: AMDGPU::OpName::dst_sel)) |
| 1278 | SDWAInst.addImm(Val: AMDGPU::SDWA::SdwaSel::DWORD); |
| 1279 | |
| 1280 | if (AMDGPU::hasNamedOperand(Opcode: SDWAOpcode, NamedIdx: AMDGPU::OpName::dst_unused)) |
| 1281 | SDWAInst.addImm(Val: AMDGPU::SDWA::DstUnused::UNUSED_PAD); |
| 1282 | |
| 1283 | assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src0_sel)); |
| 1284 | SDWAInst.addImm(Val: AMDGPU::SDWA::SdwaSel::DWORD); |
| 1285 | |
| 1286 | if (Src1) { |
| 1287 | assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src1_sel)); |
| 1288 | SDWAInst.addImm(Val: AMDGPU::SDWA::SdwaSel::DWORD); |
| 1289 | } |
| 1290 | |
| 1291 | // Check for a preserved register that needs to be copied. |
| 1292 | MachineInstr *Ret = SDWAInst.getInstr(); |
| 1293 | TII->fixImplicitOperands(MI&: *Ret); |
| 1294 | return Ret; |
| 1295 | } |
| 1296 | |
| 1297 | bool SIPeepholeSDWA::convertToSDWA(MachineInstr &MI, |
| 1298 | const SDWAOperandsVector &SDWAOperands) { |
| 1299 | LLVM_DEBUG(dbgs() << "Convert instruction:" << MI); |
| 1300 | |
| 1301 | MachineInstr *SDWAInst; |
| 1302 | if (TII->isSDWA(Opcode: MI.getOpcode())) { |
| 1303 | // Clone the instruction to allow revoking changes |
| 1304 | // made to MI during the processing of the operands |
| 1305 | // if the conversion fails. |
| 1306 | SDWAInst = MI.getMF()->CloneMachineInstr(Orig: &MI); |
| 1307 | MI.getParent()->insert(I: MI.getIterator(), M: SDWAInst); |
| 1308 | } else { |
| 1309 | SDWAInst = createSDWAVersion(MI); |
| 1310 | } |
| 1311 | |
| 1312 | // Apply all sdwa operand patterns. |
| 1313 | bool Converted = false; |
| 1314 | for (auto &Operand : SDWAOperands) { |
| 1315 | LLVM_DEBUG(dbgs() << *SDWAInst << "\nOperand: " << *Operand); |
| 1316 | // There should be no intersection between SDWA operands and potential MIs |
| 1317 | // e.g.: |
| 1318 | // v_and_b32 v0, 0xff, v1 -> src:v1 sel:BYTE_0 |
| 1319 | // v_and_b32 v2, 0xff, v0 -> src:v0 sel:BYTE_0 |
| 1320 | // v_add_u32 v3, v4, v2 |
| 1321 | // |
| 1322 | // In that example it is possible that we would fold 2nd instruction into |
| 1323 | // 3rd (v_add_u32_sdwa) and then try to fold 1st instruction into 2nd (that |
| 1324 | // was already destroyed). So if SDWAOperand is also a potential MI then do |
| 1325 | // not apply it. |
| 1326 | if (PotentialMatches.count(Key: Operand->getParentInst()) == 0) |
| 1327 | Converted |= Operand->convertToSDWA(MI&: *SDWAInst, TII); |
| 1328 | } |
| 1329 | |
| 1330 | if (!Converted) { |
| 1331 | SDWAInst->eraseFromParent(); |
| 1332 | return false; |
| 1333 | } |
| 1334 | |
| 1335 | ConvertedInstructions.push_back(Elt: SDWAInst); |
| 1336 | for (MachineOperand &MO : SDWAInst->uses()) { |
| 1337 | if (!MO.isReg()) |
| 1338 | continue; |
| 1339 | |
| 1340 | MRI->clearKillFlags(Reg: MO.getReg()); |
| 1341 | } |
| 1342 | LLVM_DEBUG(dbgs() << "\nInto:" << *SDWAInst << '\n'); |
| 1343 | ++NumSDWAInstructionsPeepholed; |
| 1344 | |
| 1345 | MI.eraseFromParent(); |
| 1346 | return true; |
| 1347 | } |
| 1348 | |
| 1349 | // If an instruction was converted to SDWA it should not have immediates or SGPR |
| 1350 | // operands (allowed one SGPR on GFX9). Copy its scalar operands into VGPRs. |
| 1351 | void SIPeepholeSDWA::legalizeScalarOperands(MachineInstr &MI, |
| 1352 | const GCNSubtarget &ST) const { |
| 1353 | const MCInstrDesc &Desc = TII->get(Opcode: MI.getOpcode()); |
| 1354 | unsigned ConstantBusCount = 0; |
| 1355 | for (MachineOperand &Op : MI.explicit_uses()) { |
| 1356 | if (Op.isReg()) { |
| 1357 | if (TRI->isVGPR(MRI: *MRI, Reg: Op.getReg())) |
| 1358 | continue; |
| 1359 | |
| 1360 | if (ST.hasSDWAScalar() && ConstantBusCount == 0) { |
| 1361 | ++ConstantBusCount; |
| 1362 | continue; |
| 1363 | } |
| 1364 | } else if (!Op.isImm()) |
| 1365 | continue; |
| 1366 | |
| 1367 | unsigned I = Op.getOperandNo(); |
| 1368 | const TargetRegisterClass *OpRC = TII->getRegClass(MCID: Desc, OpNum: I); |
| 1369 | if (!OpRC || !TRI->isVSSuperClass(RC: OpRC)) |
| 1370 | continue; |
| 1371 | |
| 1372 | Register VGPR = MRI->createVirtualRegister(RegClass: &AMDGPU::VGPR_32RegClass); |
| 1373 | auto Copy = BuildMI(BB&: *MI.getParent(), I: MI.getIterator(), MIMD: MI.getDebugLoc(), |
| 1374 | MCID: TII->get(Opcode: AMDGPU::V_MOV_B32_e32), DestReg: VGPR); |
| 1375 | if (Op.isImm()) |
| 1376 | Copy.addImm(Val: Op.getImm()); |
| 1377 | else if (Op.isReg()) |
| 1378 | Copy.addReg(RegNo: Op.getReg(), Flags: getKillRegState(B: Op.isKill()), SubReg: Op.getSubReg()); |
| 1379 | Op.ChangeToRegister(Reg: VGPR, isDef: false); |
| 1380 | } |
| 1381 | } |
| 1382 | |
| 1383 | // Re-fold the masked high-half pack (hi << 16) | (z & 0xffff) into a single |
| 1384 | // v_or_b32_sdwa src1_sel:WORD_0, which ISel's fused v_lshl_or_b32 blocks. |
| 1385 | bool SIPeepholeSDWA::splitLshlOrForSDWA(MachineBasicBlock &MBB) { |
| 1386 | struct Candidate { |
| 1387 | MachineInstr *LshlOr; |
| 1388 | MachineInstr *AndMI; |
| 1389 | MachineOperand *Hi; |
| 1390 | MachineOperand *ValSrc; |
| 1391 | }; |
| 1392 | SmallVector<Candidate, 4> Candidates; |
| 1393 | |
| 1394 | for (MachineInstr &MI : MBB) { |
| 1395 | if (MI.getOpcode() != AMDGPU::V_LSHL_OR_B32_e64) |
| 1396 | continue; |
| 1397 | |
| 1398 | MachineOperand *Shift = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src1); |
| 1399 | std::optional<int64_t> ShiftImm = foldToImm(Op: *Shift); |
| 1400 | if (!ShiftImm || *ShiftImm != 16) |
| 1401 | continue; |
| 1402 | |
| 1403 | MachineOperand *Hi = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src0); |
| 1404 | MachineOperand *Src2 = TII->getNamedOperand(MI, OperandName: AMDGPU::OpName::src2); |
| 1405 | // Src2 must be a virtual reg so getVRegDef below is valid. |
| 1406 | if (!Hi->isReg() || !Src2->isReg() || !Src2->getReg().isVirtual()) |
| 1407 | continue; |
| 1408 | |
| 1409 | // The 0xffff mask must come from a single-use v_and so it can be dropped. |
| 1410 | if (!MRI->hasOneNonDBGUse(RegNo: Src2->getReg())) |
| 1411 | continue; |
| 1412 | MachineInstr *AndMI = MRI->getVRegDef(Reg: Src2->getReg()); |
| 1413 | if (!AndMI) |
| 1414 | continue; |
| 1415 | std::optional<std::pair<MachineOperand *, SdwaSel>> Mask = |
| 1416 | matchAndMask(MI&: *AndMI); |
| 1417 | if (!Mask || Mask->second != WORD_0) |
| 1418 | continue; |
| 1419 | MachineOperand *ValSrc = Mask->first; |
| 1420 | if (!ValSrc->isReg() || !TRI->isVGPR(MRI: *MRI, Reg: ValSrc->getReg())) |
| 1421 | continue; |
| 1422 | |
| 1423 | Candidates.push_back(Elt: {.LshlOr: &MI, .AndMI: AndMI, .Hi: Hi, .ValSrc: ValSrc}); |
| 1424 | } |
| 1425 | |
| 1426 | for (const Candidate &C : Candidates) { |
| 1427 | MachineOperand *Dst = TII->getNamedOperand(MI&: *C.LshlOr, OperandName: AMDGPU::OpName::vdst); |
| 1428 | |
| 1429 | Register ShiftReg = MRI->createVirtualRegister(RegClass: &AMDGPU::VGPR_32RegClass); |
| 1430 | BuildMI(BB&: *C.LshlOr->getParent(), I&: *C.LshlOr, MIMD: C.LshlOr->getDebugLoc(), |
| 1431 | MCID: TII->get(Opcode: AMDGPU::V_LSHLREV_B32_e64), DestReg: ShiftReg) |
| 1432 | .addImm(Val: 16) |
| 1433 | .add(MO: *C.Hi); |
| 1434 | |
| 1435 | // vdst, src0_mods, src0, src1_mods, src1, clamp, dst_sel, dst_unused, |
| 1436 | // src0_sel, src1_sel. |
| 1437 | BuildMI(BB&: *C.LshlOr->getParent(), I&: *C.LshlOr, MIMD: C.LshlOr->getDebugLoc(), |
| 1438 | MCID: TII->get(Opcode: AMDGPU::V_OR_B32_sdwa)) |
| 1439 | .add(MO: *Dst) |
| 1440 | .addImm(Val: 0) |
| 1441 | .addReg(RegNo: ShiftReg) |
| 1442 | .addImm(Val: 0) |
| 1443 | .add(MO: *C.ValSrc) |
| 1444 | .addImm(Val: 0) |
| 1445 | .addImm(Val: DWORD) |
| 1446 | .addImm(Val: UNUSED_PAD) |
| 1447 | .addImm(Val: DWORD) |
| 1448 | .addImm(Val: WORD_0); |
| 1449 | |
| 1450 | MRI->clearKillFlags(Reg: C.ValSrc->getReg()); |
| 1451 | C.LshlOr->eraseFromParent(); |
| 1452 | C.AndMI->eraseFromParent(); |
| 1453 | } |
| 1454 | |
| 1455 | return !Candidates.empty(); |
| 1456 | } |
| 1457 | |
| 1458 | bool SIPeepholeSDWALegacy::runOnMachineFunction(MachineFunction &MF) { |
| 1459 | if (skipFunction(F: MF.getFunction())) |
| 1460 | return false; |
| 1461 | |
| 1462 | return SIPeepholeSDWA().run(MF); |
| 1463 | } |
| 1464 | |
| 1465 | bool SIPeepholeSDWA::run(MachineFunction &MF) { |
| 1466 | const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>(); |
| 1467 | |
| 1468 | if (!ST.hasSDWA()) |
| 1469 | return false; |
| 1470 | |
| 1471 | MRI = &MF.getRegInfo(); |
| 1472 | TRI = ST.getRegisterInfo(); |
| 1473 | TII = ST.getInstrInfo(); |
| 1474 | |
| 1475 | // Find all SDWA operands in MF. |
| 1476 | bool Ret = false; |
| 1477 | for (MachineBasicBlock &MBB : MF) { |
| 1478 | bool Changed = false; |
| 1479 | do { |
| 1480 | Ret |= splitLshlOrForSDWA(MBB); |
| 1481 | |
| 1482 | // Preprocess the ADD/SUB pairs so they could be SDWA'ed. |
| 1483 | // Look for a possible ADD or SUB that resulted from a previously lowered |
| 1484 | // V_{ADD|SUB}_U64_PSEUDO. The function pseudoOpConvertToVOP2 |
| 1485 | // lowers the pair of instructions into e32 form. |
| 1486 | matchSDWAOperands(MBB); |
| 1487 | for (const auto &OperandPair : SDWAOperands) { |
| 1488 | const auto &Operand = OperandPair.second; |
| 1489 | MachineInstr *PotentialMI = Operand->potentialToConvert(TII, ST); |
| 1490 | if (!PotentialMI) |
| 1491 | continue; |
| 1492 | |
| 1493 | switch (PotentialMI->getOpcode()) { |
| 1494 | case AMDGPU::V_ADD_CO_U32_e64: |
| 1495 | case AMDGPU::V_SUB_CO_U32_e64: |
| 1496 | pseudoOpConvertToVOP2(MI&: *PotentialMI, ST); |
| 1497 | break; |
| 1498 | case AMDGPU::V_CNDMASK_B32_e64: |
| 1499 | convertVcndmaskToVOP2(MI&: *PotentialMI, ST); |
| 1500 | break; |
| 1501 | }; |
| 1502 | } |
| 1503 | SDWAOperands.clear(); |
| 1504 | |
| 1505 | // Generate potential match list. |
| 1506 | matchSDWAOperands(MBB); |
| 1507 | |
| 1508 | for (const auto &OperandPair : SDWAOperands) { |
| 1509 | const auto &Operand = OperandPair.second; |
| 1510 | MachineInstr *PotentialMI = |
| 1511 | Operand->potentialToConvert(TII, ST, PotentialMatches: &PotentialMatches); |
| 1512 | |
| 1513 | if (PotentialMI && isConvertibleToSDWA(MI&: *PotentialMI, ST, TII)) |
| 1514 | PotentialMatches[PotentialMI].push_back(Elt: Operand.get()); |
| 1515 | } |
| 1516 | |
| 1517 | for (auto &PotentialPair : PotentialMatches) { |
| 1518 | MachineInstr &PotentialMI = *PotentialPair.first; |
| 1519 | convertToSDWA(MI&: PotentialMI, SDWAOperands: PotentialPair.second); |
| 1520 | } |
| 1521 | |
| 1522 | PotentialMatches.clear(); |
| 1523 | SDWAOperands.clear(); |
| 1524 | |
| 1525 | Changed = !ConvertedInstructions.empty(); |
| 1526 | |
| 1527 | if (Changed) |
| 1528 | Ret = true; |
| 1529 | while (!ConvertedInstructions.empty()) |
| 1530 | legalizeScalarOperands(MI&: *ConvertedInstructions.pop_back_val(), ST); |
| 1531 | } while (Changed); |
| 1532 | } |
| 1533 | |
| 1534 | return Ret; |
| 1535 | } |
| 1536 | |
| 1537 | PreservedAnalyses SIPeepholeSDWAPass::run(MachineFunction &MF, |
| 1538 | MachineFunctionAnalysisManager &) { |
| 1539 | if (MF.getFunction().hasOptNone() || !SIPeepholeSDWA().run(MF)) |
| 1540 | return PreservedAnalyses::all(); |
| 1541 | |
| 1542 | PreservedAnalyses PA = getMachineFunctionPassPreservedAnalyses(); |
| 1543 | PA.preserveSet<CFGAnalyses>(); |
| 1544 | return PA; |
| 1545 | } |
| 1546 | |