1//===- AMDGPURegisterBankInfo.cpp -------------------------------*- C++ -*-==//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9/// This file implements the targeting of the RegisterBankInfo class for
10/// AMDGPU.
11///
12/// \par
13///
14/// AMDGPU has unique register bank constraints that require special high level
15/// strategies to deal with. There are two main true physical register banks
16/// VGPR (vector), and SGPR (scalar). Additionally the VCC register bank is a
17/// sort of pseudo-register bank needed to represent SGPRs used in a vector
18/// boolean context. There is also the AGPR bank, which is a special purpose
19/// physical register bank present on some subtargets.
20///
21/// Copying from VGPR to SGPR is generally illegal, unless the value is known to
22/// be uniform. It is generally not valid to legalize operands by inserting
23/// copies as on other targets. Operations which require uniform, SGPR operands
24/// generally require scalarization by repeatedly executing the instruction,
25/// activating each set of lanes using a unique set of input values. This is
26/// referred to as a waterfall loop.
27///
28/// \par Booleans
29///
30/// Booleans (s1 values) requires special consideration. A vector compare result
31/// is naturally a bitmask with one bit per lane, in a 32 or 64-bit
32/// register. These are represented with the VCC bank. During selection, we need
33/// to be able to unambiguously go back from a register class to a register
34/// bank. To distinguish whether an SGPR should use the SGPR or VCC register
35/// bank, we need to know the use context type. An SGPR s1 value always means a
36/// VCC bank value, otherwise it will be the SGPR bank. A scalar compare sets
37/// SCC, which is a 1-bit unaddressable register. This will need to be copied to
38/// a 32-bit virtual register. Taken together, this means we need to adjust the
39/// type of boolean operations to be regbank legal. All SALU booleans need to be
40/// widened to 32-bits, and all VALU booleans need to be s1 values.
41///
42/// A noteworthy exception to the s1-means-vcc rule is for legalization artifact
43/// casts. G_TRUNC s1 results, and G_SEXT/G_ZEXT/G_ANYEXT sources are never vcc
44/// bank. A non-boolean source (such as a truncate from a 1-bit load from
45/// memory) will require a copy to the VCC bank which will require clearing the
46/// high bits and inserting a compare.
47///
48/// \par Constant bus restriction
49///
50/// VALU instructions have a limitation known as the constant bus
51/// restriction. Most VALU instructions can use SGPR operands, but may read at
52/// most 1 SGPR or constant literal value (this to 2 in gfx10 for most
53/// instructions). This is one unique SGPR, so the same SGPR may be used for
54/// multiple operands. From a register bank perspective, any combination of
55/// operands should be legal as an SGPR, but this is contextually dependent on
56/// the SGPR operands all being the same register. There is therefore optimal to
57/// choose the SGPR with the most uses to minimize the number of copies.
58///
59/// We avoid trying to solve this problem in RegBankSelect. Any VALU G_*
60/// operation should have its source operands all mapped to VGPRs (except for
61/// VCC), inserting copies from any SGPR operands. This the most trivial legal
62/// mapping. Anything beyond the simplest 1:1 instruction selection would be too
63/// complicated to solve here. Every optimization pattern or instruction
64/// selected to multiple outputs would have to enforce this rule, and there
65/// would be additional complexity in tracking this rule for every G_*
66/// operation. By forcing all inputs to VGPRs, it also simplifies the task of
67/// picking the optimal operand combination from a post-isel optimization pass.
68///
69//===----------------------------------------------------------------------===//
70
71#include "AMDGPURegisterBankInfo.h"
72
73#include "AMDGPUGlobalISelUtils.h"
74#include "AMDGPUInstrInfo.h"
75#include "AMDGPULaneMaskUtils.h"
76#include "GCNSubtarget.h"
77#include "SIMachineFunctionInfo.h"
78#include "SIRegisterInfo.h"
79#include "llvm/CodeGen/GlobalISel/GenericMachineInstrs.h"
80#include "llvm/CodeGen/GlobalISel/LegalizerHelper.h"
81#include "llvm/CodeGen/GlobalISel/MIPatternMatch.h"
82#include "llvm/CodeGen/GlobalISel/MachineIRBuilder.h"
83#include "llvm/CodeGen/RegisterBank.h"
84#include "llvm/IR/IntrinsicsAMDGPU.h"
85
86#define GET_TARGET_REGBANK_IMPL
87#include "AMDGPUGenRegisterBank.inc"
88
89// This file will be TableGen'ed at some point.
90#include "AMDGPUGenRegisterBankInfo.def"
91
92using namespace llvm;
93using namespace MIPatternMatch;
94
95namespace {
96
97// Observer to apply a register bank to new registers created by LegalizerHelper.
98class ApplyRegBankMapping final : public GISelChangeObserver {
99private:
100 MachineIRBuilder &B;
101 const AMDGPURegisterBankInfo &RBI;
102 MachineRegisterInfo &MRI;
103 const RegisterBank *NewBank;
104 SmallVector<MachineInstr *, 4> NewInsts;
105
106public:
107 ApplyRegBankMapping(MachineIRBuilder &B, const AMDGPURegisterBankInfo &RBI_,
108 MachineRegisterInfo &MRI_, const RegisterBank *RB)
109 : B(B), RBI(RBI_), MRI(MRI_), NewBank(RB) {
110 assert(!B.isObservingChanges());
111 B.setChangeObserver(*this);
112 }
113
114 ~ApplyRegBankMapping() override {
115 for (MachineInstr *MI : NewInsts)
116 applyBank(MI&: *MI);
117
118 B.stopObservingChanges();
119 }
120
121 /// Set any registers that don't have a set register class or bank to SALU.
122 void applyBank(MachineInstr &MI) {
123 const unsigned Opc = MI.getOpcode();
124 if (Opc == AMDGPU::G_ANYEXT || Opc == AMDGPU::G_ZEXT ||
125 Opc == AMDGPU::G_SEXT) {
126 // LegalizerHelper wants to use the basic legalization artifacts when
127 // widening etc. We don't handle selection with vcc in artifact sources,
128 // so we need to use a select instead to handle these properly.
129 Register DstReg = MI.getOperand(i: 0).getReg();
130 Register SrcReg = MI.getOperand(i: 1).getReg();
131 const RegisterBank *SrcBank = RBI.getRegBank(Reg: SrcReg, MRI, TRI: *RBI.TRI);
132 if (SrcBank == &AMDGPU::VCCRegBank) {
133 const LLT S32 = LLT::scalar(SizeInBits: 32);
134 assert(MRI.getType(SrcReg) == LLT::scalar(1));
135 assert(MRI.getType(DstReg) == S32);
136 assert(NewBank == &AMDGPU::VGPRRegBank);
137
138 // Replace the extension with a select, which really uses the boolean
139 // source.
140 B.setInsertPt(MBB&: *MI.getParent(), II: MI);
141
142 auto True = B.buildConstant(Res: S32, Val: Opc == AMDGPU::G_SEXT ? -1 : 1);
143 auto False = B.buildConstant(Res: S32, Val: 0);
144 B.buildSelect(Res: DstReg, Tst: SrcReg, Op0: True, Op1: False);
145 MRI.setRegBank(Reg: True.getReg(Idx: 0), RegBank: *NewBank);
146 MRI.setRegBank(Reg: False.getReg(Idx: 0), RegBank: *NewBank);
147 MI.eraseFromParent();
148 }
149
150 assert(!MRI.getRegClassOrRegBank(DstReg));
151 MRI.setRegBank(Reg: DstReg, RegBank: *NewBank);
152 return;
153 }
154
155#ifndef NDEBUG
156 if (Opc == AMDGPU::G_TRUNC) {
157 Register DstReg = MI.getOperand(0).getReg();
158 const RegisterBank *DstBank = RBI.getRegBank(DstReg, MRI, *RBI.TRI);
159 assert(DstBank != &AMDGPU::VCCRegBank);
160 }
161#endif
162
163 for (MachineOperand &Op : MI.operands()) {
164 if (!Op.isReg())
165 continue;
166
167 // We may see physical registers if building a real MI
168 Register Reg = Op.getReg();
169 if (Reg.isPhysical() || MRI.getRegClassOrRegBank(Reg))
170 continue;
171
172 const RegisterBank *RB = NewBank;
173 if (MRI.getType(Reg) == LLT::scalar(SizeInBits: 1)) {
174 assert(NewBank == &AMDGPU::VGPRRegBank &&
175 "s1 operands should only be used for vector bools");
176 assert((MI.getOpcode() != AMDGPU::G_TRUNC &&
177 MI.getOpcode() != AMDGPU::G_ANYEXT) &&
178 "not expecting legalization artifacts here");
179 RB = &AMDGPU::VCCRegBank;
180 }
181
182 MRI.setRegBank(Reg, RegBank: *RB);
183 }
184 }
185
186 void erasingInstr(MachineInstr &MI) override {}
187
188 void createdInstr(MachineInstr &MI) override {
189 // At this point, the instruction was just inserted and has no operands.
190 NewInsts.push_back(Elt: &MI);
191 }
192
193 void changingInstr(MachineInstr &MI) override {}
194 void changedInstr(MachineInstr &MI) override {
195 // FIXME: In principle we should probably add the instruction to NewInsts,
196 // but the way the LegalizerHelper uses the observer, we will always see the
197 // registers we need to set the regbank on also referenced in a new
198 // instruction.
199 }
200};
201
202} // anonymous namespace
203
204AMDGPURegisterBankInfo::AMDGPURegisterBankInfo(const GCNSubtarget &ST)
205 : Subtarget(ST), TRI(Subtarget.getRegisterInfo()),
206 TII(Subtarget.getInstrInfo()) {
207
208 // HACK: Until this is fully tablegen'd.
209 static llvm::once_flag InitializeRegisterBankFlag;
210
211 static auto InitializeRegisterBankOnce = [this]() {
212 assert(&getRegBank(AMDGPU::SGPRRegBankID) == &AMDGPU::SGPRRegBank &&
213 &getRegBank(AMDGPU::VGPRRegBankID) == &AMDGPU::VGPRRegBank &&
214 &getRegBank(AMDGPU::AGPRRegBankID) == &AMDGPU::AGPRRegBank);
215 (void)this;
216 };
217
218 llvm::call_once(flag&: InitializeRegisterBankFlag, F&: InitializeRegisterBankOnce);
219}
220
221static bool isVectorRegisterBank(const RegisterBank &Bank) {
222 unsigned BankID = Bank.getID();
223 return BankID == AMDGPU::VGPRRegBankID || BankID == AMDGPU::AGPRRegBankID;
224}
225
226bool AMDGPURegisterBankInfo::isDivergentRegBank(const RegisterBank *RB) const {
227 return RB != &AMDGPU::SGPRRegBank;
228}
229
230unsigned AMDGPURegisterBankInfo::copyCost(const RegisterBank &Dst,
231 const RegisterBank &Src,
232 TypeSize Size) const {
233 // TODO: Should there be a UniformVGPRRegBank which can use readfirstlane?
234 if (Dst.getID() == AMDGPU::SGPRRegBankID &&
235 (isVectorRegisterBank(Bank: Src) || Src.getID() == AMDGPU::VCCRegBankID)) {
236 return std::numeric_limits<unsigned>::max();
237 }
238
239 // Bool values are tricky, because the meaning is based on context. The SCC
240 // and VCC banks are for the natural scalar and vector conditions produced by
241 // a compare.
242 //
243 // Legalization doesn't know about the necessary context, so an s1 use may
244 // have been a truncate from an arbitrary value, in which case a copy (lowered
245 // as a compare with 0) needs to be inserted.
246 if (Size == 1 &&
247 (Dst.getID() == AMDGPU::SGPRRegBankID) &&
248 (isVectorRegisterBank(Bank: Src) ||
249 Src.getID() == AMDGPU::SGPRRegBankID ||
250 Src.getID() == AMDGPU::VCCRegBankID))
251 return std::numeric_limits<unsigned>::max();
252
253 // There is no direct copy between AGPRs.
254 if (Dst.getID() == AMDGPU::AGPRRegBankID &&
255 Src.getID() == AMDGPU::AGPRRegBankID)
256 return 4;
257
258 return RegisterBankInfo::copyCost(A: Dst, B: Src, Size);
259}
260
261unsigned AMDGPURegisterBankInfo::getBreakDownCost(
262 const ValueMapping &ValMapping,
263 const RegisterBank *CurBank) const {
264 // Check if this is a breakdown for G_LOAD to move the pointer from SGPR to
265 // VGPR.
266 // FIXME: Is there a better way to do this?
267 if (ValMapping.NumBreakDowns >= 2 || ValMapping.BreakDown[0].Length >= 64)
268 return 10; // This is expensive.
269
270 assert(ValMapping.NumBreakDowns == 2 &&
271 ValMapping.BreakDown[0].Length == 32 &&
272 ValMapping.BreakDown[0].StartIdx == 0 &&
273 ValMapping.BreakDown[1].Length == 32 &&
274 ValMapping.BreakDown[1].StartIdx == 32 &&
275 ValMapping.BreakDown[0].RegBank == ValMapping.BreakDown[1].RegBank);
276
277 // 32-bit extract of a 64-bit value is just access of a subregister, so free.
278 // TODO: Cost of 0 hits assert, though it's not clear it's what we really
279 // want.
280
281 // TODO: 32-bit insert to a 64-bit SGPR may incur a non-free copy due to SGPR
282 // alignment restrictions, but this probably isn't important.
283 return 1;
284}
285
286const RegisterBank &
287AMDGPURegisterBankInfo::getRegBankFromRegClass(const TargetRegisterClass &RC,
288 LLT Ty) const {
289 // We promote real scalar booleans to SReg_32. Any SGPR using s1 is really a
290 // VCC-like use.
291 if (TRI->isSGPRClass(RC: &RC)) {
292 // FIXME: This probably came from a copy from a physical register, which
293 // should be inferable from the copied to-type. We don't have many boolean
294 // physical register constraints so just assume a normal SGPR for now.
295 if (!Ty.isValid())
296 return AMDGPU::SGPRRegBank;
297
298 return Ty == LLT::scalar(SizeInBits: 1) ? AMDGPU::VCCRegBank : AMDGPU::SGPRRegBank;
299 }
300
301 return TRI->isAGPRClass(RC: &RC) ? AMDGPU::AGPRRegBank : AMDGPU::VGPRRegBank;
302}
303
304template <unsigned NumOps>
305RegisterBankInfo::InstructionMappings
306AMDGPURegisterBankInfo::addMappingFromTable(
307 const MachineInstr &MI, const MachineRegisterInfo &MRI,
308 const std::array<unsigned, NumOps> RegSrcOpIdx,
309 ArrayRef<OpRegBankEntry<NumOps>> Table) const {
310
311 InstructionMappings AltMappings;
312
313 SmallVector<const ValueMapping *, 10> Operands(MI.getNumOperands());
314
315 unsigned Sizes[NumOps];
316 for (unsigned I = 0; I < NumOps; ++I) {
317 Register Reg = MI.getOperand(RegSrcOpIdx[I]).getReg();
318 Sizes[I] = getSizeInBits(Reg, MRI, TRI: *TRI);
319 }
320
321 for (unsigned I = 0, E = MI.getNumExplicitDefs(); I != E; ++I) {
322 unsigned SizeI = getSizeInBits(Reg: MI.getOperand(i: I).getReg(), MRI, TRI: *TRI);
323 Operands[I] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: SizeI);
324 }
325
326 // getInstrMapping's default mapping uses ID 1, so start at 2.
327 unsigned MappingID = 2;
328 for (const auto &Entry : Table) {
329 for (unsigned I = 0; I < NumOps; ++I) {
330 int OpIdx = RegSrcOpIdx[I];
331 Operands[OpIdx] = AMDGPU::getValueMapping(BankID: Entry.RegBanks[I], Size: Sizes[I]);
332 }
333
334 AltMappings.push_back(Elt: &getInstructionMapping(ID: MappingID++, Cost: Entry.Cost,
335 OperandsMapping: getOperandsMapping(OpdsMapping: Operands),
336 NumOperands: Operands.size()));
337 }
338
339 return AltMappings;
340}
341
342RegisterBankInfo::InstructionMappings
343AMDGPURegisterBankInfo::getInstrAlternativeMappingsIntrinsic(
344 const MachineInstr &MI, const MachineRegisterInfo &MRI) const {
345 switch (cast<GIntrinsic>(Val: MI).getIntrinsicID()) {
346 case Intrinsic::amdgcn_readlane: {
347 static const OpRegBankEntry<3> Table[2] = {
348 // Perfectly legal.
349 { .RegBanks: { AMDGPU::SGPRRegBankID, AMDGPU::VGPRRegBankID, AMDGPU::SGPRRegBankID }, .Cost: 1 },
350
351 // Need a readfirstlane for the index.
352 { .RegBanks: { AMDGPU::SGPRRegBankID, AMDGPU::VGPRRegBankID, AMDGPU::VGPRRegBankID }, .Cost: 2 }
353 };
354
355 const std::array<unsigned, 3> RegSrcOpIdx = { ._M_elems: { 0, 2, 3 } };
356 return addMappingFromTable<3>(MI, MRI, RegSrcOpIdx, Table);
357 }
358 case Intrinsic::amdgcn_writelane: {
359 static const OpRegBankEntry<4> Table[4] = {
360 // Perfectly legal.
361 { .RegBanks: { AMDGPU::VGPRRegBankID, AMDGPU::SGPRRegBankID, AMDGPU::SGPRRegBankID, AMDGPU::VGPRRegBankID }, .Cost: 1 },
362
363 // Need readfirstlane of first op
364 { .RegBanks: { AMDGPU::VGPRRegBankID, AMDGPU::VGPRRegBankID, AMDGPU::SGPRRegBankID, AMDGPU::VGPRRegBankID }, .Cost: 2 },
365
366 // Need readfirstlane of second op
367 { .RegBanks: { AMDGPU::VGPRRegBankID, AMDGPU::SGPRRegBankID, AMDGPU::VGPRRegBankID, AMDGPU::VGPRRegBankID }, .Cost: 2 },
368
369 // Need readfirstlane of both ops
370 { .RegBanks: { AMDGPU::VGPRRegBankID, AMDGPU::VGPRRegBankID, AMDGPU::VGPRRegBankID, AMDGPU::VGPRRegBankID }, .Cost: 3 }
371 };
372
373 // rsrc, voffset, offset
374 const std::array<unsigned, 4> RegSrcOpIdx = { ._M_elems: { 0, 2, 3, 4 } };
375 return addMappingFromTable<4>(MI, MRI, RegSrcOpIdx, Table);
376 }
377 default:
378 return RegisterBankInfo::getInstrAlternativeMappings(MI);
379 }
380}
381
382RegisterBankInfo::InstructionMappings
383AMDGPURegisterBankInfo::getInstrAlternativeMappingsIntrinsicWSideEffects(
384 const MachineInstr &MI, const MachineRegisterInfo &MRI) const {
385
386 switch (cast<GIntrinsic>(Val: MI).getIntrinsicID()) {
387 case Intrinsic::amdgcn_s_buffer_load:
388 case Intrinsic::amdgcn_ptr_s_buffer_load: {
389 static const OpRegBankEntry<2> Table[4] = {
390 // Perfectly legal.
391 { .RegBanks: { AMDGPU::SGPRRegBankID, AMDGPU::SGPRRegBankID }, .Cost: 1 },
392
393 // Only need 1 register in loop
394 { .RegBanks: { AMDGPU::SGPRRegBankID, AMDGPU::VGPRRegBankID }, .Cost: 300 },
395
396 // Have to waterfall the resource.
397 { .RegBanks: { AMDGPU::VGPRRegBankID, AMDGPU::SGPRRegBankID }, .Cost: 1000 },
398
399 // Have to waterfall the resource, and the offset.
400 { .RegBanks: { AMDGPU::VGPRRegBankID, AMDGPU::VGPRRegBankID }, .Cost: 1500 }
401 };
402
403 // rsrc, offset
404 const std::array<unsigned, 2> RegSrcOpIdx = { ._M_elems: { 2, 3 } };
405 return addMappingFromTable<2>(MI, MRI, RegSrcOpIdx, Table);
406 }
407 case Intrinsic::amdgcn_ds_ordered_add:
408 case Intrinsic::amdgcn_ds_ordered_swap: {
409 // VGPR = M0, VGPR
410 static const OpRegBankEntry<3> Table[2] = {
411 // Perfectly legal.
412 { .RegBanks: { AMDGPU::VGPRRegBankID, AMDGPU::SGPRRegBankID, AMDGPU::VGPRRegBankID }, .Cost: 1 },
413
414 // Need a readfirstlane for m0
415 { .RegBanks: { AMDGPU::VGPRRegBankID, AMDGPU::VGPRRegBankID, AMDGPU::VGPRRegBankID }, .Cost: 2 }
416 };
417
418 const std::array<unsigned, 3> RegSrcOpIdx = { ._M_elems: { 0, 2, 3 } };
419 return addMappingFromTable<3>(MI, MRI, RegSrcOpIdx, Table);
420 }
421 case Intrinsic::amdgcn_s_sendmsg:
422 case Intrinsic::amdgcn_s_sendmsghalt: {
423 // FIXME: Should have no register for immediate
424 static const OpRegBankEntry<1> Table[2] = {
425 // Perfectly legal.
426 { .RegBanks: { AMDGPU::SGPRRegBankID }, .Cost: 1 },
427
428 // Need readlane
429 { .RegBanks: { AMDGPU::VGPRRegBankID }, .Cost: 3 }
430 };
431
432 const std::array<unsigned, 1> RegSrcOpIdx = { ._M_elems: { 2 } };
433 return addMappingFromTable<1>(MI, MRI, RegSrcOpIdx, Table);
434 }
435 default:
436 return RegisterBankInfo::getInstrAlternativeMappings(MI);
437 }
438}
439
440// FIXME: Returns uniform if there's no source value information. This is
441// probably wrong.
442bool AMDGPURegisterBankInfo::isScalarLoadLegal(const MachineInstr &MI) const {
443 if (!MI.hasOneMemOperand())
444 return false;
445
446 const MachineMemOperand *MMO = *MI.memoperands_begin();
447 const unsigned AS = MMO->getAddrSpace();
448 const bool IsConst = AS == AMDGPUAS::CONSTANT_ADDRESS ||
449 AS == AMDGPUAS::CONSTANT_ADDRESS_32BIT;
450 const unsigned MemSize = 8 * MMO->getSize().getValue();
451
452 // Require 4-byte alignment.
453 return (MMO->getAlign() >= Align(4) ||
454 (Subtarget.hasScalarSubwordLoads() &&
455 ((MemSize == 16 && MMO->getAlign() >= Align(2)) ||
456 (MemSize == 8 && MMO->getAlign() >= Align(1))))) &&
457 // Can't do a scalar atomic load.
458 !MMO->isAtomic() &&
459 // Don't use scalar loads for volatile accesses to non-constant address
460 // spaces.
461 (IsConst || !MMO->isVolatile()) &&
462 // Memory must be known constant, or not written before this load.
463 (IsConst || MMO->isInvariant() || (MMO->getFlags() & MONoClobber)) &&
464 AMDGPU::isUniformMMO(MMO);
465}
466
467RegisterBankInfo::InstructionMappings
468AMDGPURegisterBankInfo::getInstrAlternativeMappings(
469 const MachineInstr &MI) const {
470
471 const MachineFunction &MF = *MI.getMF();
472 const MachineRegisterInfo &MRI = MF.getRegInfo();
473
474
475 InstructionMappings AltMappings;
476 switch (MI.getOpcode()) {
477 case TargetOpcode::G_CONSTANT:
478 case TargetOpcode::G_IMPLICIT_DEF: {
479 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
480 if (Size == 1) {
481 static const OpRegBankEntry<1> Table[3] = {
482 { .RegBanks: { AMDGPU::VGPRRegBankID }, .Cost: 1 },
483 { .RegBanks: { AMDGPU::SGPRRegBankID }, .Cost: 1 },
484 { .RegBanks: { AMDGPU::VCCRegBankID }, .Cost: 1 }
485 };
486
487 return addMappingFromTable<1>(MI, MRI, RegSrcOpIdx: {._M_elems: { 0 }}, Table);
488 }
489
490 [[fallthrough]];
491 }
492 case TargetOpcode::G_FCONSTANT:
493 case TargetOpcode::G_FRAME_INDEX:
494 case TargetOpcode::G_GLOBAL_VALUE: {
495 static const OpRegBankEntry<1> Table[2] = {
496 { .RegBanks: { AMDGPU::VGPRRegBankID }, .Cost: 1 },
497 { .RegBanks: { AMDGPU::SGPRRegBankID }, .Cost: 1 }
498 };
499
500 return addMappingFromTable<1>(MI, MRI, RegSrcOpIdx: {._M_elems: { 0 }}, Table);
501 }
502 case TargetOpcode::G_AND:
503 case TargetOpcode::G_OR:
504 case TargetOpcode::G_XOR: {
505 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
506
507 if (Size == 1) {
508 // s_{and|or|xor}_b32 set scc when the result of the 32-bit op is not 0.
509 const InstructionMapping &SCCMapping = getInstructionMapping(
510 ID: 1, Cost: 1, OperandsMapping: getOperandsMapping(
511 OpdsMapping: {AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 32),
512 AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 32),
513 AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 32)}),
514 NumOperands: 3); // Num Operands
515 AltMappings.push_back(Elt: &SCCMapping);
516
517 const InstructionMapping &VCCMapping0 = getInstructionMapping(
518 ID: 2, Cost: 1, OperandsMapping: getOperandsMapping(
519 OpdsMapping: {AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size),
520 AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size),
521 AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size)}),
522 NumOperands: 3); // Num Operands
523 AltMappings.push_back(Elt: &VCCMapping0);
524 return AltMappings;
525 }
526
527 if (Size != 64)
528 break;
529
530 const InstructionMapping &SSMapping = getInstructionMapping(
531 ID: 1, Cost: 1, OperandsMapping: getOperandsMapping(
532 OpdsMapping: {AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size),
533 AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size),
534 AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size)}),
535 NumOperands: 3); // Num Operands
536 AltMappings.push_back(Elt: &SSMapping);
537
538 const InstructionMapping &VVMapping = getInstructionMapping(
539 ID: 2, Cost: 2, OperandsMapping: getOperandsMapping(
540 OpdsMapping: {AMDGPU::getValueMappingSGPR64Only(BankID: AMDGPU::VGPRRegBankID, Size),
541 AMDGPU::getValueMappingSGPR64Only(BankID: AMDGPU::VGPRRegBankID, Size),
542 AMDGPU::getValueMappingSGPR64Only(BankID: AMDGPU::VGPRRegBankID, Size)}),
543 NumOperands: 3); // Num Operands
544 AltMappings.push_back(Elt: &VVMapping);
545 break;
546 }
547 case TargetOpcode::G_LOAD:
548 case TargetOpcode::G_ZEXTLOAD:
549 case TargetOpcode::G_SEXTLOAD: {
550 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
551 LLT PtrTy = MRI.getType(Reg: MI.getOperand(i: 1).getReg());
552 unsigned PtrSize = PtrTy.getSizeInBits();
553 unsigned AS = PtrTy.getAddressSpace();
554
555 if ((AS != AMDGPUAS::LOCAL_ADDRESS && AS != AMDGPUAS::REGION_ADDRESS &&
556 AS != AMDGPUAS::PRIVATE_ADDRESS) &&
557 isScalarLoadLegal(MI)) {
558 const InstructionMapping &SSMapping = getInstructionMapping(
559 ID: 1, Cost: 1, OperandsMapping: getOperandsMapping(
560 OpdsMapping: {AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size),
561 AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: PtrSize)}),
562 NumOperands: 2); // Num Operands
563 AltMappings.push_back(Elt: &SSMapping);
564 }
565
566 const InstructionMapping &VVMapping = getInstructionMapping(
567 ID: 2, Cost: 1,
568 OperandsMapping: getOperandsMapping(
569 OpdsMapping: {AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size),
570 AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: PtrSize)}),
571 NumOperands: 2); // Num Operands
572 AltMappings.push_back(Elt: &VVMapping);
573
574 // It may be possible to have a vgpr = load sgpr mapping here, because
575 // the mubuf instructions support this kind of load, but probably for only
576 // gfx7 and older. However, the addressing mode matching in the instruction
577 // selector should be able to do a better job of detecting and selecting
578 // these kinds of loads from the vgpr = load vgpr mapping.
579
580 return AltMappings;
581
582 }
583 case TargetOpcode::G_SELECT: {
584 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
585 const InstructionMapping &SSMapping = getInstructionMapping(ID: 1, Cost: 1,
586 OperandsMapping: getOperandsMapping(OpdsMapping: {AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size),
587 AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 1),
588 AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size),
589 AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size)}),
590 NumOperands: 4); // Num Operands
591 AltMappings.push_back(Elt: &SSMapping);
592
593 const InstructionMapping &VVMapping = getInstructionMapping(ID: 2, Cost: 1,
594 OperandsMapping: getOperandsMapping(OpdsMapping: {AMDGPU::getValueMappingSGPR64Only(BankID: AMDGPU::VGPRRegBankID, Size),
595 AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: 1),
596 AMDGPU::getValueMappingSGPR64Only(BankID: AMDGPU::VGPRRegBankID, Size),
597 AMDGPU::getValueMappingSGPR64Only(BankID: AMDGPU::VGPRRegBankID, Size)}),
598 NumOperands: 4); // Num Operands
599 AltMappings.push_back(Elt: &VVMapping);
600
601 return AltMappings;
602 }
603 case TargetOpcode::G_UADDE:
604 case TargetOpcode::G_USUBE:
605 case TargetOpcode::G_SADDE:
606 case TargetOpcode::G_SSUBE: {
607 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
608 const InstructionMapping &SSMapping = getInstructionMapping(ID: 1, Cost: 1,
609 OperandsMapping: getOperandsMapping(
610 OpdsMapping: {AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size),
611 AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 1),
612 AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size),
613 AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size),
614 AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 1)}),
615 NumOperands: 5); // Num Operands
616 AltMappings.push_back(Elt: &SSMapping);
617
618 const InstructionMapping &VVMapping = getInstructionMapping(ID: 2, Cost: 1,
619 OperandsMapping: getOperandsMapping(OpdsMapping: {AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size),
620 AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: 1),
621 AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size),
622 AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size),
623 AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: 1)}),
624 NumOperands: 5); // Num Operands
625 AltMappings.push_back(Elt: &VVMapping);
626 return AltMappings;
627 }
628 case AMDGPU::G_BRCOND: {
629 assert(MRI.getType(MI.getOperand(0).getReg()).getSizeInBits() == 1);
630
631 // TODO: Change type to 32 for scalar
632 const InstructionMapping &SMapping = getInstructionMapping(
633 ID: 1, Cost: 1, OperandsMapping: getOperandsMapping(
634 OpdsMapping: {AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 1), nullptr}),
635 NumOperands: 2); // Num Operands
636 AltMappings.push_back(Elt: &SMapping);
637
638 const InstructionMapping &VMapping = getInstructionMapping(
639 ID: 1, Cost: 1, OperandsMapping: getOperandsMapping(
640 OpdsMapping: {AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: 1), nullptr }),
641 NumOperands: 2); // Num Operands
642 AltMappings.push_back(Elt: &VMapping);
643 return AltMappings;
644 }
645 case AMDGPU::G_INTRINSIC:
646 case AMDGPU::G_INTRINSIC_CONVERGENT:
647 return getInstrAlternativeMappingsIntrinsic(MI, MRI);
648 case AMDGPU::G_INTRINSIC_W_SIDE_EFFECTS:
649 case AMDGPU::G_INTRINSIC_CONVERGENT_W_SIDE_EFFECTS:
650 return getInstrAlternativeMappingsIntrinsicWSideEffects(MI, MRI);
651 default:
652 break;
653 }
654 return RegisterBankInfo::getInstrAlternativeMappings(MI);
655}
656
657void AMDGPURegisterBankInfo::split64BitValueForMapping(
658 MachineIRBuilder &B,
659 SmallVector<Register, 2> &Regs,
660 LLT HalfTy,
661 Register Reg) const {
662 assert(HalfTy.getSizeInBits() == 32);
663 MachineRegisterInfo *MRI = B.getMRI();
664 Register LoLHS = MRI->createGenericVirtualRegister(Ty: HalfTy);
665 Register HiLHS = MRI->createGenericVirtualRegister(Ty: HalfTy);
666 const RegisterBank *Bank = getRegBank(Reg, MRI: *MRI, TRI: *TRI);
667 MRI->setRegBank(Reg: LoLHS, RegBank: *Bank);
668 MRI->setRegBank(Reg: HiLHS, RegBank: *Bank);
669
670 Regs.push_back(Elt: LoLHS);
671 Regs.push_back(Elt: HiLHS);
672
673 B.buildInstr(Opcode: AMDGPU::G_UNMERGE_VALUES)
674 .addDef(RegNo: LoLHS)
675 .addDef(RegNo: HiLHS)
676 .addUse(RegNo: Reg);
677}
678
679/// Replace the current type each register in \p Regs has with \p NewTy
680static void setRegsToType(MachineRegisterInfo &MRI, ArrayRef<Register> Regs,
681 LLT NewTy) {
682 for (Register Reg : Regs) {
683 assert(MRI.getType(Reg).getSizeInBits() == NewTy.getSizeInBits());
684 MRI.setType(VReg: Reg, Ty: NewTy);
685 }
686}
687
688static LLT getHalfSizedType(LLT Ty) {
689 if (Ty.isVector()) {
690 assert(Ty.getElementCount().isKnownMultipleOf(2));
691 return LLT::scalarOrVector(EC: Ty.getElementCount().divideCoefficientBy(RHS: 2),
692 ScalarTy: Ty.getElementType());
693 }
694
695 assert(Ty.getScalarSizeInBits() % 2 == 0);
696 return LLT::scalar(SizeInBits: Ty.getScalarSizeInBits() / 2);
697}
698
699// Build one or more V_READFIRSTLANE_B32 instructions to move the given vector
700// source value into a scalar register.
701Register AMDGPURegisterBankInfo::buildReadFirstLane(MachineIRBuilder &B,
702 MachineRegisterInfo &MRI,
703 Register Src) const {
704 LLT Ty = MRI.getType(Reg: Src);
705 const RegisterBank *Bank = getRegBank(Reg: Src, MRI, TRI: *TRI);
706
707 if (Bank == &AMDGPU::SGPRRegBank)
708 return Src;
709
710 unsigned Bits = Ty.getSizeInBits();
711 assert(Bits % 32 == 0);
712
713 if (Bank != &AMDGPU::VGPRRegBank) {
714 // We need to copy from AGPR to VGPR
715 Src = B.buildCopy(Res: Ty, Op: Src).getReg(Idx: 0);
716 MRI.setRegBank(Reg: Src, RegBank: AMDGPU::VGPRRegBank);
717 }
718
719 LLT S32 = LLT::scalar(SizeInBits: 32);
720 unsigned NumParts = Bits / 32;
721 SmallVector<Register, 8> SrcParts;
722 SmallVector<Register, 8> DstParts;
723
724 if (Bits == 32) {
725 SrcParts.push_back(Elt: Src);
726 } else {
727 auto Unmerge = B.buildUnmerge(Res: S32, Op: Src);
728 for (unsigned i = 0; i < NumParts; ++i)
729 SrcParts.push_back(Elt: Unmerge.getReg(Idx: i));
730 }
731
732 for (unsigned i = 0; i < NumParts; ++i) {
733 Register SrcPart = SrcParts[i];
734 Register DstPart = MRI.createVirtualRegister(RegClass: &AMDGPU::SReg_32_XM0RegClass);
735 MRI.setType(VReg: DstPart, Ty: NumParts == 1 ? Ty : S32);
736
737 const TargetRegisterClass *Constrained =
738 constrainGenericRegister(Reg: SrcPart, RC: AMDGPU::VGPR_32RegClass, MRI);
739 (void)Constrained;
740 assert(Constrained && "Failed to constrain readfirstlane src reg");
741
742 B.buildInstr(Opc: AMDGPU::V_READFIRSTLANE_B32, DstOps: {DstPart}, SrcOps: {SrcPart});
743
744 DstParts.push_back(Elt: DstPart);
745 }
746
747 if (Bits == 32)
748 return DstParts[0];
749
750 Register Dst = B.buildMergeLikeInstr(Res: Ty, Ops: DstParts).getReg(Idx: 0);
751 MRI.setRegBank(Reg: Dst, RegBank: AMDGPU::SGPRRegBank);
752 return Dst;
753}
754
755/// Legalize instruction \p MI where operands in \p OpIndices must be SGPRs. If
756/// any of the required SGPR operands are VGPRs, perform a waterfall loop to
757/// execute the instruction for each unique combination of values in all lanes
758/// in the wave. The block will be split such that rest of the instructions are
759/// moved to a new block.
760///
761/// Essentially performs this loop:
762//
763/// Save Execution Mask
764/// For (Lane : Wavefront) {
765/// Enable Lane, Disable all other lanes
766/// SGPR = read SGPR value for current lane from VGPR
767/// VGPRResult[Lane] = use_op SGPR
768/// }
769/// Restore Execution Mask
770///
771/// There is additional complexity to try for compare values to identify the
772/// unique values used.
773bool AMDGPURegisterBankInfo::executeInWaterfallLoop(
774 MachineIRBuilder &B, iterator_range<MachineBasicBlock::iterator> Range,
775 SmallSet<Register, 4> &SGPROperandRegs) const {
776 // Track use registers which have already been expanded with a readfirstlane
777 // sequence. This may have multiple uses if moving a sequence.
778 DenseMap<Register, Register> WaterfalledRegMap;
779
780 MachineBasicBlock &MBB = B.getMBB();
781 MachineFunction *MF = &B.getMF();
782
783 const TargetRegisterClass *WaveRC = TRI->getWaveMaskRegClass();
784 const AMDGPU::LaneMaskConstants &LMC =
785 AMDGPU::LaneMaskConstants::get(ST: Subtarget);
786
787#ifndef NDEBUG
788 const int OrigRangeSize = std::distance(Range.begin(), Range.end());
789#endif
790
791 MachineRegisterInfo &MRI = *B.getMRI();
792 Register SaveExecReg = MRI.createVirtualRegister(RegClass: WaveRC);
793 Register InitSaveExecReg = MRI.createVirtualRegister(RegClass: WaveRC);
794
795 // Don't bother using generic instructions/registers for the exec mask.
796 B.buildInstr(Opcode: TargetOpcode::IMPLICIT_DEF)
797 .addDef(RegNo: InitSaveExecReg);
798
799 Register PhiExec = MRI.createVirtualRegister(RegClass: WaveRC);
800 Register NewExec = MRI.createVirtualRegister(RegClass: WaveRC);
801
802 // To insert the loop we need to split the block. Move everything before this
803 // point to a new block, and insert a new empty block before this instruction.
804 MachineBasicBlock *LoopBB = MF->CreateMachineBasicBlock();
805 MachineBasicBlock *BodyBB = MF->CreateMachineBasicBlock();
806 MachineBasicBlock *RemainderBB = MF->CreateMachineBasicBlock();
807 MachineBasicBlock *RestoreExecBB = MF->CreateMachineBasicBlock();
808 MachineFunction::iterator MBBI(MBB);
809 ++MBBI;
810 MF->insert(MBBI, MBB: LoopBB);
811 MF->insert(MBBI, MBB: BodyBB);
812 MF->insert(MBBI, MBB: RestoreExecBB);
813 MF->insert(MBBI, MBB: RemainderBB);
814
815 LoopBB->addSuccessor(Succ: BodyBB);
816 BodyBB->addSuccessor(Succ: RestoreExecBB);
817 BodyBB->addSuccessor(Succ: LoopBB);
818
819 // Move the rest of the block into a new block.
820 RemainderBB->transferSuccessorsAndUpdatePHIs(FromMBB: &MBB);
821 RemainderBB->splice(Where: RemainderBB->begin(), Other: &MBB, From: Range.end(), To: MBB.end());
822
823 MBB.addSuccessor(Succ: LoopBB);
824 RestoreExecBB->addSuccessor(Succ: RemainderBB);
825
826 B.setInsertPt(MBB&: *LoopBB, II: LoopBB->end());
827
828 B.buildInstr(Opcode: TargetOpcode::PHI)
829 .addDef(RegNo: PhiExec)
830 .addReg(RegNo: InitSaveExecReg)
831 .addMBB(MBB: &MBB)
832 .addReg(RegNo: NewExec)
833 .addMBB(MBB: BodyBB);
834
835 const DebugLoc &DL = B.getDL();
836
837 MachineInstr &FirstInst = *Range.begin();
838
839 // Move the instruction into the loop body. Note we moved everything after
840 // Range.end() already into a new block, so Range.end() is no longer valid.
841 BodyBB->splice(Where: BodyBB->end(), Other: &MBB, From: Range.begin(), To: MBB.end());
842
843 // Figure out the iterator range after splicing the instructions.
844 MachineBasicBlock::iterator NewBegin = FirstInst.getIterator();
845 auto NewEnd = BodyBB->end();
846
847 B.setMBB(*LoopBB);
848
849 LLT S1 = LLT::scalar(SizeInBits: 1);
850 Register CondReg;
851
852 assert(std::distance(NewBegin, NewEnd) == OrigRangeSize);
853
854 for (MachineInstr &MI : make_range(x: NewBegin, y: NewEnd)) {
855 for (MachineOperand &Op : MI.all_uses()) {
856 Register OldReg = Op.getReg();
857 if (!SGPROperandRegs.count(V: OldReg))
858 continue;
859
860 // See if we already processed this register in another instruction in the
861 // sequence.
862 auto OldVal = WaterfalledRegMap.find(Val: OldReg);
863 if (OldVal != WaterfalledRegMap.end()) {
864 Op.setReg(OldVal->second);
865 continue;
866 }
867
868 Register OpReg = Op.getReg();
869 LLT OpTy = MRI.getType(Reg: OpReg);
870
871 const RegisterBank *OpBank = getRegBank(Reg: OpReg, MRI, TRI: *TRI);
872 if (OpBank != &AMDGPU::VGPRRegBank) {
873 // Insert copy from AGPR to VGPR before the loop.
874 B.setMBB(MBB);
875 OpReg = B.buildCopy(Res: OpTy, Op: OpReg).getReg(Idx: 0);
876 MRI.setRegBank(Reg: OpReg, RegBank: AMDGPU::VGPRRegBank);
877 B.setMBB(*LoopBB);
878 }
879
880 Register CurrentLaneReg = buildReadFirstLane(B, MRI, Src: OpReg);
881
882 // Build the comparison(s).
883 unsigned OpSize = OpTy.getSizeInBits();
884 bool Is64 = OpSize % 64 == 0;
885 unsigned PartSize = Is64 ? 64 : 32;
886 LLT PartTy = LLT::scalar(SizeInBits: PartSize);
887 unsigned NumParts = OpSize / PartSize;
888 SmallVector<Register, 8> OpParts;
889 SmallVector<Register, 8> CurrentLaneParts;
890
891 if (NumParts == 1) {
892 OpParts.push_back(Elt: OpReg);
893 CurrentLaneParts.push_back(Elt: CurrentLaneReg);
894 } else {
895 auto UnmergeOp = B.buildUnmerge(Res: PartTy, Op: OpReg);
896 auto UnmergeCurrentLane = B.buildUnmerge(Res: PartTy, Op: CurrentLaneReg);
897 for (unsigned i = 0; i < NumParts; ++i) {
898 OpParts.push_back(Elt: UnmergeOp.getReg(Idx: i));
899 CurrentLaneParts.push_back(Elt: UnmergeCurrentLane.getReg(Idx: i));
900 MRI.setRegBank(Reg: OpParts[i], RegBank: AMDGPU::VGPRRegBank);
901 MRI.setRegBank(Reg: CurrentLaneParts[i], RegBank: AMDGPU::SGPRRegBank);
902 }
903 }
904
905 for (unsigned i = 0; i < NumParts; ++i) {
906 auto CmpReg = B.buildICmp(Pred: CmpInst::ICMP_EQ, Res: S1, Op0: CurrentLaneParts[i],
907 Op1: OpParts[i]).getReg(Idx: 0);
908 MRI.setRegBank(Reg: CmpReg, RegBank: AMDGPU::VCCRegBank);
909
910 if (!CondReg) {
911 CondReg = CmpReg;
912 } else {
913 CondReg = B.buildAnd(Dst: S1, Src0: CondReg, Src1: CmpReg).getReg(Idx: 0);
914 MRI.setRegBank(Reg: CondReg, RegBank: AMDGPU::VCCRegBank);
915 }
916 }
917
918 Op.setReg(CurrentLaneReg);
919
920 // Make sure we don't re-process this register again.
921 WaterfalledRegMap.insert(KV: std::pair(OldReg, Op.getReg()));
922 }
923 }
924
925 // The ballot becomes a no-op during instruction selection.
926 CondReg = B.buildIntrinsic(ID: Intrinsic::amdgcn_ballot,
927 Res: {LLT::scalar(SizeInBits: Subtarget.isWave32() ? 32 : 64)})
928 .addReg(RegNo: CondReg)
929 .getReg(Idx: 0);
930 MRI.setRegClass(Reg: CondReg, RC: WaveRC);
931
932 // Update EXEC, save the original EXEC value to VCC.
933 B.buildInstr(Opcode: LMC.AndSaveExecOpc)
934 .addDef(RegNo: NewExec)
935 .addReg(RegNo: CondReg, Flags: RegState::Kill)
936 .setOperandDead(3);
937
938 MRI.setSimpleHint(VReg: NewExec, PrefReg: CondReg);
939
940 B.setInsertPt(MBB&: *BodyBB, II: BodyBB->end());
941
942 // Update EXEC, switch all done bits to 0 and all todo bits to 1.
943 B.buildInstr(Opcode: LMC.XorTermOpc)
944 .addDef(RegNo: LMC.ExecReg)
945 .addReg(RegNo: LMC.ExecReg)
946 .addReg(RegNo: NewExec)
947 .setOperandDead(3);
948
949 // XXX - s_xor_b64 sets scc to 1 if the result is nonzero, so can we use
950 // s_cbranch_scc0?
951
952 // Loop back to V_READFIRSTLANE_B32 if there are still variants to cover.
953 B.buildInstr(Opcode: AMDGPU::SI_WATERFALL_LOOP).addMBB(MBB: LoopBB);
954
955 // Save the EXEC mask before the loop.
956 BuildMI(BB&: MBB, I: MBB.end(), MIMD: DL, MCID: TII->get(Opcode: LMC.MovOpc), DestReg: SaveExecReg)
957 .addReg(RegNo: LMC.ExecReg);
958
959 // Restore the EXEC mask after the loop.
960 B.setMBB(*RestoreExecBB);
961 B.buildInstr(Opcode: LMC.MovTermOpc).addDef(RegNo: LMC.ExecReg).addReg(RegNo: SaveExecReg);
962
963 // Set the insert point after the original instruction, so any new
964 // instructions will be in the remainder.
965 B.setInsertPt(MBB&: *RemainderBB, II: RemainderBB->begin());
966
967 return true;
968}
969
970// Return any unique registers used by \p MI at \p OpIndices that need to be
971// handled in a waterfall loop. Returns these registers in \p
972// SGPROperandRegs. Returns true if there are any operands to handle and a
973// waterfall loop is necessary.
974bool AMDGPURegisterBankInfo::collectWaterfallOperands(
975 SmallSet<Register, 4> &SGPROperandRegs, MachineInstr &MI,
976 MachineRegisterInfo &MRI, ArrayRef<unsigned> OpIndices) const {
977 for (unsigned Op : OpIndices) {
978 assert(MI.getOperand(Op).isUse());
979 Register Reg = MI.getOperand(i: Op).getReg();
980 const RegisterBank *OpBank = getRegBank(Reg, MRI, TRI: *TRI);
981 if (OpBank->getID() != AMDGPU::SGPRRegBankID)
982 SGPROperandRegs.insert(V: Reg);
983 }
984
985 // No operands need to be replaced, so no need to loop.
986 return !SGPROperandRegs.empty();
987}
988
989bool AMDGPURegisterBankInfo::executeInWaterfallLoop(
990 MachineIRBuilder &B, MachineInstr &MI, ArrayRef<unsigned> OpIndices) const {
991 // Use a set to avoid extra readfirstlanes in the case where multiple operands
992 // are the same register.
993 SmallSet<Register, 4> SGPROperandRegs;
994
995 if (!collectWaterfallOperands(SGPROperandRegs, MI, MRI&: *B.getMRI(), OpIndices))
996 return false;
997
998 MachineBasicBlock::iterator I = MI.getIterator();
999 return executeInWaterfallLoop(B, Range: make_range(x: I, y: std::next(x: I)),
1000 SGPROperandRegs);
1001}
1002
1003// Legalize an operand that must be an SGPR by inserting a readfirstlane.
1004void AMDGPURegisterBankInfo::constrainOpWithReadfirstlane(
1005 MachineIRBuilder &B, MachineInstr &MI, unsigned OpIdx) const {
1006 Register Reg = MI.getOperand(i: OpIdx).getReg();
1007 MachineRegisterInfo &MRI = *B.getMRI();
1008 const RegisterBank *Bank = getRegBank(Reg, MRI, TRI: *TRI);
1009 if (Bank == &AMDGPU::SGPRRegBank)
1010 return;
1011
1012 Reg = buildReadFirstLane(B, MRI, Src: Reg);
1013 MI.getOperand(i: OpIdx).setReg(Reg);
1014}
1015
1016/// Split \p Ty into 2 pieces. The first will have \p FirstSize bits, and the
1017/// rest will be in the remainder.
1018static std::pair<LLT, LLT> splitUnequalType(LLT Ty, unsigned FirstSize) {
1019 unsigned TotalSize = Ty.getSizeInBits();
1020 if (!Ty.isVector())
1021 return {LLT::scalar(SizeInBits: FirstSize), LLT::scalar(SizeInBits: TotalSize - FirstSize)};
1022
1023 LLT EltTy = Ty.getElementType();
1024 unsigned EltSize = EltTy.getSizeInBits();
1025 assert(FirstSize % EltSize == 0);
1026
1027 unsigned FirstPartNumElts = FirstSize / EltSize;
1028 unsigned RemainderElts = (TotalSize - FirstSize) / EltSize;
1029
1030 return {LLT::scalarOrVector(EC: ElementCount::getFixed(MinVal: FirstPartNumElts), ScalarTy: EltTy),
1031 LLT::scalarOrVector(EC: ElementCount::getFixed(MinVal: RemainderElts), ScalarTy: EltTy)};
1032}
1033
1034static LLT widen96To128(LLT Ty) {
1035 if (!Ty.isVector())
1036 return LLT::scalar(SizeInBits: 128);
1037
1038 LLT EltTy = Ty.getElementType();
1039 assert(128 % EltTy.getSizeInBits() == 0);
1040 return LLT::fixed_vector(NumElements: 128 / EltTy.getSizeInBits(), ScalarTy: EltTy);
1041}
1042
1043bool AMDGPURegisterBankInfo::applyMappingLoad(
1044 MachineIRBuilder &B,
1045 const AMDGPURegisterBankInfo::OperandsMapper &OpdMapper,
1046 MachineInstr &MI) const {
1047 MachineRegisterInfo &MRI = *B.getMRI();
1048 Register DstReg = MI.getOperand(i: 0).getReg();
1049 const LLT LoadTy = MRI.getType(Reg: DstReg);
1050 unsigned LoadSize = LoadTy.getSizeInBits();
1051 MachineMemOperand *MMO = *MI.memoperands_begin();
1052 const unsigned MaxNonSmrdLoadSize = 128;
1053
1054 const RegisterBank *DstBank =
1055 OpdMapper.getInstrMapping().getOperandMapping(i: 0).BreakDown[0].RegBank;
1056 if (DstBank == &AMDGPU::SGPRRegBank) {
1057 // There are some special cases that we need to look at for 32 bit and 96
1058 // bit SGPR loads otherwise we have nothing to do.
1059 if (LoadSize != 32 && (LoadSize != 96 || Subtarget.hasScalarDwordx3Loads()))
1060 return false;
1061
1062 const unsigned MemSize = 8 * MMO->getSize().getValue();
1063 // Scalar loads of size 8 or 16 bit with proper alignment may be widened to
1064 // 32 bit. Check to see if we need to widen the memory access, 8 or 16 bit
1065 // scalar loads should have a load size of 32 but memory access size of less
1066 // than 32.
1067 if (LoadSize == 32 &&
1068 (MemSize == 32 || LoadTy.isVector() || !isScalarLoadLegal(MI)))
1069 return false;
1070
1071 if (LoadSize == 32 &&
1072 ((MemSize == 8 && MMO->getAlign() >= Align(1)) ||
1073 (MemSize == 16 && MMO->getAlign() >= Align(2))) &&
1074 isScalarLoadLegal(MI) &&
1075 Subtarget.getGeneration() >= AMDGPUSubtarget::GFX12)
1076 return false;
1077
1078 Register PtrReg = MI.getOperand(i: 1).getReg();
1079
1080 ApplyRegBankMapping ApplyBank(B, *this, MRI, DstBank);
1081
1082 if (LoadSize == 32) {
1083 // This is an extending load from a sub-dword size. Widen the memory
1084 // access size to 4 bytes and clear the extra high bits appropriately
1085 const LLT S32 = LLT::scalar(SizeInBits: 32);
1086 if (MI.getOpcode() == AMDGPU::G_SEXTLOAD) {
1087 // Must extend the sign bit into higher bits for a G_SEXTLOAD
1088 auto WideLoad = B.buildLoadFromOffset(Dst: S32, BasePtr: PtrReg, BaseMMO&: *MMO, Offset: 0);
1089 B.buildSExtInReg(Res: MI.getOperand(i: 0), Op: WideLoad, ImmOp: MemSize);
1090 } else if (MI.getOpcode() == AMDGPU::G_ZEXTLOAD) {
1091 // Must extend zero into higher bits with an AND for a G_ZEXTLOAD
1092 auto WideLoad = B.buildLoadFromOffset(Dst: S32, BasePtr: PtrReg, BaseMMO&: *MMO, Offset: 0);
1093 B.buildZExtInReg(Res: MI.getOperand(i: 0), Op: WideLoad, ImmOp: MemSize);
1094 } else
1095 // We do not need to touch the higher bits for regular loads.
1096 B.buildLoadFromOffset(Dst: MI.getOperand(i: 0), BasePtr: PtrReg, BaseMMO&: *MMO, Offset: 0);
1097 } else {
1098 // 96-bit loads are only available for vector loads. We need to split this
1099 // into a 64-bit part, and 32 (unless we can widen to a 128-bit load).
1100 if (MMO->getAlign() < Align(16)) {
1101 LegalizerHelper Helper(B.getMF(), ApplyBank, B);
1102 LLT Part64, Part32;
1103 std::tie(args&: Part64, args&: Part32) = splitUnequalType(Ty: LoadTy, FirstSize: 64);
1104 if (Helper.reduceLoadStoreWidth(MI&: cast<GAnyLoad>(Val&: MI), TypeIdx: 0, NarrowTy: Part64) !=
1105 LegalizerHelper::Legalized)
1106 return false;
1107 return true;
1108 }
1109 LLT WiderTy = widen96To128(Ty: LoadTy);
1110 auto WideLoad = B.buildLoadFromOffset(Dst: WiderTy, BasePtr: PtrReg, BaseMMO&: *MMO, Offset: 0);
1111 if (WiderTy.isScalar()) {
1112 B.buildTrunc(Res: MI.getOperand(i: 0), Op: WideLoad);
1113 } else {
1114 B.buildDeleteTrailingVectorElements(Res: MI.getOperand(i: 0).getReg(),
1115 Op0: WideLoad);
1116 }
1117 }
1118
1119 MI.eraseFromParent();
1120 return true;
1121 }
1122
1123 // 128-bit loads are supported for all instruction types.
1124 if (LoadSize <= MaxNonSmrdLoadSize)
1125 return false;
1126
1127 SmallVector<Register, 1> SrcRegs(OpdMapper.getVRegs(OpIdx: 1));
1128
1129 if (SrcRegs.empty())
1130 SrcRegs.push_back(Elt: MI.getOperand(i: 1).getReg());
1131
1132 // RegBankSelect only emits scalar types, so we need to reset the pointer
1133 // operand to a pointer type.
1134 Register BasePtrReg = SrcRegs[0];
1135 LLT PtrTy = MRI.getType(Reg: MI.getOperand(i: 1).getReg());
1136 MRI.setType(VReg: BasePtrReg, Ty: PtrTy);
1137
1138 // The following are the loads not splitted enough during legalization
1139 // because it was not clear they are smem-load or vmem-load
1140 if (AMDGPU::isExtendedGlobalAddrSpace(AS: MMO->getAddrSpace()) ||
1141 MMO->getAddrSpace() == AMDGPUAS::BUFFER_RESOURCE) {
1142 assert(LoadSize % MaxNonSmrdLoadSize == 0);
1143 unsigned NumSplitParts = LoadTy.getSizeInBits() / MaxNonSmrdLoadSize;
1144 const LLT LoadSplitTy = LoadTy.divide(Factor: NumSplitParts);
1145 ApplyRegBankMapping O(B, *this, MRI, &AMDGPU::VGPRRegBank);
1146 LegalizerHelper Helper(B.getMF(), O, B);
1147 if (LoadTy.isVector()) {
1148 if (Helper.fewerElementsVector(MI, TypeIdx: 0, NarrowTy: LoadSplitTy) !=
1149 LegalizerHelper::Legalized)
1150 return false;
1151 } else {
1152 if (Helper.narrowScalar(MI, TypeIdx: 0, NarrowTy: LoadSplitTy) != LegalizerHelper::Legalized)
1153 return false;
1154 }
1155 }
1156
1157 MRI.setRegBank(Reg: DstReg, RegBank: AMDGPU::VGPRRegBank);
1158 return true;
1159}
1160
1161bool AMDGPURegisterBankInfo::applyMappingDynStackAlloc(
1162 MachineIRBuilder &B,
1163 const AMDGPURegisterBankInfo::OperandsMapper &OpdMapper,
1164 MachineInstr &MI) const {
1165 MachineRegisterInfo &MRI = *B.getMRI();
1166 const MachineFunction &MF = B.getMF();
1167 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
1168 const auto &TFI = *ST.getFrameLowering();
1169
1170 // Guard in case the stack growth direction ever changes with scratch
1171 // instructions.
1172 assert(TFI.getStackGrowthDirection() == TargetFrameLowering::StackGrowsUp &&
1173 "Stack grows upwards for AMDGPU");
1174
1175 Register Dst = MI.getOperand(i: 0).getReg();
1176 Register AllocSize = MI.getOperand(i: 1).getReg();
1177 Align Alignment = assumeAligned(Value: MI.getOperand(i: 2).getImm());
1178
1179 // When using flat-scratch, the stack offset is unscaled.
1180 const bool HasFlatScratch = ST.hasFlatScratchEnabled();
1181 const unsigned WavefrontSizeLog2 = ST.getWavefrontSizeLog2();
1182
1183 const RegisterBank *SizeBank = getRegBank(Reg: AllocSize, MRI, TRI: *TRI);
1184
1185 if (SizeBank != &AMDGPU::SGPRRegBank) {
1186 auto WaveReduction =
1187 B.buildIntrinsic(ID: Intrinsic::amdgcn_wave_reduce_umax, Res: {LLT::scalar(SizeInBits: 32)})
1188 .addUse(RegNo: AllocSize)
1189 .addImm(Val: 0);
1190 AllocSize = WaveReduction.getReg(Idx: 0);
1191 }
1192
1193 LLT PtrTy = MRI.getType(Reg: Dst);
1194 LLT IntPtrTy = LLT::scalar(SizeInBits: PtrTy.getSizeInBits());
1195
1196 const SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
1197 Register SPReg = Info->getStackPtrOffsetReg();
1198 ApplyRegBankMapping ApplyBank(B, *this, MRI, &AMDGPU::SGPRRegBank);
1199
1200 Register ScaledSize = AllocSize;
1201 if (!HasFlatScratch) {
1202 auto WaveSize = B.buildConstant(Res: LLT::scalar(SizeInBits: 32), Val: WavefrontSizeLog2);
1203 ScaledSize = B.buildShl(Dst: IntPtrTy, Src0: AllocSize, Src1: WaveSize).getReg(Idx: 0);
1204 }
1205
1206 auto OldSP = B.buildCopy(Res: PtrTy, Op: SPReg);
1207 if (Alignment > TFI.getStackAlign()) {
1208 const uint64_t ScaledAlignment =
1209 HasFlatScratch ? Alignment.value()
1210 : (Alignment.value() << WavefrontSizeLog2);
1211 const uint64_t StackAlignMask = ScaledAlignment - 1;
1212 auto Tmp1 = B.buildPtrAdd(Res: PtrTy, Op0: OldSP,
1213 Op1: B.buildConstant(Res: LLT::scalar(SizeInBits: 32), Val: StackAlignMask));
1214 B.buildMaskLowPtrBits(Res: Dst, Op0: Tmp1,
1215 NumBits: (HasFlatScratch
1216 ? Log2(A: Alignment)
1217 : Log2(A: Alignment) + WavefrontSizeLog2));
1218 } else {
1219 B.buildCopy(Res: Dst, Op: OldSP);
1220 }
1221 auto PtrAdd = B.buildPtrAdd(Res: PtrTy, Op0: Dst, Op1: ScaledSize);
1222 B.buildCopy(Res: SPReg, Op: PtrAdd);
1223 MI.eraseFromParent();
1224 return true;
1225}
1226
1227bool AMDGPURegisterBankInfo::applyMappingImage(
1228 MachineIRBuilder &B, MachineInstr &MI,
1229 const AMDGPURegisterBankInfo::OperandsMapper &OpdMapper,
1230 int RsrcIdx) const {
1231 const int NumDefs = MI.getNumExplicitDefs();
1232
1233 // The reported argument index is relative to the IR intrinsic call arguments,
1234 // so we need to shift by the number of defs and the intrinsic ID.
1235 RsrcIdx += NumDefs + 1;
1236
1237 // Insert copies to VGPR arguments.
1238 applyDefaultMapping(OpdMapper);
1239
1240 // Fixup any SGPR arguments.
1241 SmallVector<unsigned, 4> SGPRIndexes;
1242 for (int I = NumDefs, NumOps = MI.getNumOperands(); I != NumOps; ++I) {
1243 if (!MI.getOperand(i: I).isReg())
1244 continue;
1245
1246 // If this intrinsic has a sampler, it immediately follows rsrc.
1247 if (I == RsrcIdx || I == RsrcIdx + 1)
1248 SGPRIndexes.push_back(Elt: I);
1249 }
1250
1251 executeInWaterfallLoop(B, MI, OpIndices: SGPRIndexes);
1252 return true;
1253}
1254
1255// Analyze a combined offset from an llvm.amdgcn.s.buffer intrinsic and store
1256// the three offsets (voffset, soffset and instoffset)
1257unsigned AMDGPURegisterBankInfo::setBufferOffsets(
1258 MachineIRBuilder &B, Register CombinedOffset, Register &VOffsetReg,
1259 Register &SOffsetReg, int64_t &InstOffsetVal, Align Alignment) const {
1260 const LLT S32 = LLT::scalar(SizeInBits: 32);
1261 MachineRegisterInfo *MRI = B.getMRI();
1262
1263 if (std::optional<int64_t> Imm =
1264 getIConstantVRegSExtVal(VReg: CombinedOffset, MRI: *MRI)) {
1265 uint32_t SOffset, ImmOffset;
1266 if (TII->splitMUBUFOffset(Imm: *Imm, SOffset, ImmOffset, Alignment)) {
1267 VOffsetReg = B.buildConstant(Res: S32, Val: 0).getReg(Idx: 0);
1268 SOffsetReg = B.buildConstant(Res: S32, Val: SOffset).getReg(Idx: 0);
1269 InstOffsetVal = ImmOffset;
1270
1271 B.getMRI()->setRegBank(Reg: VOffsetReg, RegBank: AMDGPU::VGPRRegBank);
1272 B.getMRI()->setRegBank(Reg: SOffsetReg, RegBank: AMDGPU::SGPRRegBank);
1273 return SOffset + ImmOffset;
1274 }
1275 }
1276
1277 const bool CheckNUW = Subtarget.hasGFX1250Insts();
1278 Register Base;
1279 unsigned Offset;
1280
1281 std::tie(args&: Base, args&: Offset) =
1282 AMDGPU::getBaseWithConstantOffset(MRI&: *MRI, Reg: CombinedOffset,
1283 /*KnownBits=*/ValueTracking: nullptr,
1284 /*CheckNUW=*/CheckNUW);
1285
1286 uint32_t SOffset, ImmOffset;
1287 if (static_cast<int32_t>(Offset) > 0 &&
1288 TII->splitMUBUFOffset(Imm: Offset, SOffset, ImmOffset, Alignment)) {
1289 if (getRegBank(Reg: Base, MRI: *MRI, TRI: *TRI) == &AMDGPU::VGPRRegBank) {
1290 VOffsetReg = Base;
1291 SOffsetReg = B.buildConstant(Res: S32, Val: SOffset).getReg(Idx: 0);
1292 B.getMRI()->setRegBank(Reg: SOffsetReg, RegBank: AMDGPU::SGPRRegBank);
1293 InstOffsetVal = ImmOffset;
1294 return 0; // XXX - Why is this 0?
1295 }
1296
1297 // If we have SGPR base, we can use it for soffset.
1298 if (SOffset == 0) {
1299 VOffsetReg = B.buildConstant(Res: S32, Val: 0).getReg(Idx: 0);
1300 B.getMRI()->setRegBank(Reg: VOffsetReg, RegBank: AMDGPU::VGPRRegBank);
1301 SOffsetReg = Base;
1302 InstOffsetVal = ImmOffset;
1303 return 0; // XXX - Why is this 0?
1304 }
1305 }
1306
1307 // Handle the variable sgpr + vgpr case.
1308 MachineInstr *Add = getOpcodeDef(Opcode: AMDGPU::G_ADD, Reg: CombinedOffset, MRI: *MRI);
1309 if (Add && static_cast<int32_t>(Offset) >= 0 &&
1310 (!CheckNUW || Add->getFlag(Flag: MachineInstr::NoUWrap))) {
1311 Register Src0 = getSrcRegIgnoringCopies(Reg: Add->getOperand(i: 1).getReg(), MRI: *MRI);
1312 Register Src1 = getSrcRegIgnoringCopies(Reg: Add->getOperand(i: 2).getReg(), MRI: *MRI);
1313
1314 const RegisterBank *Src0Bank = getRegBank(Reg: Src0, MRI: *MRI, TRI: *TRI);
1315 const RegisterBank *Src1Bank = getRegBank(Reg: Src1, MRI: *MRI, TRI: *TRI);
1316
1317 if (Src0Bank == &AMDGPU::VGPRRegBank && Src1Bank == &AMDGPU::SGPRRegBank) {
1318 VOffsetReg = Src0;
1319 SOffsetReg = Src1;
1320 return 0;
1321 }
1322
1323 if (Src0Bank == &AMDGPU::SGPRRegBank && Src1Bank == &AMDGPU::VGPRRegBank) {
1324 VOffsetReg = Src1;
1325 SOffsetReg = Src0;
1326 return 0;
1327 }
1328 }
1329
1330 // Ensure we have a VGPR for the combined offset. This could be an issue if we
1331 // have an SGPR offset and a VGPR resource.
1332 if (getRegBank(Reg: CombinedOffset, MRI: *MRI, TRI: *TRI) == &AMDGPU::VGPRRegBank) {
1333 VOffsetReg = CombinedOffset;
1334 } else {
1335 VOffsetReg = B.buildCopy(Res: S32, Op: CombinedOffset).getReg(Idx: 0);
1336 B.getMRI()->setRegBank(Reg: VOffsetReg, RegBank: AMDGPU::VGPRRegBank);
1337 }
1338
1339 SOffsetReg = B.buildConstant(Res: S32, Val: 0).getReg(Idx: 0);
1340 B.getMRI()->setRegBank(Reg: SOffsetReg, RegBank: AMDGPU::SGPRRegBank);
1341 return 0;
1342}
1343
1344static unsigned getSBufferLoadCorrespondingBufferLoadOpcode(unsigned Opc) {
1345 switch (Opc) {
1346 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD:
1347 return AMDGPU::G_AMDGPU_BUFFER_LOAD;
1348 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_UBYTE:
1349 return AMDGPU::G_AMDGPU_BUFFER_LOAD_UBYTE;
1350 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_SBYTE:
1351 return AMDGPU::G_AMDGPU_BUFFER_LOAD_SBYTE;
1352 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_USHORT:
1353 return AMDGPU::G_AMDGPU_BUFFER_LOAD_USHORT;
1354 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_SSHORT:
1355 return AMDGPU::G_AMDGPU_BUFFER_LOAD_SSHORT;
1356 default:
1357 break;
1358 }
1359 llvm_unreachable("Unexpected s_buffer_load opcode");
1360}
1361
1362bool AMDGPURegisterBankInfo::applyMappingSBufferLoad(
1363 MachineIRBuilder &B, const OperandsMapper &OpdMapper) const {
1364 MachineInstr &MI = OpdMapper.getMI();
1365 MachineRegisterInfo &MRI = OpdMapper.getMRI();
1366
1367 const LLT S32 = LLT::scalar(SizeInBits: 32);
1368 Register Dst = MI.getOperand(i: 0).getReg();
1369 LLT Ty = MRI.getType(Reg: Dst);
1370
1371 const RegisterBank *RSrcBank =
1372 OpdMapper.getInstrMapping().getOperandMapping(i: 1).BreakDown[0].RegBank;
1373 const RegisterBank *OffsetBank =
1374 OpdMapper.getInstrMapping().getOperandMapping(i: 2).BreakDown[0].RegBank;
1375 if (RSrcBank == &AMDGPU::SGPRRegBank &&
1376 OffsetBank == &AMDGPU::SGPRRegBank)
1377 return true; // Legal mapping
1378
1379 // FIXME: 96-bit case was widened during legalize. We need to narrow it back
1380 // here but don't have an MMO.
1381
1382 unsigned LoadSize = Ty.getSizeInBits();
1383 int NumLoads = 1;
1384 if (LoadSize == 256 || LoadSize == 512) {
1385 NumLoads = LoadSize / 128;
1386 Ty = Ty.divide(Factor: NumLoads);
1387 }
1388
1389 // Use the alignment to ensure that the required offsets will fit into the
1390 // immediate offsets.
1391 const Align Alignment = NumLoads > 1 ? Align(16 * NumLoads) : Align(1);
1392
1393 MachineFunction &MF = B.getMF();
1394
1395 Register SOffset;
1396 Register VOffset;
1397 int64_t ImmOffset = 0;
1398
1399 unsigned MMOOffset = setBufferOffsets(B, CombinedOffset: MI.getOperand(i: 2).getReg(), VOffsetReg&: VOffset,
1400 SOffsetReg&: SOffset, InstOffsetVal&: ImmOffset, Alignment);
1401
1402 // TODO: 96-bit loads were widened to 128-bit results. Shrink the result if we
1403 // can, but we need to track an MMO for that.
1404 const unsigned MemSize = (Ty.getSizeInBits() + 7) / 8;
1405 const Align MemAlign(4); // FIXME: ABI type alignment?
1406 MachineMemOperand *BaseMMO = MF.getMachineMemOperand(
1407 PtrInfo: MachinePointerInfo(),
1408 F: MachineMemOperand::MOLoad | MachineMemOperand::MODereferenceable |
1409 MachineMemOperand::MOInvariant,
1410 Size: MemSize, BaseAlignment: MemAlign);
1411 if (MMOOffset != 0)
1412 BaseMMO = MF.getMachineMemOperand(MMO: BaseMMO, Offset: MMOOffset, Size: MemSize);
1413
1414 // If only the offset is divergent, emit a MUBUF buffer load instead. We can
1415 // assume that the buffer is unswizzled.
1416
1417 Register RSrc = MI.getOperand(i: 1).getReg();
1418 Register VIndex = B.buildConstant(Res: S32, Val: 0).getReg(Idx: 0);
1419 B.getMRI()->setRegBank(Reg: VIndex, RegBank: AMDGPU::VGPRRegBank);
1420 unsigned CachePolicy = MI.getOperand(i: 3).getImm();
1421
1422 SmallVector<Register, 4> LoadParts(NumLoads);
1423
1424 MachineBasicBlock::iterator MII = MI.getIterator();
1425 MachineInstrSpan Span(MII, &B.getMBB());
1426
1427 for (int i = 0; i < NumLoads; ++i) {
1428 if (NumLoads == 1) {
1429 LoadParts[i] = Dst;
1430 } else {
1431 LoadParts[i] = MRI.createGenericVirtualRegister(Ty);
1432 MRI.setRegBank(Reg: LoadParts[i], RegBank: AMDGPU::VGPRRegBank);
1433 }
1434
1435 if (i != 0)
1436 BaseMMO = MF.getMachineMemOperand(MMO: BaseMMO, Offset: 16, Size: MemSize);
1437
1438 B.buildInstr(Opcode: getSBufferLoadCorrespondingBufferLoadOpcode(Opc: MI.getOpcode()))
1439 .addDef(RegNo: LoadParts[i]) // vdata
1440 .addUse(RegNo: RSrc) // rsrc
1441 .addUse(RegNo: VIndex) // vindex
1442 .addUse(RegNo: VOffset) // voffset
1443 .addUse(RegNo: SOffset) // soffset
1444 .addImm(Val: ImmOffset + 16 * i) // offset(imm)
1445 .addImm(Val: CachePolicy) // cachepolicy, swizzled buffer(imm)
1446 .addImm(Val: 0) // idxen(imm)
1447 .addMemOperand(MMO: BaseMMO);
1448 }
1449
1450 // TODO: If only the resource is a VGPR, it may be better to execute the
1451 // scalar load in the waterfall loop if the resource is expected to frequently
1452 // be dynamically uniform.
1453 if (RSrcBank != &AMDGPU::SGPRRegBank) {
1454 // Remove the original instruction to avoid potentially confusing the
1455 // waterfall loop logic.
1456 B.setInstr(*Span.begin());
1457 MI.eraseFromParent();
1458
1459 SmallSet<Register, 4> OpsToWaterfall;
1460
1461 OpsToWaterfall.insert(V: RSrc);
1462 executeInWaterfallLoop(B, Range: make_range(x: Span.begin(), y: Span.end()),
1463 SGPROperandRegs&: OpsToWaterfall);
1464 }
1465
1466 if (NumLoads != 1) {
1467 if (Ty.isVector())
1468 B.buildConcatVectors(Res: Dst, Ops: LoadParts);
1469 else
1470 B.buildMergeLikeInstr(Res: Dst, Ops: LoadParts);
1471 }
1472
1473 // We removed the instruction earlier with a waterfall loop.
1474 if (RSrcBank == &AMDGPU::SGPRRegBank)
1475 MI.eraseFromParent();
1476
1477 return true;
1478}
1479
1480bool AMDGPURegisterBankInfo::applyMappingBFE(MachineIRBuilder &B,
1481 const OperandsMapper &OpdMapper,
1482 bool Signed) const {
1483 MachineInstr &MI = OpdMapper.getMI();
1484 MachineRegisterInfo &MRI = OpdMapper.getMRI();
1485
1486 // Insert basic copies
1487 applyDefaultMapping(OpdMapper);
1488
1489 Register DstReg = MI.getOperand(i: 0).getReg();
1490 LLT Ty = MRI.getType(Reg: DstReg);
1491
1492 const LLT S32 = LLT::scalar(SizeInBits: 32);
1493
1494 unsigned FirstOpnd = isa<GIntrinsic>(Val: MI) ? 2 : 1;
1495 Register SrcReg = MI.getOperand(i: FirstOpnd).getReg();
1496 Register OffsetReg = MI.getOperand(i: FirstOpnd + 1).getReg();
1497 Register WidthReg = MI.getOperand(i: FirstOpnd + 2).getReg();
1498
1499 const RegisterBank *DstBank =
1500 OpdMapper.getInstrMapping().getOperandMapping(i: 0).BreakDown[0].RegBank;
1501 if (DstBank == &AMDGPU::VGPRRegBank) {
1502 if (Ty == S32)
1503 return true;
1504
1505 // There is no 64-bit vgpr bitfield extract instructions so the operation
1506 // is expanded to a sequence of instructions that implement the operation.
1507 ApplyRegBankMapping ApplyBank(B, *this, MRI, &AMDGPU::VGPRRegBank);
1508
1509 const LLT S64 = LLT::scalar(SizeInBits: 64);
1510 // Shift the source operand so that extracted bits start at bit 0.
1511 auto ShiftOffset = Signed ? B.buildAShr(Dst: S64, Src0: SrcReg, Src1: OffsetReg)
1512 : B.buildLShr(Dst: S64, Src0: SrcReg, Src1: OffsetReg);
1513 auto UnmergeSOffset = B.buildUnmerge(Res: {S32, S32}, Op: ShiftOffset);
1514
1515 // A 64-bit bitfield extract uses the 32-bit bitfield extract instructions
1516 // if the width is a constant.
1517 if (auto ConstWidth = getIConstantVRegValWithLookThrough(VReg: WidthReg, MRI)) {
1518 // Use the 32-bit bitfield extract instruction if the width is a constant.
1519 // Depending on the width size, use either the low or high 32-bits.
1520 auto Zero = B.buildConstant(Res: S32, Val: 0);
1521 auto WidthImm = ConstWidth->Value.getZExtValue();
1522 if (WidthImm <= 32) {
1523 // Use bitfield extract on the lower 32-bit source, and then sign-extend
1524 // or clear the upper 32-bits.
1525 auto Extract =
1526 Signed ? B.buildSbfx(Dst: S32, Src: UnmergeSOffset.getReg(Idx: 0), LSB: Zero, Width: WidthReg)
1527 : B.buildUbfx(Dst: S32, Src: UnmergeSOffset.getReg(Idx: 0), LSB: Zero, Width: WidthReg);
1528 auto Extend =
1529 Signed ? B.buildAShr(Dst: S32, Src0: Extract, Src1: B.buildConstant(Res: S32, Val: 31)) : Zero;
1530 B.buildMergeLikeInstr(Res: DstReg, Ops: {Extract, Extend});
1531 } else {
1532 // Use bitfield extract on upper 32-bit source, and combine with lower
1533 // 32-bit source.
1534 auto UpperWidth = B.buildConstant(Res: S32, Val: WidthImm - 32);
1535 auto Extract =
1536 Signed
1537 ? B.buildSbfx(Dst: S32, Src: UnmergeSOffset.getReg(Idx: 1), LSB: Zero, Width: UpperWidth)
1538 : B.buildUbfx(Dst: S32, Src: UnmergeSOffset.getReg(Idx: 1), LSB: Zero, Width: UpperWidth);
1539 B.buildMergeLikeInstr(Res: DstReg, Ops: {UnmergeSOffset.getReg(Idx: 0), Extract});
1540 }
1541 MI.eraseFromParent();
1542 return true;
1543 }
1544
1545 // Expand to Src >> Offset << (64 - Width) >> (64 - Width) using 64-bit
1546 // operations.
1547 auto ExtShift = B.buildSub(Dst: S32, Src0: B.buildConstant(Res: S32, Val: 64), Src1: WidthReg);
1548 auto SignBit = B.buildShl(Dst: S64, Src0: ShiftOffset, Src1: ExtShift);
1549 if (Signed)
1550 B.buildAShr(Dst: S64, Src0: SignBit, Src1: ExtShift);
1551 else
1552 B.buildLShr(Dst: S64, Src0: SignBit, Src1: ExtShift);
1553 MI.eraseFromParent();
1554 return true;
1555 }
1556
1557 // The scalar form packs the offset and width in a single operand.
1558
1559 ApplyRegBankMapping ApplyBank(B, *this, MRI, &AMDGPU::SGPRRegBank);
1560
1561 // Ensure the high bits are clear to insert the offset.
1562 auto OffsetMask = B.buildConstant(Res: S32, Val: maskTrailingOnes<unsigned>(N: 6));
1563 auto ClampOffset = B.buildAnd(Dst: S32, Src0: OffsetReg, Src1: OffsetMask);
1564
1565 // Zeros out the low bits, so don't bother clamping the input value.
1566 auto ShiftWidth = B.buildShl(Dst: S32, Src0: WidthReg, Src1: B.buildConstant(Res: S32, Val: 16));
1567
1568 // Transformation function, pack the offset and width of a BFE into
1569 // the format expected by the S_BFE_I32 / S_BFE_U32. In the second
1570 // source, bits [5:0] contain the offset and bits [22:16] the width.
1571 auto MergedInputs = B.buildOr(Dst: S32, Src0: ClampOffset, Src1: ShiftWidth);
1572
1573 // TODO: It might be worth using a pseudo here to avoid scc clobber and
1574 // register class constraints.
1575 unsigned Opc = Ty == S32 ? (Signed ? AMDGPU::S_BFE_I32 : AMDGPU::S_BFE_U32) :
1576 (Signed ? AMDGPU::S_BFE_I64 : AMDGPU::S_BFE_U64);
1577
1578 auto MIB = B.buildInstr(Opc, DstOps: {DstReg}, SrcOps: {SrcReg, MergedInputs});
1579 constrainSelectedInstRegOperands(I&: *MIB, TII: *TII, TRI: *TRI, RBI: *this);
1580
1581 MI.eraseFromParent();
1582 return true;
1583}
1584
1585bool AMDGPURegisterBankInfo::applyMappingMAD_64_32(
1586 MachineIRBuilder &B, const OperandsMapper &OpdMapper) const {
1587 MachineInstr &MI = OpdMapper.getMI();
1588 MachineRegisterInfo &MRI = OpdMapper.getMRI();
1589
1590 // Insert basic copies.
1591 applyDefaultMapping(OpdMapper);
1592
1593 Register Dst0 = MI.getOperand(i: 0).getReg();
1594 Register Dst1 = MI.getOperand(i: 1).getReg();
1595 Register Src0 = MI.getOperand(i: 2).getReg();
1596 Register Src1 = MI.getOperand(i: 3).getReg();
1597 Register Src2 = MI.getOperand(i: 4).getReg();
1598
1599 if (MRI.getRegBankOrNull(Reg: Src0) == &AMDGPU::VGPRRegBank)
1600 return true;
1601
1602 bool IsUnsigned = MI.getOpcode() == AMDGPU::G_AMDGPU_MAD_U64_U32;
1603 LLT S1 = LLT::scalar(SizeInBits: 1);
1604 LLT S32 = LLT::scalar(SizeInBits: 32);
1605
1606 bool DstOnValu = MRI.getRegBankOrNull(Reg: Src2) == &AMDGPU::VGPRRegBank;
1607 bool Accumulate = true;
1608
1609 if (!DstOnValu) {
1610 if (mi_match(R: Src2, MRI, P: m_ZeroInt()))
1611 Accumulate = false;
1612 }
1613
1614 // Keep the multiplication on the SALU.
1615 Register DstHi;
1616 Register DstLo = B.buildMul(Dst: S32, Src0, Src1).getReg(Idx: 0);
1617 bool MulHiInVgpr = false;
1618
1619 MRI.setRegBank(Reg: DstLo, RegBank: AMDGPU::SGPRRegBank);
1620
1621 if (Subtarget.hasSMulHi()) {
1622 DstHi = IsUnsigned ? B.buildUMulH(Dst: S32, Src0, Src1).getReg(Idx: 0)
1623 : B.buildSMulH(Dst: S32, Src0, Src1).getReg(Idx: 0);
1624 MRI.setRegBank(Reg: DstHi, RegBank: AMDGPU::SGPRRegBank);
1625 } else {
1626 Register VSrc0 = B.buildCopy(Res: S32, Op: Src0).getReg(Idx: 0);
1627 Register VSrc1 = B.buildCopy(Res: S32, Op: Src1).getReg(Idx: 0);
1628
1629 MRI.setRegBank(Reg: VSrc0, RegBank: AMDGPU::VGPRRegBank);
1630 MRI.setRegBank(Reg: VSrc1, RegBank: AMDGPU::VGPRRegBank);
1631
1632 DstHi = IsUnsigned ? B.buildUMulH(Dst: S32, Src0: VSrc0, Src1: VSrc1).getReg(Idx: 0)
1633 : B.buildSMulH(Dst: S32, Src0: VSrc0, Src1: VSrc1).getReg(Idx: 0);
1634 MRI.setRegBank(Reg: DstHi, RegBank: AMDGPU::VGPRRegBank);
1635
1636 if (!DstOnValu) {
1637 DstHi = buildReadFirstLane(B, MRI, Src: DstHi);
1638 } else {
1639 MulHiInVgpr = true;
1640 }
1641 }
1642
1643 // Accumulate and produce the "carry-out" bit.
1644 //
1645 // The "carry-out" is defined as bit 64 of the result when computed as a
1646 // big integer. For unsigned multiply-add, this matches the usual definition
1647 // of carry-out. For signed multiply-add, bit 64 is the sign bit of the
1648 // result, which is determined as:
1649 // sign(Src0 * Src1) + sign(Src2) + carry-out from unsigned 64-bit add
1650 LLT CarryType = DstOnValu ? S1 : S32;
1651 const RegisterBank &CarryBank =
1652 DstOnValu ? AMDGPU::VCCRegBank : AMDGPU::SGPRRegBank;
1653 const RegisterBank &DstBank =
1654 DstOnValu ? AMDGPU::VGPRRegBank : AMDGPU::SGPRRegBank;
1655 Register Carry;
1656 Register Zero;
1657
1658 if (!IsUnsigned) {
1659 Zero = B.buildConstant(Res: S32, Val: 0).getReg(Idx: 0);
1660 MRI.setRegBank(Reg: Zero,
1661 RegBank: MulHiInVgpr ? AMDGPU::VGPRRegBank : AMDGPU::SGPRRegBank);
1662
1663 Carry = B.buildICmp(Pred: CmpInst::ICMP_SLT, Res: MulHiInVgpr ? S1 : S32, Op0: DstHi, Op1: Zero)
1664 .getReg(Idx: 0);
1665 MRI.setRegBank(Reg: Carry, RegBank: MulHiInVgpr ? AMDGPU::VCCRegBank
1666 : AMDGPU::SGPRRegBank);
1667
1668 if (DstOnValu && !MulHiInVgpr) {
1669 Carry = B.buildTrunc(Res: S1, Op: Carry).getReg(Idx: 0);
1670 MRI.setRegBank(Reg: Carry, RegBank: AMDGPU::VCCRegBank);
1671 }
1672 }
1673
1674 if (Accumulate) {
1675 if (DstOnValu) {
1676 DstLo = B.buildCopy(Res: S32, Op: DstLo).getReg(Idx: 0);
1677 DstHi = B.buildCopy(Res: S32, Op: DstHi).getReg(Idx: 0);
1678 MRI.setRegBank(Reg: DstLo, RegBank: AMDGPU::VGPRRegBank);
1679 MRI.setRegBank(Reg: DstHi, RegBank: AMDGPU::VGPRRegBank);
1680 }
1681
1682 auto Unmerge = B.buildUnmerge(Res: S32, Op: Src2);
1683 Register Src2Lo = Unmerge.getReg(Idx: 0);
1684 Register Src2Hi = Unmerge.getReg(Idx: 1);
1685 MRI.setRegBank(Reg: Src2Lo, RegBank: DstBank);
1686 MRI.setRegBank(Reg: Src2Hi, RegBank: DstBank);
1687
1688 if (!IsUnsigned) {
1689 auto Src2Sign = B.buildICmp(Pred: CmpInst::ICMP_SLT, Res: CarryType, Op0: Src2Hi, Op1: Zero);
1690 MRI.setRegBank(Reg: Src2Sign.getReg(Idx: 0), RegBank: CarryBank);
1691
1692 Carry = B.buildXor(Dst: CarryType, Src0: Carry, Src1: Src2Sign).getReg(Idx: 0);
1693 MRI.setRegBank(Reg: Carry, RegBank: CarryBank);
1694 }
1695
1696 auto AddLo = B.buildUAddo(Res: S32, CarryOut: CarryType, Op0: DstLo, Op1: Src2Lo);
1697 DstLo = AddLo.getReg(Idx: 0);
1698 Register CarryLo = AddLo.getReg(Idx: 1);
1699 MRI.setRegBank(Reg: DstLo, RegBank: DstBank);
1700 MRI.setRegBank(Reg: CarryLo, RegBank: CarryBank);
1701
1702 auto AddHi = B.buildUAdde(Res: S32, CarryOut: CarryType, Op0: DstHi, Op1: Src2Hi, CarryIn: CarryLo);
1703 DstHi = AddHi.getReg(Idx: 0);
1704 MRI.setRegBank(Reg: DstHi, RegBank: DstBank);
1705
1706 Register CarryHi = AddHi.getReg(Idx: 1);
1707 MRI.setRegBank(Reg: CarryHi, RegBank: CarryBank);
1708
1709 if (IsUnsigned) {
1710 Carry = CarryHi;
1711 } else {
1712 Carry = B.buildXor(Dst: CarryType, Src0: Carry, Src1: CarryHi).getReg(Idx: 0);
1713 MRI.setRegBank(Reg: Carry, RegBank: CarryBank);
1714 }
1715 } else {
1716 if (IsUnsigned) {
1717 Carry = B.buildConstant(Res: CarryType, Val: 0).getReg(Idx: 0);
1718 MRI.setRegBank(Reg: Carry, RegBank: CarryBank);
1719 }
1720 }
1721
1722 B.buildMergeLikeInstr(Res: Dst0, Ops: {DstLo, DstHi});
1723
1724 if (DstOnValu) {
1725 B.buildCopy(Res: Dst1, Op: Carry);
1726 } else {
1727 B.buildTrunc(Res: Dst1, Op: Carry);
1728 }
1729
1730 MI.eraseFromParent();
1731 return true;
1732}
1733
1734// Return a suitable opcode for extending the operands of Opc when widening.
1735static unsigned getExtendOp(unsigned Opc) {
1736 switch (Opc) {
1737 case TargetOpcode::G_ASHR:
1738 case TargetOpcode::G_SMIN:
1739 case TargetOpcode::G_SMAX:
1740 return TargetOpcode::G_SEXT;
1741 case TargetOpcode::G_LSHR:
1742 case TargetOpcode::G_UMIN:
1743 case TargetOpcode::G_UMAX:
1744 return TargetOpcode::G_ZEXT;
1745 default:
1746 return TargetOpcode::G_ANYEXT;
1747 }
1748}
1749
1750// Emit a legalized extension from <2 x s16> to 2 32-bit components, avoiding
1751// any illegal vector extend or unmerge operations.
1752static std::pair<Register, Register>
1753unpackV2S16ToS32(MachineIRBuilder &B, Register Src, unsigned ExtOpcode) {
1754 const LLT S32 = LLT::scalar(SizeInBits: 32);
1755 auto Bitcast = B.buildBitcast(Dst: S32, Src);
1756
1757 if (ExtOpcode == TargetOpcode::G_SEXT) {
1758 auto ExtLo = B.buildSExtInReg(Res: S32, Op: Bitcast, ImmOp: 16);
1759 auto ShiftHi = B.buildAShr(Dst: S32, Src0: Bitcast, Src1: B.buildConstant(Res: S32, Val: 16));
1760 return std::pair(ExtLo.getReg(Idx: 0), ShiftHi.getReg(Idx: 0));
1761 }
1762
1763 auto ShiftHi = B.buildLShr(Dst: S32, Src0: Bitcast, Src1: B.buildConstant(Res: S32, Val: 16));
1764 if (ExtOpcode == TargetOpcode::G_ZEXT) {
1765 auto ExtLo = B.buildAnd(Dst: S32, Src0: Bitcast, Src1: B.buildConstant(Res: S32, Val: 0xffff));
1766 return std::pair(ExtLo.getReg(Idx: 0), ShiftHi.getReg(Idx: 0));
1767 }
1768
1769 assert(ExtOpcode == TargetOpcode::G_ANYEXT);
1770 return std::pair(Bitcast.getReg(Idx: 0), ShiftHi.getReg(Idx: 0));
1771}
1772
1773// For cases where only a single copy is inserted for matching register banks.
1774// Replace the register in the instruction operand
1775static bool substituteSimpleCopyRegs(
1776 const AMDGPURegisterBankInfo::OperandsMapper &OpdMapper, unsigned OpIdx) {
1777 SmallVector<unsigned, 1> SrcReg(OpdMapper.getVRegs(OpIdx));
1778 if (!SrcReg.empty()) {
1779 assert(SrcReg.size() == 1);
1780 OpdMapper.getMI().getOperand(i: OpIdx).setReg(SrcReg[0]);
1781 return true;
1782 }
1783
1784 return false;
1785}
1786
1787/// Handle register layout difference for f16 images for some subtargets.
1788Register AMDGPURegisterBankInfo::handleD16VData(MachineIRBuilder &B,
1789 MachineRegisterInfo &MRI,
1790 Register Reg) const {
1791 if (!Subtarget.hasUnpackedD16VMem())
1792 return Reg;
1793
1794 const LLT S16 = LLT::scalar(SizeInBits: 16);
1795 LLT StoreVT = MRI.getType(Reg);
1796 if (!StoreVT.isVector() || StoreVT.getElementType() != S16)
1797 return Reg;
1798
1799 auto Unmerge = B.buildUnmerge(Res: S16, Op: Reg);
1800
1801
1802 SmallVector<Register, 4> WideRegs;
1803 for (int I = 0, E = Unmerge->getNumOperands() - 1; I != E; ++I)
1804 WideRegs.push_back(Elt: Unmerge.getReg(Idx: I));
1805
1806 const LLT S32 = LLT::scalar(SizeInBits: 32);
1807 int NumElts = StoreVT.getNumElements();
1808
1809 return B.buildMergeLikeInstr(Res: LLT::fixed_vector(NumElements: NumElts, ScalarTy: S32), Ops: WideRegs)
1810 .getReg(Idx: 0);
1811}
1812
1813static std::pair<Register, unsigned>
1814getBaseWithConstantOffset(MachineRegisterInfo &MRI, Register Reg) {
1815 int64_t Const;
1816 if (mi_match(R: Reg, MRI, P: m_ICst(Cst&: Const)))
1817 return std::pair(Register(), Const);
1818
1819 Register Base;
1820 if (mi_match(R: Reg, MRI, P: m_GAdd(L: m_Reg(R&: Base), R: m_ICst(Cst&: Const))))
1821 return std::pair(Base, Const);
1822
1823 // TODO: Handle G_OR used for add case
1824 return std::pair(Reg, 0);
1825}
1826
1827std::pair<Register, unsigned>
1828AMDGPURegisterBankInfo::splitBufferOffsets(MachineIRBuilder &B,
1829 Register OrigOffset) const {
1830 const unsigned MaxImm = SIInstrInfo::getMaxMUBUFImmOffset(ST: Subtarget);
1831 Register BaseReg;
1832 unsigned ImmOffset;
1833 const LLT S32 = LLT::scalar(SizeInBits: 32);
1834
1835 // TODO: Use AMDGPU::getBaseWithConstantOffset() instead.
1836 std::tie(args&: BaseReg, args&: ImmOffset) = getBaseWithConstantOffset(MRI&: *B.getMRI(),
1837 Reg: OrigOffset);
1838
1839 unsigned C1 = 0;
1840 if (ImmOffset != 0) {
1841 // If the immediate value is too big for the immoffset field, put only bits
1842 // that would normally fit in the immoffset field. The remaining value that
1843 // is copied/added for the voffset field is a large power of 2, and it
1844 // stands more chance of being CSEd with the copy/add for another similar
1845 // load/store.
1846 // However, do not do that rounding down if that is a negative
1847 // number, as it appears to be illegal to have a negative offset in the
1848 // vgpr, even if adding the immediate offset makes it positive.
1849 unsigned Overflow = ImmOffset & ~MaxImm;
1850 ImmOffset -= Overflow;
1851 if (static_cast<int32_t>(Overflow) < 0) {
1852 Overflow += ImmOffset;
1853 ImmOffset = 0;
1854 }
1855
1856 C1 = ImmOffset;
1857 if (Overflow != 0) {
1858 if (!BaseReg)
1859 BaseReg = B.buildConstant(Res: S32, Val: Overflow).getReg(Idx: 0);
1860 else {
1861 auto OverflowVal = B.buildConstant(Res: S32, Val: Overflow);
1862 BaseReg = B.buildAdd(Dst: S32, Src0: BaseReg, Src1: OverflowVal).getReg(Idx: 0);
1863 }
1864 }
1865 }
1866
1867 if (!BaseReg)
1868 BaseReg = B.buildConstant(Res: S32, Val: 0).getReg(Idx: 0);
1869
1870 return {BaseReg, C1};
1871}
1872
1873bool AMDGPURegisterBankInfo::buildVCopy(MachineIRBuilder &B, Register DstReg,
1874 Register SrcReg) const {
1875 MachineRegisterInfo &MRI = *B.getMRI();
1876 LLT SrcTy = MRI.getType(Reg: SrcReg);
1877 if (SrcTy.getSizeInBits() == 32) {
1878 // Use a v_mov_b32 here to make the exec dependency explicit.
1879 B.buildInstr(Opcode: AMDGPU::V_MOV_B32_e32)
1880 .addDef(RegNo: DstReg)
1881 .addUse(RegNo: SrcReg);
1882 return constrainGenericRegister(Reg: DstReg, RC: AMDGPU::VGPR_32RegClass, MRI) &&
1883 constrainGenericRegister(Reg: SrcReg, RC: AMDGPU::SReg_32RegClass, MRI);
1884 }
1885
1886 Register TmpReg0 = MRI.createVirtualRegister(RegClass: &AMDGPU::VGPR_32RegClass);
1887 Register TmpReg1 = MRI.createVirtualRegister(RegClass: &AMDGPU::VGPR_32RegClass);
1888
1889 B.buildInstr(Opcode: AMDGPU::V_MOV_B32_e32)
1890 .addDef(RegNo: TmpReg0)
1891 .addUse(RegNo: SrcReg, Flags: {}, SubReg: AMDGPU::sub0);
1892 B.buildInstr(Opcode: AMDGPU::V_MOV_B32_e32)
1893 .addDef(RegNo: TmpReg1)
1894 .addUse(RegNo: SrcReg, Flags: {}, SubReg: AMDGPU::sub1);
1895 B.buildInstr(Opcode: AMDGPU::REG_SEQUENCE)
1896 .addDef(RegNo: DstReg)
1897 .addUse(RegNo: TmpReg0)
1898 .addImm(Val: AMDGPU::sub0)
1899 .addUse(RegNo: TmpReg1)
1900 .addImm(Val: AMDGPU::sub1);
1901
1902 return constrainGenericRegister(Reg: SrcReg, RC: AMDGPU::SReg_64RegClass, MRI) &&
1903 constrainGenericRegister(Reg: DstReg, RC: AMDGPU::VReg_64RegClass, MRI);
1904}
1905
1906/// Utility function for pushing dynamic vector indexes with a constant offset
1907/// into waterfall loops.
1908static void reinsertVectorIndexAdd(MachineIRBuilder &B,
1909 MachineInstr &IdxUseInstr,
1910 unsigned OpIdx,
1911 unsigned ConstOffset) {
1912 MachineRegisterInfo &MRI = *B.getMRI();
1913 const LLT S32 = LLT::scalar(SizeInBits: 32);
1914 Register WaterfallIdx = IdxUseInstr.getOperand(i: OpIdx).getReg();
1915 B.setInsertPt(MBB&: *IdxUseInstr.getParent(), II: IdxUseInstr.getIterator());
1916
1917 auto MaterializedOffset = B.buildConstant(Res: S32, Val: ConstOffset);
1918
1919 auto Add = B.buildAdd(Dst: S32, Src0: WaterfallIdx, Src1: MaterializedOffset);
1920 MRI.setRegBank(Reg: MaterializedOffset.getReg(Idx: 0), RegBank: AMDGPU::SGPRRegBank);
1921 MRI.setRegBank(Reg: Add.getReg(Idx: 0), RegBank: AMDGPU::SGPRRegBank);
1922 IdxUseInstr.getOperand(i: OpIdx).setReg(Add.getReg(Idx: 0));
1923}
1924
1925/// Implement extending a 32-bit value to a 64-bit value. \p Lo32Reg is the
1926/// original 32-bit source value (to be inserted in the low part of the combined
1927/// 64-bit result), and \p Hi32Reg is the high half of the combined 64-bit
1928/// value.
1929static void extendLow32IntoHigh32(MachineIRBuilder &B,
1930 Register Hi32Reg, Register Lo32Reg,
1931 unsigned ExtOpc,
1932 const RegisterBank &RegBank,
1933 bool IsBooleanSrc = false) {
1934 if (ExtOpc == AMDGPU::G_ZEXT) {
1935 B.buildConstant(Res: Hi32Reg, Val: 0);
1936 } else if (ExtOpc == AMDGPU::G_SEXT) {
1937 if (IsBooleanSrc) {
1938 // If we know the original source was an s1, the high half is the same as
1939 // the low.
1940 B.buildCopy(Res: Hi32Reg, Op: Lo32Reg);
1941 } else {
1942 // Replicate sign bit from 32-bit extended part.
1943 auto ShiftAmt = B.buildConstant(Res: LLT::scalar(SizeInBits: 32), Val: 31);
1944 B.getMRI()->setRegBank(Reg: ShiftAmt.getReg(Idx: 0), RegBank);
1945 B.buildAShr(Dst: Hi32Reg, Src0: Lo32Reg, Src1: ShiftAmt);
1946 }
1947 } else {
1948 assert(ExtOpc == AMDGPU::G_ANYEXT && "not an integer extension");
1949 B.buildUndef(Res: Hi32Reg);
1950 }
1951}
1952
1953bool AMDGPURegisterBankInfo::foldExtractEltToCmpSelect(
1954 MachineIRBuilder &B, MachineInstr &MI,
1955 const OperandsMapper &OpdMapper) const {
1956 MachineRegisterInfo &MRI = *B.getMRI();
1957
1958 Register VecReg = MI.getOperand(i: 1).getReg();
1959 Register Idx = MI.getOperand(i: 2).getReg();
1960
1961 const RegisterBank &IdxBank =
1962 *OpdMapper.getInstrMapping().getOperandMapping(i: 2).BreakDown[0].RegBank;
1963
1964 bool IsDivergentIdx = IdxBank != AMDGPU::SGPRRegBank;
1965
1966 LLT VecTy = MRI.getType(Reg: VecReg);
1967 unsigned EltSize = VecTy.getScalarSizeInBits();
1968 unsigned NumElem = VecTy.getNumElements();
1969
1970 if (!SITargetLowering::shouldExpandVectorDynExt(EltSize, NumElem,
1971 IsDivergentIdx, Subtarget: &Subtarget))
1972 return false;
1973
1974 LLT S32 = LLT::scalar(SizeInBits: 32);
1975
1976 const RegisterBank &DstBank =
1977 *OpdMapper.getInstrMapping().getOperandMapping(i: 0).BreakDown[0].RegBank;
1978 const RegisterBank &SrcBank =
1979 *OpdMapper.getInstrMapping().getOperandMapping(i: 1).BreakDown[0].RegBank;
1980
1981 const RegisterBank &CCBank =
1982 (DstBank == AMDGPU::SGPRRegBank &&
1983 SrcBank == AMDGPU::SGPRRegBank &&
1984 IdxBank == AMDGPU::SGPRRegBank) ? AMDGPU::SGPRRegBank
1985 : AMDGPU::VCCRegBank;
1986 LLT CCTy = (CCBank == AMDGPU::SGPRRegBank) ? S32 : LLT::scalar(SizeInBits: 1);
1987
1988 if (CCBank == AMDGPU::VCCRegBank && IdxBank == AMDGPU::SGPRRegBank) {
1989 Idx = B.buildCopy(Res: S32, Op: Idx)->getOperand(i: 0).getReg();
1990 MRI.setRegBank(Reg: Idx, RegBank: AMDGPU::VGPRRegBank);
1991 }
1992
1993 LLT EltTy = VecTy.getScalarType();
1994 SmallVector<Register, 2> DstRegs(OpdMapper.getVRegs(OpIdx: 0));
1995 unsigned NumLanes = DstRegs.size();
1996 if (!NumLanes)
1997 NumLanes = 1;
1998 else
1999 EltTy = MRI.getType(Reg: DstRegs[0]);
2000
2001 auto UnmergeToEltTy = B.buildUnmerge(Res: EltTy, Op: VecReg);
2002 SmallVector<Register, 2> Res(NumLanes);
2003 for (unsigned L = 0; L < NumLanes; ++L)
2004 Res[L] = UnmergeToEltTy.getReg(Idx: L);
2005
2006 for (unsigned I = 1; I < NumElem; ++I) {
2007 auto IC = B.buildConstant(Res: S32, Val: I);
2008 MRI.setRegBank(Reg: IC->getOperand(i: 0).getReg(), RegBank: AMDGPU::SGPRRegBank);
2009 auto Cmp = B.buildICmp(Pred: CmpInst::ICMP_EQ, Res: CCTy, Op0: Idx, Op1: IC);
2010 MRI.setRegBank(Reg: Cmp->getOperand(i: 0).getReg(), RegBank: CCBank);
2011
2012 for (unsigned L = 0; L < NumLanes; ++L) {
2013 auto S = B.buildSelect(Res: EltTy, Tst: Cmp,
2014 Op0: UnmergeToEltTy.getReg(Idx: I * NumLanes + L), Op1: Res[L]);
2015
2016 for (unsigned N : { 0, 2, 3 })
2017 MRI.setRegBank(Reg: S->getOperand(i: N).getReg(), RegBank: DstBank);
2018
2019 Res[L] = S->getOperand(i: 0).getReg();
2020 }
2021 }
2022
2023 for (unsigned L = 0; L < NumLanes; ++L) {
2024 Register DstReg = (NumLanes == 1) ? MI.getOperand(i: 0).getReg() : DstRegs[L];
2025 B.buildCopy(Res: DstReg, Op: Res[L]);
2026 MRI.setRegBank(Reg: DstReg, RegBank: DstBank);
2027 }
2028
2029 MRI.setRegBank(Reg: MI.getOperand(i: 0).getReg(), RegBank: DstBank);
2030 MI.eraseFromParent();
2031
2032 return true;
2033}
2034
2035// Insert a cross regbank copy for a register if it already has a bank that
2036// differs from the one we want to set.
2037static Register constrainRegToBank(MachineRegisterInfo &MRI,
2038 MachineIRBuilder &B, Register &Reg,
2039 const RegisterBank &Bank) {
2040 const RegisterBank *CurrBank = MRI.getRegBankOrNull(Reg);
2041 if (CurrBank && *CurrBank != Bank) {
2042 Register Copy = B.buildCopy(Res: MRI.getType(Reg), Op: Reg).getReg(Idx: 0);
2043 MRI.setRegBank(Reg: Copy, RegBank: Bank);
2044 return Copy;
2045 }
2046
2047 MRI.setRegBank(Reg, RegBank: Bank);
2048 return Reg;
2049}
2050
2051bool AMDGPURegisterBankInfo::foldInsertEltToCmpSelect(
2052 MachineIRBuilder &B, MachineInstr &MI,
2053 const OperandsMapper &OpdMapper) const {
2054
2055 MachineRegisterInfo &MRI = *B.getMRI();
2056 Register VecReg = MI.getOperand(i: 1).getReg();
2057 Register Idx = MI.getOperand(i: 3).getReg();
2058
2059 const RegisterBank &IdxBank =
2060 *OpdMapper.getInstrMapping().getOperandMapping(i: 3).BreakDown[0].RegBank;
2061
2062 bool IsDivergentIdx = IdxBank != AMDGPU::SGPRRegBank;
2063
2064 LLT VecTy = MRI.getType(Reg: VecReg);
2065 unsigned EltSize = VecTy.getScalarSizeInBits();
2066 unsigned NumElem = VecTy.getNumElements();
2067
2068 if (!SITargetLowering::shouldExpandVectorDynExt(EltSize, NumElem,
2069 IsDivergentIdx, Subtarget: &Subtarget))
2070 return false;
2071
2072 LLT S32 = LLT::scalar(SizeInBits: 32);
2073
2074 const RegisterBank &DstBank =
2075 *OpdMapper.getInstrMapping().getOperandMapping(i: 0).BreakDown[0].RegBank;
2076 const RegisterBank &SrcBank =
2077 *OpdMapper.getInstrMapping().getOperandMapping(i: 1).BreakDown[0].RegBank;
2078 const RegisterBank &InsBank =
2079 *OpdMapper.getInstrMapping().getOperandMapping(i: 2).BreakDown[0].RegBank;
2080
2081 const RegisterBank &CCBank =
2082 (DstBank == AMDGPU::SGPRRegBank &&
2083 SrcBank == AMDGPU::SGPRRegBank &&
2084 InsBank == AMDGPU::SGPRRegBank &&
2085 IdxBank == AMDGPU::SGPRRegBank) ? AMDGPU::SGPRRegBank
2086 : AMDGPU::VCCRegBank;
2087 LLT CCTy = (CCBank == AMDGPU::SGPRRegBank) ? S32 : LLT::scalar(SizeInBits: 1);
2088
2089 if (CCBank == AMDGPU::VCCRegBank && IdxBank == AMDGPU::SGPRRegBank) {
2090 Idx = B.buildCopy(Res: S32, Op: Idx)->getOperand(i: 0).getReg();
2091 MRI.setRegBank(Reg: Idx, RegBank: AMDGPU::VGPRRegBank);
2092 }
2093
2094 LLT EltTy = VecTy.getScalarType();
2095 SmallVector<Register, 2> InsRegs(OpdMapper.getVRegs(OpIdx: 2));
2096 unsigned NumLanes = InsRegs.size();
2097 if (!NumLanes) {
2098 NumLanes = 1;
2099 InsRegs.push_back(Elt: MI.getOperand(i: 2).getReg());
2100 } else {
2101 EltTy = MRI.getType(Reg: InsRegs[0]);
2102 }
2103
2104 auto UnmergeToEltTy = B.buildUnmerge(Res: EltTy, Op: VecReg);
2105 SmallVector<Register, 16> Ops(NumElem * NumLanes);
2106
2107 for (unsigned I = 0; I < NumElem; ++I) {
2108 auto IC = B.buildConstant(Res: S32, Val: I);
2109 MRI.setRegBank(Reg: IC->getOperand(i: 0).getReg(), RegBank: AMDGPU::SGPRRegBank);
2110 auto Cmp = B.buildICmp(Pred: CmpInst::ICMP_EQ, Res: CCTy, Op0: Idx, Op1: IC);
2111 MRI.setRegBank(Reg: Cmp->getOperand(i: 0).getReg(), RegBank: CCBank);
2112
2113 for (unsigned L = 0; L < NumLanes; ++L) {
2114 Register Op0 = constrainRegToBank(MRI, B, Reg&: InsRegs[L], Bank: DstBank);
2115 Register Op1 = UnmergeToEltTy.getReg(Idx: I * NumLanes + L);
2116 Op1 = constrainRegToBank(MRI, B, Reg&: Op1, Bank: DstBank);
2117
2118 Register Select = B.buildSelect(Res: EltTy, Tst: Cmp, Op0, Op1).getReg(Idx: 0);
2119 MRI.setRegBank(Reg: Select, RegBank: DstBank);
2120
2121 Ops[I * NumLanes + L] = Select;
2122 }
2123 }
2124
2125 LLT MergeTy = LLT::fixed_vector(NumElements: Ops.size(), ScalarTy: EltTy);
2126 if (MergeTy == MRI.getType(Reg: MI.getOperand(i: 0).getReg())) {
2127 B.buildBuildVector(Res: MI.getOperand(i: 0), Ops);
2128 } else {
2129 auto Vec = B.buildBuildVector(Res: MergeTy, Ops);
2130 MRI.setRegBank(Reg: Vec->getOperand(i: 0).getReg(), RegBank: DstBank);
2131 B.buildBitcast(Dst: MI.getOperand(i: 0).getReg(), Src: Vec);
2132 }
2133
2134 MRI.setRegBank(Reg: MI.getOperand(i: 0).getReg(), RegBank: DstBank);
2135 MI.eraseFromParent();
2136
2137 return true;
2138}
2139
2140// Break s_mul_u64 into 32-bit vector operations.
2141void AMDGPURegisterBankInfo::applyMappingSMULU64(
2142 MachineIRBuilder &B, const OperandsMapper &OpdMapper) const {
2143 SmallVector<Register, 2> DefRegs(OpdMapper.getVRegs(OpIdx: 0));
2144 SmallVector<Register, 2> Src0Regs(OpdMapper.getVRegs(OpIdx: 1));
2145 SmallVector<Register, 2> Src1Regs(OpdMapper.getVRegs(OpIdx: 2));
2146
2147 // All inputs are SGPRs, nothing special to do.
2148 if (DefRegs.empty()) {
2149 assert(Src0Regs.empty() && Src1Regs.empty());
2150 applyDefaultMapping(OpdMapper);
2151 return;
2152 }
2153
2154 assert(DefRegs.size() == 2);
2155 assert(Src0Regs.size() == Src1Regs.size() &&
2156 (Src0Regs.empty() || Src0Regs.size() == 2));
2157
2158 MachineRegisterInfo &MRI = OpdMapper.getMRI();
2159 MachineInstr &MI = OpdMapper.getMI();
2160 Register DstReg = MI.getOperand(i: 0).getReg();
2161 LLT HalfTy = LLT::scalar(SizeInBits: 32);
2162
2163 // Depending on where the source registers came from, the generic code may
2164 // have decided to split the inputs already or not. If not, we still need to
2165 // extract the values.
2166
2167 if (Src0Regs.empty())
2168 split64BitValueForMapping(B, Regs&: Src0Regs, HalfTy, Reg: MI.getOperand(i: 1).getReg());
2169 else
2170 setRegsToType(MRI, Regs: Src0Regs, NewTy: HalfTy);
2171
2172 if (Src1Regs.empty())
2173 split64BitValueForMapping(B, Regs&: Src1Regs, HalfTy, Reg: MI.getOperand(i: 2).getReg());
2174 else
2175 setRegsToType(MRI, Regs: Src1Regs, NewTy: HalfTy);
2176
2177 setRegsToType(MRI, Regs: DefRegs, NewTy: HalfTy);
2178
2179 // The multiplication is done as follows:
2180 //
2181 // Op1H Op1L
2182 // * Op0H Op0L
2183 // --------------------
2184 // Op1H*Op0L Op1L*Op0L
2185 // + Op1H*Op0H Op1L*Op0H
2186 // -----------------------------------------
2187 // (Op1H*Op0L + Op1L*Op0H + carry) Op1L*Op0L
2188 //
2189 // We drop Op1H*Op0H because the result of the multiplication is a 64-bit
2190 // value and that would overflow.
2191 // The low 32-bit value is Op1L*Op0L.
2192 // The high 32-bit value is Op1H*Op0L + Op1L*Op0H + carry (from
2193 // Op1L*Op0L).
2194
2195 ApplyRegBankMapping ApplyBank(B, *this, MRI, &AMDGPU::VGPRRegBank);
2196
2197 Register Hi = B.buildUMulH(Dst: HalfTy, Src0: Src0Regs[0], Src1: Src1Regs[0]).getReg(Idx: 0);
2198 Register MulLoHi = B.buildMul(Dst: HalfTy, Src0: Src0Regs[0], Src1: Src1Regs[1]).getReg(Idx: 0);
2199 Register Add = B.buildAdd(Dst: HalfTy, Src0: Hi, Src1: MulLoHi).getReg(Idx: 0);
2200 Register MulHiLo = B.buildMul(Dst: HalfTy, Src0: Src0Regs[1], Src1: Src1Regs[0]).getReg(Idx: 0);
2201 B.buildAdd(Dst: DefRegs[1], Src0: Add, Src1: MulHiLo);
2202 B.buildMul(Dst: DefRegs[0], Src0: Src0Regs[0], Src1: Src1Regs[0]);
2203
2204 MRI.setRegBank(Reg: DstReg, RegBank: AMDGPU::VGPRRegBank);
2205 MI.eraseFromParent();
2206}
2207
2208void AMDGPURegisterBankInfo::applyMappingImpl(
2209 MachineIRBuilder &B, const OperandsMapper &OpdMapper) const {
2210 MachineInstr &MI = OpdMapper.getMI();
2211 B.setInstrAndDebugLoc(MI);
2212 unsigned Opc = MI.getOpcode();
2213 MachineRegisterInfo &MRI = OpdMapper.getMRI();
2214 switch (Opc) {
2215 case AMDGPU::G_CONSTANT:
2216 case AMDGPU::G_IMPLICIT_DEF: {
2217 Register DstReg = MI.getOperand(i: 0).getReg();
2218 LLT DstTy = MRI.getType(Reg: DstReg);
2219 if (DstTy != LLT::scalar(SizeInBits: 1))
2220 break;
2221
2222 const RegisterBank *DstBank =
2223 OpdMapper.getInstrMapping().getOperandMapping(i: 0).BreakDown[0].RegBank;
2224 if (DstBank == &AMDGPU::VCCRegBank)
2225 break;
2226 SmallVector<Register, 1> DefRegs(OpdMapper.getVRegs(OpIdx: 0));
2227 if (DefRegs.empty())
2228 DefRegs.push_back(Elt: DstReg);
2229
2230 B.setInsertPt(MBB&: *MI.getParent(), II: ++MI.getIterator());
2231
2232 Register NewDstReg = MRI.createGenericVirtualRegister(Ty: LLT::scalar(SizeInBits: 32));
2233 LLVMContext &Ctx = B.getMF().getFunction().getContext();
2234
2235 MI.getOperand(i: 0).setReg(NewDstReg);
2236 if (Opc != AMDGPU::G_IMPLICIT_DEF) {
2237 uint64_t ConstVal = MI.getOperand(i: 1).getCImm()->getZExtValue();
2238 MI.getOperand(i: 1).setCImm(
2239 ConstantInt::get(Ty: IntegerType::getInt32Ty(C&: Ctx), V: ConstVal));
2240 }
2241
2242 MRI.setRegBank(Reg: NewDstReg, RegBank: *DstBank);
2243 B.buildTrunc(Res: DefRegs[0], Op: NewDstReg);
2244 return;
2245 }
2246 case AMDGPU::G_PHI: {
2247 Register DstReg = MI.getOperand(i: 0).getReg();
2248 LLT DstTy = MRI.getType(Reg: DstReg);
2249 if (DstTy != LLT::scalar(SizeInBits: 1))
2250 break;
2251
2252 const LLT S32 = LLT::scalar(SizeInBits: 32);
2253 const RegisterBank *DstBank =
2254 OpdMapper.getInstrMapping().getOperandMapping(i: 0).BreakDown[0].RegBank;
2255 if (DstBank == &AMDGPU::VCCRegBank) {
2256 applyDefaultMapping(OpdMapper);
2257 // The standard handling only considers the result register bank for
2258 // phis. For VCC, blindly inserting a copy when the phi is lowered will
2259 // produce an invalid copy. We can only copy with some kind of compare to
2260 // get a vector boolean result. Insert a register bank copy that will be
2261 // correctly lowered to a compare.
2262 for (unsigned I = 1, E = MI.getNumOperands(); I != E; I += 2) {
2263 Register SrcReg = MI.getOperand(i: I).getReg();
2264 const RegisterBank *SrcBank = getRegBank(Reg: SrcReg, MRI, TRI: *TRI);
2265
2266 if (SrcBank != &AMDGPU::VCCRegBank) {
2267 MachineBasicBlock *SrcMBB = MI.getOperand(i: I + 1).getMBB();
2268 B.setInsertPt(MBB&: *SrcMBB, II: SrcMBB->getFirstTerminator());
2269
2270 auto Copy = B.buildCopy(Res: LLT::scalar(SizeInBits: 1), Op: SrcReg);
2271 MRI.setRegBank(Reg: Copy.getReg(Idx: 0), RegBank: AMDGPU::VCCRegBank);
2272 MI.getOperand(i: I).setReg(Copy.getReg(Idx: 0));
2273 }
2274 }
2275
2276 return;
2277 }
2278
2279 // Phi handling is strange and only considers the bank of the destination.
2280 substituteSimpleCopyRegs(OpdMapper, OpIdx: 0);
2281
2282 // Promote SGPR/VGPR booleans to s32
2283 ApplyRegBankMapping ApplyBank(B, *this, MRI, DstBank);
2284 B.setInsertPt(MBB&: B.getMBB(), II: MI);
2285 LegalizerHelper Helper(B.getMF(), ApplyBank, B);
2286
2287 if (Helper.widenScalar(MI, TypeIdx: 0, WideTy: S32) != LegalizerHelper::Legalized)
2288 llvm_unreachable("widen scalar should have succeeded");
2289
2290 return;
2291 }
2292 case AMDGPU::G_FCMP:
2293 if (!Subtarget.hasSALUFloatInsts())
2294 break;
2295 [[fallthrough]];
2296 case AMDGPU::G_ICMP:
2297 case AMDGPU::G_UADDO:
2298 case AMDGPU::G_USUBO:
2299 case AMDGPU::G_UADDE:
2300 case AMDGPU::G_SADDE:
2301 case AMDGPU::G_USUBE:
2302 case AMDGPU::G_SSUBE: {
2303 unsigned BoolDstOp =
2304 (Opc == AMDGPU::G_ICMP || Opc == AMDGPU::G_FCMP) ? 0 : 1;
2305 Register DstReg = MI.getOperand(i: BoolDstOp).getReg();
2306
2307 const RegisterBank *DstBank =
2308 OpdMapper.getInstrMapping().getOperandMapping(i: 0).BreakDown[0].RegBank;
2309 if (DstBank != &AMDGPU::SGPRRegBank)
2310 break;
2311
2312 const bool HasCarryIn = MI.getNumOperands() == 5;
2313
2314 // If this is a scalar compare, promote the result to s32, as the selection
2315 // will end up using a copy to a 32-bit vreg.
2316 const LLT S32 = LLT::scalar(SizeInBits: 32);
2317 Register NewDstReg = MRI.createGenericVirtualRegister(Ty: S32);
2318 MRI.setRegBank(Reg: NewDstReg, RegBank: AMDGPU::SGPRRegBank);
2319 MI.getOperand(i: BoolDstOp).setReg(NewDstReg);
2320
2321 if (HasCarryIn) {
2322 Register NewSrcReg = MRI.createGenericVirtualRegister(Ty: S32);
2323 MRI.setRegBank(Reg: NewSrcReg, RegBank: AMDGPU::SGPRRegBank);
2324 B.buildZExt(Res: NewSrcReg, Op: MI.getOperand(i: 4).getReg());
2325 MI.getOperand(i: 4).setReg(NewSrcReg);
2326 }
2327
2328 MachineBasicBlock *MBB = MI.getParent();
2329 B.setInsertPt(MBB&: *MBB, II: std::next(x: MI.getIterator()));
2330
2331 // If we had a constrained VCC result register, a copy was inserted to VCC
2332 // from SGPR.
2333 SmallVector<Register, 1> DefRegs(OpdMapper.getVRegs(OpIdx: 0));
2334 if (DefRegs.empty())
2335 DefRegs.push_back(Elt: DstReg);
2336 B.buildTrunc(Res: DefRegs[0], Op: NewDstReg);
2337 return;
2338 }
2339 case AMDGPU::G_SELECT: {
2340 Register DstReg = MI.getOperand(i: 0).getReg();
2341 LLT DstTy = MRI.getType(Reg: DstReg);
2342
2343 SmallVector<Register, 1> CondRegs(OpdMapper.getVRegs(OpIdx: 1));
2344 if (CondRegs.empty())
2345 CondRegs.push_back(Elt: MI.getOperand(i: 1).getReg());
2346 else {
2347 assert(CondRegs.size() == 1);
2348 }
2349
2350 const RegisterBank *CondBank = getRegBank(Reg: CondRegs[0], MRI, TRI: *TRI);
2351 if (CondBank == &AMDGPU::SGPRRegBank) {
2352 const LLT S32 = LLT::scalar(SizeInBits: 32);
2353 Register NewCondReg = MRI.createGenericVirtualRegister(Ty: S32);
2354 MRI.setRegBank(Reg: NewCondReg, RegBank: AMDGPU::SGPRRegBank);
2355
2356 MI.getOperand(i: 1).setReg(NewCondReg);
2357 B.buildZExt(Res: NewCondReg, Op: CondRegs[0]);
2358 }
2359
2360 if (DstTy.getSizeInBits() != 64)
2361 break;
2362
2363 LLT HalfTy = getHalfSizedType(Ty: DstTy);
2364
2365 SmallVector<Register, 2> DefRegs(OpdMapper.getVRegs(OpIdx: 0));
2366 SmallVector<Register, 2> Src1Regs(OpdMapper.getVRegs(OpIdx: 2));
2367 SmallVector<Register, 2> Src2Regs(OpdMapper.getVRegs(OpIdx: 3));
2368
2369 // All inputs are SGPRs, nothing special to do.
2370 if (DefRegs.empty()) {
2371 assert(Src1Regs.empty() && Src2Regs.empty());
2372 break;
2373 }
2374
2375 if (Src1Regs.empty())
2376 split64BitValueForMapping(B, Regs&: Src1Regs, HalfTy, Reg: MI.getOperand(i: 2).getReg());
2377 else {
2378 setRegsToType(MRI, Regs: Src1Regs, NewTy: HalfTy);
2379 }
2380
2381 if (Src2Regs.empty())
2382 split64BitValueForMapping(B, Regs&: Src2Regs, HalfTy, Reg: MI.getOperand(i: 3).getReg());
2383 else
2384 setRegsToType(MRI, Regs: Src2Regs, NewTy: HalfTy);
2385
2386 setRegsToType(MRI, Regs: DefRegs, NewTy: HalfTy);
2387
2388 auto Flags = MI.getFlags();
2389 B.buildSelect(Res: DefRegs[0], Tst: CondRegs[0], Op0: Src1Regs[0], Op1: Src2Regs[0], Flags);
2390 B.buildSelect(Res: DefRegs[1], Tst: CondRegs[0], Op0: Src1Regs[1], Op1: Src2Regs[1], Flags);
2391
2392 MRI.setRegBank(Reg: DstReg, RegBank: AMDGPU::VGPRRegBank);
2393 MI.eraseFromParent();
2394 return;
2395 }
2396 case AMDGPU::G_BRCOND: {
2397 Register CondReg = MI.getOperand(i: 0).getReg();
2398 // FIXME: Should use legalizer helper, but should change bool ext type.
2399 const RegisterBank *CondBank =
2400 OpdMapper.getInstrMapping().getOperandMapping(i: 0).BreakDown[0].RegBank;
2401
2402 if (CondBank == &AMDGPU::SGPRRegBank) {
2403 const LLT S32 = LLT::scalar(SizeInBits: 32);
2404 Register NewCondReg = MRI.createGenericVirtualRegister(Ty: S32);
2405 MRI.setRegBank(Reg: NewCondReg, RegBank: AMDGPU::SGPRRegBank);
2406
2407 MI.getOperand(i: 0).setReg(NewCondReg);
2408 B.buildZExt(Res: NewCondReg, Op: CondReg);
2409 return;
2410 }
2411
2412 break;
2413 }
2414 case AMDGPU::G_AND:
2415 case AMDGPU::G_OR:
2416 case AMDGPU::G_XOR: {
2417 // 64-bit and is only available on the SALU, so split into 2 32-bit ops if
2418 // there is a VGPR input.
2419 Register DstReg = MI.getOperand(i: 0).getReg();
2420 LLT DstTy = MRI.getType(Reg: DstReg);
2421
2422 const RegisterBank *DstBank =
2423 OpdMapper.getInstrMapping().getOperandMapping(i: 0).BreakDown[0].RegBank;
2424
2425 if (DstTy.getSizeInBits() == 1) {
2426 if (DstBank == &AMDGPU::VCCRegBank)
2427 break;
2428
2429 MachineFunction *MF = MI.getMF();
2430 ApplyRegBankMapping ApplyBank(B, *this, MRI, DstBank);
2431 LegalizerHelper Helper(*MF, ApplyBank, B);
2432
2433 if (Helper.widenScalar(MI, TypeIdx: 0, WideTy: LLT::scalar(SizeInBits: 32)) !=
2434 LegalizerHelper::Legalized)
2435 llvm_unreachable("widen scalar should have succeeded");
2436 return;
2437 }
2438
2439 if (DstTy.getSizeInBits() == 16 && DstBank == &AMDGPU::SGPRRegBank) {
2440 const LLT S32 = LLT::scalar(SizeInBits: 32);
2441 MachineBasicBlock *MBB = MI.getParent();
2442 MachineFunction *MF = MBB->getParent();
2443 ApplyRegBankMapping ApplySALU(B, *this, MRI, &AMDGPU::SGPRRegBank);
2444 LegalizerHelper Helper(*MF, ApplySALU, B);
2445 // Widen to S32, but handle `G_XOR x, -1` differently. Legalizer widening
2446 // will use a G_ANYEXT to extend the -1 which prevents matching G_XOR -1
2447 // as "not".
2448 if (MI.getOpcode() == AMDGPU::G_XOR &&
2449 mi_match(R: MI.getOperand(i: 2).getReg(), MRI, P: m_SpecificICstOrSplat(RequestedValue: -1))) {
2450 Helper.widenScalarSrc(MI, WideTy: S32, OpIdx: 1, ExtOpcode: AMDGPU::G_ANYEXT);
2451 Helper.widenScalarSrc(MI, WideTy: S32, OpIdx: 2, ExtOpcode: AMDGPU::G_SEXT);
2452 Helper.widenScalarDst(MI, WideTy: S32);
2453 } else {
2454 if (Helper.widenScalar(MI, TypeIdx: 0, WideTy: S32) != LegalizerHelper::Legalized)
2455 llvm_unreachable("widen scalar should have succeeded");
2456 }
2457 return;
2458 }
2459
2460 if (DstTy.getSizeInBits() != 64)
2461 break;
2462
2463 LLT HalfTy = getHalfSizedType(Ty: DstTy);
2464 SmallVector<Register, 2> DefRegs(OpdMapper.getVRegs(OpIdx: 0));
2465 SmallVector<Register, 2> Src0Regs(OpdMapper.getVRegs(OpIdx: 1));
2466 SmallVector<Register, 2> Src1Regs(OpdMapper.getVRegs(OpIdx: 2));
2467
2468 // All inputs are SGPRs, nothing special to do.
2469 if (DefRegs.empty()) {
2470 assert(Src0Regs.empty() && Src1Regs.empty());
2471 break;
2472 }
2473
2474 assert(DefRegs.size() == 2);
2475 assert(Src0Regs.size() == Src1Regs.size() &&
2476 (Src0Regs.empty() || Src0Regs.size() == 2));
2477
2478 // Depending on where the source registers came from, the generic code may
2479 // have decided to split the inputs already or not. If not, we still need to
2480 // extract the values.
2481
2482 if (Src0Regs.empty())
2483 split64BitValueForMapping(B, Regs&: Src0Regs, HalfTy, Reg: MI.getOperand(i: 1).getReg());
2484 else
2485 setRegsToType(MRI, Regs: Src0Regs, NewTy: HalfTy);
2486
2487 if (Src1Regs.empty())
2488 split64BitValueForMapping(B, Regs&: Src1Regs, HalfTy, Reg: MI.getOperand(i: 2).getReg());
2489 else
2490 setRegsToType(MRI, Regs: Src1Regs, NewTy: HalfTy);
2491
2492 setRegsToType(MRI, Regs: DefRegs, NewTy: HalfTy);
2493
2494 auto Flags = MI.getFlags();
2495 B.buildInstr(Opc, DstOps: {DefRegs[0]}, SrcOps: {Src0Regs[0], Src1Regs[0]}, Flags);
2496 B.buildInstr(Opc, DstOps: {DefRegs[1]}, SrcOps: {Src0Regs[1], Src1Regs[1]}, Flags);
2497
2498 MRI.setRegBank(Reg: DstReg, RegBank: AMDGPU::VGPRRegBank);
2499 MI.eraseFromParent();
2500 return;
2501 }
2502 case AMDGPU::G_ABS: {
2503 Register SrcReg = MI.getOperand(i: 1).getReg();
2504 const RegisterBank *SrcBank = MRI.getRegBankOrNull(Reg: SrcReg);
2505
2506 // There is no VALU abs instruction so we need to replace it with a sub and
2507 // max combination.
2508 if (SrcBank && SrcBank == &AMDGPU::VGPRRegBank) {
2509 MachineFunction *MF = MI.getMF();
2510 ApplyRegBankMapping Apply(B, *this, MRI, &AMDGPU::VGPRRegBank);
2511 LegalizerHelper Helper(*MF, Apply, B);
2512
2513 if (Helper.lowerAbsToMaxNeg(MI) != LegalizerHelper::Legalized)
2514 llvm_unreachable("lowerAbsToMaxNeg should have succeeded");
2515 return;
2516 }
2517 [[fallthrough]];
2518 }
2519 case AMDGPU::G_ADD:
2520 case AMDGPU::G_SUB:
2521 case AMDGPU::G_MUL:
2522 case AMDGPU::G_SHL:
2523 case AMDGPU::G_LSHR:
2524 case AMDGPU::G_ASHR:
2525 case AMDGPU::G_SMIN:
2526 case AMDGPU::G_SMAX:
2527 case AMDGPU::G_UMIN:
2528 case AMDGPU::G_UMAX: {
2529 Register DstReg = MI.getOperand(i: 0).getReg();
2530 LLT DstTy = MRI.getType(Reg: DstReg);
2531
2532 // Special case for s_mul_u64. There is not a vector equivalent of
2533 // s_mul_u64. Hence, we have to break down s_mul_u64 into 32-bit vector
2534 // multiplications.
2535 if (!Subtarget.useVMulU64Inst() && Opc == AMDGPU::G_MUL &&
2536 DstTy.getSizeInBits() == 64) {
2537 applyMappingSMULU64(B, OpdMapper);
2538 return;
2539 }
2540
2541 // 16-bit operations are VALU only, but can be promoted to 32-bit SALU.
2542 // Packed 16-bit operations need to be scalarized and promoted.
2543 if (DstTy != LLT::scalar(SizeInBits: 16) && DstTy != LLT::fixed_vector(NumElements: 2, ScalarSizeInBits: 16))
2544 break;
2545
2546 const RegisterBank *DstBank =
2547 OpdMapper.getInstrMapping().getOperandMapping(i: 0).BreakDown[0].RegBank;
2548 if (DstBank == &AMDGPU::VGPRRegBank)
2549 break;
2550
2551 const LLT S32 = LLT::scalar(SizeInBits: 32);
2552 MachineBasicBlock *MBB = MI.getParent();
2553 MachineFunction *MF = MBB->getParent();
2554 ApplyRegBankMapping ApplySALU(B, *this, MRI, &AMDGPU::SGPRRegBank);
2555
2556 if (DstTy.isVector() && Opc == AMDGPU::G_ABS) {
2557 Register WideSrcLo, WideSrcHi;
2558
2559 std::tie(args&: WideSrcLo, args&: WideSrcHi) =
2560 unpackV2S16ToS32(B, Src: MI.getOperand(i: 1).getReg(), ExtOpcode: TargetOpcode::G_SEXT);
2561 auto Lo = B.buildInstr(Opc: AMDGPU::G_ABS, DstOps: {S32}, SrcOps: {WideSrcLo});
2562 auto Hi = B.buildInstr(Opc: AMDGPU::G_ABS, DstOps: {S32}, SrcOps: {WideSrcHi});
2563 B.buildBuildVectorTrunc(Res: DstReg, Ops: {Lo.getReg(Idx: 0), Hi.getReg(Idx: 0)});
2564 MI.eraseFromParent();
2565 return;
2566 }
2567
2568 if (DstTy.isVector()) {
2569 Register WideSrc0Lo, WideSrc0Hi;
2570 Register WideSrc1Lo, WideSrc1Hi;
2571
2572 unsigned ExtendOp = getExtendOp(Opc: MI.getOpcode());
2573 std::tie(args&: WideSrc0Lo, args&: WideSrc0Hi)
2574 = unpackV2S16ToS32(B, Src: MI.getOperand(i: 1).getReg(), ExtOpcode: ExtendOp);
2575 std::tie(args&: WideSrc1Lo, args&: WideSrc1Hi)
2576 = unpackV2S16ToS32(B, Src: MI.getOperand(i: 2).getReg(), ExtOpcode: ExtendOp);
2577 auto Lo = B.buildInstr(Opc: MI.getOpcode(), DstOps: {S32}, SrcOps: {WideSrc0Lo, WideSrc1Lo});
2578 auto Hi = B.buildInstr(Opc: MI.getOpcode(), DstOps: {S32}, SrcOps: {WideSrc0Hi, WideSrc1Hi});
2579 B.buildBuildVectorTrunc(Res: DstReg, Ops: {Lo.getReg(Idx: 0), Hi.getReg(Idx: 0)});
2580 MI.eraseFromParent();
2581 } else {
2582 LegalizerHelper Helper(*MF, ApplySALU, B);
2583
2584 if (Helper.widenScalar(MI, TypeIdx: 0, WideTy: S32) != LegalizerHelper::Legalized)
2585 llvm_unreachable("widen scalar should have succeeded");
2586
2587 // FIXME: s16 shift amounts should be legal.
2588 if (Opc == AMDGPU::G_SHL || Opc == AMDGPU::G_LSHR ||
2589 Opc == AMDGPU::G_ASHR) {
2590 B.setInsertPt(MBB&: *MBB, II: MI.getIterator());
2591 if (Helper.widenScalar(MI, TypeIdx: 1, WideTy: S32) != LegalizerHelper::Legalized)
2592 llvm_unreachable("widen scalar should have succeeded");
2593 }
2594 }
2595
2596 return;
2597 }
2598 case AMDGPU::G_AMDGPU_S_MUL_I64_I32:
2599 case AMDGPU::G_AMDGPU_S_MUL_U64_U32: {
2600 // This is a special case for s_mul_u64. We use
2601 // G_AMDGPU_S_MUL_I64_I32 opcode to represent an s_mul_u64 operation
2602 // where the 33 higher bits are sign-extended and
2603 // G_AMDGPU_S_MUL_U64_U32 opcode to represent an s_mul_u64 operation
2604 // where the 32 higher bits are zero-extended. In case scalar registers are
2605 // selected, both opcodes are lowered as s_mul_u64. If the vector registers
2606 // are selected, then G_AMDGPU_S_MUL_I64_I32 and
2607 // G_AMDGPU_S_MUL_U64_U32 are lowered with a vector mad instruction.
2608
2609 // Insert basic copies.
2610 applyDefaultMapping(OpdMapper);
2611
2612 Register DstReg = MI.getOperand(i: 0).getReg();
2613 Register SrcReg0 = MI.getOperand(i: 1).getReg();
2614 Register SrcReg1 = MI.getOperand(i: 2).getReg();
2615 const LLT S32 = LLT::scalar(SizeInBits: 32);
2616 const LLT S64 = LLT::scalar(SizeInBits: 64);
2617 assert(MRI.getType(DstReg) == S64 && "This is a special case for s_mul_u64 "
2618 "that handles only 64-bit operands.");
2619 const RegisterBank *DstBank =
2620 OpdMapper.getInstrMapping().getOperandMapping(i: 0).BreakDown[0].RegBank;
2621
2622 // Replace G_AMDGPU_S_MUL_I64_I32 and G_AMDGPU_S_MUL_U64_U32
2623 // with s_mul_u64 operation.
2624 if (DstBank == &AMDGPU::SGPRRegBank) {
2625 MI.setDesc(TII->get(Opcode: AMDGPU::S_MUL_U64));
2626 MRI.setRegClass(Reg: DstReg, RC: &AMDGPU::SGPR_64RegClass);
2627 MRI.setRegClass(Reg: SrcReg0, RC: &AMDGPU::SGPR_64RegClass);
2628 MRI.setRegClass(Reg: SrcReg1, RC: &AMDGPU::SGPR_64RegClass);
2629 return;
2630 }
2631
2632 // Replace G_AMDGPU_S_MUL_I64_I32 and G_AMDGPU_S_MUL_U64_U32
2633 // with a vector mad.
2634 assert(MRI.getRegBankOrNull(DstReg) == &AMDGPU::VGPRRegBank &&
2635 "The destination operand should be in vector registers.");
2636
2637 // Extract the lower subregister from the first operand.
2638 Register Op0L = MRI.createVirtualRegister(RegClass: &AMDGPU::VGPR_32RegClass);
2639 MRI.setRegClass(Reg: Op0L, RC: &AMDGPU::VGPR_32RegClass);
2640 MRI.setType(VReg: Op0L, Ty: S32);
2641 B.buildTrunc(Res: Op0L, Op: SrcReg0);
2642
2643 // Extract the lower subregister from the second operand.
2644 Register Op1L = MRI.createVirtualRegister(RegClass: &AMDGPU::VGPR_32RegClass);
2645 MRI.setRegClass(Reg: Op1L, RC: &AMDGPU::VGPR_32RegClass);
2646 MRI.setType(VReg: Op1L, Ty: S32);
2647 B.buildTrunc(Res: Op1L, Op: SrcReg1);
2648
2649 unsigned NewOpc = Opc == AMDGPU::G_AMDGPU_S_MUL_U64_U32
2650 ? AMDGPU::G_AMDGPU_MAD_U64_U32
2651 : AMDGPU::G_AMDGPU_MAD_I64_I32;
2652
2653 MachineIRBuilder B(MI);
2654 Register Zero64 = B.buildConstant(Res: S64, Val: 0).getReg(Idx: 0);
2655 MRI.setRegClass(Reg: Zero64, RC: &AMDGPU::VReg_64RegClass);
2656 Register CarryOut = MRI.createVirtualRegister(RegClass: &AMDGPU::VReg_64RegClass);
2657 MRI.setRegClass(Reg: CarryOut, RC: &AMDGPU::VReg_64RegClass);
2658 B.buildInstr(Opc: NewOpc, DstOps: {DstReg, CarryOut}, SrcOps: {Op0L, Op1L, Zero64});
2659 MI.eraseFromParent();
2660 return;
2661 }
2662 case AMDGPU::G_SEXT_INREG: {
2663 SmallVector<Register, 2> SrcRegs(OpdMapper.getVRegs(OpIdx: 1));
2664 if (SrcRegs.empty())
2665 break; // Nothing to repair
2666
2667 const LLT S32 = LLT::scalar(SizeInBits: 32);
2668 ApplyRegBankMapping O(B, *this, MRI, &AMDGPU::VGPRRegBank);
2669
2670 // Don't use LegalizerHelper's narrowScalar. It produces unwanted G_SEXTs
2671 // we would need to further expand, and doesn't let us directly set the
2672 // result registers.
2673 SmallVector<Register, 2> DstRegs(OpdMapper.getVRegs(OpIdx: 0));
2674
2675 int Amt = MI.getOperand(i: 2).getImm();
2676 if (Amt <= 32) {
2677 // Downstream users have expectations for the high bit behavior, so freeze
2678 // incoming undefined bits.
2679 if (Amt == 32) {
2680 // The low bits are unchanged.
2681 B.buildFreeze(Dst: DstRegs[0], Src: SrcRegs[0]);
2682 } else {
2683 auto Freeze = B.buildFreeze(Dst: S32, Src: SrcRegs[0]);
2684 // Extend in the low bits and propagate the sign bit to the high half.
2685 B.buildSExtInReg(Res: DstRegs[0], Op: Freeze, ImmOp: Amt);
2686 }
2687
2688 B.buildAShr(Dst: DstRegs[1], Src0: DstRegs[0], Src1: B.buildConstant(Res: S32, Val: 31));
2689 } else {
2690 // The low bits are unchanged, and extend in the high bits.
2691 // No freeze required
2692 B.buildCopy(Res: DstRegs[0], Op: SrcRegs[0]);
2693 B.buildSExtInReg(Res: DstRegs[1], Op: DstRegs[0], ImmOp: Amt - 32);
2694 }
2695
2696 Register DstReg = MI.getOperand(i: 0).getReg();
2697 MRI.setRegBank(Reg: DstReg, RegBank: AMDGPU::VGPRRegBank);
2698 MI.eraseFromParent();
2699 return;
2700 }
2701 case AMDGPU::G_CTPOP:
2702 case AMDGPU::G_BITREVERSE: {
2703 const RegisterBank *DstBank =
2704 OpdMapper.getInstrMapping().getOperandMapping(i: 0).BreakDown[0].RegBank;
2705 if (DstBank == &AMDGPU::SGPRRegBank)
2706 break;
2707
2708 Register SrcReg = MI.getOperand(i: 1).getReg();
2709 const LLT S32 = LLT::scalar(SizeInBits: 32);
2710 LLT Ty = MRI.getType(Reg: SrcReg);
2711 if (Ty == S32)
2712 break;
2713
2714 ApplyRegBankMapping ApplyVALU(B, *this, MRI, &AMDGPU::VGPRRegBank);
2715
2716 MachineFunction &MF = B.getMF();
2717 LegalizerHelper Helper(MF, ApplyVALU, B);
2718
2719 if (Helper.narrowScalar(MI, TypeIdx: 1, NarrowTy: S32) != LegalizerHelper::Legalized)
2720 llvm_unreachable("narrowScalar should have succeeded");
2721 return;
2722 }
2723 case AMDGPU::G_AMDGPU_FFBH_U32:
2724 case AMDGPU::G_AMDGPU_FFBL_B32:
2725 case AMDGPU::G_CTLZ_ZERO_POISON:
2726 case AMDGPU::G_CTTZ_ZERO_POISON: {
2727 const RegisterBank *DstBank =
2728 OpdMapper.getInstrMapping().getOperandMapping(i: 0).BreakDown[0].RegBank;
2729 if (DstBank == &AMDGPU::SGPRRegBank)
2730 break;
2731
2732 Register SrcReg = MI.getOperand(i: 1).getReg();
2733 const LLT S32 = LLT::scalar(SizeInBits: 32);
2734 LLT Ty = MRI.getType(Reg: SrcReg);
2735 if (Ty == S32)
2736 break;
2737
2738 // We can narrow this more efficiently than Helper can by using ffbh/ffbl
2739 // which return -1 when the input is zero:
2740 // (ctlz_zero_poison hi:lo) -> (umin (ffbh hi), (add (ffbh lo), 32))
2741 // (cttz_zero_poison hi:lo) -> (umin (add (ffbl hi), 32), (ffbl lo))
2742 // (ffbh hi:lo) -> (umin (ffbh hi), (uaddsat (ffbh lo), 32))
2743 // (ffbl hi:lo) -> (umin (uaddsat (ffbh hi), 32), (ffbh lo))
2744 ApplyRegBankMapping ApplyVALU(B, *this, MRI, &AMDGPU::VGPRRegBank);
2745 SmallVector<Register, 2> SrcRegs(OpdMapper.getVRegs(OpIdx: 1));
2746 unsigned NewOpc = Opc == AMDGPU::G_CTLZ_ZERO_POISON
2747 ? (unsigned)AMDGPU::G_AMDGPU_FFBH_U32
2748 : Opc == AMDGPU::G_CTTZ_ZERO_POISON
2749 ? (unsigned)AMDGPU::G_AMDGPU_FFBL_B32
2750 : Opc;
2751 unsigned Idx = NewOpc == AMDGPU::G_AMDGPU_FFBH_U32;
2752 auto X = B.buildInstr(Opc: NewOpc, DstOps: {S32}, SrcOps: {SrcRegs[Idx]});
2753 auto Y = B.buildInstr(Opc: NewOpc, DstOps: {S32}, SrcOps: {SrcRegs[Idx ^ 1]});
2754 unsigned AddOpc =
2755 Opc == AMDGPU::G_CTLZ_ZERO_POISON || Opc == AMDGPU::G_CTTZ_ZERO_POISON
2756 ? AMDGPU::G_ADD
2757 : AMDGPU::G_UADDSAT;
2758 Y = B.buildInstr(Opc: AddOpc, DstOps: {S32}, SrcOps: {Y, B.buildConstant(Res: S32, Val: 32)});
2759 Register DstReg = MI.getOperand(i: 0).getReg();
2760 B.buildUMin(Dst: DstReg, Src0: X, Src1: Y);
2761 MI.eraseFromParent();
2762 return;
2763 }
2764 case AMDGPU::G_SEXT:
2765 case AMDGPU::G_ZEXT:
2766 case AMDGPU::G_ANYEXT: {
2767 Register SrcReg = MI.getOperand(i: 1).getReg();
2768 LLT SrcTy = MRI.getType(Reg: SrcReg);
2769 const bool Signed = Opc == AMDGPU::G_SEXT;
2770
2771 assert(OpdMapper.getVRegs(1).empty());
2772
2773 const RegisterBank *SrcBank =
2774 OpdMapper.getInstrMapping().getOperandMapping(i: 1).BreakDown[0].RegBank;
2775
2776 Register DstReg = MI.getOperand(i: 0).getReg();
2777 LLT DstTy = MRI.getType(Reg: DstReg);
2778 if (DstTy.isScalar() &&
2779 SrcBank != &AMDGPU::SGPRRegBank &&
2780 SrcBank != &AMDGPU::VCCRegBank &&
2781 // FIXME: Should handle any type that round to s64 when irregular
2782 // breakdowns supported.
2783 DstTy.getSizeInBits() == 64 &&
2784 SrcTy.getSizeInBits() <= 32) {
2785 SmallVector<Register, 2> DefRegs(OpdMapper.getVRegs(OpIdx: 0));
2786
2787 // Extend to 32-bit, and then extend the low half.
2788 if (Signed) {
2789 // TODO: Should really be buildSExtOrCopy
2790 B.buildSExtOrTrunc(Res: DefRegs[0], Op: SrcReg);
2791 } else if (Opc == AMDGPU::G_ZEXT) {
2792 B.buildZExtOrTrunc(Res: DefRegs[0], Op: SrcReg);
2793 } else {
2794 B.buildAnyExtOrTrunc(Res: DefRegs[0], Op: SrcReg);
2795 }
2796
2797 extendLow32IntoHigh32(B, Hi32Reg: DefRegs[1], Lo32Reg: DefRegs[0], ExtOpc: Opc, RegBank: *SrcBank);
2798 MRI.setRegBank(Reg: DstReg, RegBank: *SrcBank);
2799 MI.eraseFromParent();
2800 return;
2801 }
2802
2803 if (SrcTy != LLT::scalar(SizeInBits: 1))
2804 return;
2805
2806 // It is not legal to have a legalization artifact with a VCC source. Rather
2807 // than introducing a copy, insert the select we would have to select the
2808 // copy to.
2809 if (SrcBank == &AMDGPU::VCCRegBank) {
2810 SmallVector<Register, 2> DefRegs(OpdMapper.getVRegs(OpIdx: 0));
2811
2812 const RegisterBank *DstBank = &AMDGPU::VGPRRegBank;
2813
2814 unsigned DstSize = DstTy.getSizeInBits();
2815 // 64-bit select is SGPR only
2816 const bool UseSel64 = DstSize > 32 &&
2817 SrcBank->getID() == AMDGPU::SGPRRegBankID;
2818
2819 // TODO: Should s16 select be legal?
2820 LLT SelType = UseSel64 ? LLT::scalar(SizeInBits: 64) : LLT::scalar(SizeInBits: 32);
2821 auto True = B.buildConstant(Res: SelType, Val: Signed ? -1 : 1);
2822 auto False = B.buildConstant(Res: SelType, Val: 0);
2823
2824 MRI.setRegBank(Reg: True.getReg(Idx: 0), RegBank: *DstBank);
2825 MRI.setRegBank(Reg: False.getReg(Idx: 0), RegBank: *DstBank);
2826 MRI.setRegBank(Reg: DstReg, RegBank: *DstBank);
2827
2828 if (DstSize > 32) {
2829 B.buildSelect(Res: DefRegs[0], Tst: SrcReg, Op0: True, Op1: False);
2830 extendLow32IntoHigh32(B, Hi32Reg: DefRegs[1], Lo32Reg: DefRegs[0], ExtOpc: Opc, RegBank: *SrcBank, IsBooleanSrc: true);
2831 } else if (DstSize < 32) {
2832 auto Sel = B.buildSelect(Res: SelType, Tst: SrcReg, Op0: True, Op1: False);
2833 MRI.setRegBank(Reg: Sel.getReg(Idx: 0), RegBank: *DstBank);
2834 B.buildTrunc(Res: DstReg, Op: Sel);
2835 } else {
2836 B.buildSelect(Res: DstReg, Tst: SrcReg, Op0: True, Op1: False);
2837 }
2838
2839 MI.eraseFromParent();
2840 return;
2841 }
2842
2843 break;
2844 }
2845 case AMDGPU::G_EXTRACT_VECTOR_ELT: {
2846 SmallVector<Register, 2> DstRegs(OpdMapper.getVRegs(OpIdx: 0));
2847
2848 assert(OpdMapper.getVRegs(1).empty() && OpdMapper.getVRegs(2).empty());
2849
2850 Register DstReg = MI.getOperand(i: 0).getReg();
2851 Register SrcReg = MI.getOperand(i: 1).getReg();
2852
2853 const LLT S32 = LLT::scalar(SizeInBits: 32);
2854 LLT DstTy = MRI.getType(Reg: DstReg);
2855 LLT SrcTy = MRI.getType(Reg: SrcReg);
2856
2857 if (foldExtractEltToCmpSelect(B, MI, OpdMapper))
2858 return;
2859
2860 const ValueMapping &DstMapping
2861 = OpdMapper.getInstrMapping().getOperandMapping(i: 0);
2862 const RegisterBank *DstBank = DstMapping.BreakDown[0].RegBank;
2863 const RegisterBank *SrcBank =
2864 OpdMapper.getInstrMapping().getOperandMapping(i: 1).BreakDown[0].RegBank;
2865 const RegisterBank *IdxBank =
2866 OpdMapper.getInstrMapping().getOperandMapping(i: 2).BreakDown[0].RegBank;
2867
2868 Register BaseIdxReg;
2869 unsigned ConstOffset;
2870 std::tie(args&: BaseIdxReg, args&: ConstOffset) =
2871 AMDGPU::getBaseWithConstantOffset(MRI, Reg: MI.getOperand(i: 2).getReg());
2872
2873 // See if the index is an add of a constant which will be foldable by moving
2874 // the base register of the index later if this is going to be executed in a
2875 // waterfall loop. This is essentially to reassociate the add of a constant
2876 // with the readfirstlane.
2877 bool ShouldMoveIndexIntoLoop = IdxBank != &AMDGPU::SGPRRegBank &&
2878 ConstOffset > 0 &&
2879 ConstOffset < SrcTy.getNumElements();
2880
2881 // Move the base register. We'll re-insert the add later.
2882 if (ShouldMoveIndexIntoLoop)
2883 MI.getOperand(i: 2).setReg(BaseIdxReg);
2884
2885 // If this is a VGPR result only because the index was a VGPR result, the
2886 // actual indexing will be done on the SGPR source vector, which will
2887 // produce a scalar result. We need to copy to the VGPR result inside the
2888 // waterfall loop.
2889 const bool NeedCopyToVGPR = DstBank == &AMDGPU::VGPRRegBank &&
2890 SrcBank == &AMDGPU::SGPRRegBank;
2891 if (DstRegs.empty()) {
2892 applyDefaultMapping(OpdMapper);
2893
2894 executeInWaterfallLoop(B, MI, OpIndices: {2});
2895
2896 if (NeedCopyToVGPR) {
2897 // We don't want a phi for this temporary reg.
2898 Register TmpReg = MRI.createGenericVirtualRegister(Ty: DstTy);
2899 MRI.setRegBank(Reg: TmpReg, RegBank: AMDGPU::SGPRRegBank);
2900 MI.getOperand(i: 0).setReg(TmpReg);
2901 B.setInsertPt(MBB&: *MI.getParent(), II: ++MI.getIterator());
2902
2903 // Use a v_mov_b32 here to make the exec dependency explicit.
2904 buildVCopy(B, DstReg, SrcReg: TmpReg);
2905 }
2906
2907 // Re-insert the constant offset add inside the waterfall loop.
2908 if (ShouldMoveIndexIntoLoop)
2909 reinsertVectorIndexAdd(B, IdxUseInstr&: MI, OpIdx: 2, ConstOffset);
2910
2911 return;
2912 }
2913
2914 assert(DstTy.getSizeInBits() == 64);
2915
2916 LLT Vec32 = LLT::fixed_vector(NumElements: 2 * SrcTy.getNumElements(), ScalarSizeInBits: 32);
2917
2918 auto CastSrc = B.buildBitcast(Dst: Vec32, Src: SrcReg);
2919 auto One = B.buildConstant(Res: S32, Val: 1);
2920
2921 MachineBasicBlock::iterator MII = MI.getIterator();
2922
2923 // Split the vector index into 32-bit pieces. Prepare to move all of the
2924 // new instructions into a waterfall loop if necessary.
2925 //
2926 // Don't put the bitcast or constant in the loop.
2927 MachineInstrSpan Span(MII, &B.getMBB());
2928
2929 // Compute 32-bit element indices, (2 * OrigIdx, 2 * OrigIdx + 1).
2930 auto IdxLo = B.buildShl(Dst: S32, Src0: BaseIdxReg, Src1: One);
2931 auto IdxHi = B.buildAdd(Dst: S32, Src0: IdxLo, Src1: One);
2932
2933 auto Extract0 = B.buildExtractVectorElement(Res: DstRegs[0], Val: CastSrc, Idx: IdxLo);
2934 auto Extract1 = B.buildExtractVectorElement(Res: DstRegs[1], Val: CastSrc, Idx: IdxHi);
2935
2936 MRI.setRegBank(Reg: DstReg, RegBank: *DstBank);
2937 MRI.setRegBank(Reg: CastSrc.getReg(Idx: 0), RegBank: *SrcBank);
2938 MRI.setRegBank(Reg: One.getReg(Idx: 0), RegBank: AMDGPU::SGPRRegBank);
2939 MRI.setRegBank(Reg: IdxLo.getReg(Idx: 0), RegBank: AMDGPU::SGPRRegBank);
2940 MRI.setRegBank(Reg: IdxHi.getReg(Idx: 0), RegBank: AMDGPU::SGPRRegBank);
2941
2942 SmallSet<Register, 4> OpsToWaterfall;
2943 if (!collectWaterfallOperands(SGPROperandRegs&: OpsToWaterfall, MI, MRI, OpIndices: { 2 })) {
2944 MI.eraseFromParent();
2945 return;
2946 }
2947
2948 // Remove the original instruction to avoid potentially confusing the
2949 // waterfall loop logic.
2950 B.setInstr(*Span.begin());
2951 MI.eraseFromParent();
2952 executeInWaterfallLoop(B, Range: make_range(x: Span.begin(), y: Span.end()),
2953 SGPROperandRegs&: OpsToWaterfall);
2954
2955 if (NeedCopyToVGPR) {
2956 MachineBasicBlock *LoopBB = Extract1->getParent();
2957 Register TmpReg0 = MRI.createGenericVirtualRegister(Ty: S32);
2958 Register TmpReg1 = MRI.createGenericVirtualRegister(Ty: S32);
2959 MRI.setRegBank(Reg: TmpReg0, RegBank: AMDGPU::SGPRRegBank);
2960 MRI.setRegBank(Reg: TmpReg1, RegBank: AMDGPU::SGPRRegBank);
2961
2962 Extract0->getOperand(i: 0).setReg(TmpReg0);
2963 Extract1->getOperand(i: 0).setReg(TmpReg1);
2964
2965 B.setInsertPt(MBB&: *LoopBB, II: ++Extract1->getIterator());
2966
2967 buildVCopy(B, DstReg: DstRegs[0], SrcReg: TmpReg0);
2968 buildVCopy(B, DstReg: DstRegs[1], SrcReg: TmpReg1);
2969 }
2970
2971 if (ShouldMoveIndexIntoLoop)
2972 reinsertVectorIndexAdd(B, IdxUseInstr&: *IdxLo, OpIdx: 1, ConstOffset);
2973
2974 return;
2975 }
2976 case AMDGPU::G_INSERT_VECTOR_ELT: {
2977 SmallVector<Register, 2> InsRegs(OpdMapper.getVRegs(OpIdx: 2));
2978
2979 Register DstReg = MI.getOperand(i: 0).getReg();
2980 LLT VecTy = MRI.getType(Reg: DstReg);
2981
2982 assert(OpdMapper.getVRegs(0).empty());
2983 assert(OpdMapper.getVRegs(3).empty());
2984
2985 if (substituteSimpleCopyRegs(OpdMapper, OpIdx: 1))
2986 MRI.setType(VReg: MI.getOperand(i: 1).getReg(), Ty: VecTy);
2987
2988 if (foldInsertEltToCmpSelect(B, MI, OpdMapper))
2989 return;
2990
2991 const RegisterBank *IdxBank =
2992 OpdMapper.getInstrMapping().getOperandMapping(i: 3).BreakDown[0].RegBank;
2993
2994 Register SrcReg = MI.getOperand(i: 1).getReg();
2995 Register InsReg = MI.getOperand(i: 2).getReg();
2996 LLT InsTy = MRI.getType(Reg: InsReg);
2997 (void)InsTy;
2998
2999 Register BaseIdxReg;
3000 unsigned ConstOffset;
3001 std::tie(args&: BaseIdxReg, args&: ConstOffset) =
3002 AMDGPU::getBaseWithConstantOffset(MRI, Reg: MI.getOperand(i: 3).getReg());
3003
3004 // See if the index is an add of a constant which will be foldable by moving
3005 // the base register of the index later if this is going to be executed in a
3006 // waterfall loop. This is essentially to reassociate the add of a constant
3007 // with the readfirstlane.
3008 bool ShouldMoveIndexIntoLoop = IdxBank != &AMDGPU::SGPRRegBank &&
3009 ConstOffset > 0 &&
3010 ConstOffset < VecTy.getNumElements();
3011
3012 // Move the base register. We'll re-insert the add later.
3013 if (ShouldMoveIndexIntoLoop)
3014 MI.getOperand(i: 3).setReg(BaseIdxReg);
3015
3016
3017 if (InsRegs.empty()) {
3018 executeInWaterfallLoop(B, MI, OpIndices: {3});
3019
3020 // Re-insert the constant offset add inside the waterfall loop.
3021 if (ShouldMoveIndexIntoLoop) {
3022 reinsertVectorIndexAdd(B, IdxUseInstr&: MI, OpIdx: 3, ConstOffset);
3023 }
3024
3025 return;
3026 }
3027
3028 assert(InsTy.getSizeInBits() == 64);
3029
3030 const LLT S32 = LLT::scalar(SizeInBits: 32);
3031 LLT Vec32 = LLT::fixed_vector(NumElements: 2 * VecTy.getNumElements(), ScalarSizeInBits: 32);
3032
3033 auto CastSrc = B.buildBitcast(Dst: Vec32, Src: SrcReg);
3034 auto One = B.buildConstant(Res: S32, Val: 1);
3035
3036 // Split the vector index into 32-bit pieces. Prepare to move all of the
3037 // new instructions into a waterfall loop if necessary.
3038 //
3039 // Don't put the bitcast or constant in the loop.
3040 MachineInstrSpan Span(MachineBasicBlock::iterator(&MI), &B.getMBB());
3041
3042 // Compute 32-bit element indices, (2 * OrigIdx, 2 * OrigIdx + 1).
3043 auto IdxLo = B.buildShl(Dst: S32, Src0: BaseIdxReg, Src1: One);
3044 auto IdxHi = B.buildAdd(Dst: S32, Src0: IdxLo, Src1: One);
3045
3046 auto InsLo = B.buildInsertVectorElement(Res: Vec32, Val: CastSrc, Elt: InsRegs[0], Idx: IdxLo);
3047 auto InsHi = B.buildInsertVectorElement(Res: Vec32, Val: InsLo, Elt: InsRegs[1], Idx: IdxHi);
3048
3049 const RegisterBank *DstBank =
3050 OpdMapper.getInstrMapping().getOperandMapping(i: 0).BreakDown[0].RegBank;
3051 const RegisterBank *SrcBank =
3052 OpdMapper.getInstrMapping().getOperandMapping(i: 1).BreakDown[0].RegBank;
3053 const RegisterBank *InsSrcBank =
3054 OpdMapper.getInstrMapping().getOperandMapping(i: 2).BreakDown[0].RegBank;
3055
3056 MRI.setRegBank(Reg: InsReg, RegBank: *InsSrcBank);
3057 MRI.setRegBank(Reg: CastSrc.getReg(Idx: 0), RegBank: *SrcBank);
3058 MRI.setRegBank(Reg: InsLo.getReg(Idx: 0), RegBank: *DstBank);
3059 MRI.setRegBank(Reg: InsHi.getReg(Idx: 0), RegBank: *DstBank);
3060 MRI.setRegBank(Reg: One.getReg(Idx: 0), RegBank: AMDGPU::SGPRRegBank);
3061 MRI.setRegBank(Reg: IdxLo.getReg(Idx: 0), RegBank: AMDGPU::SGPRRegBank);
3062 MRI.setRegBank(Reg: IdxHi.getReg(Idx: 0), RegBank: AMDGPU::SGPRRegBank);
3063
3064
3065 SmallSet<Register, 4> OpsToWaterfall;
3066 if (!collectWaterfallOperands(SGPROperandRegs&: OpsToWaterfall, MI, MRI, OpIndices: { 3 })) {
3067 B.setInsertPt(MBB&: B.getMBB(), II: MI);
3068 B.buildBitcast(Dst: DstReg, Src: InsHi);
3069 MI.eraseFromParent();
3070 return;
3071 }
3072
3073 B.setInstr(*Span.begin());
3074 MI.eraseFromParent();
3075
3076 // Figure out the point after the waterfall loop before mangling the control
3077 // flow.
3078 executeInWaterfallLoop(B, Range: make_range(x: Span.begin(), y: Span.end()),
3079 SGPROperandRegs&: OpsToWaterfall);
3080
3081 // The insertion point is now right after the original instruction.
3082 //
3083 // Keep the bitcast to the original vector type out of the loop. Doing this
3084 // saved an extra phi we don't need inside the loop.
3085 B.buildBitcast(Dst: DstReg, Src: InsHi);
3086
3087 // Re-insert the constant offset add inside the waterfall loop.
3088 if (ShouldMoveIndexIntoLoop)
3089 reinsertVectorIndexAdd(B, IdxUseInstr&: *IdxLo, OpIdx: 1, ConstOffset);
3090
3091 return;
3092 }
3093 case AMDGPU::G_AMDGPU_BUFFER_LOAD:
3094 case AMDGPU::G_AMDGPU_BUFFER_LOAD_USHORT:
3095 case AMDGPU::G_AMDGPU_BUFFER_LOAD_SSHORT:
3096 case AMDGPU::G_AMDGPU_BUFFER_LOAD_UBYTE:
3097 case AMDGPU::G_AMDGPU_BUFFER_LOAD_SBYTE:
3098 case AMDGPU::G_AMDGPU_BUFFER_LOAD_TFE:
3099 case AMDGPU::G_AMDGPU_BUFFER_LOAD_USHORT_TFE:
3100 case AMDGPU::G_AMDGPU_BUFFER_LOAD_SSHORT_TFE:
3101 case AMDGPU::G_AMDGPU_BUFFER_LOAD_UBYTE_TFE:
3102 case AMDGPU::G_AMDGPU_BUFFER_LOAD_SBYTE_TFE:
3103 case AMDGPU::G_AMDGPU_BUFFER_LOAD_FORMAT:
3104 case AMDGPU::G_AMDGPU_BUFFER_LOAD_FORMAT_TFE:
3105 case AMDGPU::G_AMDGPU_BUFFER_LOAD_FORMAT_D16:
3106 case AMDGPU::G_AMDGPU_BUFFER_LOAD_FORMAT_D16_TFE:
3107 case AMDGPU::G_AMDGPU_TBUFFER_LOAD_FORMAT:
3108 case AMDGPU::G_AMDGPU_TBUFFER_LOAD_FORMAT_D16:
3109 case AMDGPU::G_AMDGPU_BUFFER_STORE:
3110 case AMDGPU::G_AMDGPU_BUFFER_STORE_BYTE:
3111 case AMDGPU::G_AMDGPU_BUFFER_STORE_SHORT:
3112 case AMDGPU::G_AMDGPU_BUFFER_STORE_FORMAT:
3113 case AMDGPU::G_AMDGPU_BUFFER_STORE_FORMAT_D16:
3114 case AMDGPU::G_AMDGPU_TBUFFER_STORE_FORMAT:
3115 case AMDGPU::G_AMDGPU_TBUFFER_STORE_FORMAT_D16: {
3116 applyDefaultMapping(OpdMapper);
3117 executeInWaterfallLoop(B, MI, OpIndices: {1, 4});
3118 return;
3119 }
3120 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SWAP:
3121 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_ADD:
3122 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SUB:
3123 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SMIN:
3124 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_UMIN:
3125 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SMAX:
3126 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_UMAX:
3127 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_AND:
3128 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_OR:
3129 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_XOR:
3130 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_INC:
3131 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_DEC:
3132 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SUB_CLAMP_U32:
3133 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_COND_SUB_U32:
3134 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_FADD:
3135 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_FMIN:
3136 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_FMAX: {
3137 applyDefaultMapping(OpdMapper);
3138 executeInWaterfallLoop(B, MI, OpIndices: {2, 5});
3139 return;
3140 }
3141 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_CMPSWAP: {
3142 applyDefaultMapping(OpdMapper);
3143 executeInWaterfallLoop(B, MI, OpIndices: {3, 6});
3144 return;
3145 }
3146 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD:
3147 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_UBYTE:
3148 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_SBYTE:
3149 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_USHORT:
3150 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_SSHORT: {
3151 applyMappingSBufferLoad(B, OpdMapper);
3152 return;
3153 }
3154 case AMDGPU::G_AMDGPU_S_BUFFER_PREFETCH:
3155 constrainOpWithReadfirstlane(B, MI, OpIdx: 0);
3156 constrainOpWithReadfirstlane(B, MI, OpIdx: 2);
3157 return;
3158 case AMDGPU::G_INTRINSIC:
3159 case AMDGPU::G_INTRINSIC_CONVERGENT: {
3160 switch (cast<GIntrinsic>(Val&: MI).getIntrinsicID()) {
3161 case Intrinsic::amdgcn_readlane: {
3162 substituteSimpleCopyRegs(OpdMapper, OpIdx: 2);
3163
3164 assert(OpdMapper.getVRegs(0).empty());
3165 assert(OpdMapper.getVRegs(3).empty());
3166
3167 // Make sure the index is an SGPR. It doesn't make sense to run this in a
3168 // waterfall loop, so assume it's a uniform value.
3169 constrainOpWithReadfirstlane(B, MI, OpIdx: 3); // Index
3170 return;
3171 }
3172 case Intrinsic::amdgcn_writelane: {
3173 assert(OpdMapper.getVRegs(0).empty());
3174 assert(OpdMapper.getVRegs(2).empty());
3175 assert(OpdMapper.getVRegs(3).empty());
3176
3177 substituteSimpleCopyRegs(OpdMapper, OpIdx: 4); // VGPR input val
3178 constrainOpWithReadfirstlane(B, MI, OpIdx: 2); // Source value
3179 constrainOpWithReadfirstlane(B, MI, OpIdx: 3); // Index
3180 return;
3181 }
3182 case Intrinsic::amdgcn_interp_p1:
3183 case Intrinsic::amdgcn_interp_p2:
3184 case Intrinsic::amdgcn_interp_mov:
3185 case Intrinsic::amdgcn_interp_p1_f16:
3186 case Intrinsic::amdgcn_interp_p2_f16:
3187 case Intrinsic::amdgcn_lds_param_load: {
3188 applyDefaultMapping(OpdMapper);
3189
3190 // Readlane for m0 value, which is always the last operand.
3191 // FIXME: Should this be a waterfall loop instead?
3192 constrainOpWithReadfirstlane(B, MI, OpIdx: MI.getNumOperands() - 1); // Index
3193 return;
3194 }
3195 case Intrinsic::amdgcn_interp_inreg_p10:
3196 case Intrinsic::amdgcn_interp_inreg_p2:
3197 case Intrinsic::amdgcn_interp_inreg_p10_f16:
3198 case Intrinsic::amdgcn_interp_inreg_p2_f16:
3199 case Intrinsic::amdgcn_interp_p10_rtz_f16:
3200 case Intrinsic::amdgcn_interp_p2_rtz_f16:
3201 case Intrinsic::amdgcn_permlane16_swap:
3202 case Intrinsic::amdgcn_permlane32_swap:
3203 applyDefaultMapping(OpdMapper);
3204 return;
3205 case Intrinsic::amdgcn_permlane16:
3206 case Intrinsic::amdgcn_permlanex16: {
3207 // Doing a waterfall loop over these wouldn't make any sense.
3208 substituteSimpleCopyRegs(OpdMapper, OpIdx: 2);
3209 substituteSimpleCopyRegs(OpdMapper, OpIdx: 3);
3210 constrainOpWithReadfirstlane(B, MI, OpIdx: 4);
3211 constrainOpWithReadfirstlane(B, MI, OpIdx: 5);
3212 return;
3213 }
3214 case Intrinsic::amdgcn_permlane_bcast:
3215 case Intrinsic::amdgcn_permlane_up:
3216 case Intrinsic::amdgcn_permlane_down:
3217 case Intrinsic::amdgcn_permlane_xor:
3218 // Doing a waterfall loop over these wouldn't make any sense.
3219 constrainOpWithReadfirstlane(B, MI, OpIdx: 3);
3220 constrainOpWithReadfirstlane(B, MI, OpIdx: 4);
3221 return;
3222 case Intrinsic::amdgcn_permlane_idx_gen: {
3223 constrainOpWithReadfirstlane(B, MI, OpIdx: 3);
3224 return;
3225 }
3226 case Intrinsic::amdgcn_sbfe:
3227 applyMappingBFE(B, OpdMapper, Signed: true);
3228 return;
3229 case Intrinsic::amdgcn_ubfe:
3230 applyMappingBFE(B, OpdMapper, Signed: false);
3231 return;
3232 case Intrinsic::amdgcn_inverse_ballot:
3233 case Intrinsic::amdgcn_s_bitreplicate:
3234 case Intrinsic::amdgcn_s_quadmask:
3235 case Intrinsic::amdgcn_s_wqm:
3236 applyDefaultMapping(OpdMapper);
3237 constrainOpWithReadfirstlane(B, MI, OpIdx: 2); // Mask
3238 return;
3239 case Intrinsic::amdgcn_ballot:
3240 // Use default handling and insert copy to vcc source.
3241 break;
3242 }
3243 break;
3244 }
3245 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_LOAD:
3246 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_LOAD_D16:
3247 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_LOAD_NORET:
3248 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_STORE:
3249 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_STORE_D16: {
3250 const AMDGPU::RsrcIntrinsic *RSrcIntrin =
3251 AMDGPU::lookupRsrcIntrinsic(Intr: AMDGPU::getIntrinsicID(I: MI));
3252 assert(RSrcIntrin && RSrcIntrin->IsImage);
3253 // Non-images can have complications from operands that allow both SGPR
3254 // and VGPR. For now it's too complicated to figure out the final opcode
3255 // to derive the register bank from the MCInstrDesc.
3256 applyMappingImage(B, MI, OpdMapper, RsrcIdx: RSrcIntrin->RsrcArg);
3257 return;
3258 }
3259 case AMDGPU::G_AMDGPU_BVH_INTERSECT_RAY:
3260 case AMDGPU::G_AMDGPU_BVH8_INTERSECT_RAY:
3261 case AMDGPU::G_AMDGPU_BVH_DUAL_INTERSECT_RAY: {
3262 bool IsDualOrBVH8 =
3263 MI.getOpcode() == AMDGPU::G_AMDGPU_BVH_DUAL_INTERSECT_RAY ||
3264 MI.getOpcode() == AMDGPU::G_AMDGPU_BVH8_INTERSECT_RAY;
3265 unsigned NumMods = IsDualOrBVH8 ? 0 : 1; // Has A16 modifier
3266 unsigned LastRegOpIdx = MI.getNumExplicitOperands() - 1 - NumMods;
3267 applyDefaultMapping(OpdMapper);
3268 executeInWaterfallLoop(B, MI, OpIndices: {LastRegOpIdx});
3269 return;
3270 }
3271 case AMDGPU::G_INTRINSIC_W_SIDE_EFFECTS:
3272 case AMDGPU::G_INTRINSIC_CONVERGENT_W_SIDE_EFFECTS: {
3273 auto IntrID = cast<GIntrinsic>(Val&: MI).getIntrinsicID();
3274 switch (IntrID) {
3275 case Intrinsic::amdgcn_ds_ordered_add:
3276 case Intrinsic::amdgcn_ds_ordered_swap: {
3277 // This is only allowed to execute with 1 lane, so readfirstlane is safe.
3278 assert(OpdMapper.getVRegs(0).empty());
3279 substituteSimpleCopyRegs(OpdMapper, OpIdx: 3);
3280 constrainOpWithReadfirstlane(B, MI, OpIdx: 2); // M0
3281 return;
3282 }
3283 case Intrinsic::amdgcn_ds_gws_init:
3284 case Intrinsic::amdgcn_ds_gws_barrier:
3285 case Intrinsic::amdgcn_ds_gws_sema_br: {
3286 // Only the first lane is executes, so readfirstlane is safe.
3287 substituteSimpleCopyRegs(OpdMapper, OpIdx: 1);
3288 constrainOpWithReadfirstlane(B, MI, OpIdx: 2); // M0
3289 return;
3290 }
3291 case Intrinsic::amdgcn_ds_gws_sema_v:
3292 case Intrinsic::amdgcn_ds_gws_sema_p:
3293 case Intrinsic::amdgcn_ds_gws_sema_release_all: {
3294 // Only the first lane is executes, so readfirstlane is safe.
3295 constrainOpWithReadfirstlane(B, MI, OpIdx: 1); // M0
3296 return;
3297 }
3298 case Intrinsic::amdgcn_ds_append:
3299 case Intrinsic::amdgcn_ds_consume: {
3300 constrainOpWithReadfirstlane(B, MI, OpIdx: 2); // M0
3301 return;
3302 }
3303 case Intrinsic::amdgcn_s_alloc_vgpr:
3304 constrainOpWithReadfirstlane(B, MI, OpIdx: 2);
3305 return;
3306 case Intrinsic::amdgcn_s_sendmsg:
3307 case Intrinsic::amdgcn_s_sendmsghalt: {
3308 // FIXME: Should this use a waterfall loop?
3309 constrainOpWithReadfirstlane(B, MI, OpIdx: 2); // M0
3310 return;
3311 }
3312 case Intrinsic::amdgcn_s_setreg: {
3313 constrainOpWithReadfirstlane(B, MI, OpIdx: 2);
3314 return;
3315 }
3316 case Intrinsic::amdgcn_s_ttracedata:
3317 constrainOpWithReadfirstlane(B, MI, OpIdx: 1); // M0
3318 return;
3319 case Intrinsic::amdgcn_raw_buffer_load_lds:
3320 case Intrinsic::amdgcn_raw_buffer_load_async_lds:
3321 case Intrinsic::amdgcn_raw_ptr_buffer_load_lds:
3322 case Intrinsic::amdgcn_raw_ptr_buffer_load_async_lds: {
3323 applyDefaultMapping(OpdMapper);
3324 constrainOpWithReadfirstlane(B, MI, OpIdx: 1); // rsrc
3325 constrainOpWithReadfirstlane(B, MI, OpIdx: 2); // M0
3326 constrainOpWithReadfirstlane(B, MI, OpIdx: 5); // soffset
3327 return;
3328 }
3329 case Intrinsic::amdgcn_struct_buffer_load_lds:
3330 case Intrinsic::amdgcn_struct_buffer_load_async_lds:
3331 case Intrinsic::amdgcn_struct_ptr_buffer_load_lds:
3332 case Intrinsic::amdgcn_struct_ptr_buffer_load_async_lds: {
3333 applyDefaultMapping(OpdMapper);
3334 constrainOpWithReadfirstlane(B, MI, OpIdx: 1); // rsrc
3335 constrainOpWithReadfirstlane(B, MI, OpIdx: 2); // M0
3336 constrainOpWithReadfirstlane(B, MI, OpIdx: 6); // soffset
3337 return;
3338 }
3339 case Intrinsic::amdgcn_cluster_load_async_to_lds_b8:
3340 case Intrinsic::amdgcn_cluster_load_async_to_lds_b32:
3341 case Intrinsic::amdgcn_cluster_load_async_to_lds_b64:
3342 case Intrinsic::amdgcn_cluster_load_async_to_lds_b128: {
3343 applyDefaultMapping(OpdMapper);
3344 constrainOpWithReadfirstlane(B, MI, OpIdx: 5);
3345 return;
3346 }
3347 case Intrinsic::amdgcn_load_to_lds:
3348 case Intrinsic::amdgcn_load_async_to_lds:
3349 case Intrinsic::amdgcn_global_load_lds:
3350 case Intrinsic::amdgcn_global_load_async_lds: {
3351 applyDefaultMapping(OpdMapper);
3352 constrainOpWithReadfirstlane(B, MI, OpIdx: 2);
3353 return;
3354 }
3355 case Intrinsic::amdgcn_lds_direct_load: {
3356 applyDefaultMapping(OpdMapper);
3357 // Readlane for m0 value, which is always the last operand.
3358 constrainOpWithReadfirstlane(B, MI, OpIdx: MI.getNumOperands() - 1); // Index
3359 return;
3360 }
3361 case Intrinsic::amdgcn_exp_row:
3362 applyDefaultMapping(OpdMapper);
3363 constrainOpWithReadfirstlane(B, MI, OpIdx: 8); // M0
3364 return;
3365 case Intrinsic::amdgcn_cluster_load_b32:
3366 case Intrinsic::amdgcn_cluster_load_b64:
3367 case Intrinsic::amdgcn_cluster_load_b128: {
3368 applyDefaultMapping(OpdMapper);
3369 constrainOpWithReadfirstlane(B, MI, OpIdx: 4); // M0
3370 return;
3371 }
3372 case Intrinsic::amdgcn_s_sleep_var:
3373 assert(OpdMapper.getVRegs(1).empty());
3374 constrainOpWithReadfirstlane(B, MI, OpIdx: 1);
3375 return;
3376 case Intrinsic::amdgcn_s_barrier_join:
3377 case Intrinsic::amdgcn_s_wakeup_barrier:
3378 constrainOpWithReadfirstlane(B, MI, OpIdx: 1);
3379 return;
3380 case Intrinsic::amdgcn_s_barrier_init:
3381 case Intrinsic::amdgcn_s_barrier_signal_var:
3382 constrainOpWithReadfirstlane(B, MI, OpIdx: 1);
3383 constrainOpWithReadfirstlane(B, MI, OpIdx: 2);
3384 return;
3385 case Intrinsic::amdgcn_s_get_barrier_state:
3386 case Intrinsic::amdgcn_s_get_named_barrier_state: {
3387 constrainOpWithReadfirstlane(B, MI, OpIdx: 2);
3388 return;
3389 }
3390 case Intrinsic::amdgcn_s_prefetch_data:
3391 case Intrinsic::amdgcn_s_prefetch_inst: {
3392 Register PtrReg = MI.getOperand(i: 1).getReg();
3393 unsigned AS = MRI.getType(Reg: PtrReg).getAddressSpace();
3394 if (AMDGPU::isFlatGlobalAddrSpace(AS)) {
3395 constrainOpWithReadfirstlane(B, MI, OpIdx: 1);
3396 constrainOpWithReadfirstlane(B, MI, OpIdx: 2);
3397 } else
3398 MI.eraseFromParent();
3399 return;
3400 }
3401 case Intrinsic::amdgcn_tensor_load_to_lds:
3402 case Intrinsic::amdgcn_tensor_store_from_lds: {
3403 constrainOpWithReadfirstlane(B, MI, OpIdx: 1);
3404 constrainOpWithReadfirstlane(B, MI, OpIdx: 2);
3405 constrainOpWithReadfirstlane(B, MI, OpIdx: 3);
3406 constrainOpWithReadfirstlane(B, MI, OpIdx: 4);
3407 constrainOpWithReadfirstlane(B, MI, OpIdx: 5);
3408 return;
3409 }
3410 default: {
3411 if (const AMDGPU::RsrcIntrinsic *RSrcIntrin =
3412 AMDGPU::lookupRsrcIntrinsic(Intr: IntrID)) {
3413 // Non-images can have complications from operands that allow both SGPR
3414 // and VGPR. For now it's too complicated to figure out the final opcode
3415 // to derive the register bank from the MCInstrDesc.
3416 if (RSrcIntrin->IsImage) {
3417 applyMappingImage(B, MI, OpdMapper, RsrcIdx: RSrcIntrin->RsrcArg);
3418 return;
3419 }
3420 }
3421
3422 break;
3423 }
3424 }
3425 break;
3426 }
3427 case AMDGPU::G_SI_CALL: {
3428 // Use a set to avoid extra readfirstlanes in the case where multiple
3429 // operands are the same register.
3430 SmallSet<Register, 4> SGPROperandRegs;
3431
3432 if (!collectWaterfallOperands(SGPROperandRegs, MI, MRI, OpIndices: {1}))
3433 break;
3434
3435 // Move all copies to physical SGPRs that are used by the call instruction
3436 // into the loop block. Start searching for these copies until the
3437 // ADJCALLSTACKUP.
3438 unsigned FrameSetupOpcode = AMDGPU::ADJCALLSTACKUP;
3439 unsigned FrameDestroyOpcode = AMDGPU::ADJCALLSTACKDOWN;
3440
3441 // Move all non-copies before the copies, so that a complete range can be
3442 // moved into the waterfall loop.
3443 SmallVector<MachineInstr *, 4> NonCopyInstrs;
3444 // Count of NonCopyInstrs found until the current LastCopy.
3445 unsigned NonCopyInstrsLen = 0;
3446 MachineBasicBlock::iterator Start(&MI);
3447 MachineBasicBlock::iterator LastCopy = Start;
3448 MachineBasicBlock *MBB = MI.getParent();
3449 const SIMachineFunctionInfo *Info =
3450 MBB->getParent()->getInfo<SIMachineFunctionInfo>();
3451 while (Start->getOpcode() != FrameSetupOpcode) {
3452 --Start;
3453 bool IsCopy = false;
3454 if (Start->getOpcode() == AMDGPU::COPY) {
3455 auto &Dst = Start->getOperand(i: 0);
3456 if (Dst.isReg()) {
3457 Register Reg = Dst.getReg();
3458 if (Reg.isPhysical() && MI.readsRegister(Reg, TRI)) {
3459 IsCopy = true;
3460 } else {
3461 // Also move the copy from the scratch rsrc descriptor into the loop
3462 // to allow it to be optimized away.
3463 auto &Src = Start->getOperand(i: 1);
3464 if (Src.isReg()) {
3465 Reg = Src.getReg();
3466 IsCopy = Info->getScratchRSrcReg() == Reg;
3467 }
3468 }
3469 }
3470 }
3471
3472 if (IsCopy) {
3473 LastCopy = Start;
3474 NonCopyInstrsLen = NonCopyInstrs.size();
3475 } else {
3476 NonCopyInstrs.push_back(Elt: &*Start);
3477 }
3478 }
3479 NonCopyInstrs.resize(N: NonCopyInstrsLen);
3480
3481 for (auto *NonCopy : reverse(C&: NonCopyInstrs)) {
3482 MBB->splice(Where: LastCopy, Other: MBB, From: NonCopy->getIterator());
3483 }
3484 Start = LastCopy;
3485
3486 // Do the same for copies after the loop
3487 NonCopyInstrs.clear();
3488 NonCopyInstrsLen = 0;
3489 MachineBasicBlock::iterator End(&MI);
3490 LastCopy = End;
3491 while (End->getOpcode() != FrameDestroyOpcode) {
3492 ++End;
3493 bool IsCopy = false;
3494 if (End->getOpcode() == AMDGPU::COPY) {
3495 auto &Src = End->getOperand(i: 1);
3496 if (Src.isReg()) {
3497 Register Reg = Src.getReg();
3498 IsCopy = Reg.isPhysical() && MI.modifiesRegister(Reg, TRI);
3499 }
3500 }
3501
3502 if (IsCopy) {
3503 LastCopy = End;
3504 NonCopyInstrsLen = NonCopyInstrs.size();
3505 } else {
3506 NonCopyInstrs.push_back(Elt: &*End);
3507 }
3508 }
3509 NonCopyInstrs.resize(N: NonCopyInstrsLen);
3510
3511 End = LastCopy;
3512 ++LastCopy;
3513 for (auto *NonCopy : reverse(C&: NonCopyInstrs)) {
3514 MBB->splice(Where: LastCopy, Other: MBB, From: NonCopy->getIterator());
3515 }
3516
3517 ++End;
3518 B.setInsertPt(MBB&: B.getMBB(), II: Start);
3519 executeInWaterfallLoop(B, Range: make_range(x: Start, y: End), SGPROperandRegs);
3520 break;
3521 }
3522 case AMDGPU::G_AMDGPU_FLAT_LOAD_MONITOR:
3523 case AMDGPU::G_AMDGPU_GLOBAL_LOAD_MONITOR:
3524 case AMDGPU::G_LOAD:
3525 case AMDGPU::G_ZEXTLOAD:
3526 case AMDGPU::G_SEXTLOAD: {
3527 if (applyMappingLoad(B, OpdMapper, MI))
3528 return;
3529 break;
3530 }
3531 case AMDGPU::G_DYN_STACKALLOC:
3532 applyMappingDynStackAlloc(B, OpdMapper, MI);
3533 return;
3534 case AMDGPU::G_STACKRESTORE: {
3535 applyDefaultMapping(OpdMapper);
3536 constrainOpWithReadfirstlane(B, MI, OpIdx: 0);
3537 return;
3538 }
3539 case AMDGPU::G_SBFX:
3540 applyMappingBFE(B, OpdMapper, /*Signed*/ true);
3541 return;
3542 case AMDGPU::G_UBFX:
3543 applyMappingBFE(B, OpdMapper, /*Signed*/ false);
3544 return;
3545 case AMDGPU::G_AMDGPU_MAD_U64_U32:
3546 case AMDGPU::G_AMDGPU_MAD_I64_I32:
3547 applyMappingMAD_64_32(B, OpdMapper);
3548 return;
3549 case AMDGPU::G_PREFETCH: {
3550 if (!Subtarget.hasSafeSmemPrefetch() && !Subtarget.hasVmemPrefInsts()) {
3551 MI.eraseFromParent();
3552 return;
3553 }
3554 Register PtrReg = MI.getOperand(i: 0).getReg();
3555 unsigned PtrBank = getRegBankID(Reg: PtrReg, MRI, Default: AMDGPU::SGPRRegBankID);
3556 if (PtrBank == AMDGPU::VGPRRegBankID &&
3557 (!Subtarget.hasVmemPrefInsts() || !MI.getOperand(i: 3).getImm())) {
3558 // Cannot do I$ prefetch with divergent pointer.
3559 MI.eraseFromParent();
3560 return;
3561 }
3562 unsigned AS = MRI.getType(Reg: PtrReg).getAddressSpace();
3563 if ((!AMDGPU::isFlatGlobalAddrSpace(AS) &&
3564 AS != AMDGPUAS::CONSTANT_ADDRESS_32BIT) ||
3565 (!Subtarget.hasSafeSmemPrefetch() &&
3566 (AS == AMDGPUAS::CONSTANT_ADDRESS_32BIT ||
3567 !MI.getOperand(i: 3).getImm() /* I$ prefetch */))) {
3568 MI.eraseFromParent();
3569 return;
3570 }
3571 applyDefaultMapping(OpdMapper);
3572 return;
3573 }
3574 default:
3575 break;
3576 }
3577
3578 return applyDefaultMapping(OpdMapper);
3579}
3580
3581// vgpr, sgpr -> vgpr
3582// vgpr, agpr -> vgpr
3583// agpr, agpr -> agpr
3584// agpr, sgpr -> vgpr
3585static unsigned regBankUnion(unsigned RB0, unsigned RB1) {
3586 if (RB0 == AMDGPU::InvalidRegBankID)
3587 return RB1;
3588 if (RB1 == AMDGPU::InvalidRegBankID)
3589 return RB0;
3590
3591 if (RB0 == AMDGPU::SGPRRegBankID && RB1 == AMDGPU::SGPRRegBankID)
3592 return AMDGPU::SGPRRegBankID;
3593
3594 if (RB0 == AMDGPU::AGPRRegBankID && RB1 == AMDGPU::AGPRRegBankID)
3595 return AMDGPU::AGPRRegBankID;
3596
3597 return AMDGPU::VGPRRegBankID;
3598}
3599
3600static unsigned regBankBoolUnion(unsigned RB0, unsigned RB1) {
3601 if (RB0 == AMDGPU::InvalidRegBankID)
3602 return RB1;
3603 if (RB1 == AMDGPU::InvalidRegBankID)
3604 return RB0;
3605
3606 // vcc, vcc -> vcc
3607 // vcc, sgpr -> vcc
3608 // vcc, vgpr -> vcc
3609 if (RB0 == AMDGPU::VCCRegBankID || RB1 == AMDGPU::VCCRegBankID)
3610 return AMDGPU::VCCRegBankID;
3611
3612 // vcc, vgpr -> vgpr
3613 return regBankUnion(RB0, RB1);
3614}
3615
3616unsigned AMDGPURegisterBankInfo::getMappingType(const MachineRegisterInfo &MRI,
3617 const MachineInstr &MI) const {
3618 unsigned RegBank = AMDGPU::InvalidRegBankID;
3619
3620 for (const MachineOperand &MO : MI.operands()) {
3621 if (!MO.isReg())
3622 continue;
3623 Register Reg = MO.getReg();
3624 if (const RegisterBank *Bank = getRegBank(Reg, MRI, TRI: *TRI)) {
3625 RegBank = regBankUnion(RB0: RegBank, RB1: Bank->getID());
3626 if (RegBank == AMDGPU::VGPRRegBankID)
3627 break;
3628 }
3629 }
3630
3631 return RegBank;
3632}
3633
3634bool AMDGPURegisterBankInfo::isSALUMapping(const MachineInstr &MI) const {
3635 const MachineFunction &MF = *MI.getMF();
3636 const MachineRegisterInfo &MRI = MF.getRegInfo();
3637 for (const MachineOperand &MO : MI.operands()) {
3638 if (!MO.isReg())
3639 continue;
3640 Register Reg = MO.getReg();
3641 if (const RegisterBank *Bank = getRegBank(Reg, MRI, TRI: *TRI)) {
3642 if (Bank->getID() != AMDGPU::SGPRRegBankID)
3643 return false;
3644 }
3645 }
3646 return true;
3647}
3648
3649const RegisterBankInfo::InstructionMapping &
3650AMDGPURegisterBankInfo::getDefaultMappingSOP(const MachineInstr &MI) const {
3651 const MachineFunction &MF = *MI.getMF();
3652 const MachineRegisterInfo &MRI = MF.getRegInfo();
3653 SmallVector<const ValueMapping*, 8> OpdsMapping(MI.getNumOperands());
3654
3655 for (unsigned i = 0, e = MI.getNumOperands(); i != e; ++i) {
3656 const MachineOperand &SrcOp = MI.getOperand(i);
3657 if (!SrcOp.isReg())
3658 continue;
3659
3660 unsigned Size = getSizeInBits(Reg: SrcOp.getReg(), MRI, TRI: *TRI);
3661 OpdsMapping[i] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size);
3662 }
3663 return getInstructionMapping(ID: 1, Cost: 1, OperandsMapping: getOperandsMapping(OpdsMapping),
3664 NumOperands: MI.getNumOperands());
3665}
3666
3667const RegisterBankInfo::InstructionMapping &
3668AMDGPURegisterBankInfo::getDefaultMappingVOP(const MachineInstr &MI) const {
3669 const MachineFunction &MF = *MI.getMF();
3670 const MachineRegisterInfo &MRI = MF.getRegInfo();
3671 SmallVector<const ValueMapping*, 8> OpdsMapping(MI.getNumOperands());
3672
3673 // Even though we technically could use SGPRs, this would require knowledge of
3674 // the constant bus restriction. Force all sources to VGPR (except for VCC).
3675 //
3676 // TODO: Unary ops are trivially OK, so accept SGPRs?
3677 for (unsigned i = 0, e = MI.getNumOperands(); i != e; ++i) {
3678 const MachineOperand &Src = MI.getOperand(i);
3679 if (!Src.isReg())
3680 continue;
3681
3682 unsigned Size = getSizeInBits(Reg: Src.getReg(), MRI, TRI: *TRI);
3683 unsigned BankID = Size == 1 ? AMDGPU::VCCRegBankID : AMDGPU::VGPRRegBankID;
3684 OpdsMapping[i] = AMDGPU::getValueMapping(BankID, Size);
3685 }
3686
3687 return getInstructionMapping(ID: 1, Cost: 1, OperandsMapping: getOperandsMapping(OpdsMapping),
3688 NumOperands: MI.getNumOperands());
3689}
3690
3691const RegisterBankInfo::InstructionMapping &
3692AMDGPURegisterBankInfo::getDefaultMappingAllVGPR(const MachineInstr &MI) const {
3693 const MachineFunction &MF = *MI.getMF();
3694 const MachineRegisterInfo &MRI = MF.getRegInfo();
3695 SmallVector<const ValueMapping*, 8> OpdsMapping(MI.getNumOperands());
3696
3697 for (unsigned I = 0, E = MI.getNumOperands(); I != E; ++I) {
3698 const MachineOperand &Op = MI.getOperand(i: I);
3699 if (!Op.isReg())
3700 continue;
3701
3702 unsigned Size = getSizeInBits(Reg: Op.getReg(), MRI, TRI: *TRI);
3703 OpdsMapping[I] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
3704 }
3705
3706 return getInstructionMapping(ID: 1, Cost: 1, OperandsMapping: getOperandsMapping(OpdsMapping),
3707 NumOperands: MI.getNumOperands());
3708}
3709
3710const RegisterBankInfo::InstructionMapping &
3711AMDGPURegisterBankInfo::getImageMapping(const MachineRegisterInfo &MRI,
3712 const MachineInstr &MI,
3713 int RsrcIdx) const {
3714 // The reported argument index is relative to the IR intrinsic call arguments,
3715 // so we need to shift by the number of defs and the intrinsic ID.
3716 RsrcIdx += MI.getNumExplicitDefs() + 1;
3717
3718 const int NumOps = MI.getNumOperands();
3719 SmallVector<const ValueMapping *, 8> OpdsMapping(NumOps);
3720
3721 // TODO: Should packed/unpacked D16 difference be reported here as part of
3722 // the value mapping?
3723 for (int I = 0; I != NumOps; ++I) {
3724 if (!MI.getOperand(i: I).isReg())
3725 continue;
3726
3727 Register OpReg = MI.getOperand(i: I).getReg();
3728 // We replace some dead address operands with $noreg
3729 if (!OpReg)
3730 continue;
3731
3732 unsigned Size = getSizeInBits(Reg: OpReg, MRI, TRI: *TRI);
3733
3734 // FIXME: Probably need a new intrinsic register bank searchable table to
3735 // handle arbitrary intrinsics easily.
3736 //
3737 // If this has a sampler, it immediately follows rsrc.
3738 const bool MustBeSGPR = I == RsrcIdx || I == RsrcIdx + 1;
3739
3740 if (MustBeSGPR) {
3741 // If this must be an SGPR, so we must report whatever it is as legal.
3742 unsigned NewBank = getRegBankID(Reg: OpReg, MRI, Default: AMDGPU::SGPRRegBankID);
3743 OpdsMapping[I] = AMDGPU::getValueMapping(BankID: NewBank, Size);
3744 } else {
3745 // Some operands must be VGPR, and these are easy to copy to.
3746 OpdsMapping[I] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
3747 }
3748 }
3749
3750 return getInstructionMapping(ID: 1, Cost: 1, OperandsMapping: getOperandsMapping(OpdsMapping), NumOperands: NumOps);
3751}
3752
3753/// Return the mapping for a pointer argument.
3754const RegisterBankInfo::ValueMapping *
3755AMDGPURegisterBankInfo::getValueMappingForPtr(const MachineRegisterInfo &MRI,
3756 Register PtrReg) const {
3757 LLT PtrTy = MRI.getType(Reg: PtrReg);
3758 unsigned Size = PtrTy.getSizeInBits();
3759 if (Subtarget.useFlatForGlobal() ||
3760 !AMDGPU::isFlatGlobalAddrSpace(AS: PtrTy.getAddressSpace()))
3761 return AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
3762
3763 // If we're using MUBUF instructions for global memory, an SGPR base register
3764 // is possible. Otherwise this needs to be a VGPR.
3765 const RegisterBank *PtrBank = getRegBank(Reg: PtrReg, MRI, TRI: *TRI);
3766 return AMDGPU::getValueMapping(BankID: PtrBank->getID(), Size);
3767}
3768
3769const RegisterBankInfo::InstructionMapping &
3770AMDGPURegisterBankInfo::getInstrMappingForLoad(const MachineInstr &MI) const {
3771
3772 const MachineFunction &MF = *MI.getMF();
3773 const MachineRegisterInfo &MRI = MF.getRegInfo();
3774 SmallVector<const ValueMapping*, 2> OpdsMapping(2);
3775 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
3776 Register PtrReg = MI.getOperand(i: 1).getReg();
3777 LLT PtrTy = MRI.getType(Reg: PtrReg);
3778 unsigned AS = PtrTy.getAddressSpace();
3779 unsigned PtrSize = PtrTy.getSizeInBits();
3780
3781 const ValueMapping *ValMapping;
3782 const ValueMapping *PtrMapping;
3783
3784 const RegisterBank *PtrBank = getRegBank(Reg: PtrReg, MRI, TRI: *TRI);
3785
3786 if (PtrBank == &AMDGPU::SGPRRegBank && AMDGPU::isFlatGlobalAddrSpace(AS)) {
3787 if (isScalarLoadLegal(MI)) {
3788 // We have a uniform instruction so we want to use an SMRD load
3789 ValMapping = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size);
3790 PtrMapping = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: PtrSize);
3791 } else {
3792 ValMapping = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
3793
3794 // If we're using MUBUF instructions for global memory, an SGPR base
3795 // register is possible. Otherwise this needs to be a VGPR.
3796 unsigned PtrBankID = Subtarget.useFlatForGlobal() ?
3797 AMDGPU::VGPRRegBankID : AMDGPU::SGPRRegBankID;
3798
3799 PtrMapping = AMDGPU::getValueMapping(BankID: PtrBankID, Size: PtrSize);
3800 }
3801 } else {
3802 ValMapping = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
3803 PtrMapping = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: PtrSize);
3804 }
3805
3806 OpdsMapping[0] = ValMapping;
3807 OpdsMapping[1] = PtrMapping;
3808 const RegisterBankInfo::InstructionMapping &Mapping = getInstructionMapping(
3809 ID: 1, Cost: 1, OperandsMapping: getOperandsMapping(OpdsMapping), NumOperands: MI.getNumOperands());
3810 return Mapping;
3811
3812 // FIXME: Do we want to add a mapping for FLAT load, or should we just
3813 // handle that during instruction selection?
3814}
3815
3816unsigned
3817AMDGPURegisterBankInfo::getRegBankID(Register Reg,
3818 const MachineRegisterInfo &MRI,
3819 unsigned Default) const {
3820 const RegisterBank *Bank = getRegBank(Reg, MRI, TRI: *TRI);
3821 return Bank ? Bank->getID() : Default;
3822}
3823
3824const RegisterBankInfo::ValueMapping *
3825AMDGPURegisterBankInfo::getSGPROpMapping(Register Reg,
3826 const MachineRegisterInfo &MRI,
3827 const TargetRegisterInfo &TRI) const {
3828 // Lie and claim anything is legal, even though this needs to be an SGPR
3829 // applyMapping will have to deal with it as a waterfall loop.
3830 unsigned Bank = getRegBankID(Reg, MRI, Default: AMDGPU::SGPRRegBankID);
3831 unsigned Size = getSizeInBits(Reg, MRI, TRI);
3832 return AMDGPU::getValueMapping(BankID: Bank, Size);
3833}
3834
3835const RegisterBankInfo::ValueMapping *
3836AMDGPURegisterBankInfo::getVGPROpMapping(Register Reg,
3837 const MachineRegisterInfo &MRI,
3838 const TargetRegisterInfo &TRI) const {
3839 unsigned Size = getSizeInBits(Reg, MRI, TRI);
3840 return AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
3841}
3842
3843const RegisterBankInfo::ValueMapping *
3844AMDGPURegisterBankInfo::getAGPROpMapping(Register Reg,
3845 const MachineRegisterInfo &MRI,
3846 const TargetRegisterInfo &TRI) const {
3847 unsigned Size = getSizeInBits(Reg, MRI, TRI);
3848 return AMDGPU::getValueMapping(BankID: AMDGPU::AGPRRegBankID, Size);
3849}
3850
3851///
3852/// This function must return a legal mapping, because
3853/// AMDGPURegisterBankInfo::getInstrAlternativeMappings() is not called
3854/// in RegBankSelect::Mode::Fast. Any mapping that would cause a
3855/// VGPR to SGPR generated is illegal.
3856///
3857// Operands that must be SGPRs must accept potentially divergent VGPRs as
3858// legal. These will be dealt with in applyMappingImpl.
3859//
3860const RegisterBankInfo::InstructionMapping &
3861AMDGPURegisterBankInfo::getInstrMapping(const MachineInstr &MI) const {
3862 const MachineFunction &MF = *MI.getMF();
3863 const MachineRegisterInfo &MRI = MF.getRegInfo();
3864
3865 if (MI.isCopy() || MI.getOpcode() == AMDGPU::G_FREEZE) {
3866 Register DstReg = MI.getOperand(i: 0).getReg();
3867 Register SrcReg = MI.getOperand(i: 1).getReg();
3868
3869 // The default logic bothers to analyze impossible alternative mappings. We
3870 // want the most straightforward mapping, so just directly handle this.
3871 const RegisterBank *DstBank = getRegBank(Reg: DstReg, MRI, TRI: *TRI);
3872 const RegisterBank *SrcBank = getRegBank(Reg: SrcReg, MRI, TRI: *TRI);
3873
3874 // For COPY between a physical reg and an s1, there is no type associated so
3875 // we need to take the virtual register's type as a hint on how to interpret
3876 // s1 values.
3877 unsigned Size;
3878 if (!SrcReg.isVirtual() && !DstBank &&
3879 MRI.getType(Reg: DstReg) == LLT::scalar(SizeInBits: 1)) {
3880 DstBank = &AMDGPU::VCCRegBank;
3881 Size = 1;
3882 } else if (!DstReg.isVirtual() && MRI.getType(Reg: SrcReg) == LLT::scalar(SizeInBits: 1)) {
3883 DstBank = &AMDGPU::VCCRegBank;
3884 Size = 1;
3885 } else {
3886 Size = getSizeInBits(Reg: DstReg, MRI, TRI: *TRI);
3887 }
3888
3889 if (!DstBank)
3890 DstBank = SrcBank;
3891 else if (!SrcBank)
3892 SrcBank = DstBank;
3893
3894 if (MI.getOpcode() != AMDGPU::G_FREEZE &&
3895 cannotCopy(Dst: *DstBank, Src: *SrcBank, Size: TypeSize::getFixed(ExactSize: Size)))
3896 return getInvalidInstructionMapping();
3897
3898 const ValueMapping &ValMap = getValueMapping(StartIdx: 0, Length: Size, RegBank: *DstBank);
3899 unsigned OpdsMappingSize = MI.isCopy() ? 1 : 2;
3900 SmallVector<const ValueMapping *, 1> OpdsMapping(OpdsMappingSize);
3901 OpdsMapping[0] = &ValMap;
3902 if (MI.getOpcode() == AMDGPU::G_FREEZE)
3903 OpdsMapping[1] = &ValMap;
3904
3905 return getInstructionMapping(
3906 ID: 1, /*Cost*/ 1,
3907 /*OperandsMapping*/ getOperandsMapping(OpdsMapping), NumOperands: OpdsMappingSize);
3908 }
3909
3910 if (MI.isRegSequence()) {
3911 // If any input is a VGPR, the result must be a VGPR. The default handling
3912 // assumes any copy between banks is legal.
3913 unsigned BankID = AMDGPU::SGPRRegBankID;
3914
3915 for (unsigned I = 1, E = MI.getNumOperands(); I != E; I += 2) {
3916 auto OpBank = getRegBankID(Reg: MI.getOperand(i: I).getReg(), MRI);
3917 // It doesn't make sense to use vcc or scc banks here, so just ignore
3918 // them.
3919 if (OpBank != AMDGPU::SGPRRegBankID) {
3920 BankID = AMDGPU::VGPRRegBankID;
3921 break;
3922 }
3923 }
3924 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
3925
3926 const ValueMapping &ValMap = getValueMapping(StartIdx: 0, Length: Size, RegBank: getRegBank(ID: BankID));
3927 return getInstructionMapping(
3928 ID: 1, /*Cost*/ 1,
3929 /*OperandsMapping*/ getOperandsMapping(OpdsMapping: {&ValMap}), NumOperands: 1);
3930 }
3931
3932 // The default handling is broken and doesn't handle illegal SGPR->VGPR copies
3933 // properly.
3934 //
3935 // TODO: There are additional exec masking dependencies to analyze.
3936 if (auto *PHI = dyn_cast<GPhi>(Val: &MI)) {
3937 unsigned ResultBank = AMDGPU::InvalidRegBankID;
3938 Register DstReg = PHI->getReg(Idx: 0);
3939
3940 // Sometimes the result may have already been assigned a bank.
3941 if (const RegisterBank *DstBank = getRegBank(Reg: DstReg, MRI, TRI: *TRI))
3942 ResultBank = DstBank->getID();
3943
3944 for (unsigned I = 0; I < PHI->getNumIncomingValues(); ++I) {
3945 Register Reg = PHI->getIncomingValue(I);
3946 const RegisterBank *Bank = getRegBank(Reg, MRI, TRI: *TRI);
3947
3948 // FIXME: Assuming VGPR for any undetermined inputs.
3949 if (!Bank || Bank->getID() == AMDGPU::VGPRRegBankID) {
3950 ResultBank = AMDGPU::VGPRRegBankID;
3951 break;
3952 }
3953
3954 // FIXME: Need to promote SGPR case to s32
3955 unsigned OpBank = Bank->getID();
3956 ResultBank = regBankBoolUnion(RB0: ResultBank, RB1: OpBank);
3957 }
3958
3959 assert(ResultBank != AMDGPU::InvalidRegBankID);
3960
3961 unsigned Size = MRI.getType(Reg: DstReg).getSizeInBits();
3962
3963 const ValueMapping &ValMap =
3964 getValueMapping(StartIdx: 0, Length: Size, RegBank: getRegBank(ID: ResultBank));
3965 return getInstructionMapping(
3966 ID: 1, /*Cost*/ 1,
3967 /*OperandsMapping*/ getOperandsMapping(OpdsMapping: {&ValMap}), NumOperands: 1);
3968 }
3969
3970 const RegisterBankInfo::InstructionMapping &Mapping = getInstrMappingImpl(MI);
3971 if (Mapping.isValid())
3972 return Mapping;
3973
3974 SmallVector<const ValueMapping*, 8> OpdsMapping(MI.getNumOperands());
3975
3976 switch (MI.getOpcode()) {
3977 default:
3978 return getInvalidInstructionMapping();
3979
3980 case AMDGPU::G_AND:
3981 case AMDGPU::G_OR:
3982 case AMDGPU::G_XOR:
3983 case AMDGPU::G_MUL: {
3984 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
3985 if (Size == 1) {
3986 const RegisterBank *DstBank
3987 = getRegBank(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
3988
3989 unsigned TargetBankID = AMDGPU::InvalidRegBankID;
3990 unsigned BankLHS = AMDGPU::InvalidRegBankID;
3991 unsigned BankRHS = AMDGPU::InvalidRegBankID;
3992 if (DstBank) {
3993 TargetBankID = DstBank->getID();
3994 if (DstBank == &AMDGPU::VCCRegBank) {
3995 TargetBankID = AMDGPU::VCCRegBankID;
3996 BankLHS = AMDGPU::VCCRegBankID;
3997 BankRHS = AMDGPU::VCCRegBankID;
3998 } else {
3999 BankLHS = getRegBankID(Reg: MI.getOperand(i: 1).getReg(), MRI,
4000 Default: AMDGPU::SGPRRegBankID);
4001 BankRHS = getRegBankID(Reg: MI.getOperand(i: 2).getReg(), MRI,
4002 Default: AMDGPU::SGPRRegBankID);
4003 }
4004 } else {
4005 BankLHS = getRegBankID(Reg: MI.getOperand(i: 1).getReg(), MRI,
4006 Default: AMDGPU::VCCRegBankID);
4007 BankRHS = getRegBankID(Reg: MI.getOperand(i: 2).getReg(), MRI,
4008 Default: AMDGPU::VCCRegBankID);
4009
4010 // Both inputs should be true booleans to produce a boolean result.
4011 if (BankLHS == AMDGPU::VGPRRegBankID || BankRHS == AMDGPU::VGPRRegBankID) {
4012 TargetBankID = AMDGPU::VGPRRegBankID;
4013 } else if (BankLHS == AMDGPU::VCCRegBankID || BankRHS == AMDGPU::VCCRegBankID) {
4014 TargetBankID = AMDGPU::VCCRegBankID;
4015 BankLHS = AMDGPU::VCCRegBankID;
4016 BankRHS = AMDGPU::VCCRegBankID;
4017 } else if (BankLHS == AMDGPU::SGPRRegBankID && BankRHS == AMDGPU::SGPRRegBankID) {
4018 TargetBankID = AMDGPU::SGPRRegBankID;
4019 }
4020 }
4021
4022 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: TargetBankID, Size);
4023 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: BankLHS, Size);
4024 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: BankRHS, Size);
4025 break;
4026 }
4027
4028 if (Size == 64) {
4029
4030 if (isSALUMapping(MI)) {
4031 OpdsMapping[0] = getValueMappingSGPR64Only(BankID: AMDGPU::SGPRRegBankID, Size);
4032 OpdsMapping[1] = OpdsMapping[2] = OpdsMapping[0];
4033 } else {
4034 if (MI.getOpcode() == AMDGPU::G_MUL && Subtarget.useVMulU64Inst())
4035 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
4036 else
4037 OpdsMapping[0] =
4038 getValueMappingSGPR64Only(BankID: AMDGPU::VGPRRegBankID, Size);
4039 unsigned Bank1 = getRegBankID(Reg: MI.getOperand(i: 1).getReg(), MRI /*, DefaultBankID*/);
4040 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: Bank1, Size);
4041
4042 unsigned Bank2 = getRegBankID(Reg: MI.getOperand(i: 2).getReg(), MRI /*, DefaultBankID*/);
4043 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: Bank2, Size);
4044 }
4045
4046 break;
4047 }
4048
4049 [[fallthrough]];
4050 }
4051 case AMDGPU::G_PTR_ADD:
4052 case AMDGPU::G_PTRMASK:
4053 case AMDGPU::G_ADD:
4054 case AMDGPU::G_SUB:
4055 case AMDGPU::G_SHL:
4056 case AMDGPU::G_LSHR:
4057 case AMDGPU::G_ASHR:
4058 case AMDGPU::G_UADDO:
4059 case AMDGPU::G_USUBO:
4060 case AMDGPU::G_UADDE:
4061 case AMDGPU::G_SADDE:
4062 case AMDGPU::G_USUBE:
4063 case AMDGPU::G_SSUBE:
4064 case AMDGPU::G_ABS:
4065 case AMDGPU::G_SHUFFLE_VECTOR:
4066 case AMDGPU::G_SBFX:
4067 case AMDGPU::G_UBFX:
4068 case AMDGPU::G_AMDGPU_S_MUL_I64_I32:
4069 case AMDGPU::G_AMDGPU_S_MUL_U64_U32:
4070 if (isSALUMapping(MI)) {
4071 LLT Ty = MRI.getType(Reg: MI.getOperand(i: 0).getReg());
4072 unsigned Size = Ty.getSizeInBits();
4073 // Packed add and sub are VALU only.
4074 if (Subtarget.hasAnyPackedU64Ops() && Ty.isVector() && Size == 128)
4075 return getDefaultMappingVOP(MI);
4076 return getDefaultMappingSOP(MI);
4077 }
4078 return getDefaultMappingVOP(MI);
4079 case AMDGPU::G_SMIN:
4080 case AMDGPU::G_SMAX:
4081 case AMDGPU::G_UMIN:
4082 case AMDGPU::G_UMAX:
4083 if (isSALUMapping(MI)) {
4084 // There are no scalar 64-bit min and max, use vector instruction instead.
4085 if (MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits() == 64 &&
4086 Subtarget.useMinMaxI64Insts())
4087 return getDefaultMappingVOP(MI);
4088 return getDefaultMappingSOP(MI);
4089 }
4090 return getDefaultMappingVOP(MI);
4091 case AMDGPU::G_FADD:
4092 case AMDGPU::G_FSUB:
4093 case AMDGPU::G_FMUL:
4094 case AMDGPU::G_FMA:
4095 case AMDGPU::G_FFLOOR:
4096 case AMDGPU::G_FCEIL:
4097 case AMDGPU::G_INTRINSIC_ROUNDEVEN:
4098 case AMDGPU::G_FMINNUM:
4099 case AMDGPU::G_FMAXNUM:
4100 case AMDGPU::G_FMINIMUMNUM:
4101 case AMDGPU::G_FMAXIMUMNUM:
4102 case AMDGPU::G_INTRINSIC_TRUNC:
4103 case AMDGPU::G_STRICT_FADD:
4104 case AMDGPU::G_STRICT_FSUB:
4105 case AMDGPU::G_STRICT_FMUL:
4106 case AMDGPU::G_STRICT_FMA: {
4107 LLT Ty = MRI.getType(Reg: MI.getOperand(i: 0).getReg());
4108 unsigned Size = Ty.getSizeInBits();
4109 if (Subtarget.hasSALUFloatInsts() && Ty.isScalar() &&
4110 (Size == 32 || Size == 16) && isSALUMapping(MI))
4111 return getDefaultMappingSOP(MI);
4112 return getDefaultMappingVOP(MI);
4113 }
4114 case AMDGPU::G_FMINIMUM:
4115 case AMDGPU::G_FMAXIMUM: {
4116 LLT Ty = MRI.getType(Reg: MI.getOperand(i: 0).getReg());
4117 unsigned Size = Ty.getSizeInBits();
4118 if (Subtarget.hasSALUMinimumMaximumInsts() && Ty.isScalar() &&
4119 (Size == 32 || Size == 16) && isSALUMapping(MI))
4120 return getDefaultMappingSOP(MI);
4121 return getDefaultMappingVOP(MI);
4122 }
4123 case AMDGPU::G_FPTOSI:
4124 case AMDGPU::G_FPTOUI:
4125 case AMDGPU::G_FPTOSI_SAT:
4126 case AMDGPU::G_FPTOUI_SAT:
4127 case AMDGPU::G_SITOFP:
4128 case AMDGPU::G_UITOFP: {
4129 unsigned SizeDst = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4130 unsigned SizeSrc = MRI.getType(Reg: MI.getOperand(i: 1).getReg()).getSizeInBits();
4131 if (Subtarget.hasSALUFloatInsts() && SizeDst == 32 && SizeSrc == 32 &&
4132 isSALUMapping(MI))
4133 return getDefaultMappingSOP(MI);
4134 return getDefaultMappingVOP(MI);
4135 }
4136 case AMDGPU::G_FPTRUNC:
4137 case AMDGPU::G_FPEXT: {
4138 unsigned SizeDst = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4139 unsigned SizeSrc = MRI.getType(Reg: MI.getOperand(i: 1).getReg()).getSizeInBits();
4140 if (Subtarget.hasSALUFloatInsts() && SizeDst != 64 && SizeSrc != 64 &&
4141 isSALUMapping(MI))
4142 return getDefaultMappingSOP(MI);
4143 return getDefaultMappingVOP(MI);
4144 }
4145 case AMDGPU::G_FSQRT:
4146 case AMDGPU::G_FEXP2:
4147 case AMDGPU::G_FLOG2: {
4148 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4149 if (Subtarget.hasPseudoScalarTrans() && (Size == 16 || Size == 32) &&
4150 isSALUMapping(MI))
4151 return getDefaultMappingSOP(MI);
4152 return getDefaultMappingVOP(MI);
4153 }
4154 case AMDGPU::G_SADDSAT: // FIXME: Could lower sat ops for SALU
4155 case AMDGPU::G_SSUBSAT:
4156 case AMDGPU::G_UADDSAT:
4157 case AMDGPU::G_USUBSAT:
4158 case AMDGPU::G_FMAD:
4159 case AMDGPU::G_FLDEXP:
4160 case AMDGPU::G_FMINNUM_IEEE:
4161 case AMDGPU::G_FMAXNUM_IEEE:
4162 case AMDGPU::G_FCANONICALIZE:
4163 case AMDGPU::G_STRICT_FLDEXP:
4164 case AMDGPU::G_BSWAP: // TODO: Somehow expand for scalar?
4165 case AMDGPU::G_FSHR: // TODO: Expand for scalar
4166 case AMDGPU::G_AMDGPU_FMIN_LEGACY:
4167 case AMDGPU::G_AMDGPU_FMAX_LEGACY:
4168 case AMDGPU::G_AMDGPU_RCP_IFLAG:
4169 case AMDGPU::G_AMDGPU_CVT_F32_UBYTE0:
4170 case AMDGPU::G_AMDGPU_CVT_F32_UBYTE1:
4171 case AMDGPU::G_AMDGPU_CVT_F32_UBYTE2:
4172 case AMDGPU::G_AMDGPU_CVT_F32_UBYTE3:
4173 case AMDGPU::G_AMDGPU_CVT_PK_I16_I32:
4174 case AMDGPU::G_AMDGPU_SMED3:
4175 case AMDGPU::G_AMDGPU_FMED3:
4176 return getDefaultMappingVOP(MI);
4177 case AMDGPU::G_UMULH:
4178 case AMDGPU::G_SMULH: {
4179 if (Subtarget.hasScalarMulHiInsts() && isSALUMapping(MI))
4180 return getDefaultMappingSOP(MI);
4181 return getDefaultMappingVOP(MI);
4182 }
4183 case AMDGPU::G_AMDGPU_MAD_U64_U32:
4184 case AMDGPU::G_AMDGPU_MAD_I64_I32: {
4185 // Three possible mappings:
4186 //
4187 // - Default SOP
4188 // - Default VOP
4189 // - Scalar multiply: src0 and src1 are SGPRs, the rest is VOP.
4190 //
4191 // This allows instruction selection to keep the multiplication part of the
4192 // instruction on the SALU.
4193 bool AllSalu = true;
4194 bool MulSalu = true;
4195 for (unsigned i = 0; i < 5; ++i) {
4196 Register Reg = MI.getOperand(i).getReg();
4197 if (const RegisterBank *Bank = getRegBank(Reg, MRI, TRI: *TRI)) {
4198 if (Bank->getID() != AMDGPU::SGPRRegBankID) {
4199 AllSalu = false;
4200 if (i == 2 || i == 3) {
4201 MulSalu = false;
4202 break;
4203 }
4204 }
4205 }
4206 }
4207
4208 if (AllSalu)
4209 return getDefaultMappingSOP(MI);
4210
4211 // If the multiply-add is full-rate in VALU, use that even if the
4212 // multiplication part is scalar. Accumulating separately on the VALU would
4213 // take two instructions.
4214 if (!MulSalu || Subtarget.hasFullRate64Ops())
4215 return getDefaultMappingVOP(MI);
4216
4217 // Keep the multiplication on the SALU, then accumulate on the VALU.
4218 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 64);
4219 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: 1);
4220 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 32);
4221 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 32);
4222 OpdsMapping[4] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 64);
4223 break;
4224 }
4225 case AMDGPU::G_IMPLICIT_DEF: {
4226 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4227 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size);
4228 break;
4229 }
4230 case AMDGPU::G_FCONSTANT:
4231 case AMDGPU::G_CONSTANT:
4232 case AMDGPU::G_GLOBAL_VALUE:
4233 case AMDGPU::G_FRAME_INDEX:
4234 case AMDGPU::G_BLOCK_ADDR:
4235 case AMDGPU::G_READSTEADYCOUNTER:
4236 case AMDGPU::G_READCYCLECOUNTER: {
4237 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4238 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size);
4239 break;
4240 }
4241 case AMDGPU::G_DYN_STACKALLOC: {
4242 // Result is always uniform, and a wave reduction is needed for the source.
4243 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 32);
4244 unsigned SrcBankID = getRegBankID(Reg: MI.getOperand(i: 1).getReg(), MRI);
4245 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: SrcBankID, Size: 32);
4246 break;
4247 }
4248 case AMDGPU::G_AMDGPU_WAVE_ADDRESS: {
4249 // This case is weird because we expect a physical register in the source,
4250 // but need to set a bank anyway.
4251 //
4252 // TODO: We could select the result to SGPR or VGPR
4253 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 32);
4254 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 32);
4255 break;
4256 }
4257 case AMDGPU::G_INSERT: {
4258 unsigned BankID = getMappingType(MRI, MI);
4259 unsigned DstSize = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
4260 unsigned SrcSize = getSizeInBits(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
4261 unsigned EltSize = getSizeInBits(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
4262 OpdsMapping[0] = AMDGPU::getValueMapping(BankID, Size: DstSize);
4263 OpdsMapping[1] = AMDGPU::getValueMapping(BankID, Size: SrcSize);
4264 OpdsMapping[2] = AMDGPU::getValueMapping(BankID, Size: EltSize);
4265 OpdsMapping[3] = nullptr;
4266 break;
4267 }
4268 case AMDGPU::G_EXTRACT: {
4269 unsigned BankID = getRegBankID(Reg: MI.getOperand(i: 1).getReg(), MRI);
4270 unsigned DstSize = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
4271 unsigned SrcSize = getSizeInBits(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
4272 OpdsMapping[0] = AMDGPU::getValueMapping(BankID, Size: DstSize);
4273 OpdsMapping[1] = AMDGPU::getValueMapping(BankID, Size: SrcSize);
4274 OpdsMapping[2] = nullptr;
4275 break;
4276 }
4277 case AMDGPU::G_BUILD_VECTOR:
4278 case AMDGPU::G_BUILD_VECTOR_TRUNC: {
4279 LLT DstTy = MRI.getType(Reg: MI.getOperand(i: 0).getReg());
4280 if (DstTy == LLT::fixed_vector(NumElements: 2, ScalarSizeInBits: 16)) {
4281 unsigned DstSize = DstTy.getSizeInBits();
4282 unsigned SrcSize = MRI.getType(Reg: MI.getOperand(i: 1).getReg()).getSizeInBits();
4283 unsigned Src0BankID = getRegBankID(Reg: MI.getOperand(i: 1).getReg(), MRI);
4284 unsigned Src1BankID = getRegBankID(Reg: MI.getOperand(i: 2).getReg(), MRI);
4285 unsigned DstBankID = regBankUnion(RB0: Src0BankID, RB1: Src1BankID);
4286
4287 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: DstBankID, Size: DstSize);
4288 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: Src0BankID, Size: SrcSize);
4289 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: Src1BankID, Size: SrcSize);
4290 break;
4291 }
4292
4293 [[fallthrough]];
4294 }
4295 case AMDGPU::G_MERGE_VALUES:
4296 case AMDGPU::G_CONCAT_VECTORS: {
4297 unsigned Bank = getMappingType(MRI, MI);
4298 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4299 unsigned SrcSize = MRI.getType(Reg: MI.getOperand(i: 1).getReg()).getSizeInBits();
4300
4301 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: Bank, Size: DstSize);
4302 // Op1 and Dst should use the same register bank.
4303 for (unsigned i = 1, e = MI.getNumOperands(); i != e; ++i)
4304 OpdsMapping[i] = AMDGPU::getValueMapping(BankID: Bank, Size: SrcSize);
4305 break;
4306 }
4307 case AMDGPU::G_BITREVERSE:
4308 case AMDGPU::G_BITCAST:
4309 case AMDGPU::G_INTTOPTR:
4310 case AMDGPU::G_PTRTOINT:
4311 case AMDGPU::G_FABS:
4312 case AMDGPU::G_FNEG: {
4313 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4314 unsigned BankID = getRegBankID(Reg: MI.getOperand(i: 1).getReg(), MRI);
4315 OpdsMapping[0] = OpdsMapping[1] = AMDGPU::getValueMapping(BankID, Size);
4316 break;
4317 }
4318 case AMDGPU::G_AMDGPU_FFBH_U32:
4319 case AMDGPU::G_AMDGPU_FFBL_B32:
4320 case AMDGPU::G_CTLZ_ZERO_POISON:
4321 case AMDGPU::G_CTTZ_ZERO_POISON: {
4322 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 1).getReg()).getSizeInBits();
4323 unsigned BankID = getRegBankID(Reg: MI.getOperand(i: 1).getReg(), MRI);
4324 OpdsMapping[0] = AMDGPU::getValueMapping(BankID, Size: 32);
4325 OpdsMapping[1] = AMDGPU::getValueMappingSGPR64Only(BankID, Size);
4326 break;
4327 }
4328 case AMDGPU::G_CTPOP: {
4329 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 1).getReg()).getSizeInBits();
4330 unsigned BankID = getRegBankID(Reg: MI.getOperand(i: 1).getReg(), MRI);
4331 OpdsMapping[0] = AMDGPU::getValueMapping(BankID, Size: 32);
4332
4333 // This should really be getValueMappingSGPR64Only, but allowing the generic
4334 // code to handle the register split just makes using LegalizerHelper more
4335 // difficult.
4336 OpdsMapping[1] = AMDGPU::getValueMapping(BankID, Size);
4337 break;
4338 }
4339 case AMDGPU::G_TRUNC: {
4340 Register Dst = MI.getOperand(i: 0).getReg();
4341 Register Src = MI.getOperand(i: 1).getReg();
4342 unsigned Bank = getRegBankID(Reg: Src, MRI);
4343 unsigned DstSize = getSizeInBits(Reg: Dst, MRI, TRI: *TRI);
4344 unsigned SrcSize = getSizeInBits(Reg: Src, MRI, TRI: *TRI);
4345 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: Bank, Size: DstSize);
4346 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: Bank, Size: SrcSize);
4347 break;
4348 }
4349 case AMDGPU::G_ZEXT:
4350 case AMDGPU::G_SEXT:
4351 case AMDGPU::G_ANYEXT:
4352 case AMDGPU::G_SEXT_INREG: {
4353 Register Dst = MI.getOperand(i: 0).getReg();
4354 Register Src = MI.getOperand(i: 1).getReg();
4355 unsigned DstSize = getSizeInBits(Reg: Dst, MRI, TRI: *TRI);
4356 unsigned SrcSize = getSizeInBits(Reg: Src, MRI, TRI: *TRI);
4357
4358 unsigned DstBank;
4359 const RegisterBank *SrcBank = getRegBank(Reg: Src, MRI, TRI: *TRI);
4360 assert(SrcBank);
4361 switch (SrcBank->getID()) {
4362 case AMDGPU::SGPRRegBankID:
4363 DstBank = AMDGPU::SGPRRegBankID;
4364 break;
4365 default:
4366 DstBank = AMDGPU::VGPRRegBankID;
4367 break;
4368 }
4369
4370 // Scalar extend can use 64-bit BFE, but VGPRs require extending to
4371 // 32-bits, and then to 64.
4372 OpdsMapping[0] = AMDGPU::getValueMappingSGPR64Only(BankID: DstBank, Size: DstSize);
4373 OpdsMapping[1] = AMDGPU::getValueMappingSGPR64Only(BankID: SrcBank->getID(),
4374 Size: SrcSize);
4375 break;
4376 }
4377 case AMDGPU::G_IS_FPCLASS: {
4378 Register SrcReg = MI.getOperand(i: 1).getReg();
4379 unsigned SrcSize = MRI.getType(Reg: SrcReg).getSizeInBits();
4380 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4381 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: DstSize);
4382 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: SrcSize);
4383 break;
4384 }
4385 case AMDGPU::G_STORE: {
4386 assert(MI.getOperand(0).isReg());
4387 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4388
4389 // FIXME: We need to specify a different reg bank once scalar stores are
4390 // supported.
4391 const ValueMapping *ValMapping =
4392 AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
4393 OpdsMapping[0] = ValMapping;
4394 OpdsMapping[1] = getValueMappingForPtr(MRI, PtrReg: MI.getOperand(i: 1).getReg());
4395 break;
4396 }
4397 case AMDGPU::G_ICMP:
4398 case AMDGPU::G_FCMP: {
4399 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 2).getReg()).getSizeInBits();
4400
4401 // See if the result register has already been constrained to vcc, which may
4402 // happen due to control flow intrinsic lowering.
4403 unsigned DstBank = getRegBankID(Reg: MI.getOperand(i: 0).getReg(), MRI,
4404 Default: AMDGPU::SGPRRegBankID);
4405 unsigned Op2Bank = getRegBankID(Reg: MI.getOperand(i: 2).getReg(), MRI);
4406 unsigned Op3Bank = getRegBankID(Reg: MI.getOperand(i: 3).getReg(), MRI);
4407
4408 auto canUseSCCICMP = [&]() {
4409 auto Pred =
4410 static_cast<CmpInst::Predicate>(MI.getOperand(i: 1).getPredicate());
4411 return Size == 32 ||
4412 (Size == 64 &&
4413 (Pred == CmpInst::ICMP_EQ || Pred == CmpInst::ICMP_NE) &&
4414 Subtarget.hasScalarCompareEq64());
4415 };
4416 auto canUseSCCFCMP = [&]() {
4417 return Subtarget.hasSALUFloatInsts() && (Size == 32 || Size == 16);
4418 };
4419
4420 bool isICMP = MI.getOpcode() == AMDGPU::G_ICMP;
4421 bool CanUseSCC = DstBank == AMDGPU::SGPRRegBankID &&
4422 Op2Bank == AMDGPU::SGPRRegBankID &&
4423 Op3Bank == AMDGPU::SGPRRegBankID &&
4424 (isICMP ? canUseSCCICMP() : canUseSCCFCMP());
4425
4426 DstBank = CanUseSCC ? AMDGPU::SGPRRegBankID : AMDGPU::VCCRegBankID;
4427 unsigned SrcBank = CanUseSCC ? AMDGPU::SGPRRegBankID : AMDGPU::VGPRRegBankID;
4428
4429 // TODO: Use 32-bit for scalar output size.
4430 // SCC results will need to be copied to a 32-bit SGPR virtual register.
4431 const unsigned ResultSize = 1;
4432
4433 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: DstBank, Size: ResultSize);
4434 OpdsMapping[1] = nullptr; // Predicate Operand.
4435 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: SrcBank, Size);
4436 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: SrcBank, Size);
4437 break;
4438 }
4439 case AMDGPU::G_EXTRACT_VECTOR_ELT: {
4440 // VGPR index can be used for waterfall when indexing a SGPR vector.
4441 unsigned SrcBankID = getRegBankID(Reg: MI.getOperand(i: 1).getReg(), MRI);
4442 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4443 unsigned SrcSize = MRI.getType(Reg: MI.getOperand(i: 1).getReg()).getSizeInBits();
4444 unsigned IdxSize = MRI.getType(Reg: MI.getOperand(i: 2).getReg()).getSizeInBits();
4445 unsigned IdxBank = getRegBankID(Reg: MI.getOperand(i: 2).getReg(), MRI);
4446 unsigned OutputBankID = regBankUnion(RB0: SrcBankID, RB1: IdxBank);
4447
4448 OpdsMapping[0] = AMDGPU::getValueMappingSGPR64Only(BankID: OutputBankID, Size: DstSize);
4449 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: SrcBankID, Size: SrcSize);
4450
4451 // The index can be either if the source vector is VGPR.
4452 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: IdxBank, Size: IdxSize);
4453 break;
4454 }
4455 case AMDGPU::G_INSERT_VECTOR_ELT: {
4456 unsigned OutputBankID = isSALUMapping(MI) ?
4457 AMDGPU::SGPRRegBankID : AMDGPU::VGPRRegBankID;
4458
4459 unsigned VecSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4460 unsigned InsertSize = MRI.getType(Reg: MI.getOperand(i: 2).getReg()).getSizeInBits();
4461 unsigned IdxSize = MRI.getType(Reg: MI.getOperand(i: 3).getReg()).getSizeInBits();
4462 unsigned InsertEltBankID = getRegBankID(Reg: MI.getOperand(i: 2).getReg(), MRI);
4463 unsigned IdxBankID = getRegBankID(Reg: MI.getOperand(i: 3).getReg(), MRI);
4464
4465 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: OutputBankID, Size: VecSize);
4466 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: OutputBankID, Size: VecSize);
4467
4468 // This is a weird case, because we need to break down the mapping based on
4469 // the register bank of a different operand.
4470 if (InsertSize == 64 && OutputBankID == AMDGPU::VGPRRegBankID) {
4471 OpdsMapping[2] = AMDGPU::getValueMappingSplit64(BankID: InsertEltBankID,
4472 Size: InsertSize);
4473 } else {
4474 assert(InsertSize == 32 || InsertSize == 64);
4475 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: InsertEltBankID, Size: InsertSize);
4476 }
4477
4478 // The index can be either if the source vector is VGPR.
4479 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: IdxBankID, Size: IdxSize);
4480 break;
4481 }
4482 case AMDGPU::G_UNMERGE_VALUES: {
4483 unsigned Bank = getMappingType(MRI, MI);
4484
4485 // Op1 and Dst should use the same register bank.
4486 // FIXME: Shouldn't this be the default? Why do we need to handle this?
4487 for (unsigned i = 0, e = MI.getNumOperands(); i != e; ++i) {
4488 unsigned Size = getSizeInBits(Reg: MI.getOperand(i).getReg(), MRI, TRI: *TRI);
4489 OpdsMapping[i] = AMDGPU::getValueMapping(BankID: Bank, Size);
4490 }
4491 break;
4492 }
4493 case AMDGPU::G_AMDGPU_BUFFER_LOAD:
4494 case AMDGPU::G_AMDGPU_BUFFER_LOAD_UBYTE:
4495 case AMDGPU::G_AMDGPU_BUFFER_LOAD_SBYTE:
4496 case AMDGPU::G_AMDGPU_BUFFER_LOAD_USHORT:
4497 case AMDGPU::G_AMDGPU_BUFFER_LOAD_SSHORT:
4498 case AMDGPU::G_AMDGPU_BUFFER_LOAD_TFE:
4499 case AMDGPU::G_AMDGPU_BUFFER_LOAD_UBYTE_TFE:
4500 case AMDGPU::G_AMDGPU_BUFFER_LOAD_SBYTE_TFE:
4501 case AMDGPU::G_AMDGPU_BUFFER_LOAD_USHORT_TFE:
4502 case AMDGPU::G_AMDGPU_BUFFER_LOAD_SSHORT_TFE:
4503 case AMDGPU::G_AMDGPU_BUFFER_LOAD_FORMAT:
4504 case AMDGPU::G_AMDGPU_BUFFER_LOAD_FORMAT_TFE:
4505 case AMDGPU::G_AMDGPU_BUFFER_LOAD_FORMAT_D16:
4506 case AMDGPU::G_AMDGPU_BUFFER_LOAD_FORMAT_D16_TFE:
4507 case AMDGPU::G_AMDGPU_TBUFFER_LOAD_FORMAT:
4508 case AMDGPU::G_AMDGPU_TBUFFER_LOAD_FORMAT_D16:
4509 case AMDGPU::G_AMDGPU_TBUFFER_STORE_FORMAT:
4510 case AMDGPU::G_AMDGPU_TBUFFER_STORE_FORMAT_D16:
4511 case AMDGPU::G_AMDGPU_BUFFER_STORE:
4512 case AMDGPU::G_AMDGPU_BUFFER_STORE_BYTE:
4513 case AMDGPU::G_AMDGPU_BUFFER_STORE_SHORT:
4514 case AMDGPU::G_AMDGPU_BUFFER_STORE_FORMAT:
4515 case AMDGPU::G_AMDGPU_BUFFER_STORE_FORMAT_D16: {
4516 OpdsMapping[0] = getVGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
4517
4518 // rsrc
4519 OpdsMapping[1] = getSGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
4520
4521 // vindex
4522 OpdsMapping[2] = getVGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
4523
4524 // voffset
4525 OpdsMapping[3] = getVGPROpMapping(Reg: MI.getOperand(i: 3).getReg(), MRI, TRI: *TRI);
4526
4527 // soffset
4528 OpdsMapping[4] = getSGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI);
4529
4530 // Any remaining operands are immediates and were correctly null
4531 // initialized.
4532 break;
4533 }
4534 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SWAP:
4535 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_ADD:
4536 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SUB:
4537 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SMIN:
4538 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_UMIN:
4539 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SMAX:
4540 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_UMAX:
4541 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_AND:
4542 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_OR:
4543 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_XOR:
4544 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_INC:
4545 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_DEC:
4546 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SUB_CLAMP_U32:
4547 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_COND_SUB_U32:
4548 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_FADD:
4549 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_FMIN:
4550 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_FMAX: {
4551 // vdata_out
4552 OpdsMapping[0] = getVGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
4553
4554 // vdata_in
4555 OpdsMapping[1] = getVGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
4556
4557 // rsrc
4558 OpdsMapping[2] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
4559
4560 // vindex
4561 OpdsMapping[3] = getVGPROpMapping(Reg: MI.getOperand(i: 3).getReg(), MRI, TRI: *TRI);
4562
4563 // voffset
4564 OpdsMapping[4] = getVGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI);
4565
4566 // soffset
4567 OpdsMapping[5] = getSGPROpMapping(Reg: MI.getOperand(i: 5).getReg(), MRI, TRI: *TRI);
4568
4569 // Any remaining operands are immediates and were correctly null
4570 // initialized.
4571 break;
4572 }
4573 case AMDGPU::G_AMDGPU_BUFFER_ATOMIC_CMPSWAP: {
4574 // vdata_out
4575 OpdsMapping[0] = getVGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
4576
4577 // vdata_in
4578 OpdsMapping[1] = getVGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
4579
4580 // cmp
4581 OpdsMapping[2] = getVGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
4582
4583 // rsrc
4584 OpdsMapping[3] = getSGPROpMapping(Reg: MI.getOperand(i: 3).getReg(), MRI, TRI: *TRI);
4585
4586 // vindex
4587 OpdsMapping[4] = getVGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI);
4588
4589 // voffset
4590 OpdsMapping[5] = getVGPROpMapping(Reg: MI.getOperand(i: 5).getReg(), MRI, TRI: *TRI);
4591
4592 // soffset
4593 OpdsMapping[6] = getSGPROpMapping(Reg: MI.getOperand(i: 6).getReg(), MRI, TRI: *TRI);
4594
4595 // Any remaining operands are immediates and were correctly null
4596 // initialized.
4597 break;
4598 }
4599 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD:
4600 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_UBYTE:
4601 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_SBYTE:
4602 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_USHORT:
4603 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_SSHORT: {
4604 // Lie and claim everything is legal, even though some need to be
4605 // SGPRs. applyMapping will have to deal with it as a waterfall loop.
4606 OpdsMapping[1] = getSGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
4607 OpdsMapping[2] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
4608
4609 // We need to convert this to a MUBUF if either the resource of offset is
4610 // VGPR.
4611 unsigned RSrcBank = OpdsMapping[1]->BreakDown[0].RegBank->getID();
4612 unsigned OffsetBank = OpdsMapping[2]->BreakDown[0].RegBank->getID();
4613 unsigned ResultBank = regBankUnion(RB0: RSrcBank, RB1: OffsetBank);
4614
4615 unsigned Size0 = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4616 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: ResultBank, Size: Size0);
4617 break;
4618 }
4619 case AMDGPU::G_AMDGPU_S_BUFFER_PREFETCH:
4620 OpdsMapping[0] = getSGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
4621 OpdsMapping[2] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
4622 break;
4623 case AMDGPU::G_AMDGPU_SPONENTRY: {
4624 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4625 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size);
4626 break;
4627 }
4628 case AMDGPU::G_INTRINSIC:
4629 case AMDGPU::G_INTRINSIC_CONVERGENT: {
4630 switch (cast<GIntrinsic>(Val: MI).getIntrinsicID()) {
4631 default:
4632 return getInvalidInstructionMapping();
4633 case Intrinsic::amdgcn_div_fmas:
4634 case Intrinsic::amdgcn_div_fixup:
4635 case Intrinsic::amdgcn_trig_preop:
4636 case Intrinsic::amdgcn_sin:
4637 case Intrinsic::amdgcn_cos:
4638 case Intrinsic::amdgcn_log_clamp:
4639 case Intrinsic::amdgcn_rcp_legacy:
4640 case Intrinsic::amdgcn_rsq_legacy:
4641 case Intrinsic::amdgcn_rsq_clamp:
4642 case Intrinsic::amdgcn_tanh:
4643 case Intrinsic::amdgcn_fmul_legacy:
4644 case Intrinsic::amdgcn_fma_legacy:
4645 case Intrinsic::amdgcn_frexp_mant:
4646 case Intrinsic::amdgcn_frexp_exp:
4647 case Intrinsic::amdgcn_fract:
4648 case Intrinsic::amdgcn_cvt_pknorm_i16:
4649 case Intrinsic::amdgcn_cvt_pknorm_u16:
4650 case Intrinsic::amdgcn_cvt_pk_i16:
4651 case Intrinsic::amdgcn_cvt_pk_u16:
4652 case Intrinsic::amdgcn_cvt_sr_pk_f16_f32:
4653 case Intrinsic::amdgcn_cvt_sr_pk_bf16_f32:
4654 case Intrinsic::amdgcn_cvt_pk_f16_fp8:
4655 case Intrinsic::amdgcn_cvt_pk_f16_bf8:
4656 case Intrinsic::amdgcn_cvt_pk_fp8_f16:
4657 case Intrinsic::amdgcn_cvt_pk_bf8_f16:
4658 case Intrinsic::amdgcn_cvt_sr_fp8_f16:
4659 case Intrinsic::amdgcn_cvt_sr_bf8_f16:
4660 case Intrinsic::amdgcn_cvt_scale_pk8_f16_fp8:
4661 case Intrinsic::amdgcn_cvt_scale_pk8_bf16_fp8:
4662 case Intrinsic::amdgcn_cvt_scale_pk8_f16_bf8:
4663 case Intrinsic::amdgcn_cvt_scale_pk8_bf16_bf8:
4664 case Intrinsic::amdgcn_cvt_scale_pk8_f16_fp4:
4665 case Intrinsic::amdgcn_cvt_scale_pk8_bf16_fp4:
4666 case Intrinsic::amdgcn_cvt_scale_pk8_f32_fp8:
4667 case Intrinsic::amdgcn_cvt_scale_pk8_f32_bf8:
4668 case Intrinsic::amdgcn_cvt_scale_pk8_f32_fp4:
4669 case Intrinsic::amdgcn_cvt_scale_pk16_f16_fp6:
4670 case Intrinsic::amdgcn_cvt_scale_pk16_bf16_fp6:
4671 case Intrinsic::amdgcn_cvt_scale_pk16_f16_bf6:
4672 case Intrinsic::amdgcn_cvt_scale_pk16_bf16_bf6:
4673 case Intrinsic::amdgcn_cvt_scale_pk16_f32_fp6:
4674 case Intrinsic::amdgcn_cvt_scale_pk16_f32_bf6:
4675 case Intrinsic::amdgcn_cvt_scale_pk32_f16_fp6:
4676 case Intrinsic::amdgcn_cvt_scale_pk32_bf16_fp6:
4677 case Intrinsic::amdgcn_cvt_scale_pk32_f16_bf6:
4678 case Intrinsic::amdgcn_cvt_scale_pk32_bf16_bf6:
4679 case Intrinsic::amdgcn_cvt_scale_pk32_f32_fp6:
4680 case Intrinsic::amdgcn_cvt_scale_pk32_f32_bf6:
4681 case Intrinsic::amdgcn_cvt_scalef32_pk8_fp8_bf16:
4682 case Intrinsic::amdgcn_cvt_scalef32_pk8_bf8_bf16:
4683 case Intrinsic::amdgcn_cvt_scalef32_pk8_fp8_f16:
4684 case Intrinsic::amdgcn_cvt_scalef32_pk8_bf8_f16:
4685 case Intrinsic::amdgcn_cvt_scalef32_pk8_fp8_f32:
4686 case Intrinsic::amdgcn_cvt_scalef32_pk8_bf8_f32:
4687 case Intrinsic::amdgcn_cvt_scalef32_pk8_fp4_f32:
4688 case Intrinsic::amdgcn_cvt_scalef32_pk8_fp4_f16:
4689 case Intrinsic::amdgcn_cvt_scalef32_pk8_fp4_bf16:
4690 case Intrinsic::amdgcn_cvt_scalef32_pk16_fp6_f32:
4691 case Intrinsic::amdgcn_cvt_scalef32_pk16_bf6_f32:
4692 case Intrinsic::amdgcn_cvt_scalef32_pk16_fp6_f16:
4693 case Intrinsic::amdgcn_cvt_scalef32_pk16_bf6_f16:
4694 case Intrinsic::amdgcn_cvt_scalef32_pk16_fp6_bf16:
4695 case Intrinsic::amdgcn_cvt_scalef32_pk16_bf6_bf16:
4696 case Intrinsic::amdgcn_cvt_scalef32_sr_pk8_fp8_bf16:
4697 case Intrinsic::amdgcn_cvt_scalef32_sr_pk8_bf8_bf16:
4698 case Intrinsic::amdgcn_cvt_scalef32_sr_pk8_fp8_f16:
4699 case Intrinsic::amdgcn_cvt_scalef32_sr_pk8_bf8_f16:
4700 case Intrinsic::amdgcn_cvt_scalef32_sr_pk8_fp8_f32:
4701 case Intrinsic::amdgcn_cvt_scalef32_sr_pk8_bf8_f32:
4702 case Intrinsic::amdgcn_cvt_scalef32_sr_pk8_fp4_f32:
4703 case Intrinsic::amdgcn_cvt_scalef32_sr_pk8_fp4_f16:
4704 case Intrinsic::amdgcn_cvt_scalef32_sr_pk8_fp4_bf16:
4705 case Intrinsic::amdgcn_cvt_scalef32_sr_pk16_fp6_f32:
4706 case Intrinsic::amdgcn_cvt_scalef32_sr_pk16_bf6_f32:
4707 case Intrinsic::amdgcn_cvt_scalef32_sr_pk16_fp6_f16:
4708 case Intrinsic::amdgcn_cvt_scalef32_sr_pk16_bf6_f16:
4709 case Intrinsic::amdgcn_cvt_scalef32_sr_pk16_fp6_bf16:
4710 case Intrinsic::amdgcn_cvt_scalef32_sr_pk16_bf6_bf16:
4711 case Intrinsic::amdgcn_sat_pk4_i4_i8:
4712 case Intrinsic::amdgcn_sat_pk4_u4_u8:
4713 case Intrinsic::amdgcn_fmed3:
4714 case Intrinsic::amdgcn_cubeid:
4715 case Intrinsic::amdgcn_cubema:
4716 case Intrinsic::amdgcn_cubesc:
4717 case Intrinsic::amdgcn_cubetc:
4718 case Intrinsic::amdgcn_sffbh:
4719 case Intrinsic::amdgcn_fmad_ftz:
4720 case Intrinsic::amdgcn_mbcnt_lo:
4721 case Intrinsic::amdgcn_mbcnt_hi:
4722 case Intrinsic::amdgcn_mul_u24:
4723 case Intrinsic::amdgcn_mul_i24:
4724 case Intrinsic::amdgcn_mulhi_u24:
4725 case Intrinsic::amdgcn_mulhi_i24:
4726 case Intrinsic::amdgcn_lerp:
4727 case Intrinsic::amdgcn_sad_u8:
4728 case Intrinsic::amdgcn_msad_u8:
4729 case Intrinsic::amdgcn_sad_hi_u8:
4730 case Intrinsic::amdgcn_sad_u16:
4731 case Intrinsic::amdgcn_qsad_pk_u16_u8:
4732 case Intrinsic::amdgcn_mqsad_pk_u16_u8:
4733 case Intrinsic::amdgcn_mqsad_u32_u8:
4734 case Intrinsic::amdgcn_cvt_pk_u8_f32:
4735 case Intrinsic::amdgcn_alignbyte:
4736 case Intrinsic::amdgcn_perm:
4737 case Intrinsic::amdgcn_prng_b32:
4738 case Intrinsic::amdgcn_exclusive_scan_sum_i32:
4739 case Intrinsic::amdgcn_exclusive_scan_sum_u32:
4740 case Intrinsic::amdgcn_exclusive_scan_xor_b32:
4741 case Intrinsic::amdgcn_exclusive_scan_or_b32:
4742 case Intrinsic::amdgcn_exclusive_scan_and_b32:
4743 case Intrinsic::amdgcn_exclusive_scan_min_i16:
4744 case Intrinsic::amdgcn_exclusive_scan_min_u16:
4745 case Intrinsic::amdgcn_exclusive_scan_min_i32:
4746 case Intrinsic::amdgcn_exclusive_scan_min_u32:
4747 case Intrinsic::amdgcn_exclusive_scan_max_i16:
4748 case Intrinsic::amdgcn_exclusive_scan_max_u16:
4749 case Intrinsic::amdgcn_exclusive_scan_max_i32:
4750 case Intrinsic::amdgcn_exclusive_scan_max_u32:
4751 case Intrinsic::amdgcn_wave_match_b32:
4752 case Intrinsic::amdgcn_fdot2:
4753 case Intrinsic::amdgcn_sdot2:
4754 case Intrinsic::amdgcn_udot2:
4755 case Intrinsic::amdgcn_sdot4:
4756 case Intrinsic::amdgcn_udot4:
4757 case Intrinsic::amdgcn_sdot8:
4758 case Intrinsic::amdgcn_udot8:
4759 case Intrinsic::amdgcn_fdot2_bf16_bf16:
4760 case Intrinsic::amdgcn_fdot2_f16_f16:
4761 case Intrinsic::amdgcn_fdot2_f32_bf16:
4762 case Intrinsic::amdgcn_fdot2c_f32_bf16:
4763 case Intrinsic::amdgcn_sudot4:
4764 case Intrinsic::amdgcn_sudot8:
4765 case Intrinsic::amdgcn_dot4_f32_fp8_bf8:
4766 case Intrinsic::amdgcn_dot4_f32_bf8_fp8:
4767 case Intrinsic::amdgcn_dot4_f32_fp8_fp8:
4768 case Intrinsic::amdgcn_dot4_f32_bf8_bf8:
4769 case Intrinsic::amdgcn_cvt_f32_fp8:
4770 case Intrinsic::amdgcn_cvt_f32_fp8_e5m3:
4771 case Intrinsic::amdgcn_cvt_f32_bf8:
4772 case Intrinsic::amdgcn_cvt_off_f32_i4:
4773 case Intrinsic::amdgcn_cvt_pk_f32_fp8:
4774 case Intrinsic::amdgcn_cvt_pk_f32_bf8:
4775 case Intrinsic::amdgcn_cvt_pk_fp8_f32:
4776 case Intrinsic::amdgcn_cvt_pk_fp8_f32_e5m3:
4777 case Intrinsic::amdgcn_cvt_pk_bf8_f32:
4778 case Intrinsic::amdgcn_cvt_sr_fp8_f32:
4779 case Intrinsic::amdgcn_cvt_sr_fp8_f32_e5m3:
4780 case Intrinsic::amdgcn_cvt_sr_bf8_f32:
4781 case Intrinsic::amdgcn_cvt_sr_bf16_f32:
4782 case Intrinsic::amdgcn_cvt_sr_f16_f32:
4783 case Intrinsic::amdgcn_cvt_f16_fp8:
4784 case Intrinsic::amdgcn_cvt_f16_bf8:
4785 case Intrinsic::amdgcn_cvt_scalef32_pk32_fp6_f16:
4786 case Intrinsic::amdgcn_cvt_scalef32_pk32_bf6_f16:
4787 case Intrinsic::amdgcn_cvt_scalef32_pk32_fp6_bf16:
4788 case Intrinsic::amdgcn_cvt_scalef32_pk32_bf6_bf16:
4789 case Intrinsic::amdgcn_cvt_scalef32_pk32_fp6_f32:
4790 case Intrinsic::amdgcn_cvt_scalef32_pk32_bf6_f32:
4791 case Intrinsic::amdgcn_cvt_scalef32_f16_fp8:
4792 case Intrinsic::amdgcn_cvt_scalef32_f16_bf8:
4793 case Intrinsic::amdgcn_cvt_scalef32_f32_fp8:
4794 case Intrinsic::amdgcn_cvt_scalef32_f32_bf8:
4795 case Intrinsic::amdgcn_cvt_scalef32_pk_fp8_f32:
4796 case Intrinsic::amdgcn_cvt_scalef32_pk_bf8_f32:
4797 case Intrinsic::amdgcn_cvt_scalef32_pk_f32_fp8:
4798 case Intrinsic::amdgcn_cvt_scalef32_pk_f32_bf8:
4799 case Intrinsic::amdgcn_cvt_scalef32_pk_fp8_f16:
4800 case Intrinsic::amdgcn_cvt_scalef32_pk_fp8_bf16:
4801 case Intrinsic::amdgcn_cvt_scalef32_pk_bf8_f16:
4802 case Intrinsic::amdgcn_cvt_scalef32_pk_bf8_bf16:
4803 case Intrinsic::amdgcn_cvt_scalef32_pk_f32_fp4:
4804 case Intrinsic::amdgcn_cvt_scalef32_pk_fp4_f32:
4805 case Intrinsic::amdgcn_cvt_scalef32_pk_f16_fp4:
4806 case Intrinsic::amdgcn_cvt_scalef32_pk_bf16_fp4:
4807 case Intrinsic::amdgcn_cvt_scalef32_pk32_f32_fp6:
4808 case Intrinsic::amdgcn_cvt_scalef32_pk32_f32_bf6:
4809 case Intrinsic::amdgcn_cvt_scalef32_pk32_f16_bf6:
4810 case Intrinsic::amdgcn_cvt_scalef32_pk32_bf16_bf6:
4811 case Intrinsic::amdgcn_cvt_scalef32_pk32_f16_fp6:
4812 case Intrinsic::amdgcn_cvt_scalef32_pk32_bf16_fp6:
4813 case Intrinsic::amdgcn_cvt_scalef32_pk_f16_bf8:
4814 case Intrinsic::amdgcn_cvt_scalef32_pk_bf16_bf8:
4815 case Intrinsic::amdgcn_cvt_scalef32_pk_f16_fp8:
4816 case Intrinsic::amdgcn_cvt_scalef32_pk_bf16_fp8:
4817 case Intrinsic::amdgcn_cvt_scalef32_pk_fp4_f16:
4818 case Intrinsic::amdgcn_cvt_scalef32_pk_fp4_bf16:
4819 case Intrinsic::amdgcn_cvt_scalef32_sr_pk_fp4_f16:
4820 case Intrinsic::amdgcn_cvt_scalef32_sr_pk_fp4_bf16:
4821 case Intrinsic::amdgcn_cvt_scalef32_sr_pk_fp4_f32:
4822 case Intrinsic::amdgcn_cvt_scalef32_sr_pk32_bf6_bf16:
4823 case Intrinsic::amdgcn_cvt_scalef32_sr_pk32_bf6_f16:
4824 case Intrinsic::amdgcn_cvt_scalef32_sr_pk32_bf6_f32:
4825 case Intrinsic::amdgcn_cvt_scalef32_sr_pk32_fp6_bf16:
4826 case Intrinsic::amdgcn_cvt_scalef32_sr_pk32_fp6_f16:
4827 case Intrinsic::amdgcn_cvt_scalef32_sr_pk32_fp6_f32:
4828 case Intrinsic::amdgcn_cvt_scalef32_sr_bf8_bf16:
4829 case Intrinsic::amdgcn_cvt_scalef32_sr_bf8_f16:
4830 case Intrinsic::amdgcn_cvt_scalef32_sr_bf8_f32:
4831 case Intrinsic::amdgcn_cvt_scalef32_sr_fp8_bf16:
4832 case Intrinsic::amdgcn_cvt_scalef32_sr_fp8_f16:
4833 case Intrinsic::amdgcn_cvt_scalef32_sr_fp8_f32:
4834 case Intrinsic::amdgcn_ashr_pk_i8_i32:
4835 case Intrinsic::amdgcn_ashr_pk_u8_i32:
4836 case Intrinsic::amdgcn_cvt_scalef32_2xpk16_fp6_f32:
4837 case Intrinsic::amdgcn_cvt_scalef32_2xpk16_bf6_f32:
4838 case Intrinsic::amdgcn_wmma_bf16_16x16x16_bf16:
4839 case Intrinsic::amdgcn_wmma_f16_16x16x16_f16:
4840 case Intrinsic::amdgcn_wmma_bf16_16x16x16_bf16_tied:
4841 case Intrinsic::amdgcn_wmma_f16_16x16x16_f16_tied:
4842 case Intrinsic::amdgcn_wmma_f32_16x16x16_bf16:
4843 case Intrinsic::amdgcn_wmma_f32_16x16x16_f16:
4844 case Intrinsic::amdgcn_wmma_i32_16x16x16_iu4:
4845 case Intrinsic::amdgcn_wmma_i32_16x16x16_iu8:
4846 case Intrinsic::amdgcn_wmma_f32_16x16x16_fp8_fp8:
4847 case Intrinsic::amdgcn_wmma_f32_16x16x16_fp8_bf8:
4848 case Intrinsic::amdgcn_wmma_f32_16x16x16_bf8_fp8:
4849 case Intrinsic::amdgcn_wmma_f32_16x16x16_bf8_bf8:
4850 case Intrinsic::amdgcn_wmma_i32_16x16x32_iu4:
4851 case Intrinsic::amdgcn_swmmac_f32_16x16x32_f16:
4852 case Intrinsic::amdgcn_swmmac_f32_16x16x32_bf16:
4853 case Intrinsic::amdgcn_swmmac_f16_16x16x32_f16:
4854 case Intrinsic::amdgcn_swmmac_bf16_16x16x32_bf16:
4855 case Intrinsic::amdgcn_swmmac_i32_16x16x32_iu8:
4856 case Intrinsic::amdgcn_swmmac_i32_16x16x32_iu4:
4857 case Intrinsic::amdgcn_swmmac_i32_16x16x64_iu4:
4858 case Intrinsic::amdgcn_swmmac_f32_16x16x32_fp8_fp8:
4859 case Intrinsic::amdgcn_swmmac_f32_16x16x32_fp8_bf8:
4860 case Intrinsic::amdgcn_swmmac_f32_16x16x32_bf8_fp8:
4861 case Intrinsic::amdgcn_swmmac_f32_16x16x32_bf8_bf8:
4862 case Intrinsic::amdgcn_wmma_f64_16x16x4_f64:
4863 case Intrinsic::amdgcn_wmma_f32_16x16x4_f32:
4864 case Intrinsic::amdgcn_wmma_f32_16x16x32_bf16:
4865 case Intrinsic::amdgcn_wmma_f32_16x16x32_f16:
4866 case Intrinsic::amdgcn_wmma_f16_16x16x32_f16:
4867 case Intrinsic::amdgcn_wmma_bf16_16x16x32_bf16:
4868 case Intrinsic::amdgcn_wmma_bf16f32_16x16x32_bf16:
4869 case Intrinsic::amdgcn_wmma_f32_16x16x64_fp8_fp8:
4870 case Intrinsic::amdgcn_wmma_f32_16x16x64_fp8_bf8:
4871 case Intrinsic::amdgcn_wmma_f32_16x16x64_bf8_fp8:
4872 case Intrinsic::amdgcn_wmma_f32_16x16x64_bf8_bf8:
4873 case Intrinsic::amdgcn_wmma_f16_16x16x64_fp8_fp8:
4874 case Intrinsic::amdgcn_wmma_f16_16x16x64_fp8_bf8:
4875 case Intrinsic::amdgcn_wmma_f16_16x16x64_bf8_fp8:
4876 case Intrinsic::amdgcn_wmma_f16_16x16x64_bf8_bf8:
4877 case Intrinsic::amdgcn_wmma_f16_16x16x128_fp8_fp8:
4878 case Intrinsic::amdgcn_wmma_f16_16x16x128_fp8_bf8:
4879 case Intrinsic::amdgcn_wmma_f16_16x16x128_bf8_fp8:
4880 case Intrinsic::amdgcn_wmma_f16_16x16x128_bf8_bf8:
4881 case Intrinsic::amdgcn_wmma_f32_16x16x128_fp8_fp8:
4882 case Intrinsic::amdgcn_wmma_f32_16x16x128_fp8_bf8:
4883 case Intrinsic::amdgcn_wmma_f32_16x16x128_bf8_fp8:
4884 case Intrinsic::amdgcn_wmma_f32_16x16x128_bf8_bf8:
4885 case Intrinsic::amdgcn_wmma_i32_16x16x64_iu8:
4886 case Intrinsic::amdgcn_wmma_f32_16x16x128_f8f6f4:
4887 case Intrinsic::amdgcn_wmma_scale_f32_16x16x128_f8f6f4:
4888 case Intrinsic::amdgcn_wmma_scale16_f32_16x16x128_f8f6f4:
4889 case Intrinsic::amdgcn_wmma_f32_32x16x128_f4:
4890 case Intrinsic::amdgcn_wmma_scale_f32_32x16x128_f4:
4891 case Intrinsic::amdgcn_wmma_scale16_f32_32x16x128_f4:
4892 case Intrinsic::amdgcn_swmmac_f16_16x16x64_f16:
4893 case Intrinsic::amdgcn_swmmac_bf16_16x16x64_bf16:
4894 case Intrinsic::amdgcn_swmmac_f32_16x16x64_bf16:
4895 case Intrinsic::amdgcn_swmmac_bf16f32_16x16x64_bf16:
4896 case Intrinsic::amdgcn_swmmac_f32_16x16x64_f16:
4897 case Intrinsic::amdgcn_swmmac_f32_16x16x128_fp8_fp8:
4898 case Intrinsic::amdgcn_swmmac_f32_16x16x128_fp8_bf8:
4899 case Intrinsic::amdgcn_swmmac_f32_16x16x128_bf8_fp8:
4900 case Intrinsic::amdgcn_swmmac_f32_16x16x128_bf8_bf8:
4901 case Intrinsic::amdgcn_swmmac_f16_16x16x128_fp8_fp8:
4902 case Intrinsic::amdgcn_swmmac_f16_16x16x128_fp8_bf8:
4903 case Intrinsic::amdgcn_swmmac_f16_16x16x128_bf8_fp8:
4904 case Intrinsic::amdgcn_swmmac_f16_16x16x128_bf8_bf8:
4905 case Intrinsic::amdgcn_swmmac_i32_16x16x128_iu8:
4906 case Intrinsic::amdgcn_perm_pk16_b4_u4:
4907 case Intrinsic::amdgcn_perm_pk16_b6_u4:
4908 case Intrinsic::amdgcn_perm_pk16_b8_u4:
4909 case Intrinsic::amdgcn_add_max_i32:
4910 case Intrinsic::amdgcn_add_max_u32:
4911 case Intrinsic::amdgcn_add_min_i32:
4912 case Intrinsic::amdgcn_add_min_u32:
4913 case Intrinsic::amdgcn_pk_add_max_i16:
4914 case Intrinsic::amdgcn_pk_add_max_u16:
4915 case Intrinsic::amdgcn_pk_add_min_i16:
4916 case Intrinsic::amdgcn_pk_add_min_u16:
4917 return getDefaultMappingVOP(MI);
4918 case Intrinsic::amdgcn_log:
4919 case Intrinsic::amdgcn_exp2:
4920 case Intrinsic::amdgcn_rcp:
4921 case Intrinsic::amdgcn_rsq:
4922 case Intrinsic::amdgcn_sqrt: {
4923 LLT Ty = MRI.getType(Reg: MI.getOperand(i: 0).getReg());
4924 unsigned Size = Ty.getSizeInBits();
4925 // There is no pseudo scalar transcendental instruction for bf16.
4926 if (Subtarget.hasPseudoScalarTrans() && !Ty.isBFloat16() &&
4927 (Size == 16 || Size == 32) && isSALUMapping(MI))
4928 return getDefaultMappingSOP(MI);
4929 return getDefaultMappingVOP(MI);
4930 }
4931 case Intrinsic::amdgcn_sbfe:
4932 case Intrinsic::amdgcn_ubfe:
4933 if (isSALUMapping(MI))
4934 return getDefaultMappingSOP(MI);
4935 return getDefaultMappingVOP(MI);
4936 case Intrinsic::amdgcn_ds_swizzle:
4937 case Intrinsic::amdgcn_ds_permute:
4938 case Intrinsic::amdgcn_ds_bpermute:
4939 case Intrinsic::amdgcn_update_dpp:
4940 case Intrinsic::amdgcn_mov_dpp8:
4941 case Intrinsic::amdgcn_mov_dpp:
4942 case Intrinsic::amdgcn_strict_wwm:
4943 case Intrinsic::amdgcn_wwm:
4944 case Intrinsic::amdgcn_strict_wqm:
4945 case Intrinsic::amdgcn_wqm:
4946 case Intrinsic::amdgcn_softwqm:
4947 case Intrinsic::amdgcn_set_inactive:
4948 case Intrinsic::amdgcn_set_inactive_chain_arg:
4949 case Intrinsic::amdgcn_permlane64:
4950 case Intrinsic::amdgcn_ds_bpermute_fi_b32:
4951 return getDefaultMappingAllVGPR(MI);
4952 case Intrinsic::amdgcn_cvt_pkrtz:
4953 if (Subtarget.hasSALUFloatInsts() && isSALUMapping(MI))
4954 return getDefaultMappingSOP(MI);
4955 return getDefaultMappingVOP(MI);
4956 case Intrinsic::amdgcn_kernarg_segment_ptr:
4957 case Intrinsic::amdgcn_s_getpc:
4958 case Intrinsic::amdgcn_groupstaticsize:
4959 case Intrinsic::amdgcn_reloc_constant:
4960 case Intrinsic::returnaddress: {
4961 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4962 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size);
4963 break;
4964 }
4965 case Intrinsic::amdgcn_wqm_vote: {
4966 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4967 OpdsMapping[0] = OpdsMapping[2]
4968 = AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size);
4969 break;
4970 }
4971 case Intrinsic::amdgcn_ps_live: {
4972 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: 1);
4973 break;
4974 }
4975 case Intrinsic::amdgcn_div_scale: {
4976 unsigned Dst0Size = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4977 unsigned Dst1Size = MRI.getType(Reg: MI.getOperand(i: 1).getReg()).getSizeInBits();
4978 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: Dst0Size);
4979 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: Dst1Size);
4980
4981 unsigned SrcSize = MRI.getType(Reg: MI.getOperand(i: 3).getReg()).getSizeInBits();
4982 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: SrcSize);
4983 OpdsMapping[4] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: SrcSize);
4984 break;
4985 }
4986 case Intrinsic::amdgcn_class: {
4987 Register Src0Reg = MI.getOperand(i: 2).getReg();
4988 Register Src1Reg = MI.getOperand(i: 3).getReg();
4989 unsigned Src0Size = MRI.getType(Reg: Src0Reg).getSizeInBits();
4990 unsigned Src1Size = MRI.getType(Reg: Src1Reg).getSizeInBits();
4991 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
4992 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: DstSize);
4993 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: Src0Size);
4994 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: Src1Size);
4995 break;
4996 }
4997 case Intrinsic::amdgcn_readlane: {
4998 // This must be an SGPR, but accept a VGPR.
4999 Register IdxReg = MI.getOperand(i: 3).getReg();
5000 unsigned IdxSize = MRI.getType(Reg: IdxReg).getSizeInBits();
5001 unsigned IdxBank = getRegBankID(Reg: IdxReg, MRI, Default: AMDGPU::SGPRRegBankID);
5002 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: IdxBank, Size: IdxSize);
5003 [[fallthrough]];
5004 }
5005 case Intrinsic::amdgcn_readfirstlane: {
5006 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5007 unsigned SrcSize = MRI.getType(Reg: MI.getOperand(i: 2).getReg()).getSizeInBits();
5008 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: DstSize);
5009 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: SrcSize);
5010 break;
5011 }
5012 case Intrinsic::amdgcn_writelane: {
5013 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5014 Register SrcReg = MI.getOperand(i: 2).getReg();
5015 unsigned SrcSize = MRI.getType(Reg: SrcReg).getSizeInBits();
5016 unsigned SrcBank = getRegBankID(Reg: SrcReg, MRI, Default: AMDGPU::SGPRRegBankID);
5017 Register IdxReg = MI.getOperand(i: 3).getReg();
5018 unsigned IdxSize = MRI.getType(Reg: IdxReg).getSizeInBits();
5019 unsigned IdxBank = getRegBankID(Reg: IdxReg, MRI, Default: AMDGPU::SGPRRegBankID);
5020 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: DstSize);
5021
5022 // These 2 must be SGPRs, but accept VGPRs. Readfirstlane will be inserted
5023 // to legalize.
5024 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: SrcBank, Size: SrcSize);
5025 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: IdxBank, Size: IdxSize);
5026 OpdsMapping[4] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: SrcSize);
5027 break;
5028 }
5029 case Intrinsic::amdgcn_if_break: {
5030 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5031 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size);
5032 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: 1);
5033 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size);
5034 break;
5035 }
5036 case Intrinsic::amdgcn_permlane16:
5037 case Intrinsic::amdgcn_permlanex16: {
5038 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5039 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5040 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5041 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5042 OpdsMapping[4] = getSGPROpMapping(Reg: MI.getOperand(i: 3).getReg(), MRI, TRI: *TRI);
5043 OpdsMapping[5] = getSGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI);
5044 break;
5045 }
5046 case Intrinsic::amdgcn_permlane_bcast:
5047 case Intrinsic::amdgcn_permlane_up:
5048 case Intrinsic::amdgcn_permlane_down:
5049 case Intrinsic::amdgcn_permlane_xor: {
5050 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5051 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5052 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5053 OpdsMapping[3] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5054 OpdsMapping[4] = getSGPROpMapping(Reg: MI.getOperand(i: 3).getReg(), MRI, TRI: *TRI);
5055 break;
5056 }
5057 case Intrinsic::amdgcn_permlane_idx_gen: {
5058 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5059 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5060 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5061 OpdsMapping[3] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5062 break;
5063 }
5064 case Intrinsic::amdgcn_permlane16_var:
5065 case Intrinsic::amdgcn_permlanex16_var: {
5066 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5067 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5068 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5069 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5070 OpdsMapping[4] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5071 break;
5072 }
5073 case Intrinsic::amdgcn_mfma_f32_4x4x1f32:
5074 case Intrinsic::amdgcn_mfma_f32_4x4x4f16:
5075 case Intrinsic::amdgcn_mfma_i32_4x4x4i8:
5076 case Intrinsic::amdgcn_mfma_f32_4x4x2bf16:
5077 case Intrinsic::amdgcn_mfma_f32_16x16x1f32:
5078 case Intrinsic::amdgcn_mfma_f32_16x16x4f32:
5079 case Intrinsic::amdgcn_mfma_f32_16x16x4f16:
5080 case Intrinsic::amdgcn_mfma_f32_16x16x16f16:
5081 case Intrinsic::amdgcn_mfma_i32_16x16x4i8:
5082 case Intrinsic::amdgcn_mfma_i32_16x16x16i8:
5083 case Intrinsic::amdgcn_mfma_f32_16x16x2bf16:
5084 case Intrinsic::amdgcn_mfma_f32_16x16x8bf16:
5085 case Intrinsic::amdgcn_mfma_f32_32x32x1f32:
5086 case Intrinsic::amdgcn_mfma_f32_32x32x2f32:
5087 case Intrinsic::amdgcn_mfma_f32_32x32x4f16:
5088 case Intrinsic::amdgcn_mfma_f32_32x32x8f16:
5089 case Intrinsic::amdgcn_mfma_i32_32x32x4i8:
5090 case Intrinsic::amdgcn_mfma_i32_32x32x8i8:
5091 case Intrinsic::amdgcn_mfma_f32_32x32x2bf16:
5092 case Intrinsic::amdgcn_mfma_f32_32x32x4bf16:
5093 case Intrinsic::amdgcn_mfma_f32_32x32x4bf16_1k:
5094 case Intrinsic::amdgcn_mfma_f32_16x16x4bf16_1k:
5095 case Intrinsic::amdgcn_mfma_f32_4x4x4bf16_1k:
5096 case Intrinsic::amdgcn_mfma_f32_32x32x8bf16_1k:
5097 case Intrinsic::amdgcn_mfma_f32_16x16x16bf16_1k:
5098 case Intrinsic::amdgcn_mfma_f64_16x16x4f64:
5099 case Intrinsic::amdgcn_mfma_f64_4x4x4f64:
5100 case Intrinsic::amdgcn_mfma_i32_16x16x32_i8:
5101 case Intrinsic::amdgcn_mfma_i32_32x32x16_i8:
5102 case Intrinsic::amdgcn_mfma_f32_16x16x8_xf32:
5103 case Intrinsic::amdgcn_mfma_f32_32x32x4_xf32:
5104 case Intrinsic::amdgcn_mfma_f32_16x16x32_bf8_bf8:
5105 case Intrinsic::amdgcn_mfma_f32_16x16x32_bf8_fp8:
5106 case Intrinsic::amdgcn_mfma_f32_16x16x32_fp8_bf8:
5107 case Intrinsic::amdgcn_mfma_f32_16x16x32_fp8_fp8:
5108 case Intrinsic::amdgcn_mfma_f32_32x32x16_bf8_bf8:
5109 case Intrinsic::amdgcn_mfma_f32_32x32x16_bf8_fp8:
5110 case Intrinsic::amdgcn_mfma_f32_32x32x16_fp8_bf8:
5111 case Intrinsic::amdgcn_mfma_f32_32x32x16_fp8_fp8:
5112 case Intrinsic::amdgcn_mfma_f32_16x16x32_f16:
5113 case Intrinsic::amdgcn_mfma_f32_32x32x16_f16:
5114 case Intrinsic::amdgcn_mfma_i32_16x16x64_i8:
5115 case Intrinsic::amdgcn_mfma_i32_32x32x32_i8:
5116 case Intrinsic::amdgcn_mfma_f32_16x16x32_bf16: {
5117 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5118 unsigned MinNumRegsRequired = DstSize / 32;
5119
5120 // Default for MAI intrinsics.
5121 // srcC can also be an immediate which can be folded later.
5122 // FIXME: Should we eventually add an alternative mapping with AGPR src
5123 // for srcA/srcB?
5124 //
5125 // vdst, srcA, srcB, srcC
5126 const SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
5127
5128 bool UseAGPRForm = !Subtarget.hasGFX90AInsts() ||
5129 Info->selectAGPRFormMFMA(NumRegs: MinNumRegsRequired);
5130
5131 OpdsMapping[0] =
5132 UseAGPRForm ? getAGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI)
5133 : getVGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5134 OpdsMapping[2] = getVGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5135 OpdsMapping[3] = getVGPROpMapping(Reg: MI.getOperand(i: 3).getReg(), MRI, TRI: *TRI);
5136 OpdsMapping[4] =
5137 UseAGPRForm ? getAGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI)
5138 : getVGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI);
5139 break;
5140 }
5141 case Intrinsic::amdgcn_mfma_scale_f32_16x16x128_f8f6f4:
5142 case Intrinsic::amdgcn_mfma_scale_f32_32x32x64_f8f6f4: {
5143 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5144 unsigned MinNumRegsRequired = DstSize / 32;
5145
5146 const SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
5147 bool UseAGPRForm = Info->selectAGPRFormMFMA(NumRegs: MinNumRegsRequired);
5148
5149 OpdsMapping[0] =
5150 UseAGPRForm ? getAGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI)
5151 : getVGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5152
5153 OpdsMapping[2] = getVGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5154 OpdsMapping[3] = getVGPROpMapping(Reg: MI.getOperand(i: 3).getReg(), MRI, TRI: *TRI);
5155 OpdsMapping[4] =
5156 UseAGPRForm ? getAGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI)
5157 : getVGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI);
5158
5159 OpdsMapping[8] = getVGPROpMapping(Reg: MI.getOperand(i: 8).getReg(), MRI, TRI: *TRI);
5160 OpdsMapping[10] = getVGPROpMapping(Reg: MI.getOperand(i: 10).getReg(), MRI, TRI: *TRI);
5161 break;
5162 }
5163 case Intrinsic::amdgcn_smfmac_f32_16x16x32_f16:
5164 case Intrinsic::amdgcn_smfmac_f32_32x32x16_f16:
5165 case Intrinsic::amdgcn_smfmac_f32_16x16x32_bf16:
5166 case Intrinsic::amdgcn_smfmac_f32_32x32x16_bf16:
5167 case Intrinsic::amdgcn_smfmac_i32_16x16x64_i8:
5168 case Intrinsic::amdgcn_smfmac_i32_32x32x32_i8:
5169 case Intrinsic::amdgcn_smfmac_f32_16x16x64_bf8_bf8:
5170 case Intrinsic::amdgcn_smfmac_f32_16x16x64_bf8_fp8:
5171 case Intrinsic::amdgcn_smfmac_f32_16x16x64_fp8_bf8:
5172 case Intrinsic::amdgcn_smfmac_f32_16x16x64_fp8_fp8:
5173 case Intrinsic::amdgcn_smfmac_f32_32x32x32_bf8_bf8:
5174 case Intrinsic::amdgcn_smfmac_f32_32x32x32_bf8_fp8:
5175 case Intrinsic::amdgcn_smfmac_f32_32x32x32_fp8_bf8:
5176 case Intrinsic::amdgcn_smfmac_f32_32x32x32_fp8_fp8:
5177 case Intrinsic::amdgcn_smfmac_f32_16x16x64_f16:
5178 case Intrinsic::amdgcn_smfmac_f32_32x32x32_f16:
5179 case Intrinsic::amdgcn_smfmac_f32_16x16x64_bf16:
5180 case Intrinsic::amdgcn_smfmac_f32_32x32x32_bf16:
5181 case Intrinsic::amdgcn_smfmac_i32_16x16x128_i8:
5182 case Intrinsic::amdgcn_smfmac_i32_32x32x64_i8:
5183 case Intrinsic::amdgcn_smfmac_f32_16x16x128_bf8_bf8:
5184 case Intrinsic::amdgcn_smfmac_f32_16x16x128_bf8_fp8:
5185 case Intrinsic::amdgcn_smfmac_f32_16x16x128_fp8_bf8:
5186 case Intrinsic::amdgcn_smfmac_f32_16x16x128_fp8_fp8:
5187 case Intrinsic::amdgcn_smfmac_f32_32x32x64_bf8_bf8:
5188 case Intrinsic::amdgcn_smfmac_f32_32x32x64_bf8_fp8:
5189 case Intrinsic::amdgcn_smfmac_f32_32x32x64_fp8_bf8:
5190 case Intrinsic::amdgcn_smfmac_f32_32x32x64_fp8_fp8: {
5191 Register DstReg = MI.getOperand(i: 0).getReg();
5192 unsigned DstSize = MRI.getType(Reg: DstReg).getSizeInBits();
5193 unsigned MinNumRegsRequired = DstSize / 32;
5194 const SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
5195 bool UseAGPRForm = Info->selectAGPRFormMFMA(NumRegs: MinNumRegsRequired);
5196
5197 // vdst, srcA, srcB, srcC, idx
5198 OpdsMapping[0] = UseAGPRForm ? getAGPROpMapping(Reg: DstReg, MRI, TRI: *TRI)
5199 : getVGPROpMapping(Reg: DstReg, MRI, TRI: *TRI);
5200
5201 OpdsMapping[2] = getVGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5202 OpdsMapping[3] = getVGPROpMapping(Reg: MI.getOperand(i: 3).getReg(), MRI, TRI: *TRI);
5203 OpdsMapping[4] =
5204 UseAGPRForm ? getAGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI)
5205 : getVGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI);
5206 OpdsMapping[5] = getVGPROpMapping(Reg: MI.getOperand(i: 5).getReg(), MRI, TRI: *TRI);
5207 break;
5208 }
5209 case Intrinsic::amdgcn_interp_p1:
5210 case Intrinsic::amdgcn_interp_p2:
5211 case Intrinsic::amdgcn_interp_mov:
5212 case Intrinsic::amdgcn_interp_p1_f16:
5213 case Intrinsic::amdgcn_interp_p2_f16:
5214 case Intrinsic::amdgcn_lds_param_load: {
5215 const int M0Idx = MI.getNumOperands() - 1;
5216 Register M0Reg = MI.getOperand(i: M0Idx).getReg();
5217 unsigned M0Bank = getRegBankID(Reg: M0Reg, MRI, Default: AMDGPU::SGPRRegBankID);
5218 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5219
5220 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: DstSize);
5221 for (int I = 2; I != M0Idx && MI.getOperand(i: I).isReg(); ++I)
5222 OpdsMapping[I] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5223
5224 // Must be SGPR, but we must take whatever the original bank is and fix it
5225 // later.
5226 OpdsMapping[M0Idx] = AMDGPU::getValueMapping(BankID: M0Bank, Size: 32);
5227 break;
5228 }
5229 case Intrinsic::amdgcn_interp_inreg_p10:
5230 case Intrinsic::amdgcn_interp_inreg_p2:
5231 case Intrinsic::amdgcn_interp_inreg_p10_f16:
5232 case Intrinsic::amdgcn_interp_inreg_p2_f16:
5233 case Intrinsic::amdgcn_interp_p10_rtz_f16:
5234 case Intrinsic::amdgcn_interp_p2_rtz_f16: {
5235 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5236 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: DstSize);
5237 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5238 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5239 OpdsMapping[4] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5240 break;
5241 }
5242 case Intrinsic::amdgcn_permlane16_swap:
5243 case Intrinsic::amdgcn_permlane32_swap: {
5244 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5245 OpdsMapping[0] = OpdsMapping[1] = OpdsMapping[3] = OpdsMapping[4] =
5246 AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: DstSize);
5247 break;
5248 }
5249 case Intrinsic::amdgcn_ballot: {
5250 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5251 unsigned SrcSize = MRI.getType(Reg: MI.getOperand(i: 2).getReg()).getSizeInBits();
5252 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: DstSize);
5253 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: SrcSize);
5254 break;
5255 }
5256 case Intrinsic::amdgcn_inverse_ballot: {
5257 // This must be an SGPR, but accept a VGPR.
5258 Register MaskReg = MI.getOperand(i: 2).getReg();
5259 unsigned MaskSize = MRI.getType(Reg: MaskReg).getSizeInBits();
5260 unsigned MaskBank = getRegBankID(Reg: MaskReg, MRI, Default: AMDGPU::SGPRRegBankID);
5261 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: 1);
5262 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: MaskBank, Size: MaskSize);
5263 break;
5264 }
5265 case Intrinsic::amdgcn_bitop3: {
5266 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5267 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5268 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5269 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5270 OpdsMapping[4] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5271 break;
5272 }
5273 case Intrinsic::amdgcn_s_quadmask:
5274 case Intrinsic::amdgcn_s_wqm: {
5275 Register MaskReg = MI.getOperand(i: 2).getReg();
5276 unsigned MaskSize = MRI.getType(Reg: MaskReg).getSizeInBits();
5277 unsigned MaskBank = getRegBankID(Reg: MaskReg, MRI, Default: AMDGPU::SGPRRegBankID);
5278 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: MaskSize);
5279 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: MaskBank, Size: MaskSize);
5280 break;
5281 }
5282 case Intrinsic::amdgcn_wave_reduce_add:
5283 case Intrinsic::amdgcn_wave_reduce_fadd:
5284 case Intrinsic::amdgcn_wave_reduce_sub:
5285 case Intrinsic::amdgcn_wave_reduce_fsub:
5286 case Intrinsic::amdgcn_wave_reduce_min:
5287 case Intrinsic::amdgcn_wave_reduce_umin:
5288 case Intrinsic::amdgcn_wave_reduce_fmin:
5289 case Intrinsic::amdgcn_wave_reduce_max:
5290 case Intrinsic::amdgcn_wave_reduce_umax:
5291 case Intrinsic::amdgcn_wave_reduce_fmax:
5292 case Intrinsic::amdgcn_wave_reduce_and:
5293 case Intrinsic::amdgcn_wave_reduce_or:
5294 case Intrinsic::amdgcn_wave_reduce_xor: {
5295 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5296 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: DstSize);
5297 unsigned OpSize = MRI.getType(Reg: MI.getOperand(i: 2).getReg()).getSizeInBits();
5298 auto regBankID =
5299 isSALUMapping(MI) ? AMDGPU::SGPRRegBankID : AMDGPU::VGPRRegBankID;
5300 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: regBankID, Size: OpSize);
5301 break;
5302 }
5303 case Intrinsic::amdgcn_s_bitreplicate: {
5304 Register MaskReg = MI.getOperand(i: 2).getReg();
5305 unsigned MaskBank = getRegBankID(Reg: MaskReg, MRI, Default: AMDGPU::SGPRRegBankID);
5306 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 64);
5307 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: MaskBank, Size: 32);
5308 break;
5309 }
5310 case Intrinsic::amdgcn_wave_shuffle: {
5311 unsigned OpSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5312 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: OpSize);
5313 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: OpSize);
5314 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: OpSize);
5315 break;
5316 }
5317 }
5318 break;
5319 }
5320 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_LOAD:
5321 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_LOAD_D16:
5322 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_LOAD_NORET:
5323 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_STORE:
5324 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_STORE_D16: {
5325 auto IntrID = AMDGPU::getIntrinsicID(I: MI);
5326 const AMDGPU::RsrcIntrinsic *RSrcIntrin = AMDGPU::lookupRsrcIntrinsic(Intr: IntrID);
5327 assert(RSrcIntrin && "missing RsrcIntrinsic for image intrinsic");
5328 // Non-images can have complications from operands that allow both SGPR
5329 // and VGPR. For now it's too complicated to figure out the final opcode
5330 // to derive the register bank from the MCInstrDesc.
5331 assert(RSrcIntrin->IsImage);
5332 return getImageMapping(MRI, MI, RsrcIdx: RSrcIntrin->RsrcArg);
5333 }
5334 case AMDGPU::G_AMDGPU_BVH_INTERSECT_RAY:
5335 case AMDGPU::G_AMDGPU_BVH8_INTERSECT_RAY:
5336 case AMDGPU::G_AMDGPU_BVH_DUAL_INTERSECT_RAY: {
5337 bool IsDualOrBVH8 =
5338 MI.getOpcode() == AMDGPU::G_AMDGPU_BVH_DUAL_INTERSECT_RAY ||
5339 MI.getOpcode() == AMDGPU::G_AMDGPU_BVH8_INTERSECT_RAY;
5340 unsigned NumMods = IsDualOrBVH8 ? 0 : 1; // Has A16 modifier
5341 unsigned LastRegOpIdx = MI.getNumExplicitOperands() - 1 - NumMods;
5342 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5343 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: DstSize);
5344 if (IsDualOrBVH8) {
5345 OpdsMapping[1] = AMDGPU::getValueMapping(
5346 BankID: AMDGPU::VGPRRegBankID,
5347 Size: MRI.getType(Reg: MI.getOperand(i: 1).getReg()).getSizeInBits());
5348 OpdsMapping[2] = AMDGPU::getValueMapping(
5349 BankID: AMDGPU::VGPRRegBankID,
5350 Size: MRI.getType(Reg: MI.getOperand(i: 2).getReg()).getSizeInBits());
5351 }
5352 OpdsMapping[LastRegOpIdx] =
5353 getSGPROpMapping(Reg: MI.getOperand(i: LastRegOpIdx).getReg(), MRI, TRI: *TRI);
5354 if (LastRegOpIdx == 3) {
5355 // Sequential form: all operands combined into VGPR256/VGPR512
5356 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 2).getReg()).getSizeInBits();
5357 if (Size > 256)
5358 Size = 512;
5359 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5360 } else {
5361 // NSA form
5362 unsigned FirstSrcOpIdx = IsDualOrBVH8 ? 4 : 2;
5363 for (unsigned I = FirstSrcOpIdx; I < LastRegOpIdx; ++I) {
5364 unsigned Size = MRI.getType(Reg: MI.getOperand(i: I).getReg()).getSizeInBits();
5365 OpdsMapping[I] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5366 }
5367 }
5368 break;
5369 }
5370 case AMDGPU::G_INTRINSIC_W_SIDE_EFFECTS:
5371 case AMDGPU::G_INTRINSIC_CONVERGENT_W_SIDE_EFFECTS: {
5372 auto IntrID = cast<GIntrinsic>(Val: MI).getIntrinsicID();
5373 switch (IntrID) {
5374 case Intrinsic::amdgcn_s_getreg:
5375 case Intrinsic::amdgcn_s_memtime:
5376 case Intrinsic::amdgcn_s_memrealtime:
5377 case Intrinsic::amdgcn_s_get_waveid_in_workgroup:
5378 case Intrinsic::amdgcn_s_sendmsg_rtn: {
5379 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5380 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size);
5381 break;
5382 }
5383 case Intrinsic::amdgcn_global_atomic_fmin_num:
5384 case Intrinsic::amdgcn_global_atomic_fmax_num:
5385 case Intrinsic::amdgcn_flat_atomic_fmin_num:
5386 case Intrinsic::amdgcn_flat_atomic_fmax_num:
5387 case Intrinsic::amdgcn_global_atomic_ordered_add_b64:
5388 case Intrinsic::amdgcn_global_load_tr_b64:
5389 case Intrinsic::amdgcn_global_load_tr_b128:
5390 case Intrinsic::amdgcn_global_load_tr4_b64:
5391 case Intrinsic::amdgcn_global_load_tr6_b96:
5392 case Intrinsic::amdgcn_ds_load_tr8_b64:
5393 case Intrinsic::amdgcn_ds_load_tr16_b128:
5394 case Intrinsic::amdgcn_ds_load_tr4_b64:
5395 case Intrinsic::amdgcn_ds_load_tr6_b96:
5396 case Intrinsic::amdgcn_ds_read_tr4_b64:
5397 case Intrinsic::amdgcn_ds_read_tr6_b96:
5398 case Intrinsic::amdgcn_ds_read_tr8_b64:
5399 case Intrinsic::amdgcn_ds_read_tr16_b64:
5400 case Intrinsic::amdgcn_ds_atomic_async_barrier_arrive_b64:
5401 case Intrinsic::amdgcn_ds_atomic_barrier_arrive_rtn_b64:
5402 return getDefaultMappingAllVGPR(MI);
5403 case Intrinsic::amdgcn_ds_ordered_add:
5404 case Intrinsic::amdgcn_ds_ordered_swap: {
5405 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5406 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: DstSize);
5407 unsigned M0Bank = getRegBankID(Reg: MI.getOperand(i: 2).getReg(), MRI,
5408 Default: AMDGPU::SGPRRegBankID);
5409 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: M0Bank, Size: 32);
5410 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5411 break;
5412 }
5413 case Intrinsic::amdgcn_ds_append:
5414 case Intrinsic::amdgcn_ds_consume: {
5415 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5416 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: DstSize);
5417 OpdsMapping[2] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5418 break;
5419 }
5420 case Intrinsic::amdgcn_exp_compr:
5421 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5422 OpdsMapping[4] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5423 break;
5424 case Intrinsic::amdgcn_exp:
5425 // FIXME: Could we support packed types here?
5426 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5427 OpdsMapping[4] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5428 OpdsMapping[5] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5429 OpdsMapping[6] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5430 break;
5431 case Intrinsic::amdgcn_exp_row:
5432 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5433 OpdsMapping[4] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5434 OpdsMapping[5] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5435 OpdsMapping[6] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5436 OpdsMapping[8] = getSGPROpMapping(Reg: MI.getOperand(i: 8).getReg(), MRI, TRI: *TRI);
5437 break;
5438 case Intrinsic::amdgcn_s_alloc_vgpr:
5439 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 1);
5440 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 32);
5441 break;
5442 case Intrinsic::amdgcn_s_sendmsg:
5443 case Intrinsic::amdgcn_s_sendmsghalt: {
5444 // This must be an SGPR, but accept a VGPR.
5445 unsigned Bank = getRegBankID(Reg: MI.getOperand(i: 2).getReg(), MRI,
5446 Default: AMDGPU::SGPRRegBankID);
5447 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: Bank, Size: 32);
5448 break;
5449 }
5450 case Intrinsic::amdgcn_s_setreg: {
5451 // This must be an SGPR, but accept a VGPR.
5452 unsigned Bank = getRegBankID(Reg: MI.getOperand(i: 2).getReg(), MRI,
5453 Default: AMDGPU::SGPRRegBankID);
5454 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: Bank, Size: 32);
5455 break;
5456 }
5457 case Intrinsic::amdgcn_s_ttracedata: {
5458 // This must be an SGPR, but accept a VGPR.
5459 unsigned Bank =
5460 getRegBankID(Reg: MI.getOperand(i: 1).getReg(), MRI, Default: AMDGPU::SGPRRegBankID);
5461 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: Bank, Size: 32);
5462 break;
5463 }
5464 case Intrinsic::amdgcn_end_cf: {
5465 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5466 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size);
5467 break;
5468 }
5469 case Intrinsic::amdgcn_else: {
5470 unsigned WaveSize = getSizeInBits(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5471 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: 1);
5472 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: WaveSize);
5473 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: WaveSize);
5474 break;
5475 }
5476 case Intrinsic::amdgcn_init_whole_wave:
5477 case Intrinsic::amdgcn_live_mask: {
5478 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: 1);
5479 break;
5480 }
5481 case Intrinsic::amdgcn_wqm_demote:
5482 case Intrinsic::amdgcn_kill: {
5483 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: 1);
5484 break;
5485 }
5486 case Intrinsic::amdgcn_ptr_s_buffer_load: {
5487 OpdsMapping[0] = getSGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5488 OpdsMapping[2] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5489 OpdsMapping[3] = getSGPROpMapping(Reg: MI.getOperand(i: 3).getReg(), MRI, TRI: *TRI);
5490 break;
5491 }
5492 case Intrinsic::amdgcn_raw_buffer_load:
5493 case Intrinsic::amdgcn_raw_ptr_buffer_load:
5494 case Intrinsic::amdgcn_raw_atomic_buffer_load:
5495 case Intrinsic::amdgcn_raw_ptr_atomic_buffer_load:
5496 case Intrinsic::amdgcn_raw_tbuffer_load:
5497 case Intrinsic::amdgcn_raw_ptr_tbuffer_load: {
5498 // FIXME: Should make intrinsic ID the last operand of the instruction,
5499 // then this would be the same as store
5500 OpdsMapping[0] = getVGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5501 OpdsMapping[2] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5502 OpdsMapping[3] = getVGPROpMapping(Reg: MI.getOperand(i: 3).getReg(), MRI, TRI: *TRI);
5503 OpdsMapping[4] = getSGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI);
5504 break;
5505 }
5506 case Intrinsic::amdgcn_raw_buffer_load_lds:
5507 case Intrinsic::amdgcn_raw_buffer_load_async_lds:
5508 case Intrinsic::amdgcn_raw_ptr_buffer_load_lds:
5509 case Intrinsic::amdgcn_raw_ptr_buffer_load_async_lds: {
5510 OpdsMapping[1] = getSGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5511 OpdsMapping[2] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5512 OpdsMapping[4] = getVGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI);
5513 OpdsMapping[5] = getSGPROpMapping(Reg: MI.getOperand(i: 5).getReg(), MRI, TRI: *TRI);
5514 break;
5515 }
5516 case Intrinsic::amdgcn_raw_buffer_store:
5517 case Intrinsic::amdgcn_raw_ptr_buffer_store:
5518 case Intrinsic::amdgcn_raw_buffer_store_format:
5519 case Intrinsic::amdgcn_raw_ptr_buffer_store_format:
5520 case Intrinsic::amdgcn_raw_tbuffer_store:
5521 case Intrinsic::amdgcn_raw_ptr_tbuffer_store: {
5522 OpdsMapping[1] = getVGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5523 OpdsMapping[2] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5524 OpdsMapping[3] = getVGPROpMapping(Reg: MI.getOperand(i: 3).getReg(), MRI, TRI: *TRI);
5525 OpdsMapping[4] = getSGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI);
5526 break;
5527 }
5528 case Intrinsic::amdgcn_struct_buffer_load:
5529 case Intrinsic::amdgcn_struct_ptr_buffer_load:
5530 case Intrinsic::amdgcn_struct_tbuffer_load:
5531 case Intrinsic::amdgcn_struct_ptr_tbuffer_load:
5532 case Intrinsic::amdgcn_struct_atomic_buffer_load:
5533 case Intrinsic::amdgcn_struct_ptr_atomic_buffer_load: {
5534 OpdsMapping[0] = getVGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5535 OpdsMapping[2] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5536 OpdsMapping[3] = getVGPROpMapping(Reg: MI.getOperand(i: 3).getReg(), MRI, TRI: *TRI);
5537 OpdsMapping[4] = getVGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI);
5538 OpdsMapping[5] = getSGPROpMapping(Reg: MI.getOperand(i: 5).getReg(), MRI, TRI: *TRI);
5539 break;
5540 }
5541 case Intrinsic::amdgcn_struct_buffer_load_lds:
5542 case Intrinsic::amdgcn_struct_buffer_load_async_lds:
5543 case Intrinsic::amdgcn_struct_ptr_buffer_load_lds:
5544 case Intrinsic::amdgcn_struct_ptr_buffer_load_async_lds: {
5545 OpdsMapping[1] = getSGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5546 OpdsMapping[2] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5547 OpdsMapping[4] = getVGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI);
5548 OpdsMapping[5] = getVGPROpMapping(Reg: MI.getOperand(i: 5).getReg(), MRI, TRI: *TRI);
5549 OpdsMapping[6] = getSGPROpMapping(Reg: MI.getOperand(i: 6).getReg(), MRI, TRI: *TRI);
5550 break;
5551 }
5552 case Intrinsic::amdgcn_struct_buffer_store:
5553 case Intrinsic::amdgcn_struct_ptr_buffer_store:
5554 case Intrinsic::amdgcn_struct_tbuffer_store:
5555 case Intrinsic::amdgcn_struct_ptr_tbuffer_store: {
5556 OpdsMapping[1] = getVGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5557 OpdsMapping[2] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5558 OpdsMapping[3] = getVGPROpMapping(Reg: MI.getOperand(i: 3).getReg(), MRI, TRI: *TRI);
5559 OpdsMapping[4] = getVGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI);
5560 OpdsMapping[5] = getSGPROpMapping(Reg: MI.getOperand(i: 5).getReg(), MRI, TRI: *TRI);
5561 break;
5562 }
5563 case Intrinsic::amdgcn_init_exec_from_input: {
5564 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5565 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size);
5566 break;
5567 }
5568 case Intrinsic::amdgcn_ds_gws_init:
5569 case Intrinsic::amdgcn_ds_gws_barrier:
5570 case Intrinsic::amdgcn_ds_gws_sema_br: {
5571 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5572
5573 // This must be an SGPR, but accept a VGPR.
5574 unsigned Bank = getRegBankID(Reg: MI.getOperand(i: 2).getReg(), MRI,
5575 Default: AMDGPU::SGPRRegBankID);
5576 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: Bank, Size: 32);
5577 break;
5578 }
5579 case Intrinsic::amdgcn_ds_gws_sema_v:
5580 case Intrinsic::amdgcn_ds_gws_sema_p:
5581 case Intrinsic::amdgcn_ds_gws_sema_release_all: {
5582 // This must be an SGPR, but accept a VGPR.
5583 unsigned Bank = getRegBankID(Reg: MI.getOperand(i: 1).getReg(), MRI,
5584 Default: AMDGPU::SGPRRegBankID);
5585 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: Bank, Size: 32);
5586 break;
5587 }
5588 case Intrinsic::amdgcn_cluster_load_b32:
5589 case Intrinsic::amdgcn_cluster_load_b64:
5590 case Intrinsic::amdgcn_cluster_load_b128: {
5591 OpdsMapping[0] = getVGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5592 OpdsMapping[2] = getVGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5593 unsigned M0Bank =
5594 getRegBankID(Reg: MI.getOperand(i: 4).getReg(), MRI, Default: AMDGPU::SGPRRegBankID);
5595 OpdsMapping[4] = AMDGPU::getValueMapping(BankID: M0Bank, Size: 32);
5596 break;
5597 }
5598 case Intrinsic::amdgcn_cluster_load_async_to_lds_b8:
5599 case Intrinsic::amdgcn_cluster_load_async_to_lds_b32:
5600 case Intrinsic::amdgcn_cluster_load_async_to_lds_b64:
5601 case Intrinsic::amdgcn_cluster_load_async_to_lds_b128: {
5602 OpdsMapping[1] = getVGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5603 // LDS address goes into $vdst (VGPR).
5604 OpdsMapping[2] = getVGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5605 unsigned M0Bank =
5606 getRegBankID(Reg: MI.getOperand(i: 5).getReg(), MRI, Default: AMDGPU::SGPRRegBankID);
5607 OpdsMapping[5] = AMDGPU::getValueMapping(BankID: M0Bank, Size: 32);
5608 break;
5609 }
5610 case Intrinsic::amdgcn_global_store_async_from_lds_b8:
5611 case Intrinsic::amdgcn_global_store_async_from_lds_b32:
5612 case Intrinsic::amdgcn_global_store_async_from_lds_b64:
5613 case Intrinsic::amdgcn_global_store_async_from_lds_b128:
5614 case Intrinsic::amdgcn_global_load_async_to_lds_b8:
5615 case Intrinsic::amdgcn_global_load_async_to_lds_b32:
5616 case Intrinsic::amdgcn_global_load_async_to_lds_b64:
5617 case Intrinsic::amdgcn_global_load_async_to_lds_b128: {
5618 OpdsMapping[1] = getVGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5619 // LDS address goes into $vdst/$vdata (VGPR).
5620 OpdsMapping[2] = getVGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5621 break;
5622 }
5623 case Intrinsic::amdgcn_load_to_lds:
5624 case Intrinsic::amdgcn_load_async_to_lds:
5625 case Intrinsic::amdgcn_global_load_lds:
5626 case Intrinsic::amdgcn_global_load_async_lds: {
5627 OpdsMapping[1] = getVGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5628 // LDS address goes into M0 (SGPR).
5629 OpdsMapping[2] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5630 break;
5631 }
5632 case Intrinsic::amdgcn_lds_direct_load: {
5633 const int M0Idx = MI.getNumOperands() - 1;
5634 Register M0Reg = MI.getOperand(i: M0Idx).getReg();
5635 unsigned M0Bank = getRegBankID(Reg: M0Reg, MRI, Default: AMDGPU::SGPRRegBankID);
5636 unsigned DstSize = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5637
5638 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: DstSize);
5639 for (int I = 2; I != M0Idx && MI.getOperand(i: I).isReg(); ++I)
5640 OpdsMapping[I] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: 32);
5641
5642 // Must be SGPR, but we must take whatever the original bank is and fix it
5643 // later.
5644 OpdsMapping[M0Idx] = AMDGPU::getValueMapping(BankID: M0Bank, Size: 32);
5645 break;
5646 }
5647 case Intrinsic::amdgcn_ds_add_gs_reg_rtn:
5648 case Intrinsic::amdgcn_ds_sub_gs_reg_rtn:
5649 OpdsMapping[0] = getVGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5650 OpdsMapping[2] = getVGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5651 break;
5652 case Intrinsic::amdgcn_ds_bvh_stack_rtn:
5653 case Intrinsic::amdgcn_ds_bvh_stack_push4_pop1_rtn:
5654 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop1_rtn:
5655 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop2_rtn: {
5656 OpdsMapping[0] =
5657 getVGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI); // %vdst
5658 OpdsMapping[1] =
5659 getVGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI); // %addr
5660 OpdsMapping[3] =
5661 getVGPROpMapping(Reg: MI.getOperand(i: 3).getReg(), MRI, TRI: *TRI); // %addr
5662 OpdsMapping[4] =
5663 getVGPROpMapping(Reg: MI.getOperand(i: 4).getReg(), MRI, TRI: *TRI); // %data0
5664 OpdsMapping[5] =
5665 getVGPROpMapping(Reg: MI.getOperand(i: 5).getReg(), MRI, TRI: *TRI); // %data1
5666 break;
5667 }
5668 case Intrinsic::amdgcn_s_sleep_var:
5669 OpdsMapping[1] = getSGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5670 break;
5671 case Intrinsic::amdgcn_s_barrier_join:
5672 case Intrinsic::amdgcn_s_wakeup_barrier:
5673 OpdsMapping[1] = getSGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5674 break;
5675 case Intrinsic::amdgcn_s_barrier_init:
5676 case Intrinsic::amdgcn_s_barrier_signal_var:
5677 OpdsMapping[1] = getSGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5678 OpdsMapping[2] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5679 break;
5680 case Intrinsic::amdgcn_s_barrier_signal_isfirst: {
5681 const unsigned ResultSize = 1;
5682 OpdsMapping[0] =
5683 AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: ResultSize);
5684 break;
5685 }
5686 case Intrinsic::amdgcn_s_get_barrier_state:
5687 case Intrinsic::amdgcn_s_get_named_barrier_state: {
5688 OpdsMapping[0] = getSGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5689 OpdsMapping[2] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5690 break;
5691 }
5692 case Intrinsic::amdgcn_pops_exiting_wave_id:
5693 return getDefaultMappingSOP(MI);
5694 case Intrinsic::amdgcn_tensor_load_to_lds:
5695 case Intrinsic::amdgcn_tensor_store_from_lds: {
5696 // Lie and claim everything is legal, even all operands need to be
5697 // SGPRs. applyMapping will have to deal with it with readfirstlane.
5698 for (unsigned I = 1; I < MI.getNumOperands(); ++I) {
5699 if (MI.getOperand(i: I).isReg()) {
5700 Register Reg = MI.getOperand(i: I).getReg();
5701 auto OpBank = getRegBankID(Reg, MRI);
5702 unsigned Size = getSizeInBits(Reg, MRI, TRI: *TRI);
5703 OpdsMapping[I] = AMDGPU::getValueMapping(BankID: OpBank, Size);
5704 }
5705 }
5706 break;
5707 }
5708 case Intrinsic::amdgcn_s_prefetch_data:
5709 case Intrinsic::amdgcn_s_prefetch_inst: {
5710 OpdsMapping[1] = getSGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5711 OpdsMapping[2] = getSGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5712 break;
5713 }
5714 case Intrinsic::amdgcn_flat_prefetch:
5715 case Intrinsic::amdgcn_global_prefetch:
5716 return getDefaultMappingVOP(MI);
5717 default:
5718 return getInvalidInstructionMapping();
5719 }
5720 break;
5721 }
5722 case AMDGPU::G_SELECT: {
5723 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits();
5724 unsigned Op2Bank = getRegBankID(Reg: MI.getOperand(i: 2).getReg(), MRI,
5725 Default: AMDGPU::SGPRRegBankID);
5726 unsigned Op3Bank = getRegBankID(Reg: MI.getOperand(i: 3).getReg(), MRI,
5727 Default: AMDGPU::SGPRRegBankID);
5728 bool SGPRSrcs = Op2Bank == AMDGPU::SGPRRegBankID &&
5729 Op3Bank == AMDGPU::SGPRRegBankID;
5730
5731 unsigned CondBankDefault = SGPRSrcs ?
5732 AMDGPU::SGPRRegBankID : AMDGPU::VCCRegBankID;
5733 unsigned CondBank = getRegBankID(Reg: MI.getOperand(i: 1).getReg(), MRI,
5734 Default: CondBankDefault);
5735 if (CondBank == AMDGPU::SGPRRegBankID)
5736 CondBank = SGPRSrcs ? AMDGPU::SGPRRegBankID : AMDGPU::VCCRegBankID;
5737 else if (CondBank == AMDGPU::VGPRRegBankID)
5738 CondBank = AMDGPU::VCCRegBankID;
5739
5740 unsigned Bank = SGPRSrcs && CondBank == AMDGPU::SGPRRegBankID ?
5741 AMDGPU::SGPRRegBankID : AMDGPU::VGPRRegBankID;
5742
5743 assert(CondBank == AMDGPU::VCCRegBankID || CondBank == AMDGPU::SGPRRegBankID);
5744
5745 // TODO: Should report 32-bit for scalar condition type.
5746 if (Size == 64) {
5747 OpdsMapping[0] = AMDGPU::getValueMappingSGPR64Only(BankID: Bank, Size);
5748 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: CondBank, Size: 1);
5749 OpdsMapping[2] = AMDGPU::getValueMappingSGPR64Only(BankID: Bank, Size);
5750 OpdsMapping[3] = AMDGPU::getValueMappingSGPR64Only(BankID: Bank, Size);
5751 } else {
5752 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: Bank, Size);
5753 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: CondBank, Size: 1);
5754 OpdsMapping[2] = AMDGPU::getValueMapping(BankID: Bank, Size);
5755 OpdsMapping[3] = AMDGPU::getValueMapping(BankID: Bank, Size);
5756 }
5757
5758 break;
5759 }
5760
5761 case AMDGPU::G_SI_CALL: {
5762 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::SGPRRegBankID, Size: 64);
5763 // Lie and claim everything is legal, even though some need to be
5764 // SGPRs. applyMapping will have to deal with it as a waterfall loop.
5765 OpdsMapping[1] = getSGPROpMapping(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5766
5767 // Allow anything for implicit arguments
5768 for (unsigned I = 4; I < MI.getNumOperands(); ++I) {
5769 if (MI.getOperand(i: I).isReg()) {
5770 Register Reg = MI.getOperand(i: I).getReg();
5771 auto OpBank = getRegBankID(Reg, MRI);
5772 unsigned Size = getSizeInBits(Reg, MRI, TRI: *TRI);
5773 OpdsMapping[I] = AMDGPU::getValueMapping(BankID: OpBank, Size);
5774 }
5775 }
5776 break;
5777 }
5778 case AMDGPU::G_LOAD:
5779 case AMDGPU::G_ZEXTLOAD:
5780 case AMDGPU::G_SEXTLOAD:
5781 return getInstrMappingForLoad(MI);
5782
5783 case AMDGPU::G_ATOMICRMW_XCHG:
5784 case AMDGPU::G_ATOMICRMW_ADD:
5785 case AMDGPU::G_ATOMICRMW_SUB:
5786 case AMDGPU::G_ATOMICRMW_AND:
5787 case AMDGPU::G_ATOMICRMW_OR:
5788 case AMDGPU::G_ATOMICRMW_XOR:
5789 case AMDGPU::G_ATOMICRMW_MAX:
5790 case AMDGPU::G_ATOMICRMW_MIN:
5791 case AMDGPU::G_ATOMICRMW_UMAX:
5792 case AMDGPU::G_ATOMICRMW_UMIN:
5793 case AMDGPU::G_ATOMICRMW_FADD:
5794 case AMDGPU::G_ATOMICRMW_FMIN:
5795 case AMDGPU::G_ATOMICRMW_FMAX:
5796 case AMDGPU::G_ATOMICRMW_UINC_WRAP:
5797 case AMDGPU::G_ATOMICRMW_UDEC_WRAP:
5798 case AMDGPU::G_ATOMICRMW_USUB_COND:
5799 case AMDGPU::G_ATOMICRMW_USUB_SAT:
5800 case AMDGPU::G_AMDGPU_ATOMIC_CMPXCHG: {
5801 OpdsMapping[0] = getVGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5802 OpdsMapping[1] = getValueMappingForPtr(MRI, PtrReg: MI.getOperand(i: 1).getReg());
5803 OpdsMapping[2] = getVGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5804 break;
5805 }
5806 case AMDGPU::G_ATOMIC_CMPXCHG: {
5807 OpdsMapping[0] = getVGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5808 OpdsMapping[1] = getValueMappingForPtr(MRI, PtrReg: MI.getOperand(i: 1).getReg());
5809 OpdsMapping[2] = getVGPROpMapping(Reg: MI.getOperand(i: 2).getReg(), MRI, TRI: *TRI);
5810 OpdsMapping[3] = getVGPROpMapping(Reg: MI.getOperand(i: 3).getReg(), MRI, TRI: *TRI);
5811 break;
5812 }
5813 case AMDGPU::G_BRCOND: {
5814 unsigned Bank = getRegBankID(Reg: MI.getOperand(i: 0).getReg(), MRI,
5815 Default: AMDGPU::SGPRRegBankID);
5816 assert(MRI.getType(MI.getOperand(0).getReg()).getSizeInBits() == 1);
5817 if (Bank != AMDGPU::SGPRRegBankID)
5818 Bank = AMDGPU::VCCRegBankID;
5819
5820 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: Bank, Size: 1);
5821 break;
5822 }
5823 case AMDGPU::G_INTRINSIC_FPTRUNC_ROUND:
5824 return getDefaultMappingVOP(MI);
5825 case AMDGPU::G_PREFETCH:
5826 OpdsMapping[0] = getSGPROpMapping(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5827 break;
5828 case AMDGPU::G_AMDGPU_WHOLE_WAVE_FUNC_SETUP:
5829 case AMDGPU::G_AMDGPU_WHOLE_WAVE_FUNC_RETURN:
5830 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VCCRegBankID, Size: 1);
5831 break;
5832 case AMDGPU::G_AMDGPU_FLAT_LOAD_MONITOR:
5833 case AMDGPU::G_AMDGPU_GLOBAL_LOAD_MONITOR: {
5834 unsigned Size = getSizeInBits(Reg: MI.getOperand(i: 0).getReg(), MRI, TRI: *TRI);
5835 unsigned PtrSize = getSizeInBits(Reg: MI.getOperand(i: 1).getReg(), MRI, TRI: *TRI);
5836 OpdsMapping[0] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size);
5837 OpdsMapping[1] = AMDGPU::getValueMapping(BankID: AMDGPU::VGPRRegBankID, Size: PtrSize);
5838 break;
5839 }
5840 }
5841
5842 return getInstructionMapping(/*ID*/1, /*Cost*/1,
5843 OperandsMapping: getOperandsMapping(OpdsMapping),
5844 NumOperands: MI.getNumOperands());
5845}
5846