1//===- SIInstrInfo.h - SI Instruction Info Interface ------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// Interface definition for SIInstrInfo.
11//
12//===----------------------------------------------------------------------===//
13
14#ifndef LLVM_LIB_TARGET_AMDGPU_SIINSTRINFO_H
15#define LLVM_LIB_TARGET_AMDGPU_SIINSTRINFO_H
16
17#include "AMDGPUMIRFormatter.h"
18#include "MCTargetDesc/AMDGPUMCTargetDesc.h"
19#include "SIRegisterInfo.h"
20#include "Utils/AMDGPUBaseInfo.h"
21#include "llvm/ADT/SetVector.h"
22#include "llvm/ADT/SmallPtrSet.h"
23#include "llvm/CodeGen/TargetInstrInfo.h"
24#include "llvm/CodeGen/TargetSchedule.h"
25
26#define GET_INSTRINFO_HEADER
27#include "AMDGPUGenInstrInfo.inc"
28
29namespace llvm {
30
31class APInt;
32class GCNSubtarget;
33class MachineDominatorTree;
34class MachineRegisterInfo;
35class RegScavenger;
36class SIMachineFunctionInfo;
37class MCRegisterClass;
38using TargetRegisterClass = MCRegisterClass;
39class ScheduleHazardRecognizer;
40
41constexpr unsigned DefaultMemoryClusterDWordsLimit = 8;
42
43/// Mark the MMO of a uniform load if there are no potentially clobbering stores
44/// on any path from the start of an entry function to this load.
45static const MachineMemOperand::Flags MONoClobber =
46 MachineMemOperand::MOTargetFlag1;
47
48/// Mark the MMO of a load as the last use.
49static const MachineMemOperand::Flags MOLastUse =
50 MachineMemOperand::MOTargetFlag2;
51
52/// Mark the MMO of cooperative load/store atomics.
53static const MachineMemOperand::Flags MOCooperative =
54 MachineMemOperand::MOTargetFlag3;
55
56struct V2PhysSCopyInfo {
57 // Operands that need to replaced by waterfall
58 SmallVector<MachineOperand *> MOs;
59 // Target physical registers replacing the MOs
60 SmallVector<Register> SGPRs;
61};
62/// Mark the MMO of accesses to memory locations that are
63/// never written to by other threads.
64static const MachineMemOperand::Flags MOThreadPrivate =
65 MachineMemOperand::MOTargetFlag4;
66
67/// Utility to store machine instructions worklist.
68struct SIInstrWorklist {
69 SIInstrWorklist() = default;
70
71 void insert(MachineInstr *MI);
72
73 MachineInstr *top() const { return InstrList[Front]; }
74
75 void erase_top() {
76 InSet.erase(Ptr: InstrList[Front]);
77 ++Front;
78 }
79
80 bool empty() const { return Front == InstrList.size(); }
81
82 void clear() {
83 InstrList.clear();
84 Front = 0;
85 InSet.clear();
86 DeferredList.clear();
87 }
88
89 bool isDeferred(MachineInstr *MI);
90
91 SetVector<MachineInstr *> &getDeferredList() { return DeferredList; }
92
93private:
94 /// InstrList contains the MachineInstrs.
95 SmallVector<MachineInstr *> InstrList;
96 SmallPtrSet<MachineInstr *, 8> InSet;
97 unsigned Front = 0;
98 /// Deferred instructions are specific MachineInstr
99 /// that will be added by insert method.
100 SetVector<MachineInstr *> DeferredList;
101};
102
103// In namespace llvm so ADL finds it when SIInstrFlags predicates are
104// instantiated with MachineInstr (MachineInstr is in namespace llvm).
105inline uint64_t getTSFlags(const MachineInstr &MI) {
106 return MI.getDesc().TSFlags;
107}
108
109class SIInstrInfo final : public AMDGPUGenInstrInfo {
110 struct ThreeAddressUpdates;
111
112private:
113 const SIRegisterInfo RI;
114 const GCNSubtarget &ST;
115 TargetSchedModel SchedModel;
116 mutable std::unique_ptr<AMDGPUMIRFormatter> Formatter;
117
118 // The inverse predicate should have the negative value.
119 enum BranchPredicate {
120 INVALID_BR = 0,
121 SCC_TRUE = 1,
122 SCC_FALSE = -1,
123 VCCNZ = 2,
124 VCCZ = -2,
125 EXECNZ = -3,
126 EXECZ = 3
127 };
128
129 using SetVectorType = SmallSetVector<MachineInstr *, 32>;
130
131 static unsigned getBranchOpcode(BranchPredicate Cond);
132 static BranchPredicate getBranchPredicate(unsigned Opcode);
133
134public:
135 unsigned buildExtractSubReg(MachineBasicBlock::iterator MI,
136 MachineRegisterInfo &MRI,
137 const MachineOperand &SuperReg,
138 const TargetRegisterClass *SuperRC,
139 unsigned SubIdx,
140 const TargetRegisterClass *SubRC) const;
141 MachineOperand buildExtractSubRegOrImm(
142 MachineBasicBlock::iterator MI, MachineRegisterInfo &MRI,
143 const MachineOperand &SuperReg, const TargetRegisterClass *SuperRC,
144 unsigned SubIdx, const TargetRegisterClass *SubRC) const;
145
146private:
147 bool optimizeSCC(MachineInstr *SCCValid, MachineInstr *SCCRedefine,
148 bool NeedInversion) const;
149
150 bool invertSCCUse(MachineInstr *SCCDef) const;
151
152 void swapOperands(MachineInstr &Inst) const;
153
154 std::pair<bool, MachineBasicBlock *>
155 moveScalarAddSub(SIInstrWorklist &Worklist, MachineInstr &Inst,
156 MachineDominatorTree *MDT = nullptr) const;
157
158 void lowerSelect(SIInstrWorklist &Worklist, MachineInstr &Inst,
159 MachineDominatorTree *MDT = nullptr) const;
160
161 void lowerScalarAbs(SIInstrWorklist &Worklist, MachineInstr &Inst) const;
162
163 void lowerScalarAbsDiff(SIInstrWorklist &Worklist, MachineInstr &Inst) const;
164
165 void lowerScalarXnor(SIInstrWorklist &Worklist, MachineInstr &Inst) const;
166
167 void splitScalarNotBinop(SIInstrWorklist &Worklist, MachineInstr &Inst,
168 unsigned Opcode) const;
169
170 void splitScalarBinOpN2(SIInstrWorklist &Worklist, MachineInstr &Inst,
171 unsigned Opcode) const;
172
173 void splitScalar64BitUnaryOp(SIInstrWorklist &Worklist, MachineInstr &Inst,
174 unsigned Opcode, bool Swap = false) const;
175
176 void splitScalar64BitBinaryOp(SIInstrWorklist &Worklist, MachineInstr &Inst,
177 unsigned Opcode,
178 MachineDominatorTree *MDT = nullptr) const;
179
180 void splitScalarSMulU64(SIInstrWorklist &Worklist, MachineInstr &Inst,
181 MachineDominatorTree *MDT) const;
182
183 void splitScalarSMulPseudo(SIInstrWorklist &Worklist, MachineInstr &Inst,
184 MachineDominatorTree *MDT) const;
185
186 void splitScalar64BitXnor(SIInstrWorklist &Worklist, MachineInstr &Inst,
187 MachineDominatorTree *MDT = nullptr) const;
188
189 void splitScalar64BitBCNT(SIInstrWorklist &Worklist,
190 MachineInstr &Inst) const;
191 void splitScalar64BitBFE(SIInstrWorklist &Worklist, MachineInstr &Inst) const;
192 void splitScalar64BitCountOp(SIInstrWorklist &Worklist, MachineInstr &Inst,
193 unsigned Opcode,
194 MachineDominatorTree *MDT = nullptr) const;
195 void movePackToVALU(SIInstrWorklist &Worklist, MachineRegisterInfo &MRI,
196 MachineInstr &Inst) const;
197
198 void addUsersToMoveToVALUWorklist(Register Reg, MachineRegisterInfo &MRI,
199 SIInstrWorklist &Worklist) const;
200
201 void addSCCDefUsersToVALUWorklist(const MachineOperand &Op,
202 MachineInstr &SCCDefInst,
203 SIInstrWorklist &Worklist,
204 Register NewCond = Register()) const;
205 void addSCCDefsToVALUWorklist(MachineInstr *SCCUseInst,
206 SIInstrWorklist &Worklist) const;
207
208 const TargetRegisterClass *
209 getDestEquivalentVGPRClass(const MachineInstr &Inst) const;
210
211 bool checkInstOffsetsDoNotOverlap(const MachineInstr &MIa,
212 const MachineInstr &MIb) const;
213
214 Register findUsedSGPR(const MachineInstr &MI, int OpIndices[3]) const;
215
216 bool verifyCopy(const MachineInstr &MI, const MachineRegisterInfo &MRI,
217 StringRef &ErrInfo) const;
218
219 bool resultDependsOnExec(const MachineInstr &MI) const;
220
221 MachineInstr *convertToThreeAddressImpl(MachineInstr &MI,
222 ThreeAddressUpdates &Updates) const;
223
224protected:
225 /// If the specific machine instruction is a instruction that moves/copies
226 /// value from one register to another register return destination and source
227 /// registers as machine operands.
228 std::optional<DestSourcePair>
229 isCopyInstrImpl(const MachineInstr &MI) const override;
230
231 bool swapSourceModifiers(MachineInstr &MI, MachineOperand &Src0,
232 AMDGPU::OpName Src0OpName, MachineOperand &Src1,
233 AMDGPU::OpName Src1OpName) const;
234 bool isLegalToSwap(const MachineInstr &MI, unsigned fromIdx,
235 unsigned toIdx) const;
236 bool isNonCommutableDPP(const MachineInstr &MI) const;
237 MachineInstr *commuteInstructionImpl(MachineInstr &MI, bool NewMI,
238 unsigned OpIdx0,
239 unsigned OpIdx1) const override;
240
241public:
242 enum TargetOperandFlags {
243 MO_MASK = 0xf,
244
245 MO_NONE = 0,
246 // MO_GOTPCREL -> symbol@GOTPCREL -> R_AMDGPU_GOTPCREL.
247 MO_GOTPCREL = 1,
248 // MO_GOTPCREL32_LO -> symbol@gotpcrel32@lo -> R_AMDGPU_GOTPCREL32_LO.
249 MO_GOTPCREL32 = 2,
250 MO_GOTPCREL32_LO = 2,
251 // MO_GOTPCREL32_HI -> symbol@gotpcrel32@hi -> R_AMDGPU_GOTPCREL32_HI.
252 MO_GOTPCREL32_HI = 3,
253 // MO_GOTPCREL64 -> symbol@GOTPCREL -> R_AMDGPU_GOTPCREL.
254 MO_GOTPCREL64 = 4,
255 // MO_REL32_LO -> symbol@rel32@lo -> R_AMDGPU_REL32_LO.
256 MO_REL32 = 5,
257 MO_REL32_LO = 5,
258 // MO_REL32_HI -> symbol@rel32@hi -> R_AMDGPU_REL32_HI.
259 MO_REL32_HI = 6,
260 MO_REL64 = 7,
261
262 MO_FAR_BRANCH_OFFSET = 8,
263
264 MO_ABS32_LO = 9,
265 MO_ABS32_HI = 10,
266 MO_ABS64 = 11,
267 };
268
269 explicit SIInstrInfo(const GCNSubtarget &ST);
270
271 const SIRegisterInfo &getRegisterInfo() const {
272 return RI;
273 }
274
275 // FIXME: This is inaccurate and needs to account for use context. Normal asm
276 // constraints should use 64-bit pointers.
277 const TargetRegisterClass *getInlineAsmMemoryOperandRegClass(
278 InlineAsm::ConstraintCode C) const override {
279 return &AMDGPU::VGPR_32RegClass;
280 }
281
282 const GCNSubtarget &getSubtarget() const {
283 return ST;
284 }
285
286 bool isReMaterializableImpl(const MachineInstr &MI) const override;
287
288 bool isIgnorableUse(const MachineInstr &MI, unsigned OpIdx) const override;
289
290 bool isSafeToSink(MachineInstr &MI, MachineBasicBlock *SuccToSinkTo,
291 MachineCycleInfo *CI) const override;
292
293 bool areLoadsFromSameBasePtr(SDNode *Load0, SDNode *Load1, int64_t &Offset0,
294 int64_t &Offset1) const override;
295
296 bool isGlobalMemoryObject(const MachineInstr *MI) const override;
297
298 bool getMemOperandsWithOffsetWidth(
299 const MachineInstr &LdSt,
300 SmallVectorImpl<const MachineOperand *> &BaseOps, int64_t &Offset,
301 bool &OffsetIsScalable, LocationSize &Width) const final;
302
303 bool shouldClusterMemOps(ArrayRef<const MachineOperand *> BaseOps1,
304 int64_t Offset1, bool OffsetIsScalable1,
305 ArrayRef<const MachineOperand *> BaseOps2,
306 int64_t Offset2, bool OffsetIsScalable2,
307 unsigned ClusterSize,
308 unsigned NumBytes) const override;
309
310 bool shouldScheduleLoadsNear(SDNode *Load0, SDNode *Load1, int64_t Offset0,
311 int64_t Offset1, unsigned NumLoads) const override;
312
313 void copyPhysReg(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI,
314 const DebugLoc &DL, Register DestReg, Register SrcReg,
315 bool KillSrc, bool RenamableDest = false,
316 bool RenamableSrc = false) const override;
317
318private:
319 void storeRegToStackSlotImpl(MachineBasicBlock &MBB,
320 MachineBasicBlock::iterator MI, Register SrcReg,
321 bool isKill, int FrameIndex,
322 const TargetRegisterClass *RC, Register VReg,
323 MachineInstr::MIFlag Flags, bool NeedsCFI) const;
324
325public:
326 void storeRegToStackSlotCFI(MachineBasicBlock &MBB,
327 MachineBasicBlock::iterator MI, Register SrcReg,
328 bool isKill, int FrameIndex,
329 const TargetRegisterClass *RC) const;
330
331 bool getConstValDefinedInReg(const MachineInstr &MI, const Register Reg,
332 int64_t &ImmVal) const override;
333
334 std::optional<int64_t>
335 getImmOrMaterializedImm(const MachineRegisterInfo &MRI,
336 const MachineOperand &Op,
337 MachineInstr **DefMI = nullptr) const;
338 std::optional<int64_t>
339 getImmOrMaterializedImm(const MachineRegisterInfo &MRI, Register Reg,
340 MachineInstr **DefMI = nullptr) const;
341
342 unsigned getVectorRegSpillSaveOpcode(Register Reg,
343 const TargetRegisterClass *RC,
344 unsigned Size,
345 const SIMachineFunctionInfo &MFI,
346 bool NeedsCFI) const;
347 unsigned
348 getVectorRegSpillRestoreOpcode(Register Reg, const TargetRegisterClass *RC,
349 unsigned Size,
350 const SIMachineFunctionInfo &MFI) const;
351
352 void storeRegToStackSlot(
353 MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register SrcReg,
354 bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg,
355 MachineInstr::MIFlag Flags = MachineInstr::NoFlags) const override;
356
357 void loadRegFromStackSlot(
358 MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg,
359 int FrameIndex, const TargetRegisterClass *RC, Register VReg,
360 unsigned SubReg = 0,
361 MachineInstr::MIFlag Flags = MachineInstr::NoFlags) const override;
362
363 bool expandPostRAPseudo(MachineInstr &MI) const override;
364
365 void
366 reMaterialize(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI,
367 Register DestReg, unsigned SubIdx, const MachineInstr &Orig,
368 LaneBitmask UsedLanes = LaneBitmask::getAll()) const override;
369
370 // Splits a V_MOV_B64_DPP_PSEUDO opcode into a pair of v_mov_b32_dpp
371 // instructions. Returns a pair of generated instructions.
372 // Can split either post-RA with physical registers or pre-RA with
373 // virtual registers. In latter case IR needs to be in SSA form and
374 // and a REG_SEQUENCE is produced to define original register.
375 std::pair<MachineInstr*, MachineInstr*>
376 expandMovDPP64(MachineInstr &MI) const;
377
378 // Returns an opcode that can be used to move a value to a \p DstRC
379 // register. If there is no hardware instruction that can store to \p
380 // DstRC, then AMDGPU::COPY is returned.
381 unsigned getMovOpcode(const TargetRegisterClass *DstRC) const;
382
383 const MCInstrDesc &getIndirectRegWriteMovRelPseudo(unsigned VecSize,
384 unsigned EltSize,
385 bool IsSGPR) const;
386
387 const MCInstrDesc &getIndirectGPRIDXPseudo(unsigned VecSize,
388 bool IsIndirectSrc) const;
389 LLVM_READONLY
390 int commuteOpcode(unsigned Opc) const;
391
392 LLVM_READONLY
393 inline int commuteOpcode(const MachineInstr &MI) const {
394 return commuteOpcode(Opc: MI.getOpcode());
395 }
396
397 bool findCommutedOpIndices(const MachineInstr &MI, unsigned &SrcOpIdx0,
398 unsigned &SrcOpIdx1) const override;
399
400 bool findCommutedOpIndices(const MCInstrDesc &Desc, unsigned &SrcOpIdx0,
401 unsigned &SrcOpIdx1) const;
402
403 bool isBranchOffsetInRange(unsigned BranchOpc,
404 int64_t BrOffset) const override;
405
406 MachineBasicBlock *getBranchDestBlock(const MachineInstr &MI) const override;
407
408 /// Return whether the block terminate with divergent branch.
409 /// Note this only work before lowering the pseudo control flow instructions.
410 bool hasDivergentBranch(const MachineBasicBlock *MBB) const;
411
412 void insertIndirectBranch(MachineBasicBlock &MBB,
413 MachineBasicBlock &NewDestBB,
414 MachineBasicBlock &RestoreBB, const DebugLoc &DL,
415 int64_t BrOffset, RegScavenger *RS) const override;
416
417 bool analyzeBranchImpl(MachineBasicBlock &MBB,
418 MachineBasicBlock::iterator I,
419 MachineBasicBlock *&TBB,
420 MachineBasicBlock *&FBB,
421 SmallVectorImpl<MachineOperand> &Cond,
422 bool AllowModify) const;
423
424 bool analyzeBranch(MachineBasicBlock &MBB, MachineBasicBlock *&TBB,
425 MachineBasicBlock *&FBB,
426 SmallVectorImpl<MachineOperand> &Cond,
427 bool AllowModify = false) const override;
428
429 unsigned removeBranch(MachineBasicBlock &MBB,
430 int *BytesRemoved = nullptr) const override;
431
432 unsigned insertBranch(MachineBasicBlock &MBB, MachineBasicBlock *TBB,
433 MachineBasicBlock *FBB, ArrayRef<MachineOperand> Cond,
434 const DebugLoc &DL,
435 int *BytesAdded = nullptr) const override;
436
437 bool reverseBranchCondition(
438 SmallVectorImpl<MachineOperand> &Cond) const override;
439
440 std::unique_ptr<PipelinerLoopInfo>
441 analyzeLoopForPipelining(MachineBasicBlock *LoopBB) const override;
442
443 bool canInsertSelect(const MachineBasicBlock &MBB,
444 ArrayRef<MachineOperand> Cond, Register DstReg,
445 Register TrueReg, Register FalseReg, int &CondCycles,
446 int &TrueCycles, int &FalseCycles) const override;
447
448 void insertSelect(MachineBasicBlock &MBB,
449 MachineBasicBlock::iterator I, const DebugLoc &DL,
450 Register DstReg, ArrayRef<MachineOperand> Cond,
451 Register TrueReg, Register FalseReg) const override;
452
453 bool analyzeCompare(const MachineInstr &MI, Register &SrcReg,
454 Register &SrcReg2, int64_t &CmpMask,
455 int64_t &CmpValue) const override;
456
457 bool optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg,
458 Register SrcReg2, int64_t CmpMask, int64_t CmpValue,
459 const MachineRegisterInfo *MRI) const override;
460
461 bool
462 areMemAccessesTriviallyDisjoint(const MachineInstr &MIa,
463 const MachineInstr &MIb) const override;
464
465 static bool isFoldableCopy(const MachineInstr &MI);
466 static unsigned getFoldableCopySrcIdx(const MachineInstr &MI);
467
468 void removeModOperands(MachineInstr &MI) const;
469
470 void mutateAndCleanupImplicit(MachineInstr &MI,
471 const MCInstrDesc &NewDesc) const;
472
473 /// Return the extracted immediate value in a subregister use from a constant
474 /// materialized in a super register.
475 ///
476 /// e.g. %imm = S_MOV_B64 K[0:63]
477 /// USE %imm.sub1
478 /// This will return K[32:63]
479 static std::optional<int64_t> extractSubregFromImm(int64_t ImmVal,
480 unsigned SubRegIndex);
481
482 bool foldImmediate(MachineInstr &UseMI, MachineInstr &DefMI, Register Reg,
483 MachineRegisterInfo *MRI) const final;
484
485 unsigned getMachineCSELookAheadLimit() const override { return 500; }
486
487 MachineInstr *convertToThreeAddress(MachineInstr &MI,
488 LiveIntervals *LIS) const override;
489
490 bool isSchedulingBoundary(const MachineInstr &MI,
491 const MachineBasicBlock *MBB,
492 const MachineFunction &MF) const override;
493
494 static bool isSALU(const MachineInstr &MI) {
495 return SIInstrFlags::isSALU(O: MI);
496 }
497
498 bool isSALU(uint32_t Opcode) const {
499 return SIInstrFlags::isSALU(O: get(Opcode));
500 }
501
502 static bool isVALU(const MachineInstr &MI, bool AllowLDSDMA) {
503 if (!AllowLDSDMA && isLDSDMA(MI))
504 return false;
505
506 return SIInstrFlags::isVALU(O: MI);
507 }
508
509 /// LDSDMA instructions act as both VALU and memory instructions, thus
510 /// we also tag them as VALU. However, in many places, we do not actually want
511 /// to include LDSDMA instructions in this query. By setting \p AllowLDSDMA to
512 /// false, this will return false for LDSDMA instructions.
513 bool isVALU(uint32_t Opcode, bool AllowLDSDMA) const {
514 if (!AllowLDSDMA && isLDSDMA(Opcode))
515 return false;
516
517 return SIInstrFlags::isVALU(O: get(Opcode));
518 }
519
520 static bool isImage(const MachineInstr &MI) {
521 return SIInstrFlags::isImage(O: MI);
522 }
523
524 bool isImage(uint32_t Opcode) const {
525 return SIInstrFlags::isImage(O: get(Opcode));
526 }
527
528 static bool isVMEM(const MachineInstr &MI) {
529 return SIInstrFlags::isVMEM(O: MI);
530 }
531
532 bool isVMEM(uint32_t Opcode) const {
533 return SIInstrFlags::isVMEM(O: get(Opcode));
534 }
535
536 /// True if MI implicitly drains XCNT.
537 static bool isXcntDrain(const MachineInstr &MI);
538
539 static bool isSOP1(const MachineInstr &MI) {
540 return SIInstrFlags::isSOP1(O: MI);
541 }
542
543 bool isSOP1(uint32_t Opcode) const {
544 return SIInstrFlags::isSOP1(O: get(Opcode));
545 }
546
547 static bool isSOP2(const MachineInstr &MI) {
548 return SIInstrFlags::isSOP2(O: MI);
549 }
550
551 bool isSOP2(uint32_t Opcode) const {
552 return SIInstrFlags::isSOP2(O: get(Opcode));
553 }
554
555 static bool isSOPC(const MachineInstr &MI) {
556 return SIInstrFlags::isSOPC(O: MI);
557 }
558
559 bool isSOPC(uint32_t Opcode) const {
560 return SIInstrFlags::isSOPC(O: get(Opcode));
561 }
562
563 static bool isSOPK(const MachineInstr &MI) {
564 return SIInstrFlags::isSOPK(O: MI);
565 }
566
567 bool isSOPK(uint32_t Opcode) const {
568 return SIInstrFlags::isSOPK(O: get(Opcode));
569 }
570
571 static bool isSOPP(const MachineInstr &MI) {
572 return SIInstrFlags::isSOPP(O: MI);
573 }
574
575 bool isSOPP(uint32_t Opcode) const {
576 return SIInstrFlags::isSOPP(O: get(Opcode));
577 }
578
579 static bool isPacked(const MachineInstr &MI) {
580 return SIInstrFlags::isPacked(O: MI);
581 }
582
583 bool isPacked(uint32_t Opcode) const {
584 return SIInstrFlags::isPacked(O: get(Opcode));
585 }
586
587 static bool isVOP1(const MachineInstr &MI) {
588 return SIInstrFlags::isVOP1(O: MI);
589 }
590
591 bool isVOP1(uint32_t Opcode) const {
592 return SIInstrFlags::isVOP1(O: get(Opcode));
593 }
594
595 static bool isVOP2(const MachineInstr &MI) {
596 return SIInstrFlags::isVOP2(O: MI);
597 }
598
599 bool isVOP2(uint32_t Opcode) const {
600 return SIInstrFlags::isVOP2(O: get(Opcode));
601 }
602
603 static bool isVOP3(const MCInstrDesc &Desc) {
604 return SIInstrFlags::isVOP3(O: Desc);
605 }
606
607 static bool isVOP3(const MachineInstr &MI) { return isVOP3(Desc: MI.getDesc()); }
608
609 bool isVOP3(uint32_t Opcode) const { return isVOP3(Desc: get(Opcode)); }
610
611 static bool isSDWA(const MachineInstr &MI) {
612 return SIInstrFlags::isSDWA(O: MI);
613 }
614
615 bool isSDWA(uint32_t Opcode) const {
616 return SIInstrFlags::isSDWA(O: get(Opcode));
617 }
618
619 static bool isVOPC(const MachineInstr &MI) {
620 return SIInstrFlags::isVOPC(O: MI);
621 }
622
623 bool isVOPC(uint32_t Opcode) const {
624 return SIInstrFlags::isVOPC(O: get(Opcode));
625 }
626
627 static bool isMUBUF(const MachineInstr &MI) {
628 return SIInstrFlags::isMUBUF(O: MI);
629 }
630
631 bool isMUBUF(uint32_t Opcode) const {
632 return SIInstrFlags::isMUBUF(O: get(Opcode));
633 }
634
635 static bool isMTBUF(const MachineInstr &MI) {
636 return SIInstrFlags::isMTBUF(O: MI);
637 }
638
639 bool isMTBUF(uint32_t Opcode) const {
640 return SIInstrFlags::isMTBUF(O: get(Opcode));
641 }
642
643 static bool isBUF(const MachineInstr &MI) {
644 return isMUBUF(MI) || isMTBUF(MI);
645 }
646
647 static bool isSMRD(const MachineInstr &MI) {
648 return SIInstrFlags::isSMRD(O: MI);
649 }
650
651 bool isSMRD(uint32_t Opcode) const {
652 return SIInstrFlags::isSMRD(O: get(Opcode));
653 }
654
655 bool isBufferSMRD(const MachineInstr &MI) const;
656
657 static bool isDS(const MachineInstr &MI) { return SIInstrFlags::isDS(O: MI); }
658
659 bool isDS(uint32_t Opcode) const { return SIInstrFlags::isDS(O: get(Opcode)); }
660
661 static bool isLDSDMA(const MachineInstr &MI) {
662 return (SIInstrFlags::isVALU(O: MI) && (isMUBUF(MI) || isFLAT(MI))) ||
663 SIInstrFlags::usesTENSOR_CNT(O: MI);
664 }
665
666 bool isLDSDMA(uint32_t Opcode) const {
667 return (SIInstrFlags::isVALU(O: get(Opcode)) &&
668 (isMUBUF(Opcode) || isFLAT(Opcode))) ||
669 SIInstrFlags::usesTENSOR_CNT(O: get(Opcode));
670 }
671
672 static bool isGWS(const MachineInstr &MI) { return SIInstrFlags::isGWS(O: MI); }
673
674 bool isGWS(uint32_t Opcode) const { return SIInstrFlags::isGWS(O: get(Opcode)); }
675
676 bool isAlwaysGDS(uint32_t Opcode) const;
677
678 static bool isMIMG(const MachineInstr &MI) {
679 return SIInstrFlags::isMIMG(O: MI);
680 }
681
682 bool isMIMG(uint32_t Opcode) const {
683 return SIInstrFlags::isMIMG(O: get(Opcode));
684 }
685
686 static bool isVIMAGE(const MachineInstr &MI) {
687 return SIInstrFlags::isVIMAGE(O: MI);
688 }
689
690 bool isVIMAGE(uint32_t Opcode) const {
691 return SIInstrFlags::isVIMAGE(O: get(Opcode));
692 }
693
694 static bool isVSAMPLE(const MachineInstr &MI) {
695 return SIInstrFlags::isVSAMPLE(O: MI);
696 }
697
698 bool isVSAMPLE(uint32_t Opcode) const {
699 return SIInstrFlags::isVSAMPLE(O: get(Opcode));
700 }
701
702 static bool isGather4(const MachineInstr &MI) {
703 return SIInstrFlags::isGather4(O: MI);
704 }
705
706 bool isGather4(uint32_t Opcode) const {
707 return SIInstrFlags::isGather4(O: get(Opcode));
708 }
709
710 static bool isFLAT(const MachineInstr &MI) {
711 return SIInstrFlags::isFLAT(O: MI);
712 }
713
714 // Is a FLAT encoded instruction which accesses a specific segment,
715 // i.e. global_* or scratch_*.
716 static bool isSegmentSpecificFLAT(const MachineInstr &MI) {
717 return SIInstrFlags::isSegmentSpecificFLAT(O: MI);
718 }
719
720 bool isSegmentSpecificFLAT(uint32_t Opcode) const {
721 return SIInstrFlags::isSegmentSpecificFLAT(O: get(Opcode));
722 }
723
724 static bool isFLATGlobal(const MachineInstr &MI) {
725 return SIInstrFlags::isFlatGlobal(O: MI);
726 }
727
728 bool isFLATGlobal(uint32_t Opcode) const {
729 return SIInstrFlags::isFlatGlobal(O: get(Opcode));
730 }
731
732 static bool isFLATScratch(const MachineInstr &MI) {
733 return SIInstrFlags::isFlatScratch(O: MI);
734 }
735
736 bool isFLATScratch(uint32_t Opcode) const {
737 return SIInstrFlags::isFlatScratch(O: get(Opcode));
738 }
739
740 // Any FLAT encoded instruction, including global_* and scratch_*.
741 bool isFLAT(uint32_t Opcode) const {
742 return SIInstrFlags::isFLAT(O: get(Opcode));
743 }
744
745 /// \returns true for SCRATCH_ instructions, or FLAT/BUF instructions unless
746 /// the MMOs do not include scratch.
747 /// Conservatively correct; will return true if \p MI cannot be proven
748 /// to not hit scratch.
749 bool mayAccessScratch(const MachineInstr &MI) const;
750
751 /// \returns true for FLAT instructions that can access VMEM.
752 bool mayAccessVMEMThroughFlat(const MachineInstr &MI) const;
753
754 /// \returns true for FLAT instructions that can access LDS.
755 bool mayAccessLDSThroughFlat(const MachineInstr &MI, bool TgSplit) const;
756
757 static bool isBlockLoadStore(uint32_t Opcode) {
758 switch (Opcode) {
759 case AMDGPU::SI_BLOCK_SPILL_V1024_SAVE:
760 case AMDGPU::SI_BLOCK_SPILL_V1024_CFI_SAVE:
761 case AMDGPU::SI_BLOCK_SPILL_V1024_RESTORE:
762 case AMDGPU::SCRATCH_STORE_BLOCK_SADDR:
763 case AMDGPU::SCRATCH_LOAD_BLOCK_SADDR:
764 case AMDGPU::SCRATCH_STORE_BLOCK_SVS:
765 case AMDGPU::SCRATCH_LOAD_BLOCK_SVS:
766 return true;
767 default:
768 return false;
769 }
770 }
771
772 static bool setsSCCIfResultIsNonZero(const MachineInstr &MI) {
773 switch (MI.getOpcode()) {
774 case AMDGPU::S_ABSDIFF_I32:
775 case AMDGPU::S_ABS_I32:
776 case AMDGPU::S_AND_B32:
777 case AMDGPU::S_AND_B64:
778 case AMDGPU::S_ANDN2_B32:
779 case AMDGPU::S_ANDN2_B64:
780 case AMDGPU::S_ASHR_I32:
781 case AMDGPU::S_ASHR_I64:
782 case AMDGPU::S_BCNT0_I32_B32:
783 case AMDGPU::S_BCNT0_I32_B64:
784 case AMDGPU::S_BCNT1_I32_B32:
785 case AMDGPU::S_BCNT1_I32_B64:
786 case AMDGPU::S_BFE_I32:
787 case AMDGPU::S_BFE_I64:
788 case AMDGPU::S_BFE_U32:
789 case AMDGPU::S_BFE_U64:
790 case AMDGPU::S_LSHL_B32:
791 case AMDGPU::S_LSHL_B64:
792 case AMDGPU::S_LSHR_B32:
793 case AMDGPU::S_LSHR_B64:
794 case AMDGPU::S_NAND_B32:
795 case AMDGPU::S_NAND_B64:
796 case AMDGPU::S_NOR_B32:
797 case AMDGPU::S_NOR_B64:
798 case AMDGPU::S_NOT_B32:
799 case AMDGPU::S_NOT_B64:
800 case AMDGPU::S_OR_B32:
801 case AMDGPU::S_OR_B64:
802 case AMDGPU::S_ORN2_B32:
803 case AMDGPU::S_ORN2_B64:
804 case AMDGPU::S_QUADMASK_B32:
805 case AMDGPU::S_QUADMASK_B64:
806 case AMDGPU::S_WQM_B32:
807 case AMDGPU::S_WQM_B64:
808 case AMDGPU::S_XNOR_B32:
809 case AMDGPU::S_XNOR_B64:
810 case AMDGPU::S_XOR_B32:
811 case AMDGPU::S_XOR_B64:
812 return true;
813 default:
814 return false;
815 }
816 }
817
818 static bool isEXP(const MachineInstr &MI) { return SIInstrFlags::isEXP(O: MI); }
819
820 static bool isDualSourceBlendEXP(const MachineInstr &MI) {
821 if (!isEXP(MI))
822 return false;
823 unsigned Target = MI.getOperand(i: 0).getImm();
824 return Target == AMDGPU::Exp::ET_DUAL_SRC_BLEND0 ||
825 Target == AMDGPU::Exp::ET_DUAL_SRC_BLEND1;
826 }
827
828 bool isEXP(uint32_t Opcode) const { return SIInstrFlags::isEXP(O: get(Opcode)); }
829
830 static bool isAtomicNoRet(const MachineInstr &MI) {
831 return SIInstrFlags::isAtomicNoRet(O: MI);
832 }
833
834 bool isAtomicNoRet(uint32_t Opcode) const {
835 return SIInstrFlags::isAtomicNoRet(O: get(Opcode));
836 }
837
838 static bool isAtomicRet(const MachineInstr &MI) {
839 return SIInstrFlags::isAtomicRet(O: MI);
840 }
841
842 bool isAtomicRet(uint32_t Opcode) const {
843 return SIInstrFlags::isAtomicRet(O: get(Opcode));
844 }
845
846 static bool isAtomic(const MachineInstr &MI) {
847 return SIInstrFlags::isAtomic(O: MI);
848 }
849
850 bool isAtomic(uint32_t Opcode) const {
851 return SIInstrFlags::isAtomic(O: get(Opcode));
852 }
853
854 static bool mayWriteLDSThroughDMA(const MachineInstr &MI) {
855 unsigned Opc = MI.getOpcode();
856 // Exclude instructions that read FROM LDS (not write to it)
857 return isLDSDMA(MI) && Opc != AMDGPU::BUFFER_STORE_LDS_DWORD &&
858 Opc != AMDGPU::TENSOR_STORE_FROM_LDS_d2 &&
859 Opc != AMDGPU::TENSOR_STORE_FROM_LDS_d4;
860 }
861
862 static bool isSBarrierSCCWrite(unsigned Opcode) {
863 return Opcode == AMDGPU::S_BARRIER_LEAVE ||
864 Opcode == AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM ||
865 Opcode == AMDGPU::S_BARRIER_SIGNAL_ISFIRST_M0;
866 }
867
868 static bool isCBranchVCCZRead(const MachineInstr &MI) {
869 unsigned Opc = MI.getOpcode();
870 return (Opc == AMDGPU::S_CBRANCH_VCCNZ || Opc == AMDGPU::S_CBRANCH_VCCZ) &&
871 !MI.getOperand(i: 1).isUndef();
872 }
873
874 static bool isWQM(const MachineInstr &MI) { return SIInstrFlags::isWQM(O: MI); }
875
876 bool isWQM(uint32_t Opcode) const { return SIInstrFlags::isWQM(O: get(Opcode)); }
877
878 static bool isDisableWQM(const MachineInstr &MI) {
879 return SIInstrFlags::isDisableWQM(O: MI);
880 }
881
882 bool isDisableWQM(uint32_t Opcode) const {
883 return SIInstrFlags::isDisableWQM(O: get(Opcode));
884 }
885
886 // SI_SPILL_S32_TO_VGPR and SI_RESTORE_S32_FROM_VGPR form a special case of
887 // SGPRs spilling to VGPRs which are SGPR spills but from VALU instructions
888 // therefore we need an explicit check for them since just checking if the
889 // Spill bit is set and what instruction type it came from misclassifies
890 // them.
891 static bool isVGPRSpill(const MachineInstr &MI) {
892 return MI.getOpcode() != AMDGPU::SI_SPILL_S32_TO_VGPR &&
893 MI.getOpcode() != AMDGPU::SI_RESTORE_S32_FROM_VGPR &&
894 (isSpill(MI) && isVALU(MI, /*AllowLDSDMA=*/AllowLDSDMA: false));
895 }
896
897 bool isVGPRSpill(uint32_t Opcode) const {
898 return Opcode != AMDGPU::SI_SPILL_S32_TO_VGPR &&
899 Opcode != AMDGPU::SI_RESTORE_S32_FROM_VGPR &&
900 (isSpill(Opcode) && isVALU(Opcode, /*AllowLDSDMA=*/AllowLDSDMA: false));
901 }
902
903 static bool isSGPRSpill(const MachineInstr &MI) {
904 return MI.getOpcode() == AMDGPU::SI_SPILL_S32_TO_VGPR ||
905 MI.getOpcode() == AMDGPU::SI_RESTORE_S32_FROM_VGPR ||
906 (isSpill(MI) && isSALU(MI));
907 }
908
909 bool isSGPRSpill(uint32_t Opcode) const {
910 return Opcode == AMDGPU::SI_SPILL_S32_TO_VGPR ||
911 Opcode == AMDGPU::SI_RESTORE_S32_FROM_VGPR ||
912 (isSpill(Opcode) && isSALU(Opcode));
913 }
914
915 bool isSpill(uint32_t Opcode) const {
916 return SIInstrFlags::isSpill(O: get(Opcode));
917 }
918
919 static bool isSpill(const MCInstrDesc &Desc) {
920 return SIInstrFlags::isSpill(O: Desc);
921 }
922
923 static bool isSpill(const MachineInstr &MI) { return isSpill(Desc: MI.getDesc()); }
924
925 static bool isWWMRegSpillOpcode(uint32_t Opcode) {
926 return Opcode == AMDGPU::SI_SPILL_WWM_V32_SAVE ||
927 Opcode == AMDGPU::SI_SPILL_WWM_AV32_SAVE ||
928 Opcode == AMDGPU::SI_SPILL_WWM_V32_RESTORE ||
929 Opcode == AMDGPU::SI_SPILL_WWM_AV32_RESTORE;
930 }
931
932 static bool isChainCallOpcode(uint64_t Opcode) {
933 return Opcode == AMDGPU::SI_CS_CHAIN_TC_W32 ||
934 Opcode == AMDGPU::SI_CS_CHAIN_TC_W64;
935 }
936
937 static bool isDPP(const MachineInstr &MI) { return SIInstrFlags::isDPP(O: MI); }
938
939 bool isDPP(uint32_t Opcode) const { return SIInstrFlags::isDPP(O: get(Opcode)); }
940
941 // Some opcodes use Src1 for DPP instead of Src0, because the sequencer
942 // transforms them and reverse the order of their operands at runtime.
943 //
944 // Documentation is incomplete on which instructions are effected, so
945 // the implementation is derived from experimentation.
946 //
947 // Listed as target-independent pseudos; the per-subtarget MC opcodes
948 // (V_SUBREV_NC_U32_e32_gfx11 and friends) are all reached through these.
949 // Defined out of line because GCNSubtarget is incomplete here.
950 static bool isSrc1DPPRevOpcode(const GCNSubtarget &ST, uint32_t Opcode);
951
952 static bool isTRANS(const MachineInstr &MI) {
953 return SIInstrFlags::isTRANS(O: MI);
954 }
955
956 bool isTRANS(uint32_t Opcode) const {
957 return SIInstrFlags::isTRANS(O: get(Opcode));
958 }
959
960 static bool isVOP3P(const MachineInstr &MI) {
961 return SIInstrFlags::isVOP3P(O: MI);
962 }
963
964 bool isVOP3P(uint32_t Opcode) const {
965 return SIInstrFlags::isVOP3P(O: get(Opcode));
966 }
967
968 bool isVOP3PMix(const MachineInstr &MI) const {
969 return isVOP3PMix(Opcode: MI.getOpcode());
970 }
971
972 bool isVOP3PMix(uint16_t Opcode) const {
973 switch (Opcode) {
974 case AMDGPU::V_FMA_MIXHI_F16:
975 case AMDGPU::V_FMA_MIXLO_F16:
976 case AMDGPU::V_FMA_MIX_F32:
977 case AMDGPU::V_MAD_MIXHI_F16:
978 case AMDGPU::V_MAD_MIXLO_F16:
979 case AMDGPU::V_MAD_MIX_F32:
980 return true;
981 default:
982 return false;
983 }
984 }
985
986 static bool isVINTRP(const MachineInstr &MI) {
987 return SIInstrFlags::isVINTRP(O: MI);
988 }
989
990 bool isVINTRP(uint32_t Opcode) const {
991 return SIInstrFlags::isVINTRP(O: get(Opcode));
992 }
993
994 static bool isMAI(const MCInstrDesc &Desc) {
995 return SIInstrFlags::isMAI(O: Desc);
996 }
997
998 static bool isMAI(const MachineInstr &MI) { return isMAI(Desc: MI.getDesc()); }
999
1000 bool isMAI(uint32_t Opcode) const { return isMAI(Desc: get(Opcode)); }
1001
1002 static bool isMFMA(const MachineInstr &MI) {
1003 return isMAI(MI) && MI.getOpcode() != AMDGPU::V_ACCVGPR_WRITE_B32_e64 &&
1004 MI.getOpcode() != AMDGPU::V_ACCVGPR_READ_B32_e64;
1005 }
1006
1007 bool isMFMA(uint32_t Opcode) const {
1008 return isMAI(Opcode) && Opcode != AMDGPU::V_ACCVGPR_WRITE_B32_e64 &&
1009 Opcode != AMDGPU::V_ACCVGPR_READ_B32_e64;
1010 }
1011
1012 static bool isDOT(const MachineInstr &MI) { return SIInstrFlags::isDOT(O: MI); }
1013
1014 static bool isWMMA(const MachineInstr &MI) {
1015 return SIInstrFlags::isWMMA(O: MI);
1016 }
1017
1018 bool isWMMA(uint32_t Opcode) const {
1019 return SIInstrFlags::isWMMA(O: get(Opcode));
1020 }
1021
1022 static bool isMFMAorWMMA(const MachineInstr &MI) {
1023 return isMFMA(MI) || isWMMA(MI) || isSWMMAC(MI);
1024 }
1025
1026 bool isMFMAorWMMA(uint32_t Opcode) const {
1027 return isMFMA(Opcode) || isWMMA(Opcode) || isSWMMAC(Opcode);
1028 }
1029
1030 static bool isSWMMAC(const MachineInstr &MI) {
1031 return SIInstrFlags::isSWMMAC(O: MI);
1032 }
1033
1034 bool isSWMMAC(uint32_t Opcode) const {
1035 return SIInstrFlags::isSWMMAC(O: get(Opcode));
1036 }
1037
1038 bool isDOT(uint32_t Opcode) const { return SIInstrFlags::isDOT(O: get(Opcode)); }
1039
1040 bool isXDLWMMA(const MachineInstr &MI) const;
1041
1042 bool isXDL(const MachineInstr &MI) const;
1043
1044 static bool isDGEMM(unsigned Opcode) { return AMDGPU::getMAIIsDGEMM(Opc: Opcode); }
1045
1046 static bool isLDSDIR(const MachineInstr &MI) {
1047 return SIInstrFlags::isLDSDIR(O: MI);
1048 }
1049
1050 bool isLDSDIR(uint32_t Opcode) const {
1051 return SIInstrFlags::isLDSDIR(O: get(Opcode));
1052 }
1053
1054 static bool isVINTERP(const MachineInstr &MI) {
1055 return SIInstrFlags::isVINTERP(O: MI);
1056 }
1057
1058 bool isVINTERP(uint32_t Opcode) const {
1059 return SIInstrFlags::isVINTERP(O: get(Opcode));
1060 }
1061
1062 static bool usesVM_CNT(const MachineInstr &MI) {
1063 return SIInstrFlags::usesVM_CNT(O: MI);
1064 }
1065
1066 static bool usesLGKM_CNT(const MachineInstr &MI) {
1067 return SIInstrFlags::usesLGKM_CNT(O: MI);
1068 }
1069
1070 static bool usesASYNC_CNT(const MachineInstr &MI) {
1071 return SIInstrFlags::usesASYNC_CNT(O: MI);
1072 }
1073
1074 bool usesASYNC_CNT(uint32_t Opcode) const {
1075 return SIInstrFlags::usesASYNC_CNT(O: get(Opcode));
1076 }
1077
1078 static bool usesTENSOR_CNT(const MachineInstr &MI) {
1079 return SIInstrFlags::usesTENSOR_CNT(O: MI);
1080 }
1081
1082 bool usesTENSOR_CNT(uint32_t Opcode) const {
1083 return SIInstrFlags::usesTENSOR_CNT(O: get(Opcode));
1084 }
1085
1086 // Most sopk treat the immediate as a signed 16-bit, however some
1087 // use it as unsigned.
1088 static bool sopkIsZext(unsigned Opcode) {
1089 return Opcode == AMDGPU::S_CMPK_EQ_U32 || Opcode == AMDGPU::S_CMPK_LG_U32 ||
1090 Opcode == AMDGPU::S_CMPK_GT_U32 || Opcode == AMDGPU::S_CMPK_GE_U32 ||
1091 Opcode == AMDGPU::S_CMPK_LT_U32 || Opcode == AMDGPU::S_CMPK_LE_U32 ||
1092 Opcode == AMDGPU::S_GETREG_B32 ||
1093 Opcode == AMDGPU::S_GETREG_B32_const;
1094 }
1095
1096 /// \returns true if this is an s_store_dword* instruction. This is more
1097 /// specific than isSMEM && mayStore.
1098 static bool isScalarStore(const MachineInstr &MI) {
1099 return SIInstrFlags::isScalarStore(O: MI);
1100 }
1101
1102 bool isScalarStore(uint32_t Opcode) const {
1103 return SIInstrFlags::isScalarStore(O: get(Opcode));
1104 }
1105
1106 static bool isFixedSize(const MachineInstr &MI) {
1107 return SIInstrFlags::isFixedSize(O: MI);
1108 }
1109
1110 bool isFixedSize(uint32_t Opcode) const {
1111 return SIInstrFlags::isFixedSize(O: get(Opcode));
1112 }
1113
1114 static bool hasFPClamp(const MachineInstr &MI) {
1115 return SIInstrFlags::hasFPClamp(O: MI);
1116 }
1117
1118 bool hasFPClamp(uint32_t Opcode) const {
1119 return SIInstrFlags::hasFPClamp(O: get(Opcode));
1120 }
1121
1122 static bool hasIntClamp(const MachineInstr &MI) {
1123 return SIInstrFlags::hasIntClamp(O: MI);
1124 }
1125
1126 static bool hasSameClamp(const MachineInstr &A, const MachineInstr &B) {
1127 const MCInstrDesc &DA = A.getDesc(), &DB = B.getDesc();
1128 return SIInstrFlags::hasFPClamp(O: DA) == SIInstrFlags::hasFPClamp(O: DB) &&
1129 SIInstrFlags::hasIntClamp(O: DA) == SIInstrFlags::hasIntClamp(O: DB) &&
1130 SIInstrFlags::hasClampLo(O: DA) == SIInstrFlags::hasClampLo(O: DB) &&
1131 SIInstrFlags::hasClampHi(O: DA) == SIInstrFlags::hasClampHi(O: DB);
1132 }
1133
1134 static bool usesFPDPRounding(const MachineInstr &MI) {
1135 return SIInstrFlags::usesFPDPRounding(O: MI);
1136 }
1137
1138 bool usesFPDPRounding(uint32_t Opcode) const {
1139 return SIInstrFlags::usesFPDPRounding(O: get(Opcode));
1140 }
1141
1142 static bool isFPAtomic(const MachineInstr &MI) {
1143 return SIInstrFlags::isFPAtomic(O: MI);
1144 }
1145
1146 bool isFPAtomic(uint32_t Opcode) const {
1147 return SIInstrFlags::isFPAtomic(O: get(Opcode));
1148 }
1149
1150 static bool isNeverUniform(const MachineInstr &MI) {
1151 return SIInstrFlags::isNeverUniform(O: MI);
1152 }
1153
1154 // Check to see if opcode is for a barrier start. Pre gfx12 this is just the
1155 // S_BARRIER, but after support for S_BARRIER_SIGNAL* / S_BARRIER_WAIT we want
1156 // to check for the barrier start (S_BARRIER_SIGNAL*)
1157 bool isBarrierStart(unsigned Opcode) const {
1158 return Opcode == AMDGPU::S_BARRIER ||
1159 Opcode == AMDGPU::S_BARRIER_SIGNAL_M0 ||
1160 Opcode == AMDGPU::S_BARRIER_SIGNAL_ISFIRST_M0 ||
1161 Opcode == AMDGPU::S_BARRIER_SIGNAL_IMM ||
1162 Opcode == AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM;
1163 }
1164
1165 bool isBarrier(unsigned Opcode) const {
1166 return isBarrierStart(Opcode) || Opcode == AMDGPU::S_BARRIER_WAIT ||
1167 Opcode == AMDGPU::S_BARRIER_INIT_M0 ||
1168 Opcode == AMDGPU::S_BARRIER_INIT_IMM ||
1169 Opcode == AMDGPU::S_BARRIER_JOIN_IMM ||
1170 Opcode == AMDGPU::S_BARRIER_LEAVE || Opcode == AMDGPU::DS_GWS_INIT ||
1171 Opcode == AMDGPU::DS_GWS_BARRIER;
1172 }
1173
1174 static bool isLoadMonitor(unsigned Opc) {
1175 switch (Opc) {
1176 case AMDGPU::GLOBAL_LOAD_MONITOR_B32:
1177 case AMDGPU::GLOBAL_LOAD_MONITOR_B32_SADDR:
1178 case AMDGPU::GLOBAL_LOAD_MONITOR_B64:
1179 case AMDGPU::GLOBAL_LOAD_MONITOR_B64_SADDR:
1180 case AMDGPU::GLOBAL_LOAD_MONITOR_B128:
1181 case AMDGPU::GLOBAL_LOAD_MONITOR_B128_SADDR:
1182 case AMDGPU::FLAT_LOAD_MONITOR_B32:
1183 case AMDGPU::FLAT_LOAD_MONITOR_B64:
1184 case AMDGPU::FLAT_LOAD_MONITOR_B128:
1185 return true;
1186 default:
1187 return false;
1188 }
1189 }
1190
1191 static bool isGFX12CacheInvOrWBInst(unsigned Opc) {
1192 return Opc == AMDGPU::GLOBAL_INV || Opc == AMDGPU::GLOBAL_WB ||
1193 Opc == AMDGPU::GLOBAL_WBINV;
1194 }
1195
1196 static bool isF16PseudoScalarTrans(unsigned Opcode) {
1197 return Opcode == AMDGPU::V_S_EXP_F16_e64 ||
1198 Opcode == AMDGPU::V_S_LOG_F16_e64 ||
1199 Opcode == AMDGPU::V_S_RCP_F16_e64 ||
1200 Opcode == AMDGPU::V_S_RSQ_F16_e64 ||
1201 Opcode == AMDGPU::V_S_SQRT_F16_e64;
1202 }
1203
1204 static bool isPseudoScalarTrans(unsigned Opcode) {
1205 return isF16PseudoScalarTrans(Opcode) ||
1206 Opcode == AMDGPU::V_S_EXP_F32_e64 ||
1207 Opcode == AMDGPU::V_S_LOG_F32_e64 ||
1208 Opcode == AMDGPU::V_S_RCP_F32_e64 ||
1209 Opcode == AMDGPU::V_S_RSQ_F32_e64 ||
1210 Opcode == AMDGPU::V_S_SQRT_F32_e64;
1211 }
1212
1213 static bool isVPermPk16(unsigned Opcode) {
1214 return Opcode == AMDGPU::V_PERM_PK16_B4_U4_e64 ||
1215 Opcode == AMDGPU::V_PERM_PK16_B6_U4_e64 ||
1216 Opcode == AMDGPU::V_PERM_PK16_B8_U4_e64;
1217 }
1218
1219 // \returns true if \p MI clears the V_PERM_PK16 hazard when it immediately
1220 // follows a V_PERM_PK16 (i.e. \p MI is a "safe" instruction).
1221 bool isVPermPk16SafeInstr(const MachineInstr &MI) const {
1222 unsigned Opc = MI.getOpcode();
1223
1224 // Only VALU ops issue on the pipe that clears the V_PERM_PK16 hazard.
1225 if (!isVALU(MI, /*AllowLDSDMA=*/AllowLDSDMA: false))
1226 return false;
1227 // OP_XDL: matrix (WMMA/SWMMAC/DOT) ops clear the hazard.
1228 if (isXDL(MI))
1229 return true;
1230 // Pseudo-scalar transcendentals (OP32_SCL_T) do NOT clear the hazard.
1231 if (isPseudoScalarTrans(Opcode: Opc))
1232 return false;
1233
1234 // Use the table lookup, not getBlockingCycles(): occupancy is gated off on
1235 // gfx1251 but the hazard applies to both gfx1250 and gfx1251. Gfx1250 table
1236 // is valid enough for gfx1251 w.r.t. v_perm_pk16 safety check here.
1237 // OP_32_T is in the table at 2 and is safe; other table entries (>= 2) are
1238 // not.
1239 unsigned Cycles = getGFX1250BlockingCyclesTable(MI);
1240 return Cycles < 2 || (Cycles == 2 && isTRANS(MI));
1241 }
1242
1243 static bool doesNotReadTiedSource(const MachineInstr &MI) {
1244 return SIInstrFlags::isTiedSourceNotRead(O: MI);
1245 }
1246
1247 bool doesNotReadTiedSource(uint32_t Opcode) const {
1248 return SIInstrFlags::isTiedSourceNotRead(O: get(Opcode));
1249 }
1250
1251 bool isIGLP(unsigned Opcode) const {
1252 return Opcode == AMDGPU::SCHED_BARRIER ||
1253 Opcode == AMDGPU::SCHED_GROUP_BARRIER || Opcode == AMDGPU::IGLP_OPT;
1254 }
1255
1256 bool isIGLP(const MachineInstr &MI) const { return isIGLP(Opcode: MI.getOpcode()); }
1257
1258 // Return true if the instruction is mutually exclusive with all non-IGLP DAG
1259 // mutations, requiring all other mutations to be disabled.
1260 bool isIGLPMutationOnly(unsigned Opcode) const {
1261 return Opcode == AMDGPU::SCHED_GROUP_BARRIER || Opcode == AMDGPU::IGLP_OPT;
1262 }
1263
1264 static unsigned getNonSoftWaitcntOpcode(unsigned Opcode) {
1265 switch (Opcode) {
1266 case AMDGPU::S_WAITCNT_soft:
1267 return AMDGPU::S_WAITCNT;
1268 case AMDGPU::S_WAITCNT_VSCNT_soft:
1269 return AMDGPU::S_WAITCNT_VSCNT;
1270 case AMDGPU::S_WAIT_LOADCNT_soft:
1271 return AMDGPU::S_WAIT_LOADCNT;
1272 case AMDGPU::S_WAIT_STORECNT_soft:
1273 return AMDGPU::S_WAIT_STORECNT;
1274 case AMDGPU::S_WAIT_SAMPLECNT_soft:
1275 return AMDGPU::S_WAIT_SAMPLECNT;
1276 case AMDGPU::S_WAIT_BVHCNT_soft:
1277 return AMDGPU::S_WAIT_BVHCNT;
1278 case AMDGPU::S_WAIT_DSCNT_soft:
1279 return AMDGPU::S_WAIT_DSCNT;
1280 case AMDGPU::S_WAIT_KMCNT_soft:
1281 return AMDGPU::S_WAIT_KMCNT;
1282 case AMDGPU::S_WAIT_XCNT_soft:
1283 return AMDGPU::S_WAIT_XCNT;
1284 default:
1285 return Opcode;
1286 }
1287 }
1288
1289 static bool isWaitcnt(unsigned Opcode) {
1290 switch (getNonSoftWaitcntOpcode(Opcode)) {
1291 case AMDGPU::S_WAITCNT:
1292 case AMDGPU::S_WAITCNT_VSCNT:
1293 case AMDGPU::S_WAITCNT_VMCNT:
1294 case AMDGPU::S_WAITCNT_EXPCNT:
1295 case AMDGPU::S_WAITCNT_LGKMCNT:
1296 case AMDGPU::S_WAIT_LOADCNT:
1297 case AMDGPU::S_WAIT_LOADCNT_DSCNT:
1298 case AMDGPU::S_WAIT_STORECNT:
1299 case AMDGPU::S_WAIT_STORECNT_DSCNT:
1300 case AMDGPU::S_WAIT_SAMPLECNT:
1301 case AMDGPU::S_WAIT_BVHCNT:
1302 case AMDGPU::S_WAIT_EXPCNT:
1303 case AMDGPU::S_WAIT_DSCNT:
1304 case AMDGPU::S_WAIT_KMCNT:
1305 case AMDGPU::S_WAIT_XCNT:
1306 case AMDGPU::S_WAIT_ASYNCCNT:
1307 case AMDGPU::S_WAIT_TENSORCNT:
1308 case AMDGPU::S_WAIT_IDLE:
1309 return true;
1310 default:
1311 return false;
1312 }
1313 }
1314
1315 bool isVGPRCopy(const MachineInstr &MI) const {
1316 assert(isCopyInstr(MI));
1317 Register Dest = MI.getOperand(i: 0).getReg();
1318 const MachineFunction &MF = *MI.getMF();
1319 const MachineRegisterInfo &MRI = MF.getRegInfo();
1320 return !RI.isSGPRReg(MRI, Reg: Dest);
1321 }
1322
1323 bool hasVGPRUses(const MachineInstr &MI) const {
1324 const MachineFunction &MF = *MI.getMF();
1325 const MachineRegisterInfo &MRI = MF.getRegInfo();
1326 return llvm::any_of(Range: MI.explicit_uses(),
1327 P: [&MRI, this](const MachineOperand &MO) {
1328 return MO.isReg() && RI.isVGPR(MRI, Reg: MO.getReg());});
1329 }
1330
1331 /// Return true if the instruction modifies the mode register.q
1332 static bool modifiesModeRegister(const MachineInstr &MI);
1333
1334 /// This function is used to determine if an instruction can be safely
1335 /// executed under EXEC = 0 without hardware error, indeterminate results,
1336 /// and/or visible effects on future vector execution or outside the shader.
1337 /// Note: as of 2024 the only use of this is SIPreEmitPeephole where it is
1338 /// used in removing branches over short EXEC = 0 sequences.
1339 /// As such it embeds certain assumptions which may not apply to every case
1340 /// of EXEC = 0 execution.
1341 bool hasUnwantedEffectsWhenEXECEmpty(const MachineInstr &MI) const;
1342
1343 /// Returns true if the instruction could potentially depend on the value of
1344 /// exec. If false, exec dependencies may safely be ignored.
1345 bool mayReadEXEC(const MachineRegisterInfo &MRI, const MachineInstr &MI) const;
1346
1347 bool isInlineConstant(const APInt &Imm) const;
1348
1349 bool isInlineConstant(const APFloat &Imm) const;
1350
1351 // Returns true if this non-register operand definitely does not need to be
1352 // encoded as a 32-bit literal. Note that this function handles all kinds of
1353 // operands, not just immediates.
1354 //
1355 // Some operands like FrameIndexes could resolve to an inline immediate value
1356 // that will not require an additional 4-bytes; this function assumes that it
1357 // will.
1358 bool isInlineConstant(const MachineOperand &MO, uint8_t OperandType) const {
1359 if (!MO.isImm())
1360 return false;
1361 return isInlineConstant(ImmVal: MO.getImm(), OperandType);
1362 }
1363 bool isInlineConstant(int64_t ImmVal, uint8_t OperandType) const;
1364
1365 bool isInlineConstant(const MachineOperand &MO,
1366 const MCOperandInfo &OpInfo) const {
1367 return isInlineConstant(MO, OperandType: OpInfo.OperandType);
1368 }
1369
1370 /// \p returns true if \p UseMO is substituted with \p DefMO in \p MI it would
1371 /// be an inline immediate.
1372 bool isInlineConstant(const MachineInstr &MI,
1373 const MachineOperand &UseMO,
1374 const MachineOperand &DefMO) const {
1375 assert(UseMO.getParent() == &MI);
1376 int OpIdx = UseMO.getOperandNo();
1377 if (OpIdx >= MI.getDesc().NumOperands)
1378 return false;
1379
1380 return isInlineConstant(MO: DefMO, OpInfo: MI.getDesc().operands()[OpIdx]);
1381 }
1382
1383 /// \p returns true if the operand \p OpIdx in \p MI is a valid inline
1384 /// immediate.
1385 bool isInlineConstant(const MachineInstr &MI, unsigned OpIdx) const {
1386 const MachineOperand &MO = MI.getOperand(i: OpIdx);
1387 return isInlineConstant(MO, OperandType: MI.getDesc().operands()[OpIdx].OperandType);
1388 }
1389
1390 bool isInlineConstant(const MachineInstr &MI, unsigned OpIdx,
1391 int64_t ImmVal) const {
1392 if (OpIdx >= MI.getDesc().NumOperands)
1393 return false;
1394
1395 if (isCopyInstr(MI)) {
1396 unsigned Size = getOpSize(MI, OpNo: OpIdx);
1397 assert(Size == 8 || Size == 4);
1398
1399 uint8_t OpType = (Size == 8) ?
1400 AMDGPU::OPERAND_REG_IMM_INT64 : AMDGPU::OPERAND_REG_IMM_INT32;
1401 return isInlineConstant(ImmVal, OperandType: OpType);
1402 }
1403
1404 return isInlineConstant(ImmVal, OperandType: MI.getDesc().operands()[OpIdx].OperandType);
1405 }
1406
1407 bool isInlineConstant(const MachineInstr &MI, unsigned OpIdx,
1408 const MachineOperand &MO) const {
1409 return isInlineConstant(MI, OpIdx, ImmVal: MO.getImm());
1410 }
1411
1412 bool isInlineConstant(const MachineOperand &MO) const {
1413 return isInlineConstant(MI: *MO.getParent(), OpIdx: MO.getOperandNo());
1414 }
1415
1416 bool isImmOperandLegal(const MCInstrDesc &InstDesc, unsigned OpNo,
1417 const MachineOperand &MO) const;
1418
1419 bool isLiteralOperandLegal(const MCInstrDesc &InstDesc,
1420 const MCOperandInfo &OpInfo) const;
1421
1422 bool isImmOperandLegal(const MCInstrDesc &InstDesc, unsigned OpNo,
1423 int64_t ImmVal) const;
1424
1425 bool isImmOperandLegal(const MachineInstr &MI, unsigned OpNo,
1426 const MachineOperand &MO) const {
1427 return isImmOperandLegal(InstDesc: MI.getDesc(), OpNo, MO);
1428 }
1429
1430 bool isNeverCoissue(MachineInstr &MI) const;
1431
1432 /// Check if this immediate value can be used for AV_MOV_B64_IMM_PSEUDO.
1433 bool isLegalAV64PseudoImm(uint64_t Imm) const;
1434
1435 /// Return true if this 64-bit VALU instruction has a 32-bit encoding.
1436 /// This function will return false if you pass it a 32-bit instruction.
1437 bool hasVALU32BitEncoding(unsigned Opcode) const;
1438
1439 /// Return true if \p Reg is a lane mask that already has 0 in every bit
1440 /// corresponding to a lane that is inactive in EXEC where \p Use executes,
1441 /// so that ANDing it with EXEC there would be a no-op. Requires SSA form.
1442 bool isMaskedByExec(Register Reg, const MachineInstr &Use,
1443 const MachineRegisterInfo &MRI, unsigned Depth = 0) const;
1444
1445 bool physRegUsesConstantBus(const MachineOperand &Reg) const;
1446 bool regUsesConstantBus(const MachineOperand &Reg,
1447 const MachineRegisterInfo &MRI) const;
1448
1449 /// Returns true if this operand uses the constant bus.
1450 bool usesConstantBus(const MachineRegisterInfo &MRI,
1451 const MachineOperand &MO,
1452 const MCOperandInfo &OpInfo) const;
1453
1454 bool usesConstantBus(const MachineRegisterInfo &MRI, const MachineInstr &MI,
1455 int OpIdx) const {
1456 return usesConstantBus(MRI, MO: MI.getOperand(i: OpIdx),
1457 OpInfo: MI.getDesc().operands()[OpIdx]);
1458 }
1459
1460 /// Return true if this instruction has any modifiers.
1461 /// e.g. src[012]_mod, omod, clamp.
1462 bool hasModifiers(unsigned Opcode) const;
1463
1464 bool hasModifiersSet(const MachineInstr &MI, AMDGPU::OpName OpName) const;
1465 bool hasAnyModifiersSet(const MachineInstr &MI) const;
1466
1467 bool canShrink(const MachineInstr &MI,
1468 const MachineRegisterInfo &MRI) const;
1469
1470 MachineInstr *buildShrunkInst(MachineInstr &MI,
1471 unsigned NewOpcode) const;
1472
1473 bool verifyInstruction(const MachineInstr &MI,
1474 StringRef &ErrInfo) const override;
1475
1476 unsigned getVALUOp(const MachineInstr &MI) const;
1477 unsigned getVALUOp(unsigned Opc) const;
1478
1479 void insertScratchExecCopy(MachineFunction &MF, MachineBasicBlock &MBB,
1480 MachineBasicBlock::iterator MBBI,
1481 const DebugLoc &DL, Register Reg, bool IsSCCLive,
1482 SlotIndexes *Indexes = nullptr) const;
1483
1484 void restoreExec(MachineFunction &MF, MachineBasicBlock &MBB,
1485 MachineBasicBlock::iterator MBBI, const DebugLoc &DL,
1486 Register Reg, SlotIndexes *Indexes = nullptr) const;
1487
1488 MachineInstr *getWholeWaveFunctionSetup(MachineFunction &MF) const;
1489
1490 /// Return the correct register class for \p OpNo. For target-specific
1491 /// instructions, this will return the register class that has been defined
1492 /// in tablegen. For generic instructions, like REG_SEQUENCE it will return
1493 /// the register class of its machine operand.
1494 /// to infer the correct register class base on the other operands.
1495 const TargetRegisterClass *getOpRegClass(const MachineInstr &MI,
1496 unsigned OpNo) const;
1497
1498 /// Return the size in bytes of the operand OpNo on the given
1499 // instruction opcode.
1500 unsigned getOpSize(uint32_t Opcode, unsigned OpNo) const {
1501 const MCOperandInfo &OpInfo = get(Opcode).operands()[OpNo];
1502
1503 if (OpInfo.RegClass == -1) {
1504 // If this is an immediate operand, this must be a 32-bit literal.
1505 assert(OpInfo.OperandType == MCOI::OPERAND_IMMEDIATE);
1506 return 4;
1507 }
1508
1509 return RI.getRegSizeInBits(RC: *RI.getRegClass(i: getOpRegClassID(OpInfo))) / 8;
1510 }
1511
1512 /// This form should usually be preferred since it handles operands
1513 /// with unknown register classes.
1514 unsigned getOpSize(const MachineInstr &MI, unsigned OpNo) const {
1515 const MachineOperand &MO = MI.getOperand(i: OpNo);
1516 if (MO.isReg()) {
1517 if (unsigned SubReg = MO.getSubReg()) {
1518 return RI.getSubRegIdxSize(Idx: SubReg) / 8;
1519 }
1520 }
1521 return RI.getRegSizeInBits(RC: *getOpRegClass(MI, OpNo)) / 8;
1522 }
1523
1524 /// Legalize the \p OpIndex operand of this instruction by inserting
1525 /// a MOV. For example:
1526 /// ADD_I32_e32 VGPR0, 15
1527 /// to
1528 /// MOV VGPR1, 15
1529 /// ADD_I32_e32 VGPR0, VGPR1
1530 ///
1531 /// If the operand being legalized is a register, then a COPY will be used
1532 /// instead of MOV.
1533 void legalizeOpWithMove(MachineInstr &MI, unsigned OpIdx) const;
1534
1535 /// Check if \p MO is a legal operand if it was the \p OpIdx Operand
1536 /// for \p MI.
1537 bool isOperandLegal(const MachineInstr &MI, unsigned OpIdx,
1538 const MachineOperand *MO = nullptr) const;
1539
1540 /// Check if \p MO (a register operand) is a legal register for the
1541 /// given operand description or operand index.
1542 /// The operand index version provide more legality checks
1543 bool isLegalRegOperand(const MachineRegisterInfo &MRI,
1544 const MCOperandInfo &OpInfo,
1545 const MachineOperand &MO) const;
1546 bool isLegalRegOperand(const MachineInstr &MI, unsigned OpIdx,
1547 const MachineOperand &MO) const;
1548
1549 /// Check if \p MO would be a legal operand for a single-SGPR-read
1550 /// instruction.
1551 ///
1552 /// Single-SGPR-read instructions typically accept VGPRs, SGPRs, or immediates
1553 /// as source operands. On gfx12+, if a source operand uses SGPRs, the HW can
1554 /// only read the first SGPR and replicate the value across all lanes. \p SrcN
1555 /// can be 0, 1, or 2, representing src0, src1, and src2, respectively. If \p
1556 /// MO is nullptr, the operand corresponding to \p SrcN will be used. Non-SGPR
1557 /// operands are always considered legal.
1558 bool
1559 isLegalSingleSGPRReadInstOperand(const MachineRegisterInfo &MRI,
1560 const MachineInstr &MI, unsigned SrcN,
1561 const MachineOperand *MO = nullptr) const;
1562
1563 /// Legalize operands in \p MI by either commuting it or inserting a
1564 /// copy of src1.
1565 void legalizeOperandsVOP2(MachineRegisterInfo &MRI, MachineInstr &MI) const;
1566
1567 /// Fix operands in \p MI to satisfy constant bus requirements.
1568 void legalizeOperandsVOP3(MachineRegisterInfo &MRI, MachineInstr &MI) const;
1569
1570 /// Copy a value from a VGPR (\p SrcReg) to SGPR. The desired register class
1571 /// for the dst register (\p DstRC) can be optionally supplied. This function
1572 /// can only be used when it is know that the value in SrcReg is same across
1573 /// all threads in the wave.
1574 /// \returns The SGPR register that \p SrcReg was copied to.
1575 Register readlaneVGPRToSGPR(Register SrcReg, MachineInstr &UseMI,
1576 MachineRegisterInfo &MRI,
1577 const TargetRegisterClass *DstRC = nullptr) const;
1578
1579 void legalizeOperandsSMRD(MachineRegisterInfo &MRI, MachineInstr &MI) const;
1580 void legalizeOperandsFLAT(MachineRegisterInfo &MRI, MachineInstr &MI) const;
1581
1582 void legalizeGenericOperand(MachineBasicBlock &InsertMBB,
1583 MachineBasicBlock::iterator I,
1584 const TargetRegisterClass *DstRC,
1585 MachineOperand &Op, MachineRegisterInfo &MRI,
1586 const DebugLoc &DL) const;
1587
1588 /// Legalize all operands in this instruction. This function may create new
1589 /// instructions and control-flow around \p MI. If present, \p MDT is
1590 /// updated.
1591 /// \returns A new basic block that contains \p MI if new blocks were created.
1592 MachineBasicBlock *
1593 legalizeOperands(MachineInstr &MI, MachineDominatorTree *MDT = nullptr) const;
1594
1595 /// Change SADDR form of a FLAT \p Inst to its VADDR form if saddr operand
1596 /// was moved to VGPR. \returns true if succeeded.
1597 bool moveFlatAddrToVGPR(MachineInstr &Inst) const;
1598
1599 /// Fix operands in Inst to fix 16bit SALU to VALU lowering.
1600 void legalizeOperandsVALUt16(MachineInstr &Inst,
1601 MachineRegisterInfo &MRI) const;
1602 void legalizeOperandsVALUt16(MachineInstr &Inst, unsigned OpIdx,
1603 MachineRegisterInfo &MRI) const;
1604
1605 /// Replace the instructions opcode with the equivalent VALU
1606 /// opcode. This function will also move the users of MachineInstruntions
1607 /// in the \p WorkList to the VALU if necessary. If present, \p MDT is
1608 /// updated.
1609 void moveToVALU(SIInstrWorklist &Worklist, MachineDominatorTree *MDT) const;
1610
1611 void
1612 moveToVALUImpl(SIInstrWorklist &Worklist, MachineDominatorTree *MDT,
1613 MachineInstr &Inst,
1614 DenseMap<MachineInstr *, V2PhysSCopyInfo> &WaterFalls,
1615 DenseMap<MachineInstr *, bool> &V2SPhyCopiesToErase) const;
1616 /// Wrapper function for generating waterfall for instruction \p MI
1617 /// This function take into consideration of related pre & succ instructions
1618 /// (e.g. calling process) into consideratioin
1619 void createWaterFallForSiCall(MachineInstr *MI, MachineDominatorTree *MDT,
1620 ArrayRef<MachineOperand *> ScalarOps,
1621 ArrayRef<Register> PhySGPRs = {}) const;
1622
1623 void insertNoop(MachineBasicBlock &MBB,
1624 MachineBasicBlock::iterator MI) const override;
1625
1626 void insertNoops(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI,
1627 unsigned Quantity) const override;
1628
1629 /// Build instructions that simulate the behavior of a `s_trap 2` instructions
1630 /// for hardware (namely, gfx11) that runs in PRIV=1 mode. There, s_trap is
1631 /// interpreted as a nop.
1632 MachineBasicBlock *insertSimulatedTrap(MachineRegisterInfo &MRI,
1633 MachineBasicBlock &MBB,
1634 MachineInstr &MI,
1635 const DebugLoc &DL) const;
1636
1637 /// Return the number of wait states that result from executing this
1638 /// instruction.
1639 static unsigned getNumWaitStates(const MachineInstr &MI);
1640
1641 /// Returns the operand named \p Op. If \p MI does not have an
1642 /// operand named \c Op, this function returns nullptr.
1643 LLVM_READONLY
1644 MachineOperand *getNamedOperand(MachineInstr &MI,
1645 AMDGPU::OpName OperandName) const;
1646
1647 LLVM_READONLY
1648 const MachineOperand *getNamedOperand(const MachineInstr &MI,
1649 AMDGPU::OpName OperandName) const {
1650 return getNamedOperand(MI&: const_cast<MachineInstr &>(MI), OperandName);
1651 }
1652
1653 /// Get required immediate operand
1654 int64_t getNamedImmOperand(const MachineInstr &MI,
1655 AMDGPU::OpName OperandName) const {
1656 int Idx = AMDGPU::getNamedOperandIdx(Opcode: MI.getOpcode(), Name: OperandName);
1657 return MI.getOperand(i: Idx).getImm();
1658 }
1659
1660 uint64_t getDefaultRsrcDataFormat() const;
1661 uint64_t getScratchRsrcWords23() const;
1662
1663 bool isLowLatencyInstruction(const MachineInstr &MI) const;
1664 bool isHighLatencyDef(int Opc) const override;
1665
1666 /// Return the descriptor of the target-specific machine instruction
1667 /// that corresponds to the specified pseudo or native opcode.
1668 const MCInstrDesc &getMCOpcodeFromPseudo(unsigned Opcode) const {
1669 return get(Opcode: pseudoToMCOpcode(Opcode));
1670 }
1671
1672 Register isStackAccess(const MachineInstr &MI, int &FrameIndex,
1673 TypeSize &MemBytes) const;
1674 Register isSGPRStackAccess(const MachineInstr &MI, int &FrameIndex,
1675 TypeSize &MemBytes) const;
1676
1677 Register isLoadFromStackSlot(const MachineInstr &MI,
1678 int &FrameIndex) const override {
1679 TypeSize MemBytes = TypeSize::getZero();
1680 return isLoadFromStackSlot(MI, FrameIndex, MemBytes);
1681 }
1682
1683 Register isLoadFromStackSlot(const MachineInstr &MI, int &FrameIndex,
1684 TypeSize &MemBytes) const override;
1685
1686 Register isStoreToStackSlot(const MachineInstr &MI,
1687 int &FrameIndex) const override {
1688 TypeSize MemBytes = TypeSize::getZero();
1689 return isStoreToStackSlot(MI, FrameIndex, MemBytes);
1690 }
1691
1692 Register isStoreToStackSlot(const MachineInstr &MI, int &FrameIndex,
1693 TypeSize &MemBytes) const override;
1694
1695 unsigned getInstSizeInBytes(const MachineInstr &MI) const override;
1696
1697 InstSizeVerifyMode
1698 getInstSizeVerifyMode(const MachineInstr &MI) const override;
1699
1700 std::pair<unsigned, unsigned>
1701 decomposeMachineOperandsTargetFlags(unsigned TF) const override;
1702
1703 ArrayRef<std::pair<int, const char *>>
1704 getSerializableTargetIndices() const override;
1705
1706 ArrayRef<std::pair<unsigned, const char *>>
1707 getSerializableDirectMachineOperandTargetFlags() const override;
1708
1709 ArrayRef<std::pair<MachineMemOperand::Flags, const char *>>
1710 getSerializableMachineMemOperandTargetFlags() const override;
1711
1712 ScheduleHazardRecognizer *
1713 CreateTargetPostRAHazardRecognizer(const InstrItineraryData *II,
1714 const ScheduleDAG *DAG) const override;
1715
1716 ScheduleHazardRecognizer *
1717 CreateTargetPostRAHazardRecognizer(const MachineFunction &MF,
1718 MachineLoopInfo *MLI) const override;
1719
1720 ScheduleHazardRecognizer *
1721 CreateTargetMIHazardRecognizer(const InstrItineraryData *II,
1722 const ScheduleDAGMI *DAG) const override;
1723
1724 unsigned getLiveRangeSplitOpcode(Register Reg,
1725 const MachineFunction &MF) const override;
1726
1727 bool isBasicBlockPrologue(const MachineInstr &MI,
1728 Register Reg = Register()) const override;
1729
1730 bool canAddToBBProlog(const MachineInstr &MI) const;
1731
1732 MachineInstr *createPHIDestinationCopy(MachineBasicBlock &MBB,
1733 MachineBasicBlock::iterator InsPt,
1734 const DebugLoc &DL, Register Src,
1735 Register Dst) const override;
1736
1737 MachineInstr *createPHISourceCopy(MachineBasicBlock &MBB,
1738 MachineBasicBlock::iterator InsPt,
1739 const DebugLoc &DL, Register Src,
1740 unsigned SrcSubReg,
1741 Register Dst) const override;
1742
1743 bool isWave32() const;
1744
1745 bool isVOPDAntidependencyAllowed(const MachineInstr &MI) const;
1746
1747 bool hasRAWDependency(const MachineInstr &FirstMI,
1748 const MachineInstr &SecondMI) const;
1749
1750 /// Return a partially built integer add instruction without carry.
1751 /// Caller must add source operands.
1752 /// For pre-GFX9 it will generate unused carry destination operand.
1753 /// TODO: After GFX9 it should return a no-carry operation.
1754 MachineInstrBuilder getAddNoCarry(MachineBasicBlock &MBB,
1755 MachineBasicBlock::iterator I,
1756 const DebugLoc &DL,
1757 Register DestReg) const;
1758
1759 MachineInstrBuilder getAddNoCarry(MachineBasicBlock &MBB,
1760 MachineBasicBlock::iterator I,
1761 const DebugLoc &DL,
1762 Register DestReg,
1763 RegScavenger &RS) const;
1764
1765 static bool isKillTerminator(unsigned Opcode);
1766 const MCInstrDesc &getKillTerminatorFromPseudo(unsigned Opcode) const;
1767
1768 bool isLegalMUBUFImmOffset(unsigned Imm) const;
1769
1770 static unsigned getMaxMUBUFImmOffset(const GCNSubtarget &ST);
1771
1772 bool splitMUBUFOffset(uint32_t Imm, uint32_t &SOffset, uint32_t &ImmOffset,
1773 Align Alignment = Align(4)) const;
1774
1775 /// Returns if \p Offset is legal for the subtarget as the offset to a FLAT
1776 /// encoded instruction with the given \p FlatVariant.
1777 bool isLegalFLATOffset(int64_t Offset, unsigned AddrSpace,
1778 AMDGPU::FlatAddrSpace FlatVariant) const;
1779
1780 /// Split \p COffsetVal into {immediate offset field, remainder offset}
1781 /// values.
1782 std::pair<int64_t, int64_t>
1783 splitFlatOffset(int64_t COffsetVal, unsigned AddrSpace,
1784 AMDGPU::FlatAddrSpace FlatVariant) const;
1785
1786 /// Returns true if negative offsets are allowed for the given \p FlatVariant.
1787 bool allowNegativeFlatOffset(AMDGPU::FlatAddrSpace FlatVariant) const;
1788
1789 /// \brief Return a target-specific opcode if Opcode is a pseudo instruction.
1790 /// Return -1 if the target-specific opcode for the pseudo instruction does
1791 /// not exist. If Opcode is not a pseudo instruction, this is identity.
1792 int pseudoToMCOpcode(int Opcode) const;
1793
1794 /// \brief Check if this instruction should only be used by assembler.
1795 /// Return true if this opcode should not be used by codegen.
1796 bool isAsmOnlyOpcode(int MCOp) const;
1797
1798 void fixImplicitOperands(MachineInstr &MI) const;
1799
1800 MachineInstr *foldMemoryOperandImpl(MachineFunction &MF, MachineInstr &MI,
1801 ArrayRef<unsigned> Ops, int FrameIndex,
1802 MachineInstr *&CopyMI,
1803 LiveIntervals *LIS = nullptr,
1804 VirtRegMap *VRM = nullptr) const override;
1805
1806 unsigned getInstrLatency(const InstrItineraryData *ItinData,
1807 const MachineInstr &MI,
1808 unsigned *PredCost = nullptr) const override;
1809
1810 unsigned getBlockingCycles(const MachineInstr &MI) const;
1811
1812 /// GFX1250 blocking-cycles table lookup with no occupancy subtarget gate.
1813 /// Returns 0 if \p MI is not in the table. Used as a multi-pass VALU denylist
1814 /// (e.g. V_PERM_PK16 hazard) on both gfx1250 and gfx1251.
1815 unsigned getGFX1250BlockingCyclesTable(const MachineInstr &MI) const;
1816
1817 const MachineOperand &getCalleeOperand(const MachineInstr &MI) const override;
1818
1819 ValueUniformity getValueUniformity(const MachineInstr &MI) const final;
1820
1821 ValueUniformity getGenericValueUniformity(const MachineInstr &MI) const;
1822
1823 const MIRFormatter *getMIRFormatter() const override;
1824
1825 static unsigned getDSShaderTypeValue(const MachineFunction &MF);
1826
1827 const TargetSchedModel &getSchedModel() const { return SchedModel; }
1828
1829 void createReadFirstLaneFromCopyToPhysReg(MachineRegisterInfo &MRI,
1830 Register DstReg,
1831 MachineInstr &Inst) const;
1832
1833 void handleCopyToPhysHelper(
1834 SIInstrWorklist &Worklist, Register DstReg, MachineInstr &Inst,
1835 MachineRegisterInfo &MRI,
1836 DenseMap<MachineInstr *, V2PhysSCopyInfo> &WaterFalls,
1837 DenseMap<MachineInstr *, bool> &V2SPhyCopiesToErase) const;
1838
1839 // FIXME: This should be removed
1840 // Enforce operand's \p OpName even alignment if required by target.
1841 // This is used if an operand is a 32 bit register but needs to be aligned
1842 // regardless.
1843 void enforceOperandRCAlignment(MachineInstr &MI, AMDGPU::OpName OpName) const;
1844
1845 /// Get the repeat rate for a VALU instruction from the scheduling model.
1846 /// Returns 1 for regular VALU, >1 for long-latency VALU (packed, F64, etc.)
1847 unsigned getRepeatRate(const MachineInstr &MI) const;
1848};
1849
1850/// \brief Returns true if a reg:subreg pair P has a TRC class
1851inline bool isOfRegClass(const TargetInstrInfo::RegSubRegPair &P,
1852 const TargetRegisterClass &TRC,
1853 MachineRegisterInfo &MRI) {
1854 auto *RC = MRI.getRegClass(Reg: P.Reg);
1855 if (!P.SubReg)
1856 return RC == &TRC;
1857 auto *TRI = MRI.getTargetRegisterInfo();
1858 return RC == TRI->getMatchingSuperRegClass(A: RC, B: &TRC, Idx: P.SubReg);
1859}
1860
1861/// \brief Create RegSubRegPair from a register MachineOperand
1862inline
1863TargetInstrInfo::RegSubRegPair getRegSubRegPair(const MachineOperand &O) {
1864 assert(O.isReg());
1865 return TargetInstrInfo::RegSubRegPair(O.getReg(), O.getSubReg());
1866}
1867
1868/// \brief Return the SubReg component from REG_SEQUENCE
1869TargetInstrInfo::RegSubRegPair getRegSequenceSubReg(MachineInstr &MI,
1870 unsigned SubReg);
1871
1872/// \brief Return the defining instruction for a given reg:subreg pair
1873/// skipping copy like instructions and subreg-manipulation pseudos.
1874/// Following another subreg of a reg:subreg isn't supported.
1875MachineInstr *getVRegSubRegDef(const TargetInstrInfo::RegSubRegPair &P,
1876 const MachineRegisterInfo &MRI);
1877
1878/// \brief Return false if EXEC is not changed between the def of \p VReg at \p
1879/// DefMI and the use at \p UseMI. Should be run on SSA. Currently does not
1880/// attempt to track between blocks.
1881bool execMayBeModifiedBeforeUse(const MachineRegisterInfo &MRI,
1882 Register VReg,
1883 const MachineInstr &DefMI,
1884 const MachineInstr &UseMI);
1885
1886/// \brief Return false if EXEC is not changed between the def of \p VReg at \p
1887/// DefMI and all its uses. Should be run on SSA. Currently does not attempt to
1888/// track between blocks.
1889bool execMayBeModifiedBeforeAnyUse(const MachineRegisterInfo &MRI,
1890 Register VReg,
1891 const MachineInstr &DefMI);
1892
1893namespace AMDGPU {
1894
1895 LLVM_READONLY
1896 int32_t getVOPe64(uint32_t Opcode);
1897
1898 LLVM_READONLY
1899 int32_t getVOPe32(uint32_t Opcode);
1900
1901 LLVM_READONLY
1902 int32_t getSDWAOp(uint32_t Opcode);
1903
1904 LLVM_READONLY
1905 int32_t getDPPOp32(uint32_t Opcode);
1906
1907 LLVM_READONLY
1908 int32_t getDPPOp64(uint32_t Opcode);
1909
1910 LLVM_READONLY
1911 int32_t getBasicFromSDWAOp(uint32_t Opcode);
1912
1913 LLVM_READONLY
1914 int32_t getCommuteRev(uint32_t Opcode);
1915
1916 LLVM_READONLY
1917 int32_t getCommuteOrig(uint32_t Opcode);
1918
1919 LLVM_READONLY
1920 int32_t getAddr64Inst(uint32_t Opcode);
1921
1922 /// Check if \p Opcode is an Addr64 opcode.
1923 ///
1924 /// \returns \p Opcode if it is an Addr64 opcode, otherwise -1.
1925 LLVM_READONLY
1926 int32_t getIfAddr64Inst(uint32_t Opcode);
1927
1928 LLVM_READONLY
1929 int32_t getSOPKOp(uint32_t Opcode);
1930
1931 /// \returns SADDR form of a FLAT Global instruction given an \p Opcode
1932 /// of a VADDR form.
1933 LLVM_READONLY
1934 int32_t getGlobalSaddrOp(uint32_t Opcode);
1935
1936 /// \returns VADDR form of a FLAT Global instruction given an \p Opcode
1937 /// of a SADDR form.
1938 LLVM_READONLY
1939 int32_t getGlobalVaddrOp(uint32_t Opcode);
1940
1941 /// \returns ST form with only immediate offset of a FLAT Scratch instruction
1942 /// given an \p Opcode of an SS (SADDR) form.
1943 LLVM_READONLY
1944 int32_t getFlatScratchInstSTfromSS(uint32_t Opcode);
1945
1946 /// \returns SV (VADDR) form of a FLAT Scratch instruction given an \p Opcode
1947 /// of an SVS (SADDR + VADDR) form.
1948 LLVM_READONLY
1949 int32_t getFlatScratchInstSVfromSVS(uint32_t Opcode);
1950
1951 /// \returns SS (SADDR) form of a FLAT Scratch instruction given an \p Opcode
1952 /// of an SV (VADDR) form.
1953 LLVM_READONLY
1954 int32_t getFlatScratchInstSSfromSV(uint32_t Opcode);
1955
1956 /// \returns SV (VADDR) form of a FLAT Scratch instruction given an \p Opcode
1957 /// of an SS (SADDR) form.
1958 LLVM_READONLY
1959 int32_t getFlatScratchInstSVfromSS(uint32_t Opcode);
1960
1961 /// \returns earlyclobber version of a MAC MFMA is exists.
1962 LLVM_READONLY
1963 int32_t getMFMAEarlyClobberOp(uint32_t Opcode);
1964
1965 /// \returns Version of an instruction which uses AGPRs for coupled operands
1966 /// given an \p Opcode which uses VGPRs for coupled operands.
1967 LLVM_READONLY
1968 int32_t getAGPRFormOp(uint32_t Opcode);
1969
1970 /// \returns v_cmpx version of a v_cmp instruction.
1971 LLVM_READONLY
1972 int32_t getVCMPXOpFromVCMP(uint32_t Opcode);
1973
1974 const uint64_t RSRC_DATA_FORMAT = 0xf00000000000LL;
1975 const uint64_t RSRC_ELEMENT_SIZE_SHIFT = (32 + 19);
1976 const uint64_t RSRC_INDEX_STRIDE_SHIFT = (32 + 21);
1977 const uint64_t RSRC_TID_ENABLE = UINT64_C(1) << (32 + 23);
1978
1979} // end namespace AMDGPU
1980
1981namespace AMDGPU {
1982enum AsmComments : MachineInstr::AsmPrinterFlagTy {
1983 // For sgpr to vgpr spill instructions
1984 SGPR_SPILL = MachineInstr::TAsmComments
1985};
1986} // namespace AMDGPU
1987
1988namespace SI {
1989namespace KernelInputOffsets {
1990
1991/// Offsets in bytes from the start of the input buffer
1992enum Offsets {
1993 NGROUPS_X = 0,
1994 NGROUPS_Y = 4,
1995 NGROUPS_Z = 8,
1996 GLOBAL_SIZE_X = 12,
1997 GLOBAL_SIZE_Y = 16,
1998 GLOBAL_SIZE_Z = 20,
1999 LOCAL_SIZE_X = 24,
2000 LOCAL_SIZE_Y = 28,
2001 LOCAL_SIZE_Z = 32
2002};
2003
2004} // end namespace KernelInputOffsets
2005} // end namespace SI
2006
2007} // end namespace llvm
2008
2009#endif // LLVM_LIB_TARGET_AMDGPU_SIINSTRINFO_H
2010