1//===- AArch64InstrInfo.h - AArch64 Instruction Information -----*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file contains the AArch64 implementation of the TargetInstrInfo class.
10//
11//===----------------------------------------------------------------------===//
12
13#ifndef LLVM_LIB_TARGET_AARCH64_AARCH64INSTRINFO_H
14#define LLVM_LIB_TARGET_AARCH64_AARCH64INSTRINFO_H
15
16#include "AArch64.h"
17#include "AArch64RegisterInfo.h"
18#include "llvm/CodeGen/TargetInstrInfo.h"
19#include "llvm/Support/TypeSize.h"
20#include <optional>
21
22#define GET_INSTRINFO_HEADER
23#include "AArch64GenInstrInfo.inc"
24
25namespace llvm {
26
27class AArch64Subtarget;
28
29static const MachineMemOperand::Flags MOSuppressPair =
30 MachineMemOperand::MOTargetFlag1;
31static const MachineMemOperand::Flags MOStridedAccess =
32 MachineMemOperand::MOTargetFlag2;
33
34#define FALKOR_STRIDED_ACCESS_MD "falkor.strided.access"
35
36// AArch64 MachineCombiner patterns
37enum AArch64MachineCombinerPattern : unsigned {
38 // These are patterns used to reduce the length of dependence chain.
39 SUBADD_OP1 = MachineCombinerPattern::TARGET_PATTERN_START,
40 SUBADD_OP2,
41
42 // These are multiply-add patterns matched by the AArch64 machine combiner.
43 MULADDW_OP1,
44 MULADDW_OP2,
45 MULSUBW_OP1,
46 MULSUBW_OP2,
47 MULADDWI_OP1,
48 MULSUBWI_OP1,
49 MULADDX_OP1,
50 MULADDX_OP2,
51 MULSUBX_OP1,
52 MULSUBX_OP2,
53 MULADDXI_OP1,
54 MULSUBXI_OP1,
55 // NEON integers vectors
56 MULADDv8i8_OP1,
57 MULADDv8i8_OP2,
58 MULADDv16i8_OP1,
59 MULADDv16i8_OP2,
60 MULADDv4i16_OP1,
61 MULADDv4i16_OP2,
62 MULADDv8i16_OP1,
63 MULADDv8i16_OP2,
64 MULADDv2i32_OP1,
65 MULADDv2i32_OP2,
66 MULADDv4i32_OP1,
67 MULADDv4i32_OP2,
68
69 MULSUBv8i8_OP1,
70 MULSUBv8i8_OP2,
71 MULSUBv16i8_OP1,
72 MULSUBv16i8_OP2,
73 MULSUBv4i16_OP1,
74 MULSUBv4i16_OP2,
75 MULSUBv8i16_OP1,
76 MULSUBv8i16_OP2,
77 MULSUBv2i32_OP1,
78 MULSUBv2i32_OP2,
79 MULSUBv4i32_OP1,
80 MULSUBv4i32_OP2,
81
82 MULADDv4i16_indexed_OP1,
83 MULADDv4i16_indexed_OP2,
84 MULADDv8i16_indexed_OP1,
85 MULADDv8i16_indexed_OP2,
86 MULADDv2i32_indexed_OP1,
87 MULADDv2i32_indexed_OP2,
88 MULADDv4i32_indexed_OP1,
89 MULADDv4i32_indexed_OP2,
90
91 MULSUBv4i16_indexed_OP1,
92 MULSUBv4i16_indexed_OP2,
93 MULSUBv8i16_indexed_OP1,
94 MULSUBv8i16_indexed_OP2,
95 MULSUBv2i32_indexed_OP1,
96 MULSUBv2i32_indexed_OP2,
97 MULSUBv4i32_indexed_OP1,
98 MULSUBv4i32_indexed_OP2,
99
100 // Floating Point
101 FMULADDH_OP1,
102 FMULADDH_OP2,
103 FMULSUBH_OP1,
104 FMULSUBH_OP2,
105 FMULADDS_OP1,
106 FMULADDS_OP2,
107 FMULSUBS_OP1,
108 FMULSUBS_OP2,
109 FMULADDD_OP1,
110 FMULADDD_OP2,
111 FMULSUBD_OP1,
112 FMULSUBD_OP2,
113 FNMULSUBH_OP1,
114 FNMULSUBS_OP1,
115 FNMULSUBD_OP1,
116 FMLAv1i32_indexed_OP1,
117 FMLAv1i32_indexed_OP2,
118 FMLAv1i64_indexed_OP1,
119 FMLAv1i64_indexed_OP2,
120 FMLAv4f16_OP1,
121 FMLAv4f16_OP2,
122 FMLAv8f16_OP1,
123 FMLAv8f16_OP2,
124 FMLAv2f32_OP2,
125 FMLAv2f32_OP1,
126 FMLAv2f64_OP1,
127 FMLAv2f64_OP2,
128 FMLAv4i16_indexed_OP1,
129 FMLAv4i16_indexed_OP2,
130 FMLAv8i16_indexed_OP1,
131 FMLAv8i16_indexed_OP2,
132 FMLAv2i32_indexed_OP1,
133 FMLAv2i32_indexed_OP2,
134 FMLAv2i64_indexed_OP1,
135 FMLAv2i64_indexed_OP2,
136 FMLAv4f32_OP1,
137 FMLAv4f32_OP2,
138 FMLAv4i32_indexed_OP1,
139 FMLAv4i32_indexed_OP2,
140 FMLSv1i32_indexed_OP2,
141 FMLSv1i64_indexed_OP2,
142 FMLSv4f16_OP1,
143 FMLSv4f16_OP2,
144 FMLSv8f16_OP1,
145 FMLSv8f16_OP2,
146 FMLSv2f32_OP1,
147 FMLSv2f32_OP2,
148 FMLSv2f64_OP1,
149 FMLSv2f64_OP2,
150 FMLSv4i16_indexed_OP1,
151 FMLSv4i16_indexed_OP2,
152 FMLSv8i16_indexed_OP1,
153 FMLSv8i16_indexed_OP2,
154 FMLSv2i32_indexed_OP1,
155 FMLSv2i32_indexed_OP2,
156 FMLSv2i64_indexed_OP1,
157 FMLSv2i64_indexed_OP2,
158 FMLSv4f32_OP1,
159 FMLSv4f32_OP2,
160 FMLSv4i32_indexed_OP1,
161 FMLSv4i32_indexed_OP2,
162
163 FMULv2i32_indexed_OP1,
164 FMULv2i32_indexed_OP2,
165 FMULv2i64_indexed_OP1,
166 FMULv2i64_indexed_OP2,
167 FMULv4i16_indexed_OP1,
168 FMULv4i16_indexed_OP2,
169 FMULv4i32_indexed_OP1,
170 FMULv4i32_indexed_OP2,
171 FMULv8i16_indexed_OP1,
172 FMULv8i16_indexed_OP2,
173
174 FNMADD,
175
176 GATHER_LANE_i32,
177 GATHER_LANE_i16,
178 GATHER_LANE_i8
179};
180class AArch64InstrInfo final : public AArch64GenInstrInfo {
181 const AArch64RegisterInfo RI;
182 const AArch64Subtarget &Subtarget;
183
184public:
185 explicit AArch64InstrInfo(const AArch64Subtarget &STI);
186
187 /// getRegisterInfo - TargetInstrInfo is a superset of MRegister info. As
188 /// such, whenever a client has an instance of instruction info, it should
189 /// always be able to get register info as well (through this method).
190 const AArch64RegisterInfo &getRegisterInfo() const { return RI; }
191
192 unsigned getInstSizeInBytes(const MachineInstr &MI) const override;
193
194 bool isAsCheapAsAMove(const MachineInstr &MI) const override;
195
196 bool isCoalescableExtInstr(const MachineInstr &MI, Register &SrcReg,
197 Register &DstReg, unsigned &SubIdx) const override;
198
199 bool
200 areMemAccessesTriviallyDisjoint(const MachineInstr &MIa,
201 const MachineInstr &MIb) const override;
202
203 Register isLoadFromStackSlot(const MachineInstr &MI,
204 int &FrameIndex) const override;
205 Register isStoreToStackSlot(const MachineInstr &MI,
206 int &FrameIndex) const override;
207
208 /// Check for post-frame ptr elimination stack locations as well. This uses a
209 /// heuristic so it isn't reliable for correctness.
210 Register isStoreToStackSlotPostFE(const MachineInstr &MI,
211 int &FrameIndex) const override;
212 /// Check for post-frame ptr elimination stack locations as well. This uses a
213 /// heuristic so it isn't reliable for correctness.
214 Register isLoadFromStackSlotPostFE(const MachineInstr &MI,
215 int &FrameIndex) const override;
216
217 /// Does this instruction set its full destination register to zero?
218 static bool isGPRZero(const MachineInstr &MI);
219
220 /// Does this instruction rename a GPR without modifying bits?
221 static bool isGPRCopy(const MachineInstr &MI);
222
223 /// Does this instruction rename an FPR without modifying bits?
224 static bool isFPRCopy(const MachineInstr &MI);
225
226 /// Return true if pairing the given load or store is hinted to be
227 /// unprofitable.
228 static bool isLdStPairSuppressed(const MachineInstr &MI);
229
230 /// Return true if the given load or store is a strided memory access.
231 static bool isStridedAccess(const MachineInstr &MI);
232
233 /// Return true if it has an unscaled load/store offset.
234 static bool hasUnscaledLdStOffset(unsigned Opc);
235 static bool hasUnscaledLdStOffset(MachineInstr &MI) {
236 return hasUnscaledLdStOffset(Opc: MI.getOpcode());
237 }
238
239 /// Returns the unscaled load/store for the scaled load/store opcode,
240 /// if there is a corresponding unscaled variant available.
241 static std::optional<unsigned> getUnscaledLdSt(unsigned Opc);
242
243 /// Scaling factor for (scaled or unscaled) load or store.
244 static int getMemScale(unsigned Opc);
245 static int getMemScale(const MachineInstr &MI) {
246 return getMemScale(Opc: MI.getOpcode());
247 }
248
249 /// Returns whether the instruction is a pre-indexed load.
250 static bool isPreLd(const MachineInstr &MI);
251
252 /// Returns whether the instruction is a pre-indexed store.
253 static bool isPreSt(const MachineInstr &MI);
254
255 /// Returns whether the instruction is a pre-indexed load/store.
256 static bool isPreLdSt(const MachineInstr &MI);
257
258 /// Returns whether the instruction is a zero-extending load.
259 static bool isZExtLoad(const MachineInstr &MI);
260
261 /// Returns whether the instruction is a sign-extending load.
262 static bool isSExtLoad(const MachineInstr &MI);
263
264 /// Returns whether the instruction is a paired load/store.
265 static bool isPairedLdSt(const MachineInstr &MI);
266
267 /// Returns the base register operator of a load/store.
268 static const MachineOperand &getLdStBaseOp(const MachineInstr &MI);
269
270 /// Returns the immediate offset operator of a load/store.
271 static const MachineOperand &getLdStOffsetOp(const MachineInstr &MI);
272
273 /// Returns whether the physical register is FP or NEON.
274 static bool isFpOrNEON(Register Reg);
275
276 /// Returns the shift amount operator of a load/store.
277 static const MachineOperand &getLdStAmountOp(const MachineInstr &MI);
278
279 /// Returns whether the instruction is FP or NEON.
280 static bool isFpOrNEON(const MachineInstr &MI);
281
282 /// Returns whether the instruction is in H form (16 bit operands)
283 static bool isHForm(const MachineInstr &MI);
284
285 /// Returns whether the instruction is in Q form (128 bit operands)
286 static bool isQForm(const MachineInstr &MI);
287
288 /// Returns whether the instruction can be compatible with non-zero BTYPE.
289 static bool hasBTISemantics(const MachineInstr &MI);
290
291 /// Returns the index for the immediate for a given instruction.
292 static unsigned getLoadStoreImmIdx(unsigned Opc);
293
294 /// Return true if pairing the given load or store may be paired with another.
295 static bool isPairableLdStInst(const MachineInstr &MI);
296
297 /// Returns true if MI is one of the TCRETURN* instructions.
298 static bool isTailCallReturnInst(const MachineInstr &MI);
299
300 /// Return the opcode that set flags when possible. The caller is
301 /// responsible for ensuring the opc has a flag setting equivalent.
302 static unsigned convertToFlagSettingOpc(unsigned Opc);
303
304 /// Return true if this is a load/store that can be potentially paired/merged.
305 bool isCandidateToMergeOrPair(const MachineInstr &MI) const;
306
307 /// Hint that pairing the given load or store is unprofitable.
308 static void suppressLdStPair(MachineInstr &MI);
309
310 std::optional<ExtAddrMode>
311 getAddrModeFromMemoryOp(const MachineInstr &MemI,
312 const TargetRegisterInfo *TRI) const override;
313
314 bool canFoldIntoAddrMode(const MachineInstr &MemI, Register Reg,
315 const MachineInstr &AddrI,
316 ExtAddrMode &AM) const override;
317
318 MachineInstr *emitLdStWithAddr(MachineInstr &MemI,
319 const ExtAddrMode &AM) const override;
320
321 bool getMemOperandsWithOffsetWidth(
322 const MachineInstr &MI, SmallVectorImpl<const MachineOperand *> &BaseOps,
323 int64_t &Offset, bool &OffsetIsScalable, LocationSize &Width,
324 const TargetRegisterInfo *TRI) const override;
325
326 /// If \p OffsetIsScalable is set to 'true', the offset is scaled by `vscale`.
327 /// This is true for some SVE instructions like ldr/str that have a
328 /// 'reg + imm' addressing mode where the immediate is an index to the
329 /// scalable vector located at 'reg + imm * vscale x #bytes'.
330 bool getMemOperandWithOffsetWidth(const MachineInstr &MI,
331 const MachineOperand *&BaseOp,
332 int64_t &Offset, bool &OffsetIsScalable,
333 TypeSize &Width,
334 const TargetRegisterInfo *TRI) const;
335
336 /// Return the immediate offset of the base register in a load/store \p LdSt.
337 MachineOperand &getMemOpBaseRegImmOfsOffsetOperand(MachineInstr &LdSt) const;
338
339 /// Returns true if opcode \p Opc is a memory operation. If it is, set
340 /// \p Scale, \p Width, \p MinOffset, and \p MaxOffset accordingly.
341 ///
342 /// For unscaled instructions, \p Scale is set to 1. All values are in bytes.
343 /// MinOffset/MaxOffset are the un-scaled limits of the immediate in the
344 /// instruction, the actual offset limit is [MinOffset*Scale,
345 /// MaxOffset*Scale].
346 static bool getMemOpInfo(unsigned Opcode, TypeSize &Scale, TypeSize &Width,
347 int64_t &MinOffset, int64_t &MaxOffset);
348
349 bool shouldClusterMemOps(ArrayRef<const MachineOperand *> BaseOps1,
350 int64_t Offset1, bool OffsetIsScalable1,
351 ArrayRef<const MachineOperand *> BaseOps2,
352 int64_t Offset2, bool OffsetIsScalable2,
353 unsigned ClusterSize,
354 unsigned NumBytes) const override;
355
356 void copyPhysRegTuple(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
357 const DebugLoc &DL, MCRegister DestReg,
358 MCRegister SrcReg, bool KillSrc,
359 llvm::ArrayRef<unsigned> Indices) const;
360 void copyGPRRegTuple(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
361 const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg,
362 bool KillSrc, unsigned Opcode, unsigned ZeroReg,
363 llvm::ArrayRef<unsigned> Indices) const;
364 void copyPhysRegImpl(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
365 const DebugLoc &DL, Register DestReg, Register SrcReg,
366 bool KillSrc, bool RenamableDest = false,
367 bool RenamableSrc = false) const;
368 void copyPhysReg(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
369 const DebugLoc &DL, Register DestReg, Register SrcReg,
370 bool KillSrc, bool RenamableDest = false,
371 bool RenamableSrc = false) const override;
372
373 void storeRegToStackSlot(
374 MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, Register SrcReg,
375 bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg,
376 MachineInstr::MIFlag Flags = MachineInstr::NoFlags) const override;
377
378 void loadRegFromStackSlot(
379 MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI,
380 Register DestReg, int FrameIndex, const TargetRegisterClass *RC,
381 Register VReg, unsigned SubReg = 0,
382 MachineInstr::MIFlag Flags = MachineInstr::NoFlags) const override;
383
384 // This tells target independent code that it is okay to pass instructions
385 // with subreg operands to foldMemoryOperandImpl.
386 bool isSubregFoldable() const override { return true; }
387
388 using TargetInstrInfo::foldMemoryOperandImpl;
389 MachineInstr *foldMemoryOperandImpl(MachineFunction &MF, MachineInstr &MI,
390 ArrayRef<unsigned> Ops, int FrameIndex,
391 MachineInstr *&CopyMI,
392 LiveIntervals *LIS = nullptr,
393 VirtRegMap *VRM = nullptr) const override;
394
395 /// \returns true if a branch from an instruction with opcode \p BranchOpc
396 /// bytes is capable of jumping to a position \p BrOffset bytes away.
397 bool isBranchOffsetInRange(unsigned BranchOpc,
398 int64_t BrOffset) const override;
399
400 MachineBasicBlock *getBranchDestBlock(const MachineInstr &MI) const override;
401
402 void insertIndirectBranch(MachineBasicBlock &MBB,
403 MachineBasicBlock &NewDestBB,
404 MachineBasicBlock &RestoreBB, const DebugLoc &DL,
405 int64_t BrOffset, RegScavenger *RS) const override;
406
407 bool analyzeBranch(MachineBasicBlock &MBB, MachineBasicBlock *&TBB,
408 MachineBasicBlock *&FBB,
409 SmallVectorImpl<MachineOperand> &Cond,
410 bool AllowModify = false) const override;
411 bool analyzeBranchPredicate(MachineBasicBlock &MBB,
412 MachineBranchPredicate &MBP,
413 bool AllowModify) const override;
414 unsigned removeBranch(MachineBasicBlock &MBB,
415 int *BytesRemoved = nullptr) const override;
416 unsigned insertBranch(MachineBasicBlock &MBB, MachineBasicBlock *TBB,
417 MachineBasicBlock *FBB, ArrayRef<MachineOperand> Cond,
418 const DebugLoc &DL,
419 int *BytesAdded = nullptr) const override;
420
421 std::unique_ptr<TargetInstrInfo::PipelinerLoopInfo>
422 analyzeLoopForPipelining(MachineBasicBlock *LoopBB) const override;
423
424 bool
425 reverseBranchCondition(SmallVectorImpl<MachineOperand> &Cond) const override;
426 bool canInsertSelect(const MachineBasicBlock &, ArrayRef<MachineOperand> Cond,
427 Register, Register, Register, int &, int &,
428 int &) const override;
429 void insertSelect(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI,
430 const DebugLoc &DL, Register DstReg,
431 ArrayRef<MachineOperand> Cond, Register TrueReg,
432 Register FalseReg) const override;
433
434 void insertNoop(MachineBasicBlock &MBB,
435 MachineBasicBlock::iterator MI) const override;
436
437 MCInst getNop() const override;
438
439 bool isSchedulingBoundary(const MachineInstr &MI,
440 const MachineBasicBlock *MBB,
441 const MachineFunction &MF) const override;
442
443 /// analyzeCompare - For a comparison instruction, return the source registers
444 /// in SrcReg and SrcReg2, and the value it compares against in CmpValue.
445 /// Return true if the comparison instruction can be analyzed.
446 bool analyzeCompare(const MachineInstr &MI, Register &SrcReg,
447 Register &SrcReg2, int64_t &CmpMask,
448 int64_t &CmpValue) const override;
449 /// optimizeCompareInstr - Convert the instruction supplying the argument to
450 /// the comparison into one that sets the zero bit in the flags register.
451 bool optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg,
452 Register SrcReg2, int64_t CmpMask, int64_t CmpValue,
453 const MachineRegisterInfo *MRI) const override;
454 bool optimizeCondBranch(MachineInstr &MI) const override;
455
456 CombinerObjective getCombinerObjective(unsigned Pattern) const override;
457 /// Return true when a code sequence can improve throughput. It
458 /// should be called only for instructions in loops.
459 /// \param Pattern - combiner pattern
460 bool isThroughputPattern(unsigned Pattern) const override;
461 /// Return true when there is potentially a faster code sequence
462 /// for an instruction chain ending in ``Root``. All potential patterns are
463 /// listed in the ``Patterns`` array.
464 bool getMachineCombinerPatterns(MachineInstr &Root,
465 SmallVectorImpl<unsigned> &Patterns,
466 bool DoRegPressureReduce) const override;
467 /// Return true when Inst is associative and commutative so that it can be
468 /// reassociated. If Invert is true, then the inverse of Inst operation must
469 /// be checked.
470 bool isAssociativeAndCommutative(const MachineInstr &Inst,
471 bool Invert) const override;
472
473 /// Returns true if \P Opcode is an instruction which performs accumulation
474 /// into a destination register.
475 bool isAccumulationOpcode(unsigned Opcode) const override;
476
477 /// Returns an opcode which defines the accumulator used by \P Opcode.
478 unsigned getAccumulationStartOpcode(unsigned Opcode) const override;
479
480 unsigned
481 getReduceOpcodeForAccumulator(unsigned int AccumulatorOpCode) const override;
482
483 /// When getMachineCombinerPatterns() finds patterns, this function
484 /// generates the instructions that could replace the original code
485 /// sequence
486 void genAlternativeCodeSequence(
487 MachineInstr &Root, unsigned Pattern,
488 SmallVectorImpl<MachineInstr *> &InsInstrs,
489 SmallVectorImpl<MachineInstr *> &DelInstrs,
490 DenseMap<Register, unsigned> &InstrIdxForVirtReg) const override;
491 /// AArch64 supports MachineCombiner.
492 bool useMachineCombiner() const override;
493
494 bool expandPostRAPseudo(MachineInstr &MI) const override;
495
496 std::pair<unsigned, unsigned>
497 decomposeMachineOperandsTargetFlags(unsigned TF) const override;
498 ArrayRef<std::pair<unsigned, const char *>>
499 getSerializableDirectMachineOperandTargetFlags() const override;
500 ArrayRef<std::pair<unsigned, const char *>>
501 getSerializableBitmaskMachineOperandTargetFlags() const override;
502 ArrayRef<std::pair<MachineMemOperand::Flags, const char *>>
503 getSerializableMachineMemOperandTargetFlags() const override;
504
505 bool isFunctionSafeToOutlineFrom(MachineFunction &MF,
506 bool OutlineFromLinkOnceODRs) const override;
507 std::optional<std::unique_ptr<outliner::OutlinedFunction>>
508 getOutliningCandidateInfo(
509 const MachineModuleInfo &MMI,
510 std::vector<outliner::Candidate> &RepeatedSequenceLocs,
511 unsigned MinRepeats) const override;
512 void mergeOutliningCandidateAttributes(
513 Function &F, std::vector<outliner::Candidate> &Candidates) const override;
514 outliner::InstrType getOutliningTypeImpl(const MachineModuleInfo &MMI,
515 MachineBasicBlock::iterator &MIT,
516 unsigned Flags) const override;
517 SmallVector<
518 std::pair<MachineBasicBlock::iterator, MachineBasicBlock::iterator>>
519 getOutlinableRanges(MachineBasicBlock &MBB, unsigned &Flags) const override;
520 void buildOutlinedFrame(MachineBasicBlock &MBB, MachineFunction &MF,
521 const outliner::OutlinedFunction &OF) const override;
522 MachineBasicBlock::iterator
523 insertOutlinedCall(Module &M, MachineBasicBlock &MBB,
524 MachineBasicBlock::iterator &It, MachineFunction &MF,
525 outliner::Candidate &C) const override;
526 bool shouldOutlineFromFunctionByDefault(MachineFunction &MF) const override;
527
528 void buildClearRegister(Register Reg, MachineBasicBlock &MBB,
529 MachineBasicBlock::iterator Iter, DebugLoc &DL,
530 bool AllowSideEffects = true) const override;
531
532 /// Returns the vector element size (B, H, S or D) of an SVE opcode.
533 uint64_t getElementSizeForOpcode(unsigned Opc) const;
534 /// Returns true if the opcode is for an SVE instruction that sets the
535 /// condition codes as if it's results had been fed to a PTEST instruction
536 /// along with the same general predicate.
537 bool isPTestLikeOpcode(unsigned Opc) const;
538 /// Returns true if the opcode is for an SVE WHILE## instruction.
539 bool isWhileOpcode(unsigned Opc) const;
540 /// Returns true if the instruction has a shift by immediate that can be
541 /// executed in one cycle less.
542 static bool isFalkorShiftExtFast(const MachineInstr &MI);
543 /// Return true if the instructions is a SEH instruction used for unwinding
544 /// on Windows.
545 static bool isSEHInstruction(const MachineInstr &MI);
546
547 std::optional<RegImmPair> isAddImmediate(const MachineInstr &MI,
548 Register Reg) const override;
549
550 bool isFunctionSafeToSplit(const MachineFunction &MF) const override;
551
552 bool isMBBSafeToSplitToCold(const MachineBasicBlock &MBB) const override;
553
554 std::optional<ParamLoadedValue>
555 describeLoadedValue(const MachineInstr &MI, Register Reg) const override;
556
557 unsigned int getTailDuplicateSize(CodeGenOptLevel OptLevel) const override;
558
559 bool isExtendLikelyToBeFolded(MachineInstr &ExtMI,
560 MachineRegisterInfo &MRI) const override;
561
562 static void decomposeStackOffsetForFrameOffsets(const StackOffset &Offset,
563 int64_t &NumBytes,
564 int64_t &NumPredicateVectors,
565 int64_t &NumDataVectors);
566 static void decomposeStackOffsetForDwarfOffsets(const StackOffset &Offset,
567 int64_t &ByteSized,
568 int64_t &VGSized);
569
570 // Return true if address of the form BaseReg + Scale * ScaledReg + Offset can
571 // be used for a load/store of NumBytes. BaseReg is always present and
572 // implicit.
573 bool isLegalAddressingMode(unsigned NumBytes, int64_t Offset,
574 unsigned Scale) const;
575
576 // Decrement the SP, issuing probes along the way. `TargetReg` is the new top
577 // of the stack. `FrameSetup` is passed as true, if the allocation is a part
578 // of constructing the activation frame of a function.
579 MachineBasicBlock::iterator probedStackAlloc(MachineBasicBlock::iterator MBBI,
580 Register TargetReg,
581 bool FrameSetup) const;
582
583 static int
584 findCondCodeUseOperandIdxForBranchOrSelect(const MachineInstr &Instr);
585
586 /// Insert a `PAUTH_EPILOGUE` pseudo before the first terminator in \p MBB to
587 /// authenticate the return address. Adds an implicit def of X16 when the
588 /// branch protection uses PAuthLR but the subtarget lacks PAuthLR
589 /// instructions. If the epilogue has callee-popped argument stack to restore,
590 /// it additionally implicit defines X15 and X17 to cover clobbered registers
591 /// for the required sequence on subtargets both with and without PAuthLR
592 /// instructions.
593 void createPauthEpilogueInstr(MachineBasicBlock &MBB, DebugLoc DL) const;
594
595#define GET_INSTRINFO_HELPER_DECLS
596#include "AArch64GenInstrInfo.inc"
597
598protected:
599 /// If the specific machine instruction is an instruction that moves/copies
600 /// value from one register to another register return destination and source
601 /// registers as machine operands.
602 std::optional<DestSourcePair>
603 isCopyInstrImpl(const MachineInstr &MI) const override;
604 std::optional<DestSourcePair>
605 isCopyLikeInstrImpl(const MachineInstr &MI) const override;
606
607private:
608 /// Sets the offsets on outlined instructions in \p MBB which use SP
609 /// so that they will be valid post-outlining.
610 ///
611 /// \param MBB A \p MachineBasicBlock in an outlined function.
612 void fixupPostOutline(MachineBasicBlock &MBB) const;
613
614 void instantiateCondBranch(MachineBasicBlock &MBB, const DebugLoc &DL,
615 MachineBasicBlock *TBB,
616 ArrayRef<MachineOperand> Cond) const;
617 bool substituteCmpToZero(MachineInstr &CmpInstr, unsigned SrcReg,
618 const MachineRegisterInfo &MRI) const;
619 bool removeCmpToZeroOrOne(MachineInstr &CmpInstr, unsigned SrcReg,
620 int CmpValue, const MachineRegisterInfo &MRI) const;
621
622 /// Returns an unused general-purpose register which can be used for
623 /// constructing an outlined call if one exists. Returns 0 otherwise.
624 Register findRegisterToSaveLRTo(outliner::Candidate &C) const;
625
626 /// Remove a ptest of a predicate-generating operation that already sets, or
627 /// can be made to set, the condition codes in an identical manner
628 bool optimizePTestInstr(MachineInstr *PTest, unsigned MaskReg,
629 unsigned PredReg,
630 const MachineRegisterInfo *MRI) const;
631 std::optional<unsigned>
632 canRemovePTestInstr(MachineInstr *PTest, MachineInstr *Mask,
633 MachineInstr *Pred, const MachineRegisterInfo *MRI) const;
634
635 /// verifyInstruction - Perform target specific instruction verification.
636 bool verifyInstruction(const MachineInstr &MI,
637 StringRef &ErrInfo) const override;
638};
639
640struct UsedNZCV {
641 bool N = false;
642 bool Z = false;
643 bool C = false;
644 bool V = false;
645
646 UsedNZCV() = default;
647
648 UsedNZCV &operator|=(const UsedNZCV &UsedFlags) {
649 this->N |= UsedFlags.N;
650 this->Z |= UsedFlags.Z;
651 this->C |= UsedFlags.C;
652 this->V |= UsedFlags.V;
653 return *this;
654 }
655};
656
657/// \returns Conditions flags used after \p CmpInstr in its MachineBB if NZCV
658/// flags are not alive in successors of the same \p CmpInstr and \p MI parent.
659/// \returns std::nullopt otherwise.
660///
661/// Collect instructions using that flags in \p CCUseInstrs if provided.
662std::optional<UsedNZCV>
663examineCFlagsUse(MachineInstr &MI, MachineInstr &CmpInstr,
664 const TargetRegisterInfo &TRI,
665 SmallVectorImpl<MachineInstr *> *CCUseInstrs = nullptr);
666
667/// Return true if there is an instruction /after/ \p DefMI and before \p UseMI
668/// which either reads or clobbers NZCV.
669bool isNZCVTouchedInInstructionRange(const MachineInstr &DefMI,
670 const MachineInstr &UseMI,
671 const TargetRegisterInfo *TRI);
672
673MCCFIInstruction createDefCFA(const TargetRegisterInfo &TRI, unsigned FrameReg,
674 unsigned Reg, const StackOffset &Offset,
675 bool LastAdjustmentWasScalable = true);
676MCCFIInstruction
677createCFAOffset(const TargetRegisterInfo &MRI, unsigned Reg,
678 const StackOffset &OffsetFromDefCFA,
679 std::optional<int64_t> IncomingVGOffsetFromDefCFA);
680
681/// emitFrameOffset - Emit instructions as needed to set DestReg to SrcReg
682/// plus Offset. This is intended to be used from within the prolog/epilog
683/// insertion (PEI) pass, where a virtual scratch register may be allocated
684/// if necessary, to be replaced by the scavenger at the end of PEI.
685void emitFrameOffset(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI,
686 const DebugLoc &DL, unsigned DestReg, unsigned SrcReg,
687 StackOffset Offset, const TargetInstrInfo *TII,
688 MachineInstr::MIFlag = MachineInstr::NoFlags,
689 bool SetNZCV = false, bool NeedsWinCFI = false,
690 bool *HasWinCFI = nullptr, bool EmitCFAOffset = false,
691 StackOffset InitialOffset = {},
692 unsigned FrameReg = AArch64::SP);
693
694/// rewriteAArch64FrameIndex - Rewrite MI to access 'Offset' bytes from the
695/// FP. Return false if the offset could not be handled directly in MI, and
696/// return the left-over portion by reference.
697bool rewriteAArch64FrameIndex(MachineInstr &MI, unsigned FrameRegIdx,
698 unsigned FrameReg, StackOffset &Offset,
699 const AArch64InstrInfo *TII);
700
701/// Use to report the frame offset status in isAArch64FrameOffsetLegal.
702enum AArch64FrameOffsetStatus {
703 AArch64FrameOffsetCannotUpdate = 0x0, ///< Offset cannot apply.
704 AArch64FrameOffsetIsLegal = 0x1, ///< Offset is legal.
705 AArch64FrameOffsetCanUpdate = 0x2 ///< Offset can apply, at least partly.
706};
707
708/// Check if the @p Offset is a valid frame offset for @p MI.
709/// The returned value reports the validity of the frame offset for @p MI.
710/// It uses the values defined by AArch64FrameOffsetStatus for that.
711/// If result == AArch64FrameOffsetCannotUpdate, @p MI cannot be updated to
712/// use an offset.eq
713/// If result & AArch64FrameOffsetIsLegal, @p Offset can completely be
714/// rewritten in @p MI.
715/// If result & AArch64FrameOffsetCanUpdate, @p Offset contains the
716/// amount that is off the limit of the legal offset.
717/// If set, @p OutUseUnscaledOp will contain the whether @p MI should be
718/// turned into an unscaled operator, which opcode is in @p OutUnscaledOp.
719/// If set, @p EmittableOffset contains the amount that can be set in @p MI
720/// (possibly with @p OutUnscaledOp if OutUseUnscaledOp is true) and that
721/// is a legal offset.
722int isAArch64FrameOffsetLegal(const MachineInstr &MI, StackOffset &Offset,
723 bool *OutUseUnscaledOp = nullptr,
724 unsigned *OutUnscaledOp = nullptr,
725 int64_t *EmittableOffset = nullptr);
726
727bool optimizeTerminators(MachineBasicBlock *MBB, const TargetInstrInfo &TII);
728
729static inline bool isUncondBranchOpcode(int Opc) { return Opc == AArch64::B; }
730
731static inline bool isCondBranchOpcode(int Opc) {
732 switch (Opc) {
733 case AArch64::Bcc:
734 case AArch64::CBZW:
735 case AArch64::CBZX:
736 case AArch64::CBNZW:
737 case AArch64::CBNZX:
738 case AArch64::TBZW:
739 case AArch64::TBZX:
740 case AArch64::TBNZW:
741 case AArch64::TBNZX:
742 case AArch64::CBWPri:
743 case AArch64::CBXPri:
744 case AArch64::CBBAssertExt:
745 case AArch64::CBHAssertExt:
746 case AArch64::CBWPrr:
747 case AArch64::CBXPrr:
748 return true;
749 default:
750 return false;
751 }
752}
753
754static inline bool isIndirectBranchOpcode(int Opc) {
755 switch (Opc) {
756 case AArch64::BR:
757 case AArch64::BRAA:
758 case AArch64::BRAB:
759 case AArch64::BRAAZ:
760 case AArch64::BRABZ:
761 return true;
762 }
763 return false;
764}
765
766static inline bool isIndirectCallOpcode(unsigned Opc) {
767 switch (Opc) {
768 case AArch64::BLR:
769 case AArch64::BLRAA:
770 case AArch64::BLRAB:
771 case AArch64::BLRAAZ:
772 case AArch64::BLRABZ:
773 return true;
774 default:
775 return false;
776 }
777}
778
779static inline bool isPTrueOpcode(unsigned Opc) {
780 switch (Opc) {
781 case AArch64::PTRUE_B:
782 case AArch64::PTRUE_H:
783 case AArch64::PTRUE_S:
784 case AArch64::PTRUE_D:
785 return true;
786 default:
787 return false;
788 }
789}
790
791/// Return opcode to be used for indirect calls.
792unsigned getBLRCallOpcode(const MachineFunction &MF);
793
794/// Return XPAC opcode to be used for a ptrauth strip using the given key.
795static inline unsigned getXPACOpcodeForKey(AArch64PACKey::ID K) {
796 using namespace AArch64PACKey;
797 switch (K) {
798 case IA: case IB: return AArch64::XPACI;
799 case DA: case DB: return AArch64::XPACD;
800 }
801 llvm_unreachable("Unhandled AArch64PACKey::ID enum");
802}
803
804/// Return AUT opcode to be used for a ptrauth auth using the given key, or its
805/// AUT*Z variant that doesn't take a discriminator operand, using zero instead.
806static inline unsigned getAUTOpcodeForKey(AArch64PACKey::ID K, bool Zero) {
807 using namespace AArch64PACKey;
808 switch (K) {
809 case IA: return Zero ? AArch64::AUTIZA : AArch64::AUTIA;
810 case IB: return Zero ? AArch64::AUTIZB : AArch64::AUTIB;
811 case DA: return Zero ? AArch64::AUTDZA : AArch64::AUTDA;
812 case DB: return Zero ? AArch64::AUTDZB : AArch64::AUTDB;
813 }
814 llvm_unreachable("Unhandled AArch64PACKey::ID enum");
815}
816
817/// Return PAC opcode to be used for a ptrauth sign using the given key, or its
818/// PAC*Z variant that doesn't take a discriminator operand, using zero instead.
819static inline unsigned getPACOpcodeForKey(AArch64PACKey::ID K, bool Zero) {
820 using namespace AArch64PACKey;
821 switch (K) {
822 case IA: return Zero ? AArch64::PACIZA : AArch64::PACIA;
823 case IB: return Zero ? AArch64::PACIZB : AArch64::PACIB;
824 case DA: return Zero ? AArch64::PACDZA : AArch64::PACDA;
825 case DB: return Zero ? AArch64::PACDZB : AArch64::PACDB;
826 }
827 llvm_unreachable("Unhandled AArch64PACKey::ID enum");
828}
829
830/// Return B(L)RA opcode to be used for an authenticated branch or call using
831/// the given key, or its B(L)RA*Z variant that doesn't take a discriminator
832/// operand, using zero instead.
833static inline unsigned getBranchOpcodeForKey(bool IsCall, AArch64PACKey::ID K,
834 bool Zero) {
835 using namespace AArch64PACKey;
836 static const unsigned BranchOpcode[2][2] = {
837 {AArch64::BRAA, AArch64::BRAAZ},
838 {AArch64::BRAB, AArch64::BRABZ},
839 };
840 static const unsigned CallOpcode[2][2] = {
841 {AArch64::BLRAA, AArch64::BLRAAZ},
842 {AArch64::BLRAB, AArch64::BLRABZ},
843 };
844
845 assert((K == IA || K == IB) && "B(L)RA* instructions require IA or IB key");
846 if (IsCall)
847 return CallOpcode[K == IB][Zero];
848 return BranchOpcode[K == IB][Zero];
849}
850
851// struct TSFlags {
852#define TSFLAG_ELEMENT_SIZE_TYPE(X) (X) // 3-bits
853#define TSFLAG_DESTRUCTIVE_INST_TYPE(X) ((X) << 3) // 4-bits
854#define TSFLAG_FALSE_LANE_TYPE(X) ((X) << 7) // 2-bits
855#define TSFLAG_INSTR_FLAGS(X) ((X) << 9) // 2-bits
856#define TSFLAG_SME_MATRIX_TYPE(X) ((X) << 11) // 3-bits
857// }
858
859namespace AArch64 {
860
861// clang-format off
862enum ElementSizeType {
863 ElementSizeMask = TSFLAG_ELEMENT_SIZE_TYPE(0x7),
864 ElementSizeNone = TSFLAG_ELEMENT_SIZE_TYPE(0x0),
865 ElementSizeB = TSFLAG_ELEMENT_SIZE_TYPE(0x1),
866 ElementSizeH = TSFLAG_ELEMENT_SIZE_TYPE(0x2),
867 ElementSizeS = TSFLAG_ELEMENT_SIZE_TYPE(0x3),
868 ElementSizeD = TSFLAG_ELEMENT_SIZE_TYPE(0x4),
869};
870
871enum DestructiveInstType {
872 DestructiveInstTypeMask = TSFLAG_DESTRUCTIVE_INST_TYPE(0xf),
873 NotDestructive = TSFLAG_DESTRUCTIVE_INST_TYPE(0x0),
874 DestructiveOther = TSFLAG_DESTRUCTIVE_INST_TYPE(0x1),
875 DestructiveUnary = TSFLAG_DESTRUCTIVE_INST_TYPE(0x2),
876 DestructiveBinaryImm = TSFLAG_DESTRUCTIVE_INST_TYPE(0x3),
877 DestructiveBinaryShImmUnpred = TSFLAG_DESTRUCTIVE_INST_TYPE(0x4),
878 DestructiveBinary = TSFLAG_DESTRUCTIVE_INST_TYPE(0x5),
879 DestructiveBinaryComm = TSFLAG_DESTRUCTIVE_INST_TYPE(0x6),
880 DestructiveBinaryCommWithRev = TSFLAG_DESTRUCTIVE_INST_TYPE(0x7),
881 DestructiveTernaryCommWithRev = TSFLAG_DESTRUCTIVE_INST_TYPE(0x8),
882 Destructive2xRegImmUnpred = TSFLAG_DESTRUCTIVE_INST_TYPE(0x9),
883 DestructiveUnaryPassthru = TSFLAG_DESTRUCTIVE_INST_TYPE(0xa),
884 DestructivePredicate = TSFLAG_DESTRUCTIVE_INST_TYPE(0xb),
885 DestructiveBinaryImmUnpred = TSFLAG_DESTRUCTIVE_INST_TYPE(0xc),
886};
887
888enum FalseLaneType {
889 FalseLanesMask = TSFLAG_FALSE_LANE_TYPE(0x3),
890 FalseLanesZero = TSFLAG_FALSE_LANE_TYPE(0x1),
891 FalseLanesUndef = TSFLAG_FALSE_LANE_TYPE(0x2),
892};
893
894// clang-format on
895
896// NOTE: This is a bit field.
897static const uint64_t InstrFlagIsWhile = TSFLAG_INSTR_FLAGS(0x1);
898static const uint64_t InstrFlagIsPTestLike = TSFLAG_INSTR_FLAGS(0x2);
899
900enum SMEMatrixType {
901 SMEMatrixTypeMask = TSFLAG_SME_MATRIX_TYPE(0x7),
902 SMEMatrixNone = TSFLAG_SME_MATRIX_TYPE(0x0),
903 SMEMatrixTileB = TSFLAG_SME_MATRIX_TYPE(0x1),
904 SMEMatrixTileH = TSFLAG_SME_MATRIX_TYPE(0x2),
905 SMEMatrixTileS = TSFLAG_SME_MATRIX_TYPE(0x3),
906 SMEMatrixTileD = TSFLAG_SME_MATRIX_TYPE(0x4),
907 SMEMatrixTileQ = TSFLAG_SME_MATRIX_TYPE(0x5),
908 SMEMatrixArray = TSFLAG_SME_MATRIX_TYPE(0x6),
909};
910
911#undef TSFLAG_ELEMENT_SIZE_TYPE
912#undef TSFLAG_DESTRUCTIVE_INST_TYPE
913#undef TSFLAG_FALSE_LANE_TYPE
914#undef TSFLAG_INSTR_FLAGS
915#undef TSFLAG_SME_MATRIX_TYPE
916
917int32_t getSVEPseudoMap(uint32_t Opcode);
918int32_t getSVERevInstr(uint32_t Opcode);
919int32_t getSVENonRevInstr(uint32_t Opcode);
920
921int32_t getSMEPseudoMap(uint32_t Opcode);
922}
923
924} // end namespace llvm
925
926#endif
927