1//===- AArch64InstrInfo.h - AArch64 Instruction Information -----*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file contains the AArch64 implementation of the TargetInstrInfo class.
10//
11//===----------------------------------------------------------------------===//
12
13#ifndef LLVM_LIB_TARGET_AARCH64_AARCH64INSTRINFO_H
14#define LLVM_LIB_TARGET_AARCH64_AARCH64INSTRINFO_H
15
16#include "AArch64.h"
17#include "AArch64RegisterInfo.h"
18#include "llvm/CodeGen/TargetInstrInfo.h"
19#include "llvm/Support/AArch64MemoryHints.h"
20#include "llvm/Support/TypeSize.h"
21#include <optional>
22
23#define GET_INSTRINFO_HEADER
24#include "AArch64GenInstrInfo.inc"
25
26namespace llvm {
27
28class AArch64Subtarget;
29
30static const MachineMemOperand::Flags MOSuppressPair =
31 MachineMemOperand::MOTargetFlag1;
32static const MachineMemOperand::Flags MOStridedAccess =
33 MachineMemOperand::MOTargetFlag2;
34
35#define FALKOR_STRIDED_ACCESS_MD "falkor.strided.access"
36
37// AArch64 MachineCombiner patterns
38enum AArch64MachineCombinerPattern : unsigned {
39 // These are patterns used to reduce the length of dependence chain.
40 SUBADD_OP1 = MachineCombinerPattern::TARGET_PATTERN_START,
41 SUBADD_OP2,
42
43 // These are multiply-add patterns matched by the AArch64 machine combiner.
44 MULADDW_OP1,
45 MULADDW_OP2,
46 MULSUBW_OP1,
47 MULSUBW_OP2,
48 MULADDWI_OP1,
49 MULSUBWI_OP1,
50 MULADDX_OP1,
51 MULADDX_OP2,
52 MULSUBX_OP1,
53 MULSUBX_OP2,
54 MULADDXI_OP1,
55 MULSUBXI_OP1,
56 // NEON integers vectors
57 MULADDv8i8_OP1,
58 MULADDv8i8_OP2,
59 MULADDv16i8_OP1,
60 MULADDv16i8_OP2,
61 MULADDv4i16_OP1,
62 MULADDv4i16_OP2,
63 MULADDv8i16_OP1,
64 MULADDv8i16_OP2,
65 MULADDv2i32_OP1,
66 MULADDv2i32_OP2,
67 MULADDv4i32_OP1,
68 MULADDv4i32_OP2,
69
70 MULSUBv8i8_OP1,
71 MULSUBv8i8_OP2,
72 MULSUBv16i8_OP1,
73 MULSUBv16i8_OP2,
74 MULSUBv4i16_OP1,
75 MULSUBv4i16_OP2,
76 MULSUBv8i16_OP1,
77 MULSUBv8i16_OP2,
78 MULSUBv2i32_OP1,
79 MULSUBv2i32_OP2,
80 MULSUBv4i32_OP1,
81 MULSUBv4i32_OP2,
82
83 MULADDv4i16_indexed_OP1,
84 MULADDv4i16_indexed_OP2,
85 MULADDv8i16_indexed_OP1,
86 MULADDv8i16_indexed_OP2,
87 MULADDv2i32_indexed_OP1,
88 MULADDv2i32_indexed_OP2,
89 MULADDv4i32_indexed_OP1,
90 MULADDv4i32_indexed_OP2,
91
92 MULSUBv4i16_indexed_OP1,
93 MULSUBv4i16_indexed_OP2,
94 MULSUBv8i16_indexed_OP1,
95 MULSUBv8i16_indexed_OP2,
96 MULSUBv2i32_indexed_OP1,
97 MULSUBv2i32_indexed_OP2,
98 MULSUBv4i32_indexed_OP1,
99 MULSUBv4i32_indexed_OP2,
100
101 // Floating Point
102 FMULADDH_OP1,
103 FMULADDH_OP2,
104 FMULSUBH_OP1,
105 FMULSUBH_OP2,
106 FMULADDS_OP1,
107 FMULADDS_OP2,
108 FMULSUBS_OP1,
109 FMULSUBS_OP2,
110 FMULADDD_OP1,
111 FMULADDD_OP2,
112 FMULSUBD_OP1,
113 FMULSUBD_OP2,
114 FNMULSUBH_OP1,
115 FNMULSUBS_OP1,
116 FNMULSUBD_OP1,
117 FMLAv1i32_indexed_OP1,
118 FMLAv1i32_indexed_OP2,
119 FMLAv1i64_indexed_OP1,
120 FMLAv1i64_indexed_OP2,
121 FMLAv4f16_OP1,
122 FMLAv4f16_OP2,
123 FMLAv8f16_OP1,
124 FMLAv8f16_OP2,
125 FMLAv2f32_OP2,
126 FMLAv2f32_OP1,
127 FMLAv2f64_OP1,
128 FMLAv2f64_OP2,
129 FMLAv4i16_indexed_OP1,
130 FMLAv4i16_indexed_OP2,
131 FMLAv8i16_indexed_OP1,
132 FMLAv8i16_indexed_OP2,
133 FMLAv2i32_indexed_OP1,
134 FMLAv2i32_indexed_OP2,
135 FMLAv2i64_indexed_OP1,
136 FMLAv2i64_indexed_OP2,
137 FMLAv4f32_OP1,
138 FMLAv4f32_OP2,
139 FMLAv4i32_indexed_OP1,
140 FMLAv4i32_indexed_OP2,
141 FMLSv1i32_indexed_OP2,
142 FMLSv1i64_indexed_OP2,
143 FMLSv4f16_OP1,
144 FMLSv4f16_OP2,
145 FMLSv8f16_OP1,
146 FMLSv8f16_OP2,
147 FMLSv2f32_OP1,
148 FMLSv2f32_OP2,
149 FMLSv2f64_OP1,
150 FMLSv2f64_OP2,
151 FMLSv4i16_indexed_OP1,
152 FMLSv4i16_indexed_OP2,
153 FMLSv8i16_indexed_OP1,
154 FMLSv8i16_indexed_OP2,
155 FMLSv2i32_indexed_OP1,
156 FMLSv2i32_indexed_OP2,
157 FMLSv2i64_indexed_OP1,
158 FMLSv2i64_indexed_OP2,
159 FMLSv4f32_OP1,
160 FMLSv4f32_OP2,
161 FMLSv4i32_indexed_OP1,
162 FMLSv4i32_indexed_OP2,
163
164 FMULv2i32_indexed_OP1,
165 FMULv2i32_indexed_OP2,
166 FMULv2i64_indexed_OP1,
167 FMULv2i64_indexed_OP2,
168 FMULv4i16_indexed_OP1,
169 FMULv4i16_indexed_OP2,
170 FMULv4i32_indexed_OP1,
171 FMULv4i32_indexed_OP2,
172 FMULv8i16_indexed_OP1,
173 FMULv8i16_indexed_OP2,
174
175 FNMADD,
176
177 GATHER_LANE_i32,
178 GATHER_LANE_i16,
179 GATHER_LANE_i8
180};
181class AArch64InstrInfo final : public AArch64GenInstrInfo {
182 const AArch64RegisterInfo RI;
183 const AArch64Subtarget &Subtarget;
184
185public:
186 explicit AArch64InstrInfo(const AArch64Subtarget &STI);
187
188 /// getRegisterInfo - TargetInstrInfo is a superset of MRegister info. As
189 /// such, whenever a client has an instance of instruction info, it should
190 /// always be able to get register info as well (through this method).
191 const AArch64RegisterInfo &getRegisterInfo() const { return RI; }
192
193 const TargetRegisterClass *getInlineAsmMemoryOperandRegClass(
194 InlineAsm::ConstraintCode C) const override {
195 return &AArch64::GPR64spRegClass;
196 }
197
198 unsigned getInstSizeInBytes(const MachineInstr &MI) const override;
199
200 bool isAsCheapAsAMove(const MachineInstr &MI) const override;
201
202 bool isCoalescableExtInstr(const MachineInstr &MI, Register &SrcReg,
203 Register &DstReg, unsigned &SubIdx) const override;
204
205 bool
206 areMemAccessesTriviallyDisjoint(const MachineInstr &MIa,
207 const MachineInstr &MIb) const override;
208
209 Register isLoadFromStackSlot(const MachineInstr &MI,
210 int &FrameIndex) const override;
211 Register isStoreToStackSlot(const MachineInstr &MI,
212 int &FrameIndex) const override;
213
214 /// Check for post-frame ptr elimination stack locations as well. This uses a
215 /// heuristic so it isn't reliable for correctness.
216 Register isStoreToStackSlotPostFE(const MachineInstr &MI,
217 int &FrameIndex) const override;
218 /// Check for post-frame ptr elimination stack locations as well. This uses a
219 /// heuristic so it isn't reliable for correctness.
220 Register isLoadFromStackSlotPostFE(const MachineInstr &MI,
221 int &FrameIndex) const override;
222
223 /// Does this instruction set its full destination register to zero?
224 static bool isGPRZero(const MachineInstr &MI);
225
226 /// Does this instruction rename a GPR without modifying bits?
227 static bool isGPRCopy(const MachineInstr &MI);
228
229 /// Does this instruction rename an FPR without modifying bits?
230 static bool isFPRCopy(const MachineInstr &MI);
231
232 /// Return true if pairing the given load or store is hinted to be
233 /// unprofitable.
234 static bool isLdStPairSuppressed(const MachineInstr &MI);
235
236 /// Return true if the given load or store is a strided memory access.
237 static bool isStridedAccess(const MachineInstr &MI);
238
239 /// Return true if it has an unscaled load/store offset.
240 static bool hasUnscaledLdStOffset(unsigned Opc);
241 static bool hasUnscaledLdStOffset(MachineInstr &MI) {
242 return hasUnscaledLdStOffset(Opc: MI.getOpcode());
243 }
244
245 /// Returns the unscaled load/store for the scaled load/store opcode,
246 /// if there is a corresponding unscaled variant available.
247 static std::optional<unsigned> getUnscaledLdSt(unsigned Opc);
248
249 /// Scaling factor for (scaled or unscaled) load or store.
250 static int getMemScale(unsigned Opc);
251 static int getMemScale(const MachineInstr &MI) {
252 return getMemScale(Opc: MI.getOpcode());
253 }
254
255 /// Returns whether the instruction is a pre-indexed load.
256 static bool isPreLd(const MachineInstr &MI);
257
258 /// Returns whether the instruction is a pre-indexed store.
259 static bool isPreSt(const MachineInstr &MI);
260
261 /// Returns whether the instruction is a pre-indexed load/store.
262 static bool isPreLdSt(const MachineInstr &MI);
263
264 /// Returns whether the instruction is a zero-extending load.
265 static bool isZExtLoad(const MachineInstr &MI);
266
267 /// Returns whether the instruction is a sign-extending load.
268 static bool isSExtLoad(const MachineInstr &MI);
269
270 /// Returns whether the instruction is a paired load/store.
271 static bool isPairedLdSt(const MachineInstr &MI);
272
273 /// Returns the base register operator of a load/store.
274 static const MachineOperand &getLdStBaseOp(const MachineInstr &MI);
275
276 /// Returns the immediate offset operator of a load/store.
277 static const MachineOperand &getLdStOffsetOp(const MachineInstr &MI);
278
279 /// Returns whether the physical register is FP or NEON.
280 static bool isFpOrNEON(Register Reg);
281
282 /// Returns the shift amount operator of a load/store.
283 static const MachineOperand &getLdStAmountOp(const MachineInstr &MI);
284
285 /// Returns whether the instruction is FP or NEON.
286 static bool isFpOrNEON(const MachineInstr &MI);
287
288 /// Returns whether the instruction is in H form (16 bit operands)
289 static bool isHForm(const MachineInstr &MI);
290
291 /// Returns whether the instruction is in Q form (128 bit operands)
292 static bool isQForm(const MachineInstr &MI);
293
294 /// Returns whether the instruction can be compatible with non-zero BTYPE.
295 static bool hasBTISemantics(const MachineInstr &MI);
296
297 /// Returns the index for the immediate for a given instruction.
298 static unsigned getLoadStoreImmIdx(unsigned Opc);
299
300 /// Return true if pairing the given load or store may be paired with another.
301 static bool isPairableLdStInst(const MachineInstr &MI);
302
303 /// Returns true if MI is one of the TCRETURN* instructions.
304 static bool isTailCallReturnInst(const MachineInstr &MI);
305
306 /// Return the opcode that set flags when possible. The caller is
307 /// responsible for ensuring the opc has a flag setting equivalent.
308 static unsigned convertToFlagSettingOpc(unsigned Opc);
309
310 /// Return true if this is a load/store that can be potentially paired/merged.
311 bool isCandidateToMergeOrPair(const MachineInstr &MI) const;
312
313 /// Hint that pairing the given load or store is unprofitable.
314 static void suppressLdStPair(MachineInstr &MI);
315
316 std::optional<ExtAddrMode>
317 getAddrModeFromMemoryOp(const MachineInstr &MemI) const override;
318
319 bool canFoldIntoAddrMode(const MachineInstr &MemI, Register Reg,
320 const MachineInstr &AddrI,
321 ExtAddrMode &AM) const override;
322
323 MachineInstr *emitLdStWithAddr(MachineInstr &MemI,
324 const ExtAddrMode &AM) const override;
325
326 bool getMemOperandsWithOffsetWidth(
327 const MachineInstr &MI, SmallVectorImpl<const MachineOperand *> &BaseOps,
328 int64_t &Offset, bool &OffsetIsScalable,
329 LocationSize &Width) const override;
330
331 /// If \p OffsetIsScalable is set to 'true', the offset is scaled by `vscale`.
332 /// This is true for some SVE instructions like ldr/str that have a
333 /// 'reg + imm' addressing mode where the immediate is an index to the
334 /// scalable vector located at 'reg + imm * vscale x #bytes'.
335 bool getMemOperandWithOffsetWidth(const MachineInstr &MI,
336 const MachineOperand *&BaseOp,
337 int64_t &Offset, bool &OffsetIsScalable,
338 TypeSize &Width) const;
339
340 /// Return the immediate offset of the base register in a load/store \p LdSt.
341 MachineOperand &getMemOpBaseRegImmOfsOffsetOperand(MachineInstr &LdSt) const;
342
343 /// Returns true if opcode \p Opc is a memory operation. If it is, set
344 /// \p Scale, \p Width, \p MinOffset, and \p MaxOffset accordingly.
345 ///
346 /// For unscaled instructions, \p Scale is set to 1. All values are in bytes.
347 /// MinOffset/MaxOffset are the un-scaled limits of the immediate in the
348 /// instruction, the actual offset limit is [MinOffset*Scale,
349 /// MaxOffset*Scale].
350 static bool getMemOpInfo(unsigned Opcode, TypeSize &Scale, TypeSize &Width,
351 int64_t &MinOffset, int64_t &MaxOffset);
352
353 bool shouldClusterMemOps(ArrayRef<const MachineOperand *> BaseOps1,
354 int64_t Offset1, bool OffsetIsScalable1,
355 ArrayRef<const MachineOperand *> BaseOps2,
356 int64_t Offset2, bool OffsetIsScalable2,
357 unsigned ClusterSize,
358 unsigned NumBytes) const override;
359
360 void copyPhysRegTuple(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
361 const DebugLoc &DL, MCRegister DestReg,
362 MCRegister SrcReg, bool KillSrc,
363 llvm::ArrayRef<unsigned> Indices) const;
364 void copyGPRRegTuple(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
365 const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg,
366 bool KillSrc, unsigned Opcode, unsigned ZeroReg,
367 llvm::ArrayRef<unsigned> Indices) const;
368 void copyPhysRegImpl(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
369 const DebugLoc &DL, Register DestReg, Register SrcReg,
370 bool KillSrc, bool RenamableDest = false,
371 bool RenamableSrc = false) const;
372 void copyPhysReg(MachineBasicBlock &MBB, MachineBasicBlock::iterator I,
373 const DebugLoc &DL, Register DestReg, Register SrcReg,
374 bool KillSrc, bool RenamableDest = false,
375 bool RenamableSrc = false) const override;
376
377 void storeRegToStackSlot(
378 MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, Register SrcReg,
379 bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg,
380 MachineInstr::MIFlag Flags = MachineInstr::NoFlags) const override;
381
382 void loadRegFromStackSlot(
383 MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI,
384 Register DestReg, int FrameIndex, const TargetRegisterClass *RC,
385 Register VReg, unsigned SubReg = 0,
386 MachineInstr::MIFlag Flags = MachineInstr::NoFlags) const override;
387
388 // This tells target independent code that it is okay to pass instructions
389 // with subreg operands to foldMemoryOperandImpl.
390 bool isSubregFoldable() const override { return true; }
391
392 using TargetInstrInfo::foldMemoryOperandImpl;
393 MachineInstr *foldMemoryOperandImpl(MachineFunction &MF, MachineInstr &MI,
394 ArrayRef<unsigned> Ops, int FrameIndex,
395 MachineInstr *&CopyMI,
396 LiveIntervals *LIS = nullptr,
397 VirtRegMap *VRM = nullptr) const override;
398
399 /// \returns true if a branch from an instruction with opcode \p BranchOpc
400 /// bytes is capable of jumping to a position \p BrOffset bytes away.
401 bool isBranchOffsetInRange(unsigned BranchOpc,
402 int64_t BrOffset) const override;
403
404 MachineBasicBlock *getBranchDestBlock(const MachineInstr &MI) const override;
405
406 void insertIndirectBranch(MachineBasicBlock &MBB,
407 MachineBasicBlock &NewDestBB,
408 MachineBasicBlock &RestoreBB, const DebugLoc &DL,
409 int64_t BrOffset, RegScavenger *RS) const override;
410
411 bool analyzeBranch(MachineBasicBlock &MBB, MachineBasicBlock *&TBB,
412 MachineBasicBlock *&FBB,
413 SmallVectorImpl<MachineOperand> &Cond,
414 bool AllowModify = false) const override;
415 bool analyzeBranchPredicate(MachineBasicBlock &MBB,
416 MachineBranchPredicate &MBP,
417 bool AllowModify) const override;
418 unsigned removeBranch(MachineBasicBlock &MBB,
419 int *BytesRemoved = nullptr) const override;
420 unsigned insertBranch(MachineBasicBlock &MBB, MachineBasicBlock *TBB,
421 MachineBasicBlock *FBB, ArrayRef<MachineOperand> Cond,
422 const DebugLoc &DL,
423 int *BytesAdded = nullptr) const override;
424
425 /// Inserts the compare instruction needed to un-fuse a fused conditional
426 /// branch instruction and returns the condition code of the original fused
427 /// branch.
428 AArch64CC::CondCode insertCmpForCondBr(MachineBasicBlock &MBB,
429 MachineBasicBlock::iterator MI,
430 const DebugLoc &DL,
431 ArrayRef<MachineOperand> Cond) const;
432
433 std::unique_ptr<TargetInstrInfo::PipelinerLoopInfo>
434 analyzeLoopForPipelining(MachineBasicBlock *LoopBB) const override;
435
436 bool
437 reverseBranchCondition(SmallVectorImpl<MachineOperand> &Cond) const override;
438 bool canInsertSelect(const MachineBasicBlock &, ArrayRef<MachineOperand> Cond,
439 Register, Register, Register, int &, int &,
440 int &) const override;
441 void insertSelect(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI,
442 const DebugLoc &DL, Register DstReg,
443 ArrayRef<MachineOperand> Cond, Register TrueReg,
444 Register FalseReg) const override;
445
446 void insertNoop(MachineBasicBlock &MBB,
447 MachineBasicBlock::iterator MI) const override;
448
449 MCInst getNop() const override;
450
451 bool isSchedulingBoundary(const MachineInstr &MI,
452 const MachineBasicBlock *MBB,
453 const MachineFunction &MF) const override;
454
455 /// analyzeCompare - For a comparison instruction, return the source registers
456 /// in SrcReg and SrcReg2, and the value it compares against in CmpValue.
457 /// Return true if the comparison instruction can be analyzed.
458 bool analyzeCompare(const MachineInstr &MI, Register &SrcReg,
459 Register &SrcReg2, int64_t &CmpMask,
460 int64_t &CmpValue) const override;
461 /// optimizeCompareInstr - Convert the instruction supplying the argument to
462 /// the comparison into one that sets the zero bit in the flags register.
463 bool optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg,
464 Register SrcReg2, int64_t CmpMask, int64_t CmpValue,
465 const MachineRegisterInfo *MRI) const override;
466 bool optimizeCondBranch(MachineInstr &MI) const override;
467
468 CombinerObjective getCombinerObjective(unsigned Pattern) const override;
469 /// Return true when a code sequence can improve throughput. It
470 /// should be called only for instructions in loops.
471 /// \param Pattern - combiner pattern
472 bool isThroughputPattern(unsigned Pattern) const override;
473 /// Return true when there is potentially a faster code sequence
474 /// for an instruction chain ending in ``Root``. All potential patterns are
475 /// listed in the ``Patterns`` array.
476 bool getMachineCombinerPatterns(MachineInstr &Root,
477 SmallVectorImpl<unsigned> &Patterns,
478 bool DoRegPressureReduce) const override;
479 /// Return true when Inst is associative and commutative so that it can be
480 /// reassociated. If Invert is true, then the inverse of Inst operation must
481 /// be checked.
482 bool isAssociativeAndCommutative(const MachineInstr &Inst,
483 bool Invert) const override;
484
485 /// Returns true if \P Opcode is an instruction which performs accumulation
486 /// into a destination register.
487 bool isAccumulationOpcode(unsigned Opcode) const override;
488
489 /// Returns an opcode which defines the accumulator used by \P Opcode.
490 unsigned getAccumulationStartOpcode(unsigned Opcode) const override;
491
492 unsigned
493 getReduceOpcodeForAccumulator(unsigned int AccumulatorOpCode) const override;
494
495 /// When getMachineCombinerPatterns() finds patterns, this function
496 /// generates the instructions that could replace the original code
497 /// sequence
498 void genAlternativeCodeSequence(
499 MachineInstr &Root, unsigned Pattern,
500 SmallVectorImpl<MachineInstr *> &InsInstrs,
501 SmallVectorImpl<MachineInstr *> &DelInstrs,
502 DenseMap<Register, unsigned> &InstrIdxForVirtReg) const override;
503 /// AArch64 supports MachineCombiner.
504 bool useMachineCombiner() const override;
505
506 bool expandPostRAPseudo(MachineInstr &MI) const override;
507
508 std::pair<unsigned, unsigned>
509 decomposeMachineOperandsTargetFlags(unsigned TF) const override;
510 ArrayRef<std::pair<unsigned, const char *>>
511 getSerializableDirectMachineOperandTargetFlags() const override;
512 ArrayRef<std::pair<unsigned, const char *>>
513 getSerializableBitmaskMachineOperandTargetFlags() const override;
514 ArrayRef<std::pair<MachineMemOperand::Flags, const char *>>
515 getSerializableMachineMemOperandTargetFlags() const override;
516
517 bool isFunctionSafeToOutlineFrom(MachineFunction &MF,
518 bool OutlineFromLinkOnceODRs) const override;
519 std::optional<std::unique_ptr<outliner::OutlinedFunction>>
520 getOutliningCandidateInfo(
521 const MachineModuleInfo &MMI,
522 std::vector<outliner::Candidate> &RepeatedSequenceLocs,
523 unsigned MinRepeats) const override;
524 void mergeOutliningCandidateAttributes(
525 Function &F, std::vector<outliner::Candidate> &Candidates) const override;
526 outliner::InstrType getOutliningTypeImpl(const MachineModuleInfo &MMI,
527 MachineBasicBlock::iterator &MIT,
528 unsigned Flags) const override;
529 SmallVector<
530 std::pair<MachineBasicBlock::iterator, MachineBasicBlock::iterator>>
531 getOutlinableRanges(MachineBasicBlock &MBB, unsigned &Flags) const override;
532 void buildOutlinedFrame(MachineBasicBlock &MBB, MachineFunction &MF,
533 const outliner::OutlinedFunction &OF) const override;
534 MachineBasicBlock::iterator
535 insertOutlinedCall(Module &M, MachineBasicBlock &MBB,
536 MachineBasicBlock::iterator &It, MachineFunction &MF,
537 outliner::Candidate &C) const override;
538 bool shouldOutlineFromFunctionByDefault(MachineFunction &MF) const override;
539
540 void buildClearRegister(Register Reg, MachineBasicBlock &MBB,
541 MachineBasicBlock::iterator Iter, DebugLoc &DL,
542 bool AllowSideEffects = true) const override;
543
544 /// Returns the vector element size (B, H, S or D) of an SVE opcode.
545 uint64_t getElementSizeForOpcode(unsigned Opc) const;
546 /// Returns true if the opcode is for an SVE instruction that sets the
547 /// condition codes as if it's results had been fed to a PTEST instruction
548 /// along with the same general predicate.
549 bool isPTestLikeOpcode(unsigned Opc) const;
550 /// Returns true if the opcode is for an SVE WHILE## instruction.
551 bool isWhileOpcode(unsigned Opc) const;
552 /// Returns true if the instruction has a shift by immediate that can be
553 /// executed in one cycle less.
554 static bool isFalkorShiftExtFast(const MachineInstr &MI);
555 /// Return true if the instructions is a SEH instruction used for unwinding
556 /// on Windows.
557 static bool isSEHInstruction(const MachineInstr &MI);
558
559 std::optional<RegImmPair> isAddImmediate(const MachineInstr &MI,
560 Register Reg) const override;
561
562 bool isFunctionSafeToSplit(const MachineFunction &MF) const override;
563
564 bool isMBBSafeToSplitToCold(const MachineBasicBlock &MBB) const override;
565
566 std::optional<ParamLoadedValue>
567 describeLoadedValue(const MachineInstr &MI, Register Reg) const override;
568
569 unsigned int getTailDuplicateSize(CodeGenOptLevel OptLevel) const override;
570
571 bool isExtendLikelyToBeFolded(MachineInstr &ExtMI,
572 MachineRegisterInfo &MRI) const override;
573
574 static void decomposeStackOffsetForFrameOffsets(const StackOffset &Offset,
575 int64_t &NumBytes,
576 int64_t &NumPredicateVectors,
577 int64_t &NumDataVectors);
578 static void decomposeStackOffsetForDwarfOffsets(const StackOffset &Offset,
579 int64_t &ByteSized,
580 int64_t &VGSized);
581
582 // Return true if address of the form BaseReg + Scale * ScaledReg + Offset can
583 // be used for a load/store of NumBytes. BaseReg is always present and
584 // implicit.
585 bool isLegalAddressingMode(unsigned NumBytes, int64_t Offset,
586 unsigned Scale) const;
587
588 // Decrement the SP, issuing probes along the way. `TargetReg` is the new top
589 // of the stack. `FrameSetup` is passed as true, if the allocation is a part
590 // of constructing the activation frame of a function.
591 MachineBasicBlock::iterator probedStackAlloc(MachineBasicBlock::iterator MBBI,
592 Register TargetReg,
593 bool FrameSetup) const;
594
595 static int
596 findCondCodeUseOperandIdxForBranchOrSelect(const MachineInstr &Instr);
597
598 /// Insert a `PAUTH_EPILOGUE` pseudo before the first terminator in \p MBB to
599 /// authenticate the return address. Adds an implicit def of X16 when the
600 /// branch protection uses PAuthLR but the subtarget lacks PAuthLR
601 /// instructions. If the epilogue has callee-popped argument stack to restore,
602 /// it additionally implicit defines X15 and X17 to cover clobbered registers
603 /// for the required sequence on subtargets both with and without PAuthLR
604 /// instructions.
605 void createPauthEpilogueInstr(MachineBasicBlock &MBB, DebugLoc DL) const;
606
607#define GET_INSTRINFO_HELPER_DECLS
608#include "AArch64GenInstrInfo.inc"
609
610protected:
611 /// If the specific machine instruction is an instruction that moves/copies
612 /// value from one register to another register return destination and source
613 /// registers as machine operands.
614 std::optional<DestSourcePair>
615 isCopyInstrImpl(const MachineInstr &MI) const override;
616 std::optional<DestSourcePair>
617 isCopyLikeInstrImpl(const MachineInstr &MI) const override;
618
619private:
620 /// Sets the offsets on outlined instructions in \p MBB which use SP
621 /// so that they will be valid post-outlining.
622 ///
623 /// \param MBB A \p MachineBasicBlock in an outlined function.
624 void fixupPostOutline(MachineBasicBlock &MBB) const;
625
626 void instantiateCondBranch(MachineBasicBlock &MBB, const DebugLoc &DL,
627 MachineBasicBlock *TBB,
628 ArrayRef<MachineOperand> Cond) const;
629 bool substituteCmpToZero(MachineInstr &CmpInstr, unsigned SrcReg,
630 const MachineRegisterInfo &MRI) const;
631 bool removeCmpToZeroOrOne(MachineInstr &CmpInstr, unsigned SrcReg,
632 int CmpValue, const MachineRegisterInfo &MRI) const;
633
634 /// Returns an unused general-purpose register which can be used for
635 /// constructing an outlined call if one exists. Returns 0 otherwise.
636 Register findRegisterToSaveLRTo(outliner::Candidate &C) const;
637
638 /// Remove a ptest of a predicate-generating operation that already sets, or
639 /// can be made to set, the condition codes in an identical manner
640 bool optimizePTestInstr(MachineInstr *PTest, unsigned MaskReg,
641 unsigned PredReg,
642 const MachineRegisterInfo *MRI) const;
643 std::optional<unsigned>
644 canRemovePTestInstr(MachineInstr *PTest, MachineInstr *Mask,
645 MachineInstr *Pred, const MachineRegisterInfo *MRI) const;
646
647 /// verifyInstruction - Perform target specific instruction verification.
648 bool verifyInstruction(const MachineInstr &MI,
649 StringRef &ErrInfo) const override;
650};
651
652struct UsedNZCV {
653 bool N = false;
654 bool Z = false;
655 bool C = false;
656 bool V = false;
657
658 UsedNZCV() = default;
659
660 UsedNZCV &operator|=(const UsedNZCV &UsedFlags) {
661 this->N |= UsedFlags.N;
662 this->Z |= UsedFlags.Z;
663 this->C |= UsedFlags.C;
664 this->V |= UsedFlags.V;
665 return *this;
666 }
667};
668
669/// \returns Conditions flags used after \p CmpInstr in its MachineBB if NZCV
670/// flags are not alive in successors of the same \p CmpInstr and \p MI parent.
671/// \returns std::nullopt otherwise.
672///
673/// Collect instructions using that flags in \p CCUseInstrs if provided.
674std::optional<UsedNZCV>
675examineCFlagsUse(MachineInstr &MI, MachineInstr &CmpInstr,
676 const TargetRegisterInfo &TRI,
677 SmallVectorImpl<MachineInstr *> *CCUseInstrs = nullptr);
678
679/// Return true if there is an instruction /after/ \p DefMI and before \p UseMI
680/// which either reads or clobbers NZCV.
681bool isNZCVTouchedInInstructionRange(const MachineInstr &DefMI,
682 const MachineInstr &UseMI,
683 const TargetRegisterInfo *TRI);
684
685MCCFIInstruction createDefCFA(const TargetRegisterInfo &TRI, unsigned FrameReg,
686 unsigned Reg, const StackOffset &Offset,
687 bool LastAdjustmentWasScalable = true);
688MCCFIInstruction
689createCFAOffset(const TargetRegisterInfo &MRI, unsigned Reg,
690 const StackOffset &OffsetFromDefCFA,
691 std::optional<int64_t> IncomingVGOffsetFromDefCFA);
692
693/// emitFrameOffset - Emit instructions as needed to set DestReg to SrcReg
694/// plus Offset. This is intended to be used from within the prolog/epilog
695/// insertion (PEI) pass, where a virtual scratch register may be allocated
696/// if necessary, to be replaced by the scavenger at the end of PEI.
697void emitFrameOffset(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI,
698 const DebugLoc &DL, unsigned DestReg, unsigned SrcReg,
699 StackOffset Offset, const TargetInstrInfo *TII,
700 MachineInstr::MIFlag = MachineInstr::NoFlags,
701 bool SetNZCV = false, bool NeedsWinCFI = false,
702 bool *HasWinCFI = nullptr, bool EmitCFAOffset = false,
703 StackOffset InitialOffset = {},
704 unsigned FrameReg = AArch64::SP);
705
706/// rewriteAArch64FrameIndex - Rewrite MI to access 'Offset' bytes from the
707/// FP. Return false if the offset could not be handled directly in MI, and
708/// return the left-over portion by reference.
709bool rewriteAArch64FrameIndex(MachineInstr &MI, unsigned FrameRegIdx,
710 unsigned FrameReg, StackOffset &Offset,
711 const AArch64InstrInfo *TII);
712
713/// Use to report the frame offset status in isAArch64FrameOffsetLegal.
714enum AArch64FrameOffsetStatus {
715 AArch64FrameOffsetCannotUpdate = 0x0, ///< Offset cannot apply.
716 AArch64FrameOffsetIsLegal = 0x1, ///< Offset is legal.
717 AArch64FrameOffsetCanUpdate = 0x2 ///< Offset can apply, at least partly.
718};
719
720/// Check if the @p Offset is a valid frame offset for @p MI.
721/// The returned value reports the validity of the frame offset for @p MI.
722/// It uses the values defined by AArch64FrameOffsetStatus for that.
723/// If result == AArch64FrameOffsetCannotUpdate, @p MI cannot be updated to
724/// use an offset.eq
725/// If result & AArch64FrameOffsetIsLegal, @p Offset can completely be
726/// rewritten in @p MI.
727/// If result & AArch64FrameOffsetCanUpdate, @p Offset contains the
728/// amount that is off the limit of the legal offset.
729/// If set, @p OutUseUnscaledOp will contain the whether @p MI should be
730/// turned into an unscaled operator, which opcode is in @p OutUnscaledOp.
731/// If set, @p EmittableOffset contains the amount that can be set in @p MI
732/// (possibly with @p OutUnscaledOp if OutUseUnscaledOp is true) and that
733/// is a legal offset.
734int isAArch64FrameOffsetLegal(const MachineInstr &MI, StackOffset &Offset,
735 bool *OutUseUnscaledOp = nullptr,
736 unsigned *OutUnscaledOp = nullptr,
737 int64_t *EmittableOffset = nullptr);
738
739bool optimizeTerminators(MachineBasicBlock *MBB, const TargetInstrInfo &TII);
740
741static inline bool isUncondBranchOpcode(int Opc) { return Opc == AArch64::B; }
742
743static inline bool isCondBranchOpcode(int Opc) {
744 switch (Opc) {
745 case AArch64::Bcc:
746 case AArch64::CBZW:
747 case AArch64::CBZX:
748 case AArch64::CBNZW:
749 case AArch64::CBNZX:
750 case AArch64::TBZW:
751 case AArch64::TBZX:
752 case AArch64::TBNZW:
753 case AArch64::TBNZX:
754 case AArch64::CBWPri:
755 case AArch64::CBXPri:
756 case AArch64::CBBAssertExt:
757 case AArch64::CBHAssertExt:
758 case AArch64::CBWPrr:
759 case AArch64::CBXPrr:
760 return true;
761 default:
762 return false;
763 }
764}
765
766static inline bool isIndirectBranchOpcode(int Opc) {
767 switch (Opc) {
768 case AArch64::BR:
769 case AArch64::BRAA:
770 case AArch64::BRAB:
771 case AArch64::BRAAZ:
772 case AArch64::BRABZ:
773 return true;
774 }
775 return false;
776}
777
778static inline bool isIndirectCallOpcode(unsigned Opc) {
779 switch (Opc) {
780 case AArch64::BLR:
781 case AArch64::BLRAA:
782 case AArch64::BLRAB:
783 case AArch64::BLRAAZ:
784 case AArch64::BLRABZ:
785 return true;
786 default:
787 return false;
788 }
789}
790
791static inline bool isPTrueOpcode(unsigned Opc) {
792 switch (Opc) {
793 case AArch64::PTRUE_B:
794 case AArch64::PTRUE_H:
795 case AArch64::PTRUE_S:
796 case AArch64::PTRUE_D:
797 return true;
798 default:
799 return false;
800 }
801}
802
803/// Return opcode to be used for indirect calls.
804unsigned getBLRCallOpcode(const MachineFunction &MF);
805
806/// Return XPAC opcode to be used for a ptrauth strip using the given key.
807static inline unsigned getXPACOpcodeForKey(AArch64PACKey::ID K) {
808 using namespace AArch64PACKey;
809 switch (K) {
810 case IA: case IB: return AArch64::XPACI;
811 case DA: case DB: return AArch64::XPACD;
812 }
813 llvm_unreachable("Unhandled AArch64PACKey::ID enum");
814}
815
816/// Return AUT opcode to be used for a ptrauth auth using the given key, or its
817/// AUT*Z variant that doesn't take a discriminator operand, using zero instead.
818static inline unsigned getAUTOpcodeForKey(AArch64PACKey::ID K, bool Zero) {
819 using namespace AArch64PACKey;
820 switch (K) {
821 case IA: return Zero ? AArch64::AUTIZA : AArch64::AUTIA;
822 case IB: return Zero ? AArch64::AUTIZB : AArch64::AUTIB;
823 case DA: return Zero ? AArch64::AUTDZA : AArch64::AUTDA;
824 case DB: return Zero ? AArch64::AUTDZB : AArch64::AUTDB;
825 }
826 llvm_unreachable("Unhandled AArch64PACKey::ID enum");
827}
828
829/// Return PAC opcode to be used for a ptrauth sign using the given key, or its
830/// PAC*Z variant that doesn't take a discriminator operand, using zero instead.
831static inline unsigned getPACOpcodeForKey(AArch64PACKey::ID K, bool Zero) {
832 using namespace AArch64PACKey;
833 switch (K) {
834 case IA: return Zero ? AArch64::PACIZA : AArch64::PACIA;
835 case IB: return Zero ? AArch64::PACIZB : AArch64::PACIB;
836 case DA: return Zero ? AArch64::PACDZA : AArch64::PACDA;
837 case DB: return Zero ? AArch64::PACDZB : AArch64::PACDB;
838 }
839 llvm_unreachable("Unhandled AArch64PACKey::ID enum");
840}
841
842/// Return B(L)RA opcode to be used for an authenticated branch or call using
843/// the given key, or its B(L)RA*Z variant that doesn't take a discriminator
844/// operand, using zero instead.
845static inline unsigned getBranchOpcodeForKey(bool IsCall, AArch64PACKey::ID K,
846 bool Zero) {
847 using namespace AArch64PACKey;
848 static const unsigned BranchOpcode[2][2] = {
849 {AArch64::BRAA, AArch64::BRAAZ},
850 {AArch64::BRAB, AArch64::BRABZ},
851 };
852 static const unsigned CallOpcode[2][2] = {
853 {AArch64::BLRAA, AArch64::BLRAAZ},
854 {AArch64::BLRAB, AArch64::BLRABZ},
855 };
856
857 assert((K == IA || K == IB) && "B(L)RA* instructions require IA or IB key");
858 if (IsCall)
859 return CallOpcode[K == IB][Zero];
860 return BranchOpcode[K == IB][Zero];
861}
862
863// struct TSFlags {
864#define TSFLAG_ELEMENT_SIZE_TYPE(X) (X) // 3-bits
865#define TSFLAG_DESTRUCTIVE_INST_TYPE(X) ((X) << 3) // 4-bits
866#define TSFLAG_FALSE_LANE_TYPE(X) ((X) << 7) // 2-bits
867#define TSFLAG_INSTR_FLAGS(X) ((X) << 9) // 2-bits
868#define TSFLAG_SME_MATRIX_TYPE(X) ((X) << 11) // 3-bits
869// }
870
871namespace AArch64 {
872
873// clang-format off
874enum ElementSizeType {
875 ElementSizeMask = TSFLAG_ELEMENT_SIZE_TYPE(0x7),
876 ElementSizeNone = TSFLAG_ELEMENT_SIZE_TYPE(0x0),
877 ElementSizeB = TSFLAG_ELEMENT_SIZE_TYPE(0x1),
878 ElementSizeH = TSFLAG_ELEMENT_SIZE_TYPE(0x2),
879 ElementSizeS = TSFLAG_ELEMENT_SIZE_TYPE(0x3),
880 ElementSizeD = TSFLAG_ELEMENT_SIZE_TYPE(0x4),
881};
882
883enum DestructiveInstType {
884 DestructiveInstTypeMask = TSFLAG_DESTRUCTIVE_INST_TYPE(0xf),
885 NotDestructive = TSFLAG_DESTRUCTIVE_INST_TYPE(0x0),
886 DestructiveOther = TSFLAG_DESTRUCTIVE_INST_TYPE(0x1),
887 DestructiveUnary = TSFLAG_DESTRUCTIVE_INST_TYPE(0x2),
888 DestructiveBinaryImm = TSFLAG_DESTRUCTIVE_INST_TYPE(0x3),
889 DestructiveBinaryShImmUnpred = TSFLAG_DESTRUCTIVE_INST_TYPE(0x4),
890 DestructiveBinary = TSFLAG_DESTRUCTIVE_INST_TYPE(0x5),
891 DestructiveBinaryComm = TSFLAG_DESTRUCTIVE_INST_TYPE(0x6),
892 DestructiveBinaryCommWithRev = TSFLAG_DESTRUCTIVE_INST_TYPE(0x7),
893 DestructiveTernaryCommWithRev = TSFLAG_DESTRUCTIVE_INST_TYPE(0x8),
894 Destructive2xRegImmUnpred = TSFLAG_DESTRUCTIVE_INST_TYPE(0x9),
895 DestructiveUnaryPassthru = TSFLAG_DESTRUCTIVE_INST_TYPE(0xa),
896 DestructivePredicate = TSFLAG_DESTRUCTIVE_INST_TYPE(0xb),
897 DestructiveBinaryImmUnpred = TSFLAG_DESTRUCTIVE_INST_TYPE(0xc),
898};
899
900enum FalseLaneType {
901 FalseLanesMask = TSFLAG_FALSE_LANE_TYPE(0x3),
902 FalseLanesZero = TSFLAG_FALSE_LANE_TYPE(0x1),
903 FalseLanesUndef = TSFLAG_FALSE_LANE_TYPE(0x2),
904};
905
906// clang-format on
907
908// NOTE: This is a bit field.
909static const uint64_t InstrFlagIsWhile = TSFLAG_INSTR_FLAGS(0x1);
910static const uint64_t InstrFlagIsPTestLike = TSFLAG_INSTR_FLAGS(0x2);
911
912enum SMEMatrixType {
913 SMEMatrixTypeMask = TSFLAG_SME_MATRIX_TYPE(0x7),
914 SMEMatrixNone = TSFLAG_SME_MATRIX_TYPE(0x0),
915 SMEMatrixTileB = TSFLAG_SME_MATRIX_TYPE(0x1),
916 SMEMatrixTileH = TSFLAG_SME_MATRIX_TYPE(0x2),
917 SMEMatrixTileS = TSFLAG_SME_MATRIX_TYPE(0x3),
918 SMEMatrixTileD = TSFLAG_SME_MATRIX_TYPE(0x4),
919 SMEMatrixTileQ = TSFLAG_SME_MATRIX_TYPE(0x5),
920 SMEMatrixArray = TSFLAG_SME_MATRIX_TYPE(0x6),
921};
922
923#undef TSFLAG_ELEMENT_SIZE_TYPE
924#undef TSFLAG_DESTRUCTIVE_INST_TYPE
925#undef TSFLAG_FALSE_LANE_TYPE
926#undef TSFLAG_INSTR_FLAGS
927#undef TSFLAG_SME_MATRIX_TYPE
928
929int32_t getSVEPseudoMap(uint32_t Opcode);
930int32_t getSVERevInstr(uint32_t Opcode);
931int32_t getSVENonRevInstr(uint32_t Opcode);
932
933int32_t getSMEPseudoMap(uint32_t Opcode);
934}
935
936} // end namespace llvm
937
938#endif
939