1//===--- AArch64Subtarget.h - Define Subtarget for the AArch64 -*- C++ -*--===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file declares the AArch64 specific subclass of TargetSubtarget.
10//
11//===----------------------------------------------------------------------===//
12
13#ifndef LLVM_LIB_TARGET_AARCH64_AARCH64SUBTARGET_H
14#define LLVM_LIB_TARGET_AARCH64_AARCH64SUBTARGET_H
15
16#include "AArch64FrameLowering.h"
17#include "AArch64ISelLowering.h"
18#include "AArch64InstrInfo.h"
19#include "AArch64PointerAuth.h"
20#include "AArch64RegisterInfo.h"
21#include "AArch64SelectionDAGInfo.h"
22#include "llvm/CodeGen/GlobalISel/CallLowering.h"
23#include "llvm/CodeGen/GlobalISel/InlineAsmLowering.h"
24#include "llvm/CodeGen/GlobalISel/InstructionSelector.h"
25#include "llvm/CodeGen/GlobalISel/LegalizerInfo.h"
26#include "llvm/CodeGen/RegisterBankInfo.h"
27#include "llvm/CodeGen/TargetSubtargetInfo.h"
28#include "llvm/IR/DataLayout.h"
29#include "llvm/TargetParser/Triple.h"
30#include <optional>
31
32namespace llvm::AArch64 {
33// Mode for selecting how to insert frame record info into the stack ring
34// buffer.
35enum class StackTaggingRecordStackHistoryMode {
36 // Do not record frame record info.
37 None,
38
39 // Insert instructions into the prologue for storing into the stack ring
40 // buffer directly.
41 Instr,
42};
43
44enum class UncheckedLdStMode { Never, Safe, Always };
45} // namespace llvm::AArch64
46
47#define OPTIONS_STRUCT_DECL
48#include "AArch64Options.inc"
49
50#define GET_SUBTARGETINFO_HEADER
51#include "AArch64GenSubtargetInfo.inc"
52
53namespace llvm {
54class GlobalValue;
55class StringRef;
56
57class AArch64Subtarget final : public AArch64GenSubtargetInfo {
58 const AArch64Options &CLOpts;
59
60public:
61 enum ARMProcFamilyEnum : uint8_t {
62 Generic,
63#define ARM_PROCESSOR_FAMILY(ENUM) ENUM,
64#include "llvm/TargetParser/AArch64TargetParserDef.inc"
65#undef ARM_PROCESSOR_FAMILY
66 };
67
68protected:
69 /// ARMProcFamily - ARM processor family: Cortex-A53, Cortex-A57, and others.
70 ARMProcFamilyEnum ARMProcFamily = Generic;
71
72 // Enable 64-bit vectorization in SLP.
73 unsigned MinVectorRegisterBitWidth = 64;
74
75// Bool members corresponding to the SubtargetFeatures defined in tablegen
76#define GET_SUBTARGETINFO_MACRO(ATTRIBUTE, DEFAULT, GETTER) \
77 bool ATTRIBUTE = DEFAULT;
78#include "AArch64GenSubtargetInfo.inc"
79
80 unsigned EpilogueVectorizationMinVF = 16;
81 uint8_t MaxInterleaveFactor = 2;
82 uint8_t VectorInsertExtractBaseCost = 2;
83 uint16_t CacheLineSize = 64;
84 // Default scatter/gather overhead.
85 unsigned ScatterOverhead = 10;
86 unsigned GatherOverhead = 10;
87 uint16_t PrefetchDistance = 0;
88 uint16_t MinPrefetchStride = 1;
89 unsigned MaxPrefetchIterationsAhead = UINT_MAX;
90 Align PrefFunctionAlignment;
91 Align PrefLoopAlignment;
92 unsigned MaxBytesForLoopAlignment = 0;
93 unsigned MinimumJumpTableEntries = 4;
94 unsigned MaxJumpTableSize = 0;
95 unsigned FixedLoadLatency = 0;
96
97 // ReserveXRegister[i] - X#i is not available as a general purpose register.
98 BitVector ReserveXRegister;
99
100 // ReserveXRegisterForRA[i] - X#i is not available for register allocator.
101 BitVector ReserveXRegisterForRA;
102
103 // CustomCallUsedXRegister[i] - X#i call saved.
104 BitVector CustomCallSavedXRegs;
105
106 bool IsLittle;
107
108 bool IsStreaming;
109 bool IsStreamingCompatible;
110 unsigned MinSVEVectorSizeInBits;
111 unsigned MaxSVEVectorSizeInBits;
112 bool EnableSRLTSubregToRegMitigation;
113 unsigned VScaleForTuning = 1;
114 TailFoldingOpts DefaultSVETFOpts = TailFoldingOpts::Disabled;
115
116 bool EnableSubregLiveness;
117
118 /// TargetTriple - What processor and OS we're targeting.
119 Triple TargetTriple;
120
121 AArch64FrameLowering FrameLowering;
122 AArch64InstrInfo InstrInfo;
123 AArch64SelectionDAGInfo TSInfo;
124 AArch64TargetLowering TLInfo;
125
126 /// GlobalISel related APIs.
127 std::unique_ptr<CallLowering> CallLoweringInfo;
128 std::unique_ptr<InlineAsmLowering> InlineAsmLoweringInfo;
129 std::unique_ptr<InstructionSelector> InstSelector;
130 std::unique_ptr<LegalizerInfo> Legalizer;
131 std::unique_ptr<RegisterBankInfo> RegBankInfo;
132
133private:
134 /// initializeSubtargetDependencies - Initializes using CPUString and the
135 /// passed in feature string so that we can use initializer lists for
136 /// subtarget initialization.
137 AArch64Subtarget &initializeSubtargetDependencies(StringRef FS,
138 StringRef CPUString,
139 StringRef TuneCPUString,
140 bool HasMinSize);
141
142 /// Initialize properties based on the selected processor family.
143 void initializeProperties(bool HasMinSize);
144
145public:
146 /// This constructor initializes the data members to match that
147 /// of the specified triple.
148 AArch64Subtarget(const Triple &TT, StringRef CPU, StringRef TuneCPU,
149 StringRef FS, const TargetMachine &TM, bool LittleEndian,
150 unsigned MinSVEVectorSizeInBitsOverride = 0,
151 unsigned MaxSVEVectorSizeInBitsOverride = 0,
152 bool IsStreaming = false, bool IsStreamingCompatible = false,
153 bool HasMinSize = false,
154 bool EnableSRLTSubregToRegMitigation = false);
155
156 const AArch64Options &getCLOpts() const { return CLOpts; }
157
158// Getters for SubtargetFeatures defined in tablegen
159#define GET_SUBTARGETINFO_MACRO(ATTRIBUTE, DEFAULT, GETTER) \
160 bool GETTER() const { return ATTRIBUTE; }
161#include "AArch64GenSubtargetInfo.inc"
162
163 const AArch64SelectionDAGInfo *getSelectionDAGInfo() const override {
164 return &TSInfo;
165 }
166 const AArch64FrameLowering *getFrameLowering() const override {
167 return &FrameLowering;
168 }
169 const AArch64TargetLowering *getTargetLowering() const override {
170 return &TLInfo;
171 }
172 const AArch64InstrInfo *getInstrInfo() const override { return &InstrInfo; }
173 const AArch64RegisterInfo *getRegisterInfo() const override {
174 return &getInstrInfo()->getRegisterInfo();
175 }
176 const CallLowering *getCallLowering() const override;
177 const InlineAsmLowering *getInlineAsmLowering() const override;
178 InstructionSelector *getInstructionSelector() const override;
179 const LegalizerInfo *getLegalizerInfo() const override;
180 const RegisterBankInfo *getRegBankInfo() const override;
181 const Triple &getTargetTriple() const { return TargetTriple; }
182 bool enableMachineScheduler() const override { return true; }
183 bool enablePostRAScheduler() const override { return usePostRAScheduler(); }
184 bool enableSubRegLiveness() const override { return EnableSubregLiveness; }
185
186 bool enableMachinePipeliner() const override;
187 bool useDFAforSMS() const override { return false; }
188
189 /// Returns ARM processor family.
190 /// Avoid this function! CPU specifics should be kept local to this class
191 /// and preferably modeled with SubtargetFeatures or properties in
192 /// initializeProperties().
193 ARMProcFamilyEnum getProcFamily() const {
194 return ARMProcFamily;
195 }
196
197 /// Returns true if the processor is an Apple M-series or aligned A-series
198 /// (A14 or newer).
199 bool isAppleMLike() const {
200 switch (ARMProcFamily) {
201 case AppleA14:
202 case AppleA15:
203 case AppleA16:
204 case AppleA17:
205 case AppleM4:
206 case AppleM5:
207 return true;
208 default:
209 return false;
210 }
211 }
212
213 bool isXRaySupported() const override { return true; }
214
215 /// Returns true if the function has a streaming body.
216 bool isStreaming() const { return IsStreaming; }
217
218 /// Returns true if the function has a streaming-compatible body.
219 bool isStreamingCompatible() const { return IsStreamingCompatible; }
220
221 /// Returns the size of memory region that if accessed by both the CPU and
222 /// the SME unit could result in a hazard. 0 = disabled.
223 unsigned getStreamingHazardSize() const {
224 return CLOpts.streaming_hazard_size.value_or(
225 u: !hasSMEFA64() && hasSME() && hasSVE() ? 1024 : 0);
226 }
227
228 /// Returns true if the target has NEON and the function at runtime is known
229 /// to have NEON enabled (e.g. the function is known not to be in streaming-SVE
230 /// mode, which disables NEON instructions).
231 bool isNeonAvailable() const {
232 return hasNEON() &&
233 (hasSMEFA64() || (!isStreaming() && !isStreamingCompatible()));
234 }
235
236 /// Returns true if the target has SVE and can use the full range of SVE
237 /// instructions, for example because it knows the function is known not to be
238 /// in streaming-SVE mode or when the target has FEAT_FA64 enabled.
239 bool isSVEAvailable() const {
240 return hasSVE() &&
241 (hasSMEFA64() || (!isStreaming() && !isStreamingCompatible()));
242 }
243
244 /// Returns true if the target has access to the streaming-compatible subset
245 /// of SVE instructions.
246 bool isStreamingSVEAvailable() const { return hasSME() && isStreaming(); }
247
248 /// Returns true if the target has access to either the full range of SVE
249 /// instructions, or the streaming-compatible subset of SVE instructions.
250 bool isSVEorStreamingSVEAvailable() const {
251 return hasSVE() || isStreamingSVEAvailable();
252 }
253
254 /// Returns true if the target has access to either the full range of SVE
255 /// instructions, or the streaming-compatible subset of SVE instructions
256 /// available to SME2.
257 bool isNonStreamingSVEorSME2Available() const {
258 return isSVEAvailable() || (isSVEorStreamingSVEAvailable() && hasSME2());
259 }
260
261 unsigned getMinVectorRegisterBitWidth() const {
262 // Don't assume any minimum vector size when PSTATE.SM may not be 0, because
263 // we don't yet support streaming-compatible codegen support that we trust
264 // is safe for functions that may be executed in streaming-SVE mode.
265 // By returning '0' here, we disable vectorization.
266 if (!isSVEAvailable() && !isNeonAvailable())
267 return 0;
268 return MinVectorRegisterBitWidth;
269 }
270
271 bool isXRegisterReserved(size_t i) const { return ReserveXRegister[i]; }
272 bool isXRegisterReservedForRA(size_t i) const { return ReserveXRegisterForRA[i]; }
273 unsigned getNumXRegisterReserved() const {
274 BitVector AllReservedX(AArch64::GPR64commonRegClass.getNumRegs());
275 AllReservedX |= ReserveXRegister;
276 AllReservedX |= ReserveXRegisterForRA;
277 return AllReservedX.count();
278 }
279 bool isLRReservedForRA() const { return ReserveLRForRA; }
280 bool isXRegCustomCalleeSaved(size_t i) const {
281 return CustomCallSavedXRegs[i];
282 }
283 bool hasCustomCallingConv() const { return CustomCallSavedXRegs.any(); }
284
285 /// Return true if the CPU supports any kind of instruction fusion.
286 bool hasFusion() const {
287 return hasArithmeticBccFusion() || hasArithmeticCbzFusion() ||
288 hasFuseAES() || hasFuseArithmeticLogic() || hasFuseCmpCSel() ||
289 hasFuseFCmpFCSel() || hasFuseCmpCSet() || hasFuseAdrpAdd() ||
290 hasFuseLiterals() || hasFuseAppleSMECompute() || hasFuseFMinFMax();
291 }
292
293 /// Return true if the subtarget fuses this pair of move immediate
294 /// instructions.
295 bool fusesMOVImmPair(unsigned FirstOpc, unsigned FirstShift,
296 unsigned SecondOpc, unsigned SecondShift) const;
297
298 /// Return true if the subtarget fuses this pair of move immediate
299 /// instructions. The 1st instruction is a wildcard when it is nullptr, which
300 /// tells whether the 2nd one can be fused at all.
301 bool fusesMOVImmPair(const MachineInstr *FirstMI,
302 const MachineInstr &SecondMI) const;
303
304 unsigned getEpilogueVectorizationMinVF() const {
305 return EpilogueVectorizationMinVF;
306 }
307 unsigned getMaxInterleaveFactor() const { return MaxInterleaveFactor; }
308 unsigned getVectorInsertExtractBaseCost() const;
309 unsigned getCacheLineSize() const override { return CacheLineSize; }
310 unsigned getScatterOverhead() const { return ScatterOverhead; }
311 unsigned getGatherOverhead() const { return GatherOverhead; }
312 unsigned getPrefetchDistance() const override { return PrefetchDistance; }
313 unsigned getMinPrefetchStride(unsigned NumMemAccesses,
314 unsigned NumStridedMemAccesses,
315 unsigned NumPrefetches,
316 bool HasCall) const override {
317 return MinPrefetchStride;
318 }
319 unsigned getMaxPrefetchIterationsAhead() const override {
320 return MaxPrefetchIterationsAhead;
321 }
322 Align getPrefFunctionAlignment() const {
323 return PrefFunctionAlignment;
324 }
325 Align getPrefLoopAlignment() const { return PrefLoopAlignment; }
326
327 unsigned getMaxBytesForLoopAlignment() const {
328 return MaxBytesForLoopAlignment;
329 }
330
331 unsigned getMaximumJumpTableSize() const { return MaxJumpTableSize; }
332 unsigned getMinimumJumpTableEntries() const {
333 return MinimumJumpTableEntries;
334 }
335
336 unsigned getFixedLoadLatency() const { return FixedLoadLatency; }
337
338 /// CPU has TBI (top byte of addresses is ignored during HW address
339 /// translation) and OS enables it.
340 bool supportsAddressTopByteIgnored() const;
341
342 bool isLittleEndian() const { return IsLittle; }
343
344 bool isTargetDarwin() const { return TargetTriple.isOSDarwin(); }
345 bool isTargetIOS() const { return TargetTriple.isiOS(); }
346 bool isTargetLinux() const { return TargetTriple.isOSLinux(); }
347 bool isTargetWindows() const { return TargetTriple.isOSWindows(); }
348 bool isTargetAndroid() const { return TargetTriple.isAndroid(); }
349 bool isTargetFuchsia() const { return TargetTriple.isOSFuchsia(); }
350 bool isWindowsArm64EC() const { return TargetTriple.isWindowsArm64EC(); }
351 bool isLFI() const { return TargetTriple.isLFI(); }
352
353 bool isTargetCOFF() const { return TargetTriple.isOSBinFormatCOFF(); }
354 bool isTargetELF() const { return TargetTriple.isOSBinFormatELF(); }
355 bool isTargetMachO() const { return TargetTriple.isOSBinFormatMachO(); }
356
357 bool isTargetILP32() const {
358 return TargetTriple.isArch32Bit() ||
359 TargetTriple.getEnvironment() == Triple::GNUILP32;
360 }
361
362 bool useAA() const override;
363
364 bool addrSinkUsingGEPs() const override {
365 // Keeping GEPs inbounds is important for exploiting AArch64
366 // addressing-modes in ILP32 mode.
367 return useAA() || isTargetILP32();
368 }
369
370 bool useSmallAddressing() const {
371 switch (TLInfo.getTargetMachine().getCodeModel()) {
372 case CodeModel::Kernel:
373 // Kernel is currently allowed only for Fuchsia targets,
374 // where it is the same as Small for almost all purposes.
375 case CodeModel::Small:
376 return true;
377 default:
378 return false;
379 }
380 }
381
382 /// Returns whether the operating system makes it safer to store sensitive
383 /// values in x16 and x17 as opposed to other registers.
384 bool isX16X17Safer() const;
385
386 /// ParseSubtargetFeatures - Parses features string setting specified
387 /// subtarget options. Definition of function is auto generated by tblgen.
388 void ParseSubtargetFeatures(StringRef CPU, StringRef TuneCPU, StringRef FS);
389
390 /// ClassifyGlobalReference - Find the target operand flags that describe
391 /// how a global value should be referenced for the current subtarget.
392 unsigned ClassifyGlobalReference(const GlobalValue *GV,
393 const TargetMachine &TM) const;
394
395 unsigned classifyGlobalFunctionReference(const GlobalValue *GV,
396 const TargetMachine &TM) const;
397
398 /// This function is design to compatible with the function def in other
399 /// targets and escape build error about the virtual function def in base
400 /// class TargetSubtargetInfo. Updeate me if AArch64 target need to use it.
401 unsigned char
402 classifyGlobalFunctionReference(const GlobalValue *GV) const override {
403 return 0;
404 }
405
406 void overrideSchedPolicy(MachineSchedPolicy &Policy,
407 const SchedRegion &Region) const override;
408
409 void adjustSchedDependency(SUnit *Def, int DefOpIdx, SUnit *Use, int UseOpIdx,
410 SDep &Dep,
411 const TargetSchedModel *SchedModel) const override;
412
413 bool enableEarlyIfConversion() const override;
414
415 std::unique_ptr<PBQPRAConstraint> getCustomPBQPConstraints() const override;
416
417 bool isCallingConvWin64(CallingConv::ID CC, bool IsVarArg) const {
418 switch (CC) {
419 case CallingConv::C:
420 case CallingConv::Fast:
421 case CallingConv::Swift:
422 case CallingConv::SwiftTail:
423 return isTargetWindows();
424 case CallingConv::PreserveNone:
425 return IsVarArg && isTargetWindows();
426 case CallingConv::Win64:
427 return true;
428 default:
429 return false;
430 }
431 }
432
433 /// Return whether FrameLowering should always set the "extended frame
434 /// present" bit in FP, or set it based on a symbol in the runtime.
435 bool swiftAsyncContextIsDynamicallySet() const {
436 // Older OS versions (particularly system unwinders) are confused by the
437 // Swift extended frame, so when building code that might be run on them we
438 // must dynamically query the concurrency library to determine whether
439 // extended frames should be flagged as present.
440 const Triple &TT = getTargetTriple();
441
442 unsigned Major = TT.getOSVersion().getMajor();
443 switch(TT.getOS()) {
444 default:
445 return false;
446 case Triple::IOS:
447 case Triple::TvOS:
448 return Major < 15;
449 case Triple::WatchOS:
450 return Major < 8;
451 case Triple::MacOSX:
452 case Triple::Darwin:
453 return Major < 12;
454 }
455 }
456
457 void mirFileLoaded(MachineFunction &MF) const override;
458
459 // Return the known range for the bit length of SVE data registers. A value
460 // of 0 means nothing is known about that particular limit beyond what's
461 // implied by the architecture.
462 unsigned getMaxSVEVectorSizeInBits() const {
463 assert(isSVEorStreamingSVEAvailable() &&
464 "Tried to get SVE vector length without SVE support!");
465 return MaxSVEVectorSizeInBits;
466 }
467
468 unsigned getMinSVEVectorSizeInBits() const {
469 assert(isSVEorStreamingSVEAvailable() &&
470 "Tried to get SVE vector length without SVE support!");
471 return MinSVEVectorSizeInBits;
472 }
473
474 // Return the known bit length of SVE data registers. A value of 0 means the
475 // length is unknown beyond what's implied by the architecture.
476 unsigned getSVEVectorSizeInBits() const {
477 assert(isSVEorStreamingSVEAvailable() &&
478 "Tried to get SVE vector length without SVE support!");
479 if (MinSVEVectorSizeInBits == MaxSVEVectorSizeInBits)
480 return MaxSVEVectorSizeInBits;
481 return 0;
482 }
483
484 // Return the known bit length of SVE predicate registers. A value of 0 means
485 // the length is unknown beyond what's implied by the architecture.
486 unsigned getSVEPredicateSizeInBits() const {
487 return getSVEVectorSizeInBits() / 8;
488 }
489
490 bool useSVEForFixedLengthVectors() const {
491 if (!isSVEorStreamingSVEAvailable())
492 return false;
493
494 // Prefer NEON unless larger SVE registers are available.
495 return !isNeonAvailable() || getMinSVEVectorSizeInBits() >= 256;
496 }
497
498 bool useSVEForFixedLengthVectors(EVT VT) const {
499 if (!useSVEForFixedLengthVectors() || !VT.isFixedLengthVector())
500 return false;
501 return VT.getFixedSizeInBits() > AArch64::SVEBitsPerBlock ||
502 !isNeonAvailable();
503 }
504
505 unsigned getVScaleForTuning() const { return VScaleForTuning; }
506
507 TailFoldingOpts getSVETailFoldingDefaultOpts() const {
508 return DefaultSVETFOpts;
509 }
510
511 /// Returns true to use the addvl/inc/dec instructions, as opposed to separate
512 /// add + cnt instructions.
513 bool useScalarIncVL() const;
514
515 bool enableSRLTSubregToRegMitigation() const {
516 return EnableSRLTSubregToRegMitigation;
517 }
518
519 /// Choose a method of checking LR before performing a tail call.
520 AArch64PAuth::AuthCheckMethod
521 getAuthenticatedLRCheckMethod(const MachineFunction &MF) const;
522
523 /// Compute the integer discriminator for a given BlockAddress constant, if
524 /// blockaddress signing is enabled, or std::nullopt otherwise.
525 /// Blockaddress signing is controlled by the function attribute
526 /// "ptrauth-indirect-gotos" on the parent function.
527 /// Note that this assumes the discriminator is independent of the indirect
528 /// goto branch site itself, i.e., it's the same for all BlockAddresses in
529 /// a function.
530 std::optional<uint16_t>
531 getPtrAuthBlockAddressDiscriminatorIfEnabled(const Function &ParentFn) const;
532
533 bool enableAggressiveInterleaving() const { return AggressiveInterleaving; }
534};
535} // End llvm namespace
536
537#endif
538