1//===-- AArch64Subtarget.cpp - AArch64 Subtarget Information ----*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the AArch64 specific subclass of TargetSubtarget.
10//
11//===----------------------------------------------------------------------===//
12
13#include "AArch64Subtarget.h"
14
15#include "AArch64.h"
16#include "AArch64InstrInfo.h"
17#include "AArch64PBQPRegAlloc.h"
18#include "AArch64TargetMachine.h"
19#include "GISel/AArch64CallLowering.h"
20#include "GISel/AArch64LegalizerInfo.h"
21#include "GISel/AArch64RegisterBankInfo.h"
22#include "MCTargetDesc/AArch64AddressingModes.h"
23#include "llvm/CodeGen/GlobalISel/InstructionSelect.h"
24#include "llvm/CodeGen/MachineFrameInfo.h"
25#include "llvm/CodeGen/MachineScheduler.h"
26#include "llvm/IR/GlobalValue.h"
27#include "llvm/Option/LibraryOptions.h"
28#include "llvm/Support/SipHash.h"
29#include "llvm/TargetParser/AArch64TargetParser.h"
30
31using namespace llvm;
32
33#define DEBUG_TYPE "aarch64-subtarget"
34
35#define GET_SUBTARGETINFO_CTOR
36#define GET_SUBTARGETINFO_TARGET_DESC
37#include "AArch64GenSubtargetInfo.inc"
38
39#define OPTIONS_STRUCT_DEFS
40#include "AArch64Options.inc"
41
42// Reserve a list of X# registers, so they are unavailable for register
43// allocator, but can still be used as ABI requests, such as passing arguments
44// to function call.
45static cl::list<std::string>
46ReservedRegsForRA("reserve-regs-for-regalloc", cl::desc("Reserve physical "
47 "registers, so they can't be used by register allocator. "
48 "Should only be used for testing register allocator."),
49 cl::CommaSeparated, cl::Hidden);
50
51unsigned AArch64Subtarget::getVectorInsertExtractBaseCost() const {
52 return CLOpts.insert_extract_base_cost.value_or(u: VectorInsertExtractBaseCost);
53}
54
55AArch64Subtarget &AArch64Subtarget::initializeSubtargetDependencies(
56 StringRef FS, StringRef CPUString, StringRef TuneCPUString,
57 bool HasMinSize) {
58 // Determine default and user-specified characteristics
59
60 if (CPUString.empty())
61 CPUString = "generic";
62
63 if (TuneCPUString.empty())
64 TuneCPUString = CPUString;
65
66 ParseSubtargetFeatures(CPU: CPUString, TuneCPU: TuneCPUString, FS);
67 initializeProperties(HasMinSize);
68
69 return *this;
70}
71
72void AArch64Subtarget::initializeProperties(bool HasMinSize) {
73 // Initialize CPU specific properties. We should add a tablegen feature for
74 // this in the future so we can specify it together with the subtarget
75 // features.
76 switch (ARMProcFamily) {
77 case Generic:
78 // Using TuneCPU=generic we avoid ldapur instructions to line up with the
79 // cpus that use the AvoidLDAPUR feature. We don't want this to be on
80 // forever, so it is enabled between armv8.4 and armv8.7/armv9.2.
81 if (hasV8_4aOps() && !hasV8_8aOps())
82 AvoidLDAPUR = true;
83 break;
84 case Carmel:
85 break;
86 case CortexA35:
87 case CortexA53:
88 case CortexA55:
89 case CortexR82:
90 case CortexR82AE:
91 PrefFunctionAlignment = Align(16);
92 PrefLoopAlignment = Align(16);
93 MaxBytesForLoopAlignment = 8;
94 break;
95 case CortexA57:
96 PrefFunctionAlignment = Align(16);
97 PrefLoopAlignment = Align(16);
98 MaxBytesForLoopAlignment = 8;
99 break;
100 case CortexA65:
101 PrefFunctionAlignment = Align(8);
102 break;
103 case CortexA72:
104 case CortexA73:
105 case CortexA75:
106 PrefFunctionAlignment = Align(16);
107 PrefLoopAlignment = Align(16);
108 MaxBytesForLoopAlignment = 8;
109 break;
110 case CortexA76:
111 case CortexA77:
112 case CortexA78:
113 case CortexA78AE:
114 case CortexA78C:
115 case CortexX1:
116 PrefFunctionAlignment = Align(16);
117 PrefLoopAlignment = Align(32);
118 MaxBytesForLoopAlignment = 16;
119 break;
120 case CortexA320:
121 case CortexA510:
122 case CortexA520:
123 case C1Nano:
124 PrefFunctionAlignment = Align(16);
125 VScaleForTuning = 1;
126 PrefLoopAlignment = Align(16);
127 MaxBytesForLoopAlignment = 8;
128 break;
129 case CortexA710:
130 case CortexA715:
131 case CortexA720:
132 case CortexA725:
133 case C1Pro:
134 case CortexX2:
135 case CortexX3:
136 case CortexX4:
137 case CortexX925:
138 case C1Premium:
139 case C1Ultra:
140 case C2Ultra:
141 PrefFunctionAlignment = Align(16);
142 VScaleForTuning = 1;
143 PrefLoopAlignment = Align(32);
144 MaxBytesForLoopAlignment = 16;
145 break;
146 case A64FX:
147 CacheLineSize = 256;
148 PrefFunctionAlignment = Align(8);
149 PrefLoopAlignment = Align(4);
150 PrefetchDistance = 128;
151 MinPrefetchStride = 1024;
152 MaxPrefetchIterationsAhead = 4;
153 VScaleForTuning = 4;
154 break;
155 case MONAKA:
156 VScaleForTuning = 2;
157 break;
158 case AppleA7:
159 case AppleA10:
160 case AppleA11:
161 case AppleA12:
162 case AppleA13:
163 case AppleA14:
164 case AppleA15:
165 case AppleA16:
166 case AppleA17:
167 case AppleM4:
168 case AppleM5:
169 PrefetchDistance = 280;
170 MinPrefetchStride = 2048;
171 MaxPrefetchIterationsAhead = 3;
172 break;
173 case ExynosM3:
174 MaxJumpTableSize = 20;
175 PrefFunctionAlignment = Align(32);
176 PrefLoopAlignment = Align(16);
177 break;
178 case Falkor:
179 // FIXME: remove this to enable 64-bit SLP if performance looks good.
180 MinVectorRegisterBitWidth = 128;
181 CacheLineSize = 128;
182 PrefetchDistance = 820;
183 MinPrefetchStride = 2048;
184 MaxPrefetchIterationsAhead = 8;
185 break;
186 case Kryo:
187 VectorInsertExtractBaseCost = 2;
188 CacheLineSize = 128;
189 PrefetchDistance = 740;
190 MinPrefetchStride = 1024;
191 MaxPrefetchIterationsAhead = 11;
192 // FIXME: remove this to enable 64-bit SLP if performance looks good.
193 MinVectorRegisterBitWidth = 128;
194 break;
195 case NeoverseE1:
196 PrefFunctionAlignment = Align(8);
197 break;
198 case NeoverseN1:
199 PrefFunctionAlignment = Align(16);
200 PrefLoopAlignment = Align(32);
201 MaxBytesForLoopAlignment = 16;
202 break;
203 case NeoverseV2:
204 case NeoverseV3:
205 EpilogueVectorizationMinVF = 8;
206 ScatterOverhead = 13;
207 [[fallthrough]];
208 case NeoverseN2:
209 case NeoverseN3:
210 case NeoverseV3AE:
211 PrefFunctionAlignment = Align(16);
212 PrefLoopAlignment = Align(32);
213 MaxBytesForLoopAlignment = 16;
214 VScaleForTuning = 1;
215 break;
216 case NeoverseV1:
217 PrefFunctionAlignment = Align(16);
218 PrefLoopAlignment = Align(32);
219 MaxBytesForLoopAlignment = 16;
220 VScaleForTuning = 2;
221 DefaultSVETFOpts = TailFoldingOpts::Simple;
222 break;
223 case Neoverse512TVB:
224 PrefFunctionAlignment = Align(16);
225 VScaleForTuning = 1;
226 break;
227 case Saphira:
228 // FIXME: remove this to enable 64-bit SLP if performance looks good.
229 MinVectorRegisterBitWidth = 128;
230 break;
231 case ThunderX2T99:
232 PrefFunctionAlignment = Align(8);
233 PrefLoopAlignment = Align(4);
234 PrefetchDistance = 128;
235 MinPrefetchStride = 1024;
236 MaxPrefetchIterationsAhead = 4;
237 // FIXME: remove this to enable 64-bit SLP if performance looks good.
238 MinVectorRegisterBitWidth = 128;
239 break;
240 case ThunderX:
241 case ThunderXT88:
242 case ThunderXT81:
243 case ThunderXT83:
244 CacheLineSize = 128;
245 PrefFunctionAlignment = Align(8);
246 PrefLoopAlignment = Align(4);
247 // FIXME: remove this to enable 64-bit SLP if performance looks good.
248 MinVectorRegisterBitWidth = 128;
249 break;
250 case TSV110:
251 PrefFunctionAlignment = Align(16);
252 PrefLoopAlignment = Align(4);
253 break;
254 case HIP12:
255 PrefFunctionAlignment = Align(16);
256 PrefLoopAlignment = Align(4);
257 VScaleForTuning = 2;
258 DefaultSVETFOpts = TailFoldingOpts::Simple;
259 break;
260 case ThunderX3T110:
261 PrefFunctionAlignment = Align(16);
262 PrefLoopAlignment = Align(4);
263 PrefetchDistance = 128;
264 MinPrefetchStride = 1024;
265 MaxPrefetchIterationsAhead = 4;
266 // FIXME: remove this to enable 64-bit SLP if performance looks good.
267 MinVectorRegisterBitWidth = 128;
268 break;
269 case Ampere1:
270 case Ampere1A:
271 case Ampere1B:
272 case Ampere1C:
273 PrefFunctionAlignment = Align(64);
274 PrefLoopAlignment = Align(64);
275 break;
276 case Oryon:
277 PrefFunctionAlignment = Align(16);
278 PrefetchDistance = 128;
279 MinPrefetchStride = 1024;
280 break;
281 case Olympus:
282 EpilogueVectorizationMinVF = 8;
283 ScatterOverhead = 13;
284 PrefFunctionAlignment = Align(16);
285 PrefLoopAlignment = Align(32);
286 MaxBytesForLoopAlignment = 16;
287 VScaleForTuning = 1;
288 break;
289 }
290
291 if (CLOpts.min_jump_table_entries || !HasMinSize)
292 MinimumJumpTableEntries = CLOpts.min_jump_table_entries.value_or(u: 10);
293 if (CLOpts.sve_vscale_for_tuning)
294 VScaleForTuning = *CLOpts.sve_vscale_for_tuning;
295}
296
297AArch64Subtarget::AArch64Subtarget(const Triple &TT, StringRef CPU,
298 StringRef TuneCPU, StringRef FS,
299 const TargetMachine &TM, bool LittleEndian,
300 unsigned MinSVEVectorSizeInBitsOverride,
301 unsigned MaxSVEVectorSizeInBitsOverride,
302 bool IsStreaming, bool IsStreamingCompatible,
303 bool HasMinSize,
304 bool EnableSRLTSubregToRegMitigation)
305 : AArch64GenSubtargetInfo(TT, CPU, TuneCPU, FS),
306 CLOpts(static_cast<const AArch64TargetMachine &>(TM).getCLOpts()),
307 ReserveXRegister(AArch64::GPR64commonRegClass.getNumRegs()),
308 ReserveXRegisterForRA(AArch64::GPR64commonRegClass.getNumRegs()),
309 CustomCallSavedXRegs(AArch64::GPR64commonRegClass.getNumRegs()),
310 IsLittle(LittleEndian), IsStreaming(IsStreaming),
311 IsStreamingCompatible(IsStreamingCompatible),
312 MinSVEVectorSizeInBits(MinSVEVectorSizeInBitsOverride),
313 MaxSVEVectorSizeInBits(MaxSVEVectorSizeInBitsOverride),
314 EnableSRLTSubregToRegMitigation(EnableSRLTSubregToRegMitigation),
315 // To benefit from SME2's strided-register multi-vector load/store
316 // instructions we'll need to enable subreg liveness. Our longer
317 // term aim is to make this the default, regardless of streaming
318 // mode, but there are still some outstanding issues, see:
319 // https://github.com/llvm/llvm-project/pull/174188
320 // and:
321 // https://github.com/llvm/llvm-project/pull/168353
322 EnableSubregLiveness(IsStreaming ||
323 CLOpts.enable_subreg_liveness_tracking),
324 TargetTriple(TT),
325 InstrInfo(initializeSubtargetDependencies(FS, CPUString: CPU, TuneCPUString: TuneCPU, HasMinSize)),
326 TLInfo(TM, *this) {
327 if (AArch64::isX18ReservedByDefault(TT))
328 ReserveXRegister.set(18);
329
330 CallLoweringInfo.reset(p: new AArch64CallLowering(*getTargetLowering()));
331 InlineAsmLoweringInfo.reset(p: new InlineAsmLowering(getTargetLowering()));
332 Legalizer.reset(p: new AArch64LegalizerInfo(*this));
333
334 auto *RBI = new AArch64RegisterBankInfo(*getRegisterInfo());
335
336 // FIXME: At this point, we can't rely on Subtarget having RBI.
337 // It's awkward to mix passing RBI and the Subtarget; should we pass
338 // TII/TRI as well?
339 InstSelector.reset(p: createAArch64InstructionSelector(
340 *static_cast<const AArch64TargetMachine *>(&TM), *this, *RBI));
341
342 RegBankInfo.reset(p: RBI);
343
344 auto TRI = getRegisterInfo();
345 StringSet<> ReservedRegNames(llvm::from_range, ReservedRegsForRA);
346 for (unsigned i = 0; i < 29; ++i) {
347 if (ReservedRegNames.count(Key: TRI->getName(RegNo: AArch64::X0 + i)))
348 ReserveXRegisterForRA.set(i);
349 }
350 // X30 is named LR, so we can't use TRI->getName to check X30.
351 if (ReservedRegNames.count(Key: "X30") || ReservedRegNames.count(Key: "LR"))
352 ReserveXRegisterForRA.set(30);
353 // X29 is named FP, so we can't use TRI->getName to check X29.
354 if (ReservedRegNames.count(Key: "X29") || ReservedRegNames.count(Key: "FP"))
355 ReserveXRegisterForRA.set(29);
356}
357
358const CallLowering *AArch64Subtarget::getCallLowering() const {
359 return CallLoweringInfo.get();
360}
361
362const InlineAsmLowering *AArch64Subtarget::getInlineAsmLowering() const {
363 return InlineAsmLoweringInfo.get();
364}
365
366InstructionSelector *AArch64Subtarget::getInstructionSelector() const {
367 return InstSelector.get();
368}
369
370const LegalizerInfo *AArch64Subtarget::getLegalizerInfo() const {
371 return Legalizer.get();
372}
373
374const RegisterBankInfo *AArch64Subtarget::getRegBankInfo() const {
375 return RegBankInfo.get();
376}
377
378/// Find the target operand flags that describe how a global value should be
379/// referenced for the current subtarget.
380unsigned
381AArch64Subtarget::ClassifyGlobalReference(const GlobalValue *GV,
382 const TargetMachine &TM) const {
383 // MachO large model always goes via a GOT, simply to get a single 8-byte
384 // absolute relocation on all global addresses.
385 if (TM.getCodeModel() == CodeModel::Large && isTargetMachO())
386 return AArch64II::MO_GOT;
387
388 // All globals dynamically protected by MTE must have their address tags
389 // synthesized. This is done by having the loader stash the tag in the GOT
390 // entry. Force all tagged globals (even ones with internal linkage) through
391 // the GOT.
392 if (GV->isTagged())
393 return AArch64II::MO_GOT;
394
395 if (!TM.shouldAssumeDSOLocal(GV)) {
396 if (GV->hasDLLImportStorageClass()) {
397 return AArch64II::MO_GOT | AArch64II::MO_DLLIMPORT;
398 }
399 if (getTargetTriple().isOSWindows())
400 return AArch64II::MO_GOT | AArch64II::MO_COFFSTUB;
401 return AArch64II::MO_GOT;
402 }
403
404 // The small code model's direct accesses use ADRP, which cannot
405 // necessarily produce the value 0 (if the code is above 4GB).
406 // Same for the tiny code model, where we have a pc relative LDR.
407 if ((useSmallAddressing() || TM.getCodeModel() == CodeModel::Tiny) &&
408 GV->hasExternalWeakLinkage())
409 return AArch64II::MO_GOT;
410
411 // References to tagged globals are marked with MO_NC | MO_TAGGED to indicate
412 // that their nominal addresses are tagged and outside of the code model. In
413 // AArch64ExpandPseudo::expandMI we emit an additional instruction to set the
414 // tag if necessary based on MO_TAGGED.
415 if (AllowTaggedGlobals && !isa<FunctionType>(Val: GV->getValueType()))
416 return AArch64II::MO_NC | AArch64II::MO_TAGGED;
417
418 return AArch64II::MO_NO_FLAG;
419}
420
421unsigned AArch64Subtarget::classifyGlobalFunctionReference(
422 const GlobalValue *GV, const TargetMachine &TM) const {
423 // MachO large model always goes via a GOT, because we don't have the
424 // relocations available to do anything else..
425 if (TM.getCodeModel() == CodeModel::Large && isTargetMachO() &&
426 !GV->hasInternalLinkage())
427 return AArch64II::MO_GOT;
428
429 // NonLazyBind goes via GOT unless we know it's available locally.
430 auto *F = dyn_cast<Function>(Val: GV);
431 if ((!isTargetMachO() || CLOpts.macho_enable_nonlazybind) && F &&
432 F->hasFnAttribute(Kind: Attribute::NonLazyBind) && !TM.shouldAssumeDSOLocal(GV))
433 return AArch64II::MO_GOT;
434
435 if (getTargetTriple().isOSWindows()) {
436 if (isWindowsArm64EC() && GV->getValueType()->isFunctionTy()) {
437 if (GV->hasDLLImportStorageClass()) {
438 // On Arm64EC, if we're calling a symbol from the import table
439 // directly, use MO_ARM64EC_CALLMANGLE.
440 return AArch64II::MO_GOT | AArch64II::MO_DLLIMPORT |
441 AArch64II::MO_ARM64EC_CALLMANGLE;
442 }
443 if (GV->hasExternalLinkage()) {
444 // If we're calling a symbol directly, use the mangled form in the
445 // call instruction.
446 return AArch64II::MO_ARM64EC_CALLMANGLE;
447 }
448 }
449
450 // Use ClassifyGlobalReference for setting MO_DLLIMPORT/MO_COFFSTUB.
451 return ClassifyGlobalReference(GV, TM);
452 }
453
454 return AArch64II::MO_NO_FLAG;
455}
456
457void AArch64Subtarget::overrideSchedPolicy(MachineSchedPolicy &Policy,
458 const SchedRegion &Region) const {
459 // LNT run (at least on Cyclone) showed reasonably significant gains for
460 // bi-directional scheduling. 253.perlbmk.
461 Policy.OnlyTopDown = false;
462 Policy.OnlyBottomUp = false;
463 // Enabling or Disabling the latency heuristic is a close call: It seems to
464 // help nearly no benchmark on out-of-order architectures, on the other hand
465 // it regresses register pressure on a few benchmarking.
466 Policy.DisableLatencyHeuristic = DisableLatencySchedHeuristic;
467}
468
469void AArch64Subtarget::adjustSchedDependency(
470 SUnit *Def, int DefOpIdx, SUnit *Use, int UseOpIdx, SDep &Dep,
471 const TargetSchedModel *SchedModel) const {
472 if (!SchedModel || Dep.getKind() != SDep::Kind::Data || !Dep.getReg() ||
473 !Def->isInstr() || !Use->isInstr() ||
474 (Def->getInstr()->getOpcode() != TargetOpcode::BUNDLE &&
475 Use->getInstr()->getOpcode() != TargetOpcode::BUNDLE))
476 return;
477
478 // If the Def is a BUNDLE, find the last instruction in the bundle that defs
479 // the register.
480 const MachineInstr *DefMI = Def->getInstr();
481 if (DefMI->getOpcode() == TargetOpcode::BUNDLE) {
482 Register Reg = DefMI->getOperand(i: DefOpIdx).getReg();
483 for (const auto &Op : const_mi_bundle_ops(MI: *DefMI)) {
484 if (Op.isReg() && Op.isDef() && Op.getReg() == Reg) {
485 DefMI = Op.getParent();
486 DefOpIdx = Op.getOperandNo();
487 }
488 }
489 }
490
491 // If the Use is a BUNDLE, find the first instruction that uses the Reg.
492 const MachineInstr *UseMI = Use->getInstr();
493 if (UseMI->getOpcode() == TargetOpcode::BUNDLE) {
494 Register Reg = UseMI->getOperand(i: UseOpIdx).getReg();
495 for (const auto &Op : const_mi_bundle_ops(MI: *UseMI)) {
496 if (Op.isReg() && Op.isUse() && Op.getReg() == Reg) {
497 UseMI = Op.getParent();
498 UseOpIdx = Op.getOperandNo();
499 break;
500 }
501 }
502 }
503
504 Dep.setLatency(
505 SchedModel->computeOperandLatency(DefMI, DefOperIdx: DefOpIdx, UseMI, UseOperIdx: UseOpIdx));
506}
507
508bool AArch64Subtarget::enableEarlyIfConversion() const {
509 return CLOpts.early_ifcvt;
510}
511
512bool AArch64Subtarget::supportsAddressTopByteIgnored() const {
513 if (!CLOpts.use_tbi)
514 return false;
515
516 if (TargetTriple.isDriverKit())
517 return true;
518 if (TargetTriple.isiOS()) {
519 return TargetTriple.getiOSVersion() >= VersionTuple(8);
520 }
521
522 return false;
523}
524
525std::unique_ptr<PBQPRAConstraint>
526AArch64Subtarget::getCustomPBQPConstraints() const {
527 return balanceFPOps() ? std::make_unique<A57ChainingConstraint>() : nullptr;
528}
529
530void AArch64Subtarget::mirFileLoaded(MachineFunction &MF) const {
531 // We usually compute max call frame size after ISel. Do the computation now
532 // if the .mir file didn't specify it. Note that this will probably give you
533 // bogus values after PEI has eliminated the callframe setup/destroy pseudo
534 // instructions, specify explicitly if you need it to be correct.
535 MachineFrameInfo &MFI = MF.getFrameInfo();
536 if (!MFI.isMaxCallFrameSizeComputed())
537 MFI.computeMaxCallFrameSize(MF);
538}
539
540bool AArch64Subtarget::useAA() const { return CLOpts.use_aa; }
541
542bool AArch64Subtarget::useScalarIncVL() const {
543 // If SVE2 or SME is present (we are not SVE-1 only) and
544 // -sve-use-scalar-inc-vl is not otherwise set, enable it by default.
545 return valueOr(X: CLOpts.sve_use_scalar_inc_vl, Default: hasSVE2() || hasSME());
546}
547
548// If return address signing is enabled, tail calls are emitted as follows:
549//
550// ```
551// <authenticate LR>
552// <check LR>
553// TCRETURN ; the callee may sign and spill the LR in its prologue
554// ```
555//
556// LR may require explicit checking because if FEAT_FPAC is not implemented
557// and LR was tampered with, then `<authenticate LR>` will not generate an
558// exception on its own. Later, if the callee spills the signed LR value and
559// neither FEAT_PAuth2 nor FEAT_EPAC are implemented, the valid PAC replaces
560// the higher bits of LR thus hiding the authentication failure.
561AArch64PAuth::AuthCheckMethod AArch64Subtarget::getAuthenticatedLRCheckMethod(
562 const MachineFunction &MF) const {
563 // TODO: Check subtarget for the scheme. Present variant is a default for
564 // pauthtest ABI.
565 if (MF.getFunction().hasFnAttribute(Kind: "ptrauth-returns") &&
566 MF.getFunction().hasFnAttribute(Kind: "ptrauth-auth-traps"))
567 return AArch64PAuth::AuthCheckMethod::HighBitsNoTBI;
568 // At now, use None by default because checks may introduce an unexpected
569 // performance regression or incompatibility with execute-only mappings.
570 return CLOpts.authenticated_lr_check_method.value_or(
571 u: AArch64PAuth::AuthCheckMethod::None);
572}
573
574std::optional<uint16_t>
575AArch64Subtarget::getPtrAuthBlockAddressDiscriminatorIfEnabled(
576 const Function &ParentFn) const {
577 if (!ParentFn.hasFnAttribute(Kind: "ptrauth-indirect-gotos"))
578 return std::nullopt;
579 // We currently have one simple mechanism for all targets.
580 // This isn't ABI, so we can always do better in the future.
581 return getPointerAuthStableSipHash(
582 S: (Twine(ParentFn.getName()) + " blockaddress").str());
583}
584
585bool AArch64Subtarget::isX16X17Safer() const {
586 // The Darwin kernel implements special protections for x16 and x17 so we
587 // should prefer to use those registers on that platform.
588 return isTargetDarwin();
589}
590
591bool AArch64Subtarget::enableMachinePipeliner() const {
592 return getSchedModel().hasInstrSchedModel();
593}
594
595/// Returns a MOVK's shifter operand, or 0 otherwise.
596static unsigned getMOVKShiftImm(const MachineInstr &MI) {
597 unsigned Opc = MI.getOpcode();
598 if (Opc != AArch64::MOVKWi && Opc != AArch64::MOVKXi)
599 return 0;
600 return MI.getOperand(i: 3).getImm();
601}
602
603/// \p HasFirst is false when the 1st instruction is a wildcard.
604static bool fusesMOVImmPairImpl(const AArch64Subtarget &ST, bool HasFirst,
605 unsigned FirstOpc, unsigned FirstShift,
606 unsigned SecondOpc, unsigned SecondShift) {
607 assert(ST.hasFuseLiterals() && "the subtarget doesn't fuse move immediate");
608
609 // 32 bit immediate.
610 if ((!HasFirst || FirstOpc == AArch64::MOVZWi) &&
611 SecondOpc == AArch64::MOVKWi && SecondShift == 16)
612 return true;
613
614 // Lower half of 64 bit immediate.
615 if ((!HasFirst || FirstOpc == AArch64::MOVZXi) &&
616 SecondOpc == AArch64::MOVKXi && SecondShift == 16)
617 return true;
618
619 // Upper half of 64 bit immediate.
620 if ((!HasFirst || (FirstOpc == AArch64::MOVKXi && FirstShift == 32)) &&
621 SecondOpc == AArch64::MOVKXi && SecondShift == 48)
622 return true;
623
624 return false;
625}
626
627bool AArch64Subtarget::fusesMOVImmPair(unsigned FirstOpc, unsigned FirstShift,
628 unsigned SecondOpc,
629 unsigned SecondShift) const {
630 return fusesMOVImmPairImpl(ST: *this, /*HasFirst=*/true, FirstOpc, FirstShift,
631 SecondOpc, SecondShift);
632}
633
634bool AArch64Subtarget::fusesMOVImmPair(const MachineInstr *FirstMI,
635 const MachineInstr &SecondMI) const {
636 return fusesMOVImmPairImpl(ST: *this, HasFirst: FirstMI != nullptr,
637 FirstOpc: FirstMI ? FirstMI->getOpcode() : 0,
638 FirstShift: FirstMI ? getMOVKShiftImm(MI: *FirstMI) : 0,
639 SecondOpc: SecondMI.getOpcode(), SecondShift: getMOVKShiftImm(MI: SecondMI));
640}
641