1//===-- X86Subtarget.h - Define Subtarget for the X86 ----------*- C++ -*--===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file declares the X86 specific subclass of TargetSubtargetInfo.
10//
11//===----------------------------------------------------------------------===//
12
13#ifndef LLVM_LIB_TARGET_X86_X86SUBTARGET_H
14#define LLVM_LIB_TARGET_X86_X86SUBTARGET_H
15
16#include "X86FrameLowering.h"
17#include "X86ISelLowering.h"
18#include "X86InstrInfo.h"
19#include "X86SelectionDAGInfo.h"
20#include "llvm/CodeGen/TargetSubtargetInfo.h"
21#include "llvm/IR/CallingConv.h"
22#include "llvm/TargetParser/Triple.h"
23#include <bitset>
24#include <climits>
25#include <cstdint>
26#include <memory>
27#include <optional>
28
29#define OPTIONS_STRUCT_DECL
30#include "X86Options.inc"
31
32#define GET_SUBTARGETINFO_HEADER
33#include "X86GenSubtargetInfo.inc"
34
35namespace llvm {
36
37class CallLowering;
38class GlobalValue;
39class InstructionSelector;
40class LegalizerInfo;
41class RegisterBankInfo;
42class StringRef;
43class TargetMachine;
44
45/// The X86 backend supports a number of different styles of PIC.
46///
47namespace PICStyles {
48
49enum class Style {
50 StubPIC, // Used on i386-darwin in pic mode.
51 GOT, // Used on 32 bit elf on when in pic mode.
52 RIPRel, // Used on X86-64 when in pic mode.
53 None // Set when not in pic mode.
54};
55
56} // end namespace PICStyles
57
58class X86Subtarget final : public X86GenSubtargetInfo {
59 const X86Options &CLOpts;
60
61 enum X86SSEEnum {
62 NoSSE, SSE1, SSE2, SSE3, SSSE3, SSE41, SSE42, AVX, AVX2, AVX512
63 };
64
65 /// Which PIC style to use
66 PICStyles::Style PICStyle;
67
68 const TargetMachine &TM;
69
70 /// SSE1, SSE2, SSE3, SSSE3, SSE41, SSE42, or none supported.
71 X86SSEEnum X86SSELevel = NoSSE;
72
73#define GET_SUBTARGETINFO_MACRO(ATTRIBUTE, DEFAULT, GETTER) \
74 bool ATTRIBUTE = DEFAULT;
75#include "X86GenSubtargetInfo.inc"
76 /// ReservedRReg R#i is not available as a general purpose register.
77 std::bitset<X86::NUM_TARGET_REGS> ReservedRReg;
78
79 /// The minimum alignment known to hold of the stack frame on
80 /// entry to the function and which must be maintained by every function.
81 Align stackAlignment = Align(4);
82
83 Align TileConfigAlignment = Align(4);
84
85 /// Max. memset / memcpy size that is turned into rep/movs, rep/stos ops.
86 ///
87 // FIXME: this is a known good value for Yonah. How about others?
88 unsigned MaxInlineSizeThreshold = 128;
89
90 /// What processor and OS we're targeting.
91 Triple TargetTriple;
92
93 /// GlobalISel related APIs.
94 std::unique_ptr<CallLowering> CallLoweringInfo;
95 std::unique_ptr<LegalizerInfo> Legalizer;
96 std::unique_ptr<RegisterBankInfo> RegBankInfo;
97 std::unique_ptr<InstructionSelector> InstSelector;
98
99 /// Override the stack alignment.
100 MaybeAlign StackAlignOverride;
101
102 /// Preferred vector width from function attribute.
103 unsigned PreferVectorWidthOverride;
104
105 /// Resolved preferred vector width from function attribute and subtarget
106 /// features.
107 unsigned PreferVectorWidth = UINT32_MAX;
108
109 /// Required vector width from function attribute.
110 unsigned RequiredVectorWidth;
111
112 bool HasUserReservedRegisters;
113
114 X86SelectionDAGInfo TSInfo;
115 // Ordering here is important. X86InstrInfo initializes X86RegisterInfo which
116 // X86TargetLowering needs.
117 X86InstrInfo InstrInfo;
118 X86TargetLowering TLInfo;
119 X86FrameLowering FrameLowering;
120
121public:
122 /// This constructor initializes the data members to match that
123 /// of the specified triple.
124 ///
125 X86Subtarget(const Triple &TT, StringRef CPU, StringRef TuneCPU, StringRef FS,
126 const X86TargetMachine &TM, MaybeAlign StackAlignOverride,
127 unsigned PreferVectorWidthOverride,
128 unsigned RequiredVectorWidth);
129 ~X86Subtarget() override;
130
131 const X86Options &getCLOpts() const { return CLOpts; }
132
133 const X86TargetLowering *getTargetLowering() const override {
134 return &TLInfo;
135 }
136
137 const X86InstrInfo *getInstrInfo() const override { return &InstrInfo; }
138
139 const X86FrameLowering *getFrameLowering() const override {
140 return &FrameLowering;
141 }
142
143 const X86SelectionDAGInfo *getSelectionDAGInfo() const override {
144 return &TSInfo;
145 }
146
147 const X86RegisterInfo *getRegisterInfo() const override {
148 return &getInstrInfo()->getRegisterInfo();
149 }
150
151 unsigned getTileConfigSize() const { return 64; }
152 Align getTileConfigAlignment() const { return TileConfigAlignment; }
153
154 /// Returns the minimum alignment known to hold of the
155 /// stack frame on entry to the function and which must be maintained by every
156 /// function for this subtarget.
157 Align getStackAlignment() const { return stackAlignment; }
158
159 /// Returns the maximum memset / memcpy size
160 /// that still makes it profitable to inline the call.
161 unsigned getMaxInlineSizeThreshold() const { return MaxInlineSizeThreshold; }
162
163 /// ParseSubtargetFeatures - Parses features string setting specified
164 /// subtarget options. Definition of function is auto generated by tblgen.
165 void ParseSubtargetFeatures(StringRef CPU, StringRef TuneCPU, StringRef FS);
166
167 /// Methods used by Global ISel
168 const CallLowering *getCallLowering() const override;
169 InstructionSelector *getInstructionSelector() const override;
170 const LegalizerInfo *getLegalizerInfo() const override;
171 const RegisterBankInfo *getRegBankInfo() const override;
172
173 bool isRegisterReservedByUser(Register i) const override {
174 return ReservedRReg[i.id()];
175 }
176 bool hasUserReservedRegisters() const { return HasUserReservedRegisters; }
177
178private:
179 /// Initialize the full set of dependencies so we can use an initializer
180 /// list for X86Subtarget.
181 X86Subtarget &initializeSubtargetDependencies(StringRef CPU,
182 StringRef TuneCPU,
183 StringRef FS);
184 void initSubtargetFeatures(StringRef CPU, StringRef TuneCPU, StringRef FS);
185
186public:
187
188#define GET_SUBTARGETINFO_MACRO(ATTRIBUTE, DEFAULT, GETTER) \
189 bool GETTER() const { return ATTRIBUTE; }
190#include "X86GenSubtargetInfo.inc"
191
192 /// Is this x86_64 with the ILP32 programming model (x32 ABI)?
193 bool isTarget64BitILP32() const { return Is64Bit && IsX32; }
194
195 /// Is this x86_64 with the LP64 programming model (standard AMD64, no x32)?
196 bool isTarget64BitLP64() const { return Is64Bit && !IsX32; }
197
198 PICStyles::Style getPICStyle() const { return PICStyle; }
199 void setPICStyle(PICStyles::Style Style) { PICStyle = Style; }
200
201 bool canUseCMPXCHG8B() const { return hasCX8(); }
202 bool canUseCMPXCHG16B() const {
203 // CX16 is just the CPUID bit, instruction requires 64-bit mode too.
204 return hasCX16() && is64Bit();
205 }
206 // SSE codegen depends on cmovs, and all SSE1+ processors support them.
207 // All 64-bit processors support cmov.
208 bool canUseCMOV() const { return hasCMOV() || hasSSE1() || is64Bit(); }
209 bool hasSSE1() const { return X86SSELevel >= SSE1; }
210 bool hasSSE2() const { return X86SSELevel >= SSE2; }
211 bool hasSSE3() const { return X86SSELevel >= SSE3; }
212 bool hasSSSE3() const { return X86SSELevel >= SSSE3; }
213 bool hasSSE41() const { return X86SSELevel >= SSE41; }
214 bool hasSSE42() const { return X86SSELevel >= SSE42; }
215 bool hasAVX() const { return X86SSELevel >= AVX; }
216 bool hasAVX2() const { return X86SSELevel >= AVX2; }
217 bool hasAVX512() const { return X86SSELevel >= AVX512; }
218 bool hasInt256() const { return hasAVX2(); }
219 bool hasAnyFMA() const { return hasFMA() || hasFMA4(); }
220 bool hasPrefetchW() const {
221 // The PREFETCHW instruction was added with 3DNow but later CPUs gave it
222 // its own CPUID bit as part of deprecating 3DNow.
223 return hasPRFCHW();
224 }
225 bool hasSSEPrefetch() const {
226 // We also implicitly enable these when we have a write prefix supporting
227 // cache level OR if we have prfchw.
228 return hasSSE1() || hasPRFCHW() || hasPREFETCHI();
229 }
230 bool canUseLAHFSAHF() const { return hasLAHFSAHF64() || !is64Bit(); }
231 // These are generic getters that OR together all of the thunk types
232 // supported by the subtarget. Therefore useIndirectThunk*() will return true
233 // if any respective thunk feature is enabled.
234 bool useIndirectThunkCalls() const {
235 return useRetpolineIndirectCalls() || useLVIControlFlowIntegrity();
236 }
237 bool useIndirectThunkBranches() const {
238 return useRetpolineIndirectBranches() || useLVIControlFlowIntegrity();
239 }
240
241 unsigned getPreferVectorWidth() const { return PreferVectorWidth; }
242 unsigned getRequiredVectorWidth() const { return RequiredVectorWidth; }
243
244 // Helper functions to determine when we should allow widening to 512-bit
245 // during codegen.
246 // TODO: Currently we're always allowing widening on CPUs without VLX,
247 // because for many cases we don't have a better option.
248 bool canExtendTo512DQ() const {
249 return hasAVX512() && (!hasVLX() || getPreferVectorWidth() >= 512);
250 }
251 bool canExtendTo512BW() const {
252 return hasBWI() && canExtendTo512DQ();
253 }
254
255 bool hasNoDomainDelay() const { return NoDomainDelay; }
256 bool hasNoDomainDelayMov() const {
257 return hasNoDomainDelay() || NoDomainDelayMov;
258 }
259 bool hasNoDomainDelayBlend() const {
260 return hasNoDomainDelay() || NoDomainDelayBlend;
261 }
262 bool hasNoDomainDelayShuffle() const {
263 return hasNoDomainDelay() || NoDomainDelayShuffle;
264 }
265
266 // If there are no 512-bit vectors and we prefer not to use 512-bit registers,
267 // disable them in the legalizer.
268 bool useAVX512Regs() const {
269 return hasAVX512() && (canExtendTo512DQ() || RequiredVectorWidth > 256);
270 }
271
272 bool useLight256BitInstructions() const {
273 return getPreferVectorWidth() >= 256 || AllowLight256Bit;
274 }
275
276 bool useBWIRegs() const {
277 return hasBWI() && useAVX512Regs();
278 }
279
280 // Returns true if the destination register of a BSF/BSR instruction is
281 // not touched if the source register is zero.
282 // NOTE: i32->i64 implicit zext isn't guaranteed by BSR/BSF pass through.
283 bool hasBitScanPassThrough() const { return is64Bit(); }
284
285 bool isXRaySupported() const override { return is64Bit(); }
286
287 /// Use clflush if we have SSE2 or we're on x86-64 (even if we asked for
288 /// no-sse2). There isn't any reason to disable it if the target processor
289 /// supports it.
290 bool hasCLFLUSH() const { return hasSSE2() || is64Bit(); }
291
292 /// Use mfence if we have SSE2 or we're on x86-64 (even if we asked for
293 /// no-sse2). There isn't any reason to disable it if the target processor
294 /// supports it.
295 bool hasMFence() const { return hasSSE2() || is64Bit(); }
296
297 /// Avoid use of `mfence` for`fence seq_cst`, and instead use `lock or`.
298 bool avoidMFence() const { return is64Bit(); }
299
300 const Triple &getTargetTriple() const { return TargetTriple; }
301
302 bool isTargetDarwin() const { return TargetTriple.isOSDarwin(); }
303 bool isTargetFreeBSD() const { return TargetTriple.isOSFreeBSD(); }
304 bool isTargetDragonFly() const { return TargetTriple.isOSDragonFly(); }
305 bool isTargetSolaris() const { return TargetTriple.isOSSolaris(); }
306 bool isTargetPS() const { return TargetTriple.isPS(); }
307
308 bool isTargetELF() const { return TargetTriple.isOSBinFormatELF(); }
309 bool isTargetCOFF() const { return TargetTriple.isOSBinFormatCOFF(); }
310 bool isTargetMachO() const { return TargetTriple.isOSBinFormatMachO(); }
311
312 bool isTargetLinux() const { return TargetTriple.isOSLinux(); }
313 bool isTargetKFreeBSD() const { return TargetTriple.isOSKFreeBSD(); }
314 bool isTargetHurd() const { return TargetTriple.isOSHurd(); }
315 bool isTargetGlibc() const { return TargetTriple.isOSGlibc(); }
316 bool isTargetMusl() const { return TargetTriple.isMusl(); }
317 bool isTargetAndroid() const { return TargetTriple.isAndroid(); }
318 bool isTargetMCU() const { return TargetTriple.isOSIAMCU(); }
319 bool isTargetFuchsia() const { return TargetTriple.isOSFuchsia(); }
320
321 bool isLFI() const { return TargetTriple.isLFI(); }
322
323 bool isTargetWindowsMSVC() const {
324 return TargetTriple.isWindowsMSVCEnvironment();
325 }
326
327 bool isTargetWindowsCoreCLR() const {
328 return TargetTriple.isWindowsCoreCLREnvironment();
329 }
330
331 bool isTargetWindowsCygwin() const {
332 return TargetTriple.isWindowsCygwinEnvironment();
333 }
334
335 bool isTargetWindowsGNU() const {
336 return TargetTriple.isWindowsGNUEnvironment();
337 }
338
339 bool isTargetWindowsItanium() const {
340 return TargetTriple.isWindowsItaniumEnvironment();
341 }
342
343 bool isTargetCygMing() const { return TargetTriple.isOSCygMing(); }
344
345 bool isUEFI() const { return TargetTriple.isUEFI(); }
346
347 bool isOSWindows() const { return TargetTriple.isOSWindows(); }
348
349 bool isOSWindowsOrUEFI() const { return TargetTriple.isOSWindowsOrUEFI(); }
350
351 bool isTargetUEFI64() const { return Is64Bit && isUEFI(); }
352
353 bool isTargetWin64() const { return Is64Bit && isOSWindows(); }
354
355 bool isTargetWin32() const { return !Is64Bit && isOSWindows(); }
356
357 bool isPICStyleGOT() const { return PICStyle == PICStyles::Style::GOT; }
358 bool isPICStyleRIPRel() const { return PICStyle == PICStyles::Style::RIPRel; }
359
360 bool isPICStyleStubPIC() const {
361 return PICStyle == PICStyles::Style::StubPIC;
362 }
363
364 bool isPositionIndependent() const;
365
366 bool isCallingConvWin64(CallingConv::ID CC) const {
367 switch (CC) {
368 // On Win64, all these conventions just use the default convention.
369 case CallingConv::C:
370 case CallingConv::Fast:
371 case CallingConv::Tail:
372 return isTargetWin64() || isTargetUEFI64();
373 case CallingConv::Swift:
374 case CallingConv::SwiftTail:
375 case CallingConv::X86_FastCall:
376 case CallingConv::X86_StdCall:
377 case CallingConv::X86_ThisCall:
378 case CallingConv::X86_VectorCall:
379 case CallingConv::Intel_OCL_BI:
380 return isTargetWin64();
381 // This convention allows using the Win64 convention on other targets.
382 case CallingConv::Win64:
383 return true;
384 // This convention allows using the SysV convention on Windows targets.
385 case CallingConv::X86_64_SysV:
386 return false;
387 // Otherwise, who knows what this is.
388 default:
389 return false;
390 }
391 }
392
393 /// Classify a global variable reference for the current subtarget according
394 /// to how we should reference it in a non-pcrel context.
395 unsigned char classifyLocalReference(const GlobalValue *GV) const;
396
397 unsigned char classifyGlobalReference(const GlobalValue *GV,
398 const Module &M) const;
399 unsigned char classifyGlobalReference(const GlobalValue *GV) const;
400
401 /// Classify a global function reference for the current subtarget.
402 unsigned char classifyGlobalFunctionReference(const GlobalValue *GV,
403 const Module &M) const;
404 unsigned char
405 classifyGlobalFunctionReference(const GlobalValue *GV) const override;
406
407 /// Classify a blockaddress reference for the current subtarget according to
408 /// how we should reference it in a non-pcrel context.
409 unsigned char classifyBlockAddressReference() const;
410
411 /// Return true if the subtarget allows calls to immediate address.
412 bool isLegalToCallImmediateAddr() const;
413
414 /// Return whether FrameLowering should always set the "extended frame
415 /// present" bit in FP, or set it based on a symbol in the runtime.
416 bool swiftAsyncContextIsDynamicallySet() const {
417 // Older OS versions (particularly system unwinders) are confused by the
418 // Swift extended frame, so when building code that might be run on them we
419 // must dynamically query the concurrency library to determine whether
420 // extended frames should be flagged as present.
421 const Triple &TT = getTargetTriple();
422
423 unsigned Major = TT.getOSVersion().getMajor();
424 switch(TT.getOS()) {
425 default:
426 return false;
427 case Triple::IOS:
428 case Triple::TvOS:
429 return Major < 15;
430 case Triple::WatchOS:
431 return Major < 8;
432 case Triple::MacOSX:
433 case Triple::Darwin:
434 return Major < 12;
435 }
436 }
437
438 /// If we are using indirect thunks, we need to expand indirectbr to avoid it
439 /// lowering to an actual indirect jump.
440 bool enableIndirectBrExpand() const override {
441 return useIndirectThunkBranches();
442 }
443
444 /// Enable the MachineScheduler pass for all X86 subtargets.
445 bool enableMachineScheduler() const override { return true; }
446
447 bool enableEarlyIfConversion() const override;
448
449 void getPostRAMutations(std::vector<std::unique_ptr<ScheduleDAGMutation>>
450 &Mutations) const override;
451
452 AntiDepBreakMode getAntiDepBreakMode() const override {
453 return TargetSubtargetInfo::ANTIDEP_CRITICAL;
454 }
455};
456
457} // end namespace llvm
458
459#endif // LLVM_LIB_TARGET_X86_X86SUBTARGET_H
460