1//===-- X86SelectionDAGInfo.cpp - X86 SelectionDAG Info -------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the X86SelectionDAGInfo class.
10//
11//===----------------------------------------------------------------------===//
12
13#include "X86SelectionDAGInfo.h"
14#include "X86InstrInfo.h"
15#include "X86RegisterInfo.h"
16#include "X86Subtarget.h"
17#include "llvm/CodeGen/MachineFrameInfo.h"
18#include "llvm/CodeGen/SelectionDAG.h"
19#include "llvm/CodeGen/TargetLowering.h"
20
21#define GET_SDNODE_DESC
22#include "X86GenSDNodeInfo.inc"
23
24using namespace llvm;
25
26#define DEBUG_TYPE "x86-selectiondag-info"
27
28X86SelectionDAGInfo::X86SelectionDAGInfo()
29 : SelectionDAGGenTargetInfo(X86GenSDNodeInfo) {}
30
31const char *X86SelectionDAGInfo::getTargetNodeName(unsigned Opcode) const {
32#define NODE_NAME_CASE(NODE) \
33 case X86ISD::NODE: \
34 return "X86ISD::" #NODE;
35
36 // These nodes don't have corresponding entries in *.td files yet.
37 switch (static_cast<X86ISD::NodeType>(Opcode)) {
38 NODE_NAME_CASE(POP_FROM_X87_REG)
39 NODE_NAME_CASE(GlobalBaseReg)
40 NODE_NAME_CASE(LCMPXCHG16_SAVE_RBX_DAG)
41 NODE_NAME_CASE(PCMPESTR)
42 NODE_NAME_CASE(PCMPISTR)
43 NODE_NAME_CASE(MGATHER)
44 NODE_NAME_CASE(MSCATTER)
45 NODE_NAME_CASE(AESENCWIDE128KL)
46 NODE_NAME_CASE(AESDECWIDE128KL)
47 NODE_NAME_CASE(AESENCWIDE256KL)
48 NODE_NAME_CASE(AESDECWIDE256KL)
49 }
50#undef NODE_NAME_CASE
51
52 return SelectionDAGGenTargetInfo::getTargetNodeName(Opcode);
53}
54
55bool X86SelectionDAGInfo::isTargetMemoryOpcode(unsigned Opcode) const {
56 // These nodes don't have corresponding entries in *.td files yet.
57 if (Opcode >= X86ISD::FIRST_MEMORY_OPCODE &&
58 Opcode <= X86ISD::LAST_MEMORY_OPCODE)
59 return true;
60
61 return SelectionDAGGenTargetInfo::isTargetMemoryOpcode(Opcode);
62}
63
64void X86SelectionDAGInfo::verifyTargetNode(const SelectionDAG &DAG,
65 const SDNode *N) const {
66 SelectionDAGGenTargetInfo::verifyTargetNode(DAG, N);
67
68 switch (N->getOpcode()) {
69 default:
70 break;
71 case X86ISD::CALL:
72 case X86ISD::TC_RETURN:
73 case X86ISD::TC_RETURN_GLOBALADDR: {
74 // The call target is an integer whose width depends on both the
75 // subtarget and on how the callee is addressed:
76 // * A direct call to a GlobalAddress/ExternalSymbol is i32 on the
77 // x32 ABI (as well as plain 32-bit mode) and i64 under LP64.
78 // * Anything else (register, folded load, or RIP-relative CFGuard call)
79 // uses the register width the subtarget executes in, i.e. i64 whenever
80 // the subtarget runs in 64-bit mode (including x32) and i32 otherwise.
81 const X86Subtarget &Subtarget =
82 DAG.getMachineFunction().getSubtarget<X86Subtarget>();
83 SDValue Target = N->getOperand(Num: 1);
84 bool IsDirect =
85 isa<GlobalAddressSDNode>(Val: Target) || isa<ExternalSymbolSDNode>(Val: Target);
86 bool WantI64 =
87 IsDirect ? Subtarget.isTarget64BitLP64() : Subtarget.is64Bit();
88 EVT ExpectedVT = WantI64 ? MVT::i64 : MVT::i32;
89 EVT VT = Target.getValueType();
90 if (VT != ExpectedVT)
91 report_fatal_error(reason: "invalid node: " + Twine(N->getOperationName(G: &DAG)) +
92 " operand #1 must have type " +
93 ExpectedVT.getEVTString() + ", but has type " +
94 VT.getEVTString());
95 break;
96 }
97 }
98}
99
100/// Returns the best type to use with repmovs/repstos depending on alignment.
101static MVT getOptimalRepType(const X86Subtarget &Subtarget, Align Alignment) {
102 uint64_t Align = Alignment.value();
103 assert((Align != 0) && "Align is normalized");
104 assert(isPowerOf2_64(Align) && "Align is a power of 2");
105 switch (Align) {
106 case 1:
107 return MVT::i8;
108 case 2:
109 return MVT::i16;
110 case 4:
111 return MVT::i32;
112 default:
113 return Subtarget.is64Bit() ? MVT::i64 : MVT::i32;
114 }
115}
116
117bool X86SelectionDAGInfo::isBaseRegConflictPossible(
118 SelectionDAG &DAG, ArrayRef<MCPhysReg> ClobberSet) const {
119 // We cannot use TRI->hasBasePointer() until *after* we select all basic
120 // blocks. Legalization may introduce new stack temporaries with large
121 // alignment requirements. Fall back to generic code if there are any
122 // dynamic stack adjustments (hopefully rare) and the base pointer would
123 // conflict if we had to use it.
124 MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo();
125 if (!MFI.hasVarSizedObjects() && !MFI.hasOpaqueSPAdjustment())
126 return false;
127
128 const X86RegisterInfo *TRI = static_cast<const X86RegisterInfo *>(
129 DAG.getSubtarget().getRegisterInfo());
130 return llvm::is_contained(Range&: ClobberSet, Element: TRI->getBaseRegister());
131}
132
133/// Emit a single REP STOSB instruction for a particular constant size.
134static SDValue emitRepstos(const X86Subtarget &Subtarget, SelectionDAG &DAG,
135 const SDLoc &dl, SDValue Chain, SDValue Dst,
136 SDValue Val, SDValue Size, MVT AVT) {
137 const bool Use64BitRegs = Subtarget.isTarget64BitLP64();
138 unsigned AX = X86::AL;
139 switch (AVT.getSizeInBits()) {
140 case 8:
141 AX = X86::AL;
142 break;
143 case 16:
144 AX = X86::AX;
145 break;
146 case 32:
147 AX = X86::EAX;
148 break;
149 default:
150 AX = X86::RAX;
151 break;
152 }
153
154 const unsigned CX = Use64BitRegs ? X86::RCX : X86::ECX;
155 const unsigned DI = Use64BitRegs ? X86::RDI : X86::EDI;
156
157 SDValue InGlue;
158 Chain = DAG.getCopyToReg(Chain, dl, Reg: AX, N: Val, Glue: InGlue);
159 InGlue = Chain.getValue(R: 1);
160 Chain = DAG.getCopyToReg(Chain, dl, Reg: CX, N: Size, Glue: InGlue);
161 InGlue = Chain.getValue(R: 1);
162 Chain = DAG.getCopyToReg(Chain, dl, Reg: DI, N: Dst, Glue: InGlue);
163 InGlue = Chain.getValue(R: 1);
164
165 SDVTList Tys = DAG.getVTList(VT1: MVT::Other, VT2: MVT::Glue);
166 SDValue Ops[] = {Chain, DAG.getValueType(AVT), InGlue};
167 return DAG.getNode(Opcode: X86ISD::REP_STOS, DL: dl, VTList: Tys, Ops);
168}
169
170/// Emit a single REP STOSB instruction for a particular constant size.
171static SDValue emitRepstosB(const X86Subtarget &Subtarget, SelectionDAG &DAG,
172 const SDLoc &dl, SDValue Chain, SDValue Dst,
173 SDValue Val, uint64_t Size) {
174 return emitRepstos(Subtarget, DAG, dl, Chain, Dst, Val,
175 Size: DAG.getIntPtrConstant(Val: Size, DL: dl), AVT: MVT::i8);
176}
177
178/// Returns a REP STOS instruction, possibly with a few load/stores to implement
179/// a constant size memory set. In some cases where we know REP MOVS is
180/// inefficient we return an empty SDValue so the calling code can either
181/// generate a store sequence or call the runtime memset function.
182static SDValue emitConstantSizeRepstos(SelectionDAG &DAG,
183 const X86Subtarget &Subtarget,
184 const SDLoc &dl, SDValue Chain,
185 SDValue Dst, SDValue Val, uint64_t Size,
186 EVT SizeVT, Align Alignment,
187 bool isVolatile, bool AlwaysInline,
188 MachinePointerInfo DstPtrInfo) {
189 /// In case we optimize for size, we use repstosb even if it's less efficient
190 /// so we can save the loads/stores of the leftover.
191 if (DAG.getMachineFunction().getFunction().hasMinSize()) {
192 if (auto *ValC = dyn_cast<ConstantSDNode>(Val)) {
193 // Special case 0 because otherwise we get large literals,
194 // which causes larger encoding.
195 if ((Size & 31) == 0 && (ValC->getZExtValue() & 255) == 0) {
196 MVT BlockType = MVT::i32;
197 const uint64_t BlockBits = BlockType.getSizeInBits();
198 const uint64_t BlockBytes = BlockBits / 8;
199 const uint64_t BlockCount = Size / BlockBytes;
200
201 Val = DAG.getConstant(Val: 0, DL: dl, VT: BlockType);
202 // repstosd is same size as repstosb
203 return emitRepstos(Subtarget, DAG, dl, Chain, Dst, Val,
204 Size: DAG.getIntPtrConstant(Val: BlockCount, DL: dl), AVT: BlockType);
205 }
206 }
207 return emitRepstosB(Subtarget, DAG, dl, Chain, Dst, Val, Size);
208 }
209
210 if (Size > Subtarget.getMaxInlineSizeThreshold())
211 return SDValue();
212
213 // If not DWORD aligned or size is more than the threshold, call the library.
214 // The libc version is likely to be faster for these cases. It can use the
215 // address value and run time information about the CPU.
216 if (Alignment < Align(4))
217 return SDValue();
218
219 MVT BlockType = MVT::i8;
220 uint64_t BlockCount = Size;
221 uint64_t BytesLeft = 0;
222
223 SDValue OriginalVal = Val;
224 if (auto *ValC = dyn_cast<ConstantSDNode>(Val)) {
225 BlockType = getOptimalRepType(Subtarget, Alignment);
226 uint64_t Value = ValC->getZExtValue() & 255;
227 const uint64_t BlockBits = BlockType.getSizeInBits();
228
229 if (BlockBits >= 16)
230 Value = (Value << 8) | Value;
231
232 if (BlockBits >= 32)
233 Value = (Value << 16) | Value;
234
235 if (BlockBits >= 64)
236 Value = (Value << 32) | Value;
237
238 const uint64_t BlockBytes = BlockBits / 8;
239 BlockCount = Size / BlockBytes;
240 BytesLeft = Size % BlockBytes;
241 Val = DAG.getConstant(Val: Value, DL: dl, VT: BlockType);
242 }
243
244 SDValue RepStos =
245 emitRepstos(Subtarget, DAG, dl, Chain, Dst, Val,
246 Size: DAG.getIntPtrConstant(Val: BlockCount, DL: dl), AVT: BlockType);
247 /// RepStos can process the whole length.
248 if (BytesLeft == 0)
249 return RepStos;
250
251 // Handle the last 1 - 7 bytes.
252 SmallVector<SDValue, 4> Results;
253 Results.push_back(Elt: RepStos);
254 unsigned Offset = Size - BytesLeft;
255 EVT AddrVT = Dst.getValueType();
256
257 Results.push_back(
258 Elt: DAG.getMemset(Chain, dl,
259 Dst: DAG.getNode(Opcode: ISD::ADD, DL: dl, VT: AddrVT, N1: Dst,
260 N2: DAG.getConstant(Val: Offset, DL: dl, VT: AddrVT)),
261 Src: OriginalVal, Size: DAG.getConstant(Val: BytesLeft, DL: dl, VT: SizeVT),
262 Alignment, isVol: isVolatile, AlwaysInline,
263 /* CI */ nullptr, DstPtrInfo: DstPtrInfo.getWithOffset(O: Offset)));
264
265 return DAG.getNode(Opcode: ISD::TokenFactor, DL: dl, VT: MVT::Other, Ops: Results);
266}
267
268SDValue X86SelectionDAGInfo::EmitTargetCodeForMemset(
269 SelectionDAG &DAG, const SDLoc &dl, SDValue Chain, SDValue Dst, SDValue Val,
270 SDValue Size, Align Alignment, bool isVolatile, bool AlwaysInline,
271 MachinePointerInfo DstPtrInfo) const {
272 const X86Subtarget &Subtarget =
273 DAG.getMachineFunction().getSubtarget<X86Subtarget>();
274
275 // If to a segment-relative address space, use the default lowering.
276 if (DstPtrInfo.getAddrSpace() >= 256)
277 return SDValue();
278
279 // REP STOS uses EDI on x86-32. Fall back if the user reserved EDI, so the
280 // generic expander can avoid emitting REP STOS.
281 if (!Subtarget.is64Bit() && Subtarget.isRegisterReservedByUser(i: X86::EDI))
282 return SDValue();
283
284 // If the base register might conflict with our physical registers, bail out.
285 const MCPhysReg ClobberSet[] = {X86::RCX, X86::RAX, X86::RDI,
286 X86::ECX, X86::EAX, X86::EDI};
287 if (isBaseRegConflictPossible(DAG, ClobberSet))
288 return SDValue();
289
290 ConstantSDNode *ConstantSize = dyn_cast<ConstantSDNode>(Val&: Size);
291 if (!ConstantSize)
292 return SDValue();
293
294 return emitConstantSizeRepstos(
295 DAG, Subtarget, dl, Chain, Dst, Val, Size: ConstantSize->getZExtValue(),
296 SizeVT: Size.getValueType(), Alignment, isVolatile, AlwaysInline, DstPtrInfo);
297}
298
299/// Emit a single REP MOVS{B,W,D,Q} instruction.
300static SDValue emitRepmovs(const X86Subtarget &Subtarget, SelectionDAG &DAG,
301 const SDLoc &dl, SDValue Chain, SDValue Dst,
302 SDValue Src, SDValue Size, MVT AVT) {
303 const bool Use64BitRegs = Subtarget.isTarget64BitLP64();
304 const unsigned CX = Use64BitRegs ? X86::RCX : X86::ECX;
305 const unsigned DI = Use64BitRegs ? X86::RDI : X86::EDI;
306 const unsigned SI = Use64BitRegs ? X86::RSI : X86::ESI;
307
308 SDValue InGlue;
309 Chain = DAG.getCopyToReg(Chain, dl, Reg: CX, N: Size, Glue: InGlue);
310 InGlue = Chain.getValue(R: 1);
311 Chain = DAG.getCopyToReg(Chain, dl, Reg: DI, N: Dst, Glue: InGlue);
312 InGlue = Chain.getValue(R: 1);
313 Chain = DAG.getCopyToReg(Chain, dl, Reg: SI, N: Src, Glue: InGlue);
314 InGlue = Chain.getValue(R: 1);
315
316 SDVTList Tys = DAG.getVTList(VT1: MVT::Other, VT2: MVT::Glue);
317 SDValue Ops[] = {Chain, DAG.getValueType(AVT), InGlue};
318 return DAG.getNode(Opcode: X86ISD::REP_MOVS, DL: dl, VTList: Tys, Ops);
319}
320
321/// Emit a single REP MOVSB instruction for a particular constant size.
322static SDValue emitRepmovsB(const X86Subtarget &Subtarget, SelectionDAG &DAG,
323 const SDLoc &dl, SDValue Chain, SDValue Dst,
324 SDValue Src, uint64_t Size) {
325 return emitRepmovs(Subtarget, DAG, dl, Chain, Dst, Src,
326 Size: DAG.getIntPtrConstant(Val: Size, DL: dl), AVT: MVT::i8);
327}
328
329/// Returns a REP MOVS instruction, possibly with a few load/stores to implement
330/// a constant size memory copy. In some cases where we know REP MOVS is
331/// inefficient we return an empty SDValue so the calling code can either
332/// generate a load/store sequence or call the runtime memcpy function.
333static SDValue emitConstantSizeRepmov(
334 SelectionDAG &DAG, const X86Subtarget &Subtarget, const SDLoc &dl,
335 SDValue Chain, SDValue Dst, SDValue Src, uint64_t Size, EVT SizeVT,
336 Align Alignment, bool isVolatile, bool AlwaysInline,
337 MachinePointerInfo DstPtrInfo, MachinePointerInfo SrcPtrInfo) {
338 /// In case we optimize for size, we use repmovsb even if it's less efficient
339 /// so we can save the loads/stores of the leftover.
340 if (DAG.getMachineFunction().getFunction().hasMinSize())
341 return emitRepmovsB(Subtarget, DAG, dl, Chain, Dst, Src, Size);
342
343 /// TODO: Revisit next line: big copy with ERMSB on march >= haswell are very
344 /// efficient.
345 if (!AlwaysInline && Size > Subtarget.getMaxInlineSizeThreshold())
346 return SDValue();
347
348 /// If we have enhanced repmovs we use it.
349 if (Subtarget.hasERMSB())
350 return emitRepmovsB(Subtarget, DAG, dl, Chain, Dst, Src, Size);
351
352 assert(!Subtarget.hasERMSB() && "No efficient RepMovs");
353 /// We assume runtime memcpy will do a better job for unaligned copies when
354 /// ERMS is not present.
355 if (!AlwaysInline && (Alignment < Align(4)))
356 return SDValue();
357
358 const MVT BlockType = getOptimalRepType(Subtarget, Alignment);
359 const uint64_t BlockBytes = BlockType.getSizeInBits() / 8;
360 const uint64_t BlockCount = Size / BlockBytes;
361 const uint64_t BytesLeft = Size % BlockBytes;
362 SDValue RepMovs =
363 emitRepmovs(Subtarget, DAG, dl, Chain, Dst, Src,
364 Size: DAG.getIntPtrConstant(Val: BlockCount, DL: dl), AVT: BlockType);
365
366 /// RepMov can process the whole length.
367 if (BytesLeft == 0)
368 return RepMovs;
369
370 assert(BytesLeft && "We have leftover at this point");
371
372 // Handle the last 1 - 7 bytes.
373 SmallVector<SDValue, 4> Results;
374 Results.push_back(Elt: RepMovs);
375 unsigned Offset = Size - BytesLeft;
376 EVT DstVT = Dst.getValueType();
377 EVT SrcVT = Src.getValueType();
378 Results.push_back(Elt: DAG.getMemcpy(
379 Chain, dl,
380 Dst: DAG.getNode(Opcode: ISD::ADD, DL: dl, VT: DstVT, N1: Dst, N2: DAG.getConstant(Val: Offset, DL: dl, VT: DstVT)),
381 Src: DAG.getNode(Opcode: ISD::ADD, DL: dl, VT: SrcVT, N1: Src, N2: DAG.getConstant(Val: Offset, DL: dl, VT: SrcVT)),
382 Size: DAG.getConstant(Val: BytesLeft, DL: dl, VT: SizeVT), DstAlign: Alignment, SrcAlign: Alignment, isVol: isVolatile,
383 /*AlwaysInline*/ true, /*CI=*/nullptr, OverrideTailCall: std::nullopt,
384 DstPtrInfo: DstPtrInfo.getWithOffset(O: Offset), SrcPtrInfo: SrcPtrInfo.getWithOffset(O: Offset)));
385 return DAG.getNode(Opcode: ISD::TokenFactor, DL: dl, VT: MVT::Other, Ops: Results);
386}
387
388SDValue X86SelectionDAGInfo::EmitTargetCodeForMemcpy(
389 SelectionDAG &DAG, const SDLoc &dl, SDValue Chain, SDValue Dst, SDValue Src,
390 SDValue Size, Align DstAlign, Align SrcAlign, bool isVolatile,
391 bool AlwaysInline, MachinePointerInfo DstPtrInfo,
392 MachinePointerInfo SrcPtrInfo) const {
393 const X86Subtarget &Subtarget =
394 DAG.getMachineFunction().getSubtarget<X86Subtarget>();
395
396 // If to a segment-relative address space, use the default lowering.
397 if (DstPtrInfo.getAddrSpace() >= 256 || SrcPtrInfo.getAddrSpace() >= 256)
398 return SDValue();
399
400 // REP MOVS uses EDI/ESI on x86-32. fall back only when EDI is
401 // reserved so the generic expander can avoid emitting REP MOVS.
402 if (!Subtarget.is64Bit() && Subtarget.isRegisterReservedByUser(i: X86::EDI))
403 return SDValue();
404
405 // If the base registers conflict with our physical registers, use the default
406 // lowering.
407 const MCPhysReg ClobberSet[] = {X86::RCX, X86::RSI, X86::RDI,
408 X86::ECX, X86::ESI, X86::EDI};
409 if (isBaseRegConflictPossible(DAG, ClobberSet))
410 return SDValue();
411
412 // If enabled and available, use fast short rep mov.
413 if (Subtarget.getCLOpts().use_fsrm_for_memcpy && Subtarget.hasFSRM())
414 return emitRepmovs(Subtarget, DAG, dl, Chain, Dst, Src, Size, AVT: MVT::i8);
415
416 // Handle constant sizes
417 if (ConstantSDNode *ConstantSize = dyn_cast<ConstantSDNode>(Val&: Size)) {
418 Align Alignment = std::min(a: DstAlign, b: SrcAlign);
419 return emitConstantSizeRepmov(DAG, Subtarget, dl, Chain, Dst, Src,
420 Size: ConstantSize->getZExtValue(),
421 SizeVT: Size.getValueType(), Alignment, isVolatile,
422 AlwaysInline, DstPtrInfo, SrcPtrInfo);
423 }
424
425 return SDValue();
426}
427