1//===- AArch64InstructionSelector.cpp ----------------------------*- C++ -*-==//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9/// This file implements the targeting of the InstructionSelector class for
10/// AArch64.
11/// \todo This should be generated by TableGen.
12//===----------------------------------------------------------------------===//
13
14#include "AArch64GlobalISelUtils.h"
15#include "AArch64InstrInfo.h"
16#include "AArch64MachineFunctionInfo.h"
17#include "AArch64RegisterBankInfo.h"
18#include "AArch64RegisterInfo.h"
19#include "AArch64Subtarget.h"
20#include "AArch64TargetMachine.h"
21#include "MCTargetDesc/AArch64AddressingModes.h"
22#include "MCTargetDesc/AArch64MCTargetDesc.h"
23#include "llvm/BinaryFormat/Dwarf.h"
24#include "llvm/CodeGen/GlobalISel/GIMatchTableExecutorImpl.h"
25#include "llvm/CodeGen/GlobalISel/GISelValueTracking.h"
26#include "llvm/CodeGen/GlobalISel/GenericMachineInstrs.h"
27#include "llvm/CodeGen/GlobalISel/InstructionSelector.h"
28#include "llvm/CodeGen/GlobalISel/MIPatternMatch.h"
29#include "llvm/CodeGen/GlobalISel/MachineIRBuilder.h"
30#include "llvm/CodeGen/GlobalISel/Utils.h"
31#include "llvm/CodeGen/MachineBasicBlock.h"
32#include "llvm/CodeGen/MachineConstantPool.h"
33#include "llvm/CodeGen/MachineFrameInfo.h"
34#include "llvm/CodeGen/MachineFunction.h"
35#include "llvm/CodeGen/MachineInstr.h"
36#include "llvm/CodeGen/MachineInstrBuilder.h"
37#include "llvm/CodeGen/MachineMemOperand.h"
38#include "llvm/CodeGen/MachineOperand.h"
39#include "llvm/CodeGen/MachineRegisterInfo.h"
40#include "llvm/CodeGen/TargetOpcodes.h"
41#include "llvm/CodeGen/TargetRegisterInfo.h"
42#include "llvm/IR/Constants.h"
43#include "llvm/IR/DerivedTypes.h"
44#include "llvm/IR/Instructions.h"
45#include "llvm/IR/IntrinsicsAArch64.h"
46#include "llvm/IR/Type.h"
47#include "llvm/Pass.h"
48#include "llvm/Support/Debug.h"
49#include "llvm/Support/raw_ostream.h"
50#include <optional>
51
52#define DEBUG_TYPE "aarch64-isel"
53
54using namespace llvm;
55using namespace MIPatternMatch;
56using namespace AArch64GISelUtils;
57
58namespace llvm {
59class BlockFrequencyInfo;
60class ProfileSummaryInfo;
61}
62
63namespace {
64
65#define GET_GLOBALISEL_PREDICATE_BITSET
66#include "AArch64GenGlobalISel.inc"
67#undef GET_GLOBALISEL_PREDICATE_BITSET
68
69
70class AArch64InstructionSelector : public InstructionSelector {
71public:
72 AArch64InstructionSelector(const AArch64TargetMachine &TM,
73 const AArch64Subtarget &STI,
74 const AArch64RegisterBankInfo &RBI);
75
76 bool select(MachineInstr &I) override;
77 static const char *getName() { return DEBUG_TYPE; }
78
79 void setupMF(MachineFunction &MF, GISelValueTracking *VT,
80 CodeGenCoverage *CoverageInfo, ProfileSummaryInfo *PSI,
81 BlockFrequencyInfo *BFI) override {
82 InstructionSelector::setupMF(mf&: MF, vt: VT, covinfo: CoverageInfo, psi: PSI, bfi: BFI);
83 MIB.setMF(MF);
84
85 // hasFnAttribute() is expensive to call on every BRCOND selection, so
86 // cache it here for each run of the selector.
87 ProduceNonFlagSettingCondBr =
88 !MF.getFunction().hasFnAttribute(Kind: Attribute::SpeculativeLoadHardening);
89 MFReturnAddr = Register();
90
91 processPHIs(MF);
92 }
93
94private:
95 /// tblgen-erated 'select' implementation, used as the initial selector for
96 /// the patterns that don't require complex C++.
97 bool selectImpl(MachineInstr &I, CodeGenCoverage &CoverageInfo) const;
98
99 // A lowering phase that runs before any selection attempts.
100 // Returns true if the instruction was modified.
101 bool preISelLower(MachineInstr &I);
102
103 // An early selection function that runs before the selectImpl() call.
104 bool earlySelect(MachineInstr &I);
105
106 /// Save state that is shared between select calls, call select on \p I and
107 /// then restore the saved state. This can be used to recursively call select
108 /// within a select call.
109 bool selectAndRestoreState(MachineInstr &I);
110
111 // Do some preprocessing of G_PHIs before we begin selection.
112 void processPHIs(MachineFunction &MF);
113
114 bool earlySelectSHL(MachineInstr &I, MachineRegisterInfo &MRI);
115
116 /// Eliminate same-sized cross-bank copies into stores before selectImpl().
117 bool contractCrossBankCopyIntoStore(MachineInstr &I,
118 MachineRegisterInfo &MRI);
119
120 bool convertPtrAddToAdd(MachineInstr &I, MachineRegisterInfo &MRI);
121
122 bool selectVaStartAAPCS(MachineInstr &I, MachineFunction &MF,
123 MachineRegisterInfo &MRI) const;
124 bool selectVaStartDarwin(MachineInstr &I, MachineFunction &MF,
125 MachineRegisterInfo &MRI) const;
126
127 ///@{
128 /// Helper functions for selectCompareBranch.
129 bool selectCompareBranchFedByFCmp(MachineInstr &I, MachineInstr &FCmp,
130 MachineIRBuilder &MIB) const;
131 bool selectCompareBranchFedByICmp(MachineInstr &I, MachineInstr &ICmp,
132 MachineIRBuilder &MIB) const;
133 bool tryOptCompareBranchFedByICmp(MachineInstr &I, MachineInstr &ICmp,
134 MachineIRBuilder &MIB) const;
135 bool tryOptAndIntoCompareBranch(MachineInstr &AndInst, bool Invert,
136 MachineBasicBlock *DstMBB,
137 MachineIRBuilder &MIB) const;
138 ///@}
139
140 bool selectCompareBranch(MachineInstr &I, MachineFunction &MF,
141 MachineRegisterInfo &MRI);
142
143 bool selectVectorAshrLshr(MachineInstr &I, MachineRegisterInfo &MRI);
144 bool selectVectorSHL(MachineInstr &I, MachineRegisterInfo &MRI);
145
146 // Helper to generate an equivalent of scalar_to_vector into a new register,
147 // returned via 'Dst'.
148 MachineInstr *emitScalarToVector(unsigned EltSize,
149 const TargetRegisterClass *DstRC,
150 Register Scalar,
151 MachineIRBuilder &MIRBuilder) const;
152 /// Helper to narrow vector that was widened by emitScalarToVector.
153 /// Copy lowest part of 128-bit or 64-bit vector to 64-bit or 32-bit
154 /// vector, correspondingly.
155 MachineInstr *emitNarrowVector(Register DstReg, Register SrcReg,
156 MachineIRBuilder &MIRBuilder,
157 MachineRegisterInfo &MRI) const;
158
159 /// Emit a lane insert into \p DstReg, or a new vector register if
160 /// std::nullopt is provided.
161 ///
162 /// The lane inserted into is defined by \p LaneIdx. The vector source
163 /// register is given by \p SrcReg. The register containing the element is
164 /// given by \p EltReg.
165 MachineInstr *emitLaneInsert(std::optional<Register> DstReg, Register SrcReg,
166 Register EltReg, unsigned LaneIdx,
167 const RegisterBank &RB,
168 MachineIRBuilder &MIRBuilder) const;
169
170 /// Emit a sequence of instructions representing a constant \p CV for a
171 /// vector register \p Dst. (E.g. a MOV, or a load from a constant pool.)
172 ///
173 /// \returns the last instruction in the sequence on success, and nullptr
174 /// otherwise.
175 MachineInstr *emitConstantVector(Register Dst, Constant *CV,
176 MachineIRBuilder &MIRBuilder,
177 MachineRegisterInfo &MRI);
178
179 MachineInstr *tryAdvSIMDModImm8(Register Dst, unsigned DstSize, APInt Bits,
180 MachineIRBuilder &MIRBuilder);
181
182 MachineInstr *tryAdvSIMDModImm16(Register Dst, unsigned DstSize, APInt Bits,
183 MachineIRBuilder &MIRBuilder, bool Inv);
184
185 MachineInstr *tryAdvSIMDModImm32(Register Dst, unsigned DstSize, APInt Bits,
186 MachineIRBuilder &MIRBuilder, bool Inv);
187 MachineInstr *tryAdvSIMDModImm64(Register Dst, unsigned DstSize, APInt Bits,
188 MachineIRBuilder &MIRBuilder);
189 MachineInstr *tryAdvSIMDModImm321s(Register Dst, unsigned DstSize, APInt Bits,
190 MachineIRBuilder &MIRBuilder, bool Inv);
191 MachineInstr *tryAdvSIMDModImmFP(Register Dst, unsigned DstSize, APInt Bits,
192 MachineIRBuilder &MIRBuilder);
193
194 bool tryOptConstantBuildVec(MachineInstr &MI, LLT DstTy,
195 MachineRegisterInfo &MRI);
196 /// \returns true if a G_BUILD_VECTOR instruction \p MI can be selected as a
197 /// SUBREG_TO_REG.
198 bool tryOptBuildVecToSubregToReg(MachineInstr &MI, MachineRegisterInfo &MRI);
199 bool selectBuildVector(MachineInstr &I, MachineRegisterInfo &MRI);
200 bool selectMergeValues(MachineInstr &I, MachineRegisterInfo &MRI);
201 bool selectUnmergeValues(MachineInstr &I, MachineRegisterInfo &MRI);
202
203 bool selectShuffleVector(MachineInstr &I, MachineRegisterInfo &MRI);
204 bool selectExtractElt(MachineInstr &I, MachineRegisterInfo &MRI);
205 bool selectConcatVectors(MachineInstr &I, MachineRegisterInfo &MRI);
206 bool selectSplitVectorUnmerge(MachineInstr &I, MachineRegisterInfo &MRI);
207
208 /// Helper function to select vector load intrinsics like
209 /// @llvm.aarch64.neon.ld2.*, @llvm.aarch64.neon.ld4.*, etc.
210 /// \p Opc is the opcode that the selected instruction should use.
211 /// \p NumVecs is the number of vector destinations for the instruction.
212 /// \p I is the original G_INTRINSIC_W_SIDE_EFFECTS instruction.
213 bool selectVectorLoadIntrinsic(unsigned Opc, unsigned NumVecs,
214 MachineInstr &I);
215 bool selectVectorLoadLaneIntrinsic(unsigned Opc, unsigned NumVecs,
216 MachineInstr &I);
217 void selectVectorStoreIntrinsic(MachineInstr &I, unsigned NumVecs,
218 unsigned Opc);
219 bool selectVectorStoreLaneIntrinsic(MachineInstr &I, unsigned NumVecs,
220 unsigned Opc);
221 bool selectIntrinsicWithSideEffects(MachineInstr &I,
222 MachineRegisterInfo &MRI);
223 bool selectIntrinsic(MachineInstr &I, MachineRegisterInfo &MRI);
224 bool selectJumpTable(MachineInstr &I, MachineRegisterInfo &MRI);
225 bool selectBrJT(MachineInstr &I, MachineRegisterInfo &MRI);
226 bool selectTLSGlobalValueELF(MachineInstr &I, MachineRegisterInfo &MRI);
227 bool selectTLSLocalExecELF(const GlobalValue *GV, MachineInstr &I,
228 MachineRegisterInfo &MRI);
229 bool selectTLSGlobalValueMachO(MachineInstr &I, MachineRegisterInfo &MRI);
230 bool selectTLSGlobalValue(MachineInstr &I, MachineRegisterInfo &MRI);
231 bool selectPtrAuthGlobalValue(MachineInstr &I,
232 MachineRegisterInfo &MRI) const;
233 bool selectMOPS(MachineInstr &I, MachineRegisterInfo &MRI);
234 bool selectUSMovFromExtend(MachineInstr &I, MachineRegisterInfo &MRI);
235 void SelectTable(MachineInstr &I, MachineRegisterInfo &MRI, unsigned NumVecs,
236 unsigned Opc1, unsigned Opc2, bool isExt);
237
238 bool selectIndexedExtLoad(MachineInstr &I, MachineRegisterInfo &MRI);
239 bool selectIndexedLoad(MachineInstr &I, MachineRegisterInfo &MRI);
240 bool selectIndexedStore(GIndexedStore &I, MachineRegisterInfo &MRI);
241
242 unsigned emitConstantPoolEntry(const Constant *CPVal,
243 MachineFunction &MF) const;
244 MachineInstr *emitLoadFromConstantPool(const Constant *CPVal,
245 MachineIRBuilder &MIRBuilder) const;
246
247 // Emit a vector concat operation.
248 MachineInstr *emitVectorConcat(std::optional<Register> Dst, Register Op1,
249 Register Op2,
250 MachineIRBuilder &MIRBuilder) const;
251
252 // Emit an integer compare between LHS and RHS, which checks for Predicate.
253 MachineInstr *emitIntegerCompare(MachineOperand &LHS, MachineOperand &RHS,
254 MachineOperand &Predicate,
255 MachineIRBuilder &MIRBuilder) const;
256
257 /// Emit a floating point comparison between \p LHS and \p RHS.
258 /// \p Pred if given is the intended predicate to use.
259 MachineInstr *
260 emitFPCompare(Register LHS, Register RHS, MachineIRBuilder &MIRBuilder,
261 std::optional<CmpInst::Predicate> = std::nullopt) const;
262
263 MachineInstr *
264 emitInstr(unsigned Opcode, std::initializer_list<llvm::DstOp> DstOps,
265 std::initializer_list<llvm::SrcOp> SrcOps,
266 MachineIRBuilder &MIRBuilder,
267 const ComplexRendererFns &RenderFns = std::nullopt) const;
268 /// Helper function to emit an add or sub instruction.
269 ///
270 /// \p AddrModeAndSizeToOpcode must contain each of the opcode variants above
271 /// in a specific order.
272 ///
273 /// Below is an example of the expected input to \p AddrModeAndSizeToOpcode.
274 ///
275 /// \code
276 /// const std::array<std::array<unsigned, 2>, 4> Table {
277 /// {{AArch64::ADDXri, AArch64::ADDWri},
278 /// {AArch64::ADDXrs, AArch64::ADDWrs},
279 /// {AArch64::ADDXrr, AArch64::ADDWrr},
280 /// {AArch64::SUBXri, AArch64::SUBWri},
281 /// {AArch64::ADDXrx, AArch64::ADDWrx}}};
282 /// \endcode
283 ///
284 /// Each row in the table corresponds to a different addressing mode. Each
285 /// column corresponds to a different register size.
286 ///
287 /// \attention Rows must be structured as follows:
288 /// - Row 0: The ri opcode variants
289 /// - Row 1: The rs opcode variants
290 /// - Row 2: The rr opcode variants
291 /// - Row 3: The ri opcode variants for negative immediates
292 /// - Row 4: The rx opcode variants
293 ///
294 /// \attention Columns must be structured as follows:
295 /// - Column 0: The 64-bit opcode variants
296 /// - Column 1: The 32-bit opcode variants
297 ///
298 /// \p Dst is the destination register of the binop to emit.
299 /// \p LHS is the left-hand operand of the binop to emit.
300 /// \p RHS is the right-hand operand of the binop to emit.
301 MachineInstr *emitAddSub(
302 const std::array<std::array<unsigned, 2>, 5> &AddrModeAndSizeToOpcode,
303 Register Dst, MachineOperand &LHS, MachineOperand &RHS,
304 MachineIRBuilder &MIRBuilder) const;
305 MachineInstr *emitADD(Register DefReg, MachineOperand &LHS,
306 MachineOperand &RHS,
307 MachineIRBuilder &MIRBuilder) const;
308 MachineInstr *emitADDS(Register Dst, MachineOperand &LHS, MachineOperand &RHS,
309 MachineIRBuilder &MIRBuilder) const;
310 MachineInstr *emitSUBS(Register Dst, MachineOperand &LHS, MachineOperand &RHS,
311 MachineIRBuilder &MIRBuilder) const;
312 MachineInstr *emitADCS(Register Dst, MachineOperand &LHS, MachineOperand &RHS,
313 MachineIRBuilder &MIRBuilder) const;
314 MachineInstr *emitSBCS(Register Dst, MachineOperand &LHS, MachineOperand &RHS,
315 MachineIRBuilder &MIRBuilder) const;
316 MachineInstr *emitCMP(MachineOperand &LHS, MachineOperand &RHS,
317 MachineIRBuilder &MIRBuilder) const;
318 MachineInstr *emitCMN(MachineOperand &LHS, MachineOperand &RHS,
319 MachineIRBuilder &MIRBuilder) const;
320 MachineInstr *emitTST(MachineOperand &LHS, MachineOperand &RHS,
321 MachineIRBuilder &MIRBuilder) const;
322 MachineInstr *emitSelect(Register Dst, Register LHS, Register RHS,
323 AArch64CC::CondCode CC,
324 MachineIRBuilder &MIRBuilder) const;
325 MachineInstr *emitExtractVectorElt(std::optional<Register> DstReg,
326 const RegisterBank &DstRB, LLT ScalarTy,
327 Register VecReg, unsigned LaneIdx,
328 MachineIRBuilder &MIRBuilder) const;
329 MachineInstr *emitCSINC(Register Dst, Register Src1, Register Src2,
330 AArch64CC::CondCode Pred,
331 MachineIRBuilder &MIRBuilder) const;
332 /// Emit a CSet for a FP compare.
333 ///
334 /// \p Dst is expected to be a 32-bit scalar register.
335 MachineInstr *emitCSetForFCmp(Register Dst, CmpInst::Predicate Pred,
336 MachineIRBuilder &MIRBuilder) const;
337
338 /// Emit an instruction that sets NZCV to the carry-in expected by \p I.
339 /// Might elide the instruction if the previous instruction already sets NZCV
340 /// correctly.
341 MachineInstr *emitCarryIn(MachineInstr &I, Register CarryReg);
342
343 /// Emit the overflow op for \p Opcode.
344 ///
345 /// \p Opcode is expected to be an overflow op's opcode, e.g. G_UADDO,
346 /// G_USUBO, etc.
347 std::pair<MachineInstr *, AArch64CC::CondCode>
348 emitOverflowOp(unsigned Opcode, Register Dst, MachineOperand &LHS,
349 MachineOperand &RHS, MachineIRBuilder &MIRBuilder) const;
350
351 bool selectOverflowOp(MachineInstr &I, MachineRegisterInfo &MRI);
352
353 /// Emit expression as a conjunction (a series of CCMP/CFCMP ops).
354 /// In some cases this is even possible with OR operations in the expression.
355 MachineInstr *emitConjunction(Register Val, AArch64CC::CondCode &OutCC,
356 MachineIRBuilder &MIB) const;
357 MachineInstr *emitConditionalComparison(Register LHS, Register RHS,
358 CmpInst::Predicate CC,
359 AArch64CC::CondCode Predicate,
360 AArch64CC::CondCode OutCC,
361 MachineIRBuilder &MIB) const;
362 MachineInstr *emitConjunctionRec(Register Val, AArch64CC::CondCode &OutCC,
363 bool Negate, Register CCOp,
364 AArch64CC::CondCode Predicate,
365 MachineIRBuilder &MIB) const;
366
367 /// Emit a TB(N)Z instruction which tests \p Bit in \p TestReg.
368 /// \p IsNegative is true if the test should be "not zero".
369 /// This will also optimize the test bit instruction when possible.
370 MachineInstr *emitTestBit(Register TestReg, uint64_t Bit, bool IsNegative,
371 MachineBasicBlock *DstMBB,
372 MachineIRBuilder &MIB) const;
373
374 /// Emit a CB(N)Z instruction which branches to \p DestMBB.
375 MachineInstr *emitCBZ(Register CompareReg, bool IsNegative,
376 MachineBasicBlock *DestMBB,
377 MachineIRBuilder &MIB) const;
378
379 // Equivalent to the i32shift_a and friends from AArch64InstrInfo.td.
380 // We use these manually instead of using the importer since it doesn't
381 // support SDNodeXForm.
382 ComplexRendererFns selectShiftA_32(const MachineOperand &Root) const;
383 ComplexRendererFns selectShiftB_32(const MachineOperand &Root) const;
384 ComplexRendererFns selectShiftA_64(const MachineOperand &Root) const;
385 ComplexRendererFns selectShiftB_64(const MachineOperand &Root) const;
386
387 template <unsigned ShiftWidth>
388 ComplexRendererFns selectShiftMask(MachineOperand &Root) const;
389 ComplexRendererFns select12BitValueWithLeftShift(uint64_t Immed) const;
390 ComplexRendererFns selectArithImmed(MachineOperand &Root) const;
391 ComplexRendererFns selectNegArithImmed(MachineOperand &Root) const;
392
393 ComplexRendererFns selectAddrModeUnscaled(MachineOperand &Root,
394 unsigned Size) const;
395
396 ComplexRendererFns selectAddrModeUnscaled8(MachineOperand &Root) const {
397 return selectAddrModeUnscaled(Root, Size: 1);
398 }
399 ComplexRendererFns selectAddrModeUnscaled16(MachineOperand &Root) const {
400 return selectAddrModeUnscaled(Root, Size: 2);
401 }
402 ComplexRendererFns selectAddrModeUnscaled32(MachineOperand &Root) const {
403 return selectAddrModeUnscaled(Root, Size: 4);
404 }
405 ComplexRendererFns selectAddrModeUnscaled64(MachineOperand &Root) const {
406 return selectAddrModeUnscaled(Root, Size: 8);
407 }
408 ComplexRendererFns selectAddrModeUnscaled128(MachineOperand &Root) const {
409 return selectAddrModeUnscaled(Root, Size: 16);
410 }
411
412 /// Helper to try to fold in a GISEL_ADD_LOW into an immediate, to be used
413 /// from complex pattern matchers like selectAddrModeIndexed().
414 ComplexRendererFns tryFoldAddLowIntoImm(MachineInstr &RootDef, unsigned Size,
415 MachineRegisterInfo &MRI) const;
416
417 ComplexRendererFns selectAddrModeIndexed(MachineOperand &Root,
418 unsigned Size) const;
419 template <int Width>
420 ComplexRendererFns selectAddrModeIndexed(MachineOperand &Root) const {
421 return selectAddrModeIndexed(Root, Size: Width / 8);
422 }
423
424 std::optional<bool>
425 isWorthFoldingIntoAddrMode(const MachineInstr &MI,
426 const MachineRegisterInfo &MRI) const;
427
428 bool isWorthFoldingIntoExtendedReg(const MachineInstr &MI,
429 const MachineRegisterInfo &MRI,
430 bool IsAddrOperand) const;
431 ComplexRendererFns
432 selectAddrModeShiftedExtendXReg(MachineOperand &Root,
433 unsigned SizeInBytes) const;
434
435 /// Returns a \p ComplexRendererFns which contains a base, offset, and whether
436 /// or not a shift + extend should be folded into an addressing mode. Returns
437 /// None when this is not profitable or possible.
438 ComplexRendererFns
439 selectExtendedSHL(MachineOperand &Root, MachineOperand &Base,
440 MachineOperand &Offset, unsigned SizeInBytes,
441 bool WantsExt) const;
442 ComplexRendererFns selectAddrModeRegisterOffset(MachineOperand &Root) const;
443 ComplexRendererFns selectAddrModeXRO(MachineOperand &Root,
444 unsigned SizeInBytes) const;
445 template <int Width>
446 ComplexRendererFns selectAddrModeXRO(MachineOperand &Root) const {
447 return selectAddrModeXRO(Root, SizeInBytes: Width / 8);
448 }
449
450 ComplexRendererFns selectAddrModeWRO(MachineOperand &Root,
451 unsigned SizeInBytes) const;
452 template <int Width>
453 ComplexRendererFns selectAddrModeWRO(MachineOperand &Root) const {
454 return selectAddrModeWRO(Root, SizeInBytes: Width / 8);
455 }
456
457 ComplexRendererFns selectShiftedRegister(MachineOperand &Root,
458 bool AllowROR = false) const;
459
460 ComplexRendererFns selectArithShiftedRegister(MachineOperand &Root) const {
461 return selectShiftedRegister(Root);
462 }
463
464 ComplexRendererFns selectLogicalShiftedRegister(MachineOperand &Root) const {
465 return selectShiftedRegister(Root, AllowROR: true);
466 }
467
468 /// Given an extend instruction, determine the correct shift-extend type for
469 /// that instruction.
470 ///
471 /// If the instruction is going to be used in a load or store, pass
472 /// \p IsLoadStore = true.
473 AArch64_AM::ShiftExtendType
474 getExtendTypeForInst(MachineInstr &MI, MachineRegisterInfo &MRI,
475 bool IsLoadStore = false) const;
476
477 /// Move \p Reg to \p RC if \p Reg is not already on \p RC.
478 ///
479 /// \returns Either \p Reg if no change was necessary, or the new register
480 /// created by moving \p Reg.
481 ///
482 /// Note: This uses emitCopy right now.
483 Register moveScalarRegClass(Register Reg, const TargetRegisterClass &RC,
484 MachineIRBuilder &MIB) const;
485
486 ComplexRendererFns selectArithExtendedRegister(MachineOperand &Root) const;
487
488 ComplexRendererFns selectExtractHigh(MachineOperand &Root) const;
489 template <unsigned Width>
490 ComplexRendererFns selectCVTFixedPoint(MachineOperand &Root) const;
491 template <unsigned Width>
492 ComplexRendererFns selectCVTFixedPosRecipOperand(MachineOperand &Root) const;
493 ComplexRendererFns selectCVTFixedPointBase(const MachineOperand &Root,
494 unsigned width,
495 bool isReciprocal = false) const;
496 ComplexRendererFns selectCVTFixedPointVec(MachineOperand &Root) const;
497 ComplexRendererFns
498 selectCVTFixedPosRecipOperandVec(MachineOperand &Root) const;
499 void renderFixedPointScalarXForm(MachineInstrBuilder &MIB,
500 const MachineInstr &MI, int OpIdx) const;
501 unsigned getFixedPointWidthFromOperand(const MachineOperand &Root) const;
502 void renderFixedPointXForm(MachineInstrBuilder &MIB, const MachineInstr &MI,
503 int OpIdx = -1) const;
504 void renderFixedPointRecipXForm(MachineInstrBuilder &MIB,
505 const MachineInstr &MI, int OpIdx = -1) const;
506 void renderFixedPointImm(MachineInstrBuilder &MIB, const MachineOperand &Root,
507 unsigned Width, bool isReciprocal) const;
508 void renderTruncImm(MachineInstrBuilder &MIB, const MachineInstr &MI,
509 int OpIdx = -1) const;
510 void renderLogicalImm32(MachineInstrBuilder &MIB, const MachineInstr &I,
511 int OpIdx = -1) const;
512 void renderLogicalImm64(MachineInstrBuilder &MIB, const MachineInstr &I,
513 int OpIdx = -1) const;
514 void renderUbsanTrap(MachineInstrBuilder &MIB, const MachineInstr &MI,
515 int OpIdx) const;
516 void renderFPImm16(MachineInstrBuilder &MIB, const MachineInstr &MI,
517 int OpIdx = -1) const;
518 void renderFPImm32(MachineInstrBuilder &MIB, const MachineInstr &MI,
519 int OpIdx = -1) const;
520 void renderFPImm64(MachineInstrBuilder &MIB, const MachineInstr &MI,
521 int OpIdx = -1) const;
522 void renderFPImm32SIMDModImmType4(MachineInstrBuilder &MIB,
523 const MachineInstr &MI,
524 int OpIdx = -1) const;
525
526 // Materialize a GlobalValue or BlockAddress using a movz+movk sequence.
527 void materializeLargeCMVal(MachineInstr &I, const Value *V, unsigned OpFlags);
528
529 // Optimization methods.
530 bool tryOptSelect(GSelect &Sel);
531 bool tryOptSelectConjunction(GSelect &Sel, MachineInstr &CondMI);
532 MachineInstr *tryFoldIntegerCompare(MachineOperand &LHS, MachineOperand &RHS,
533 MachineOperand &Predicate,
534 MachineIRBuilder &MIRBuilder) const;
535
536 /// Return true if \p MI is a load or store of \p NumBytes bytes.
537 bool isLoadStoreOfNumBytes(const MachineInstr &MI, unsigned NumBytes) const;
538
539 /// Returns true if \p MI is guaranteed to have the high-half of a 64-bit
540 /// register zeroed out. In other words, the result of MI has been explicitly
541 /// zero extended.
542 bool isDef32(const MachineInstr &MI) const;
543
544 const AArch64TargetMachine &TM;
545 const AArch64Subtarget &STI;
546 const AArch64InstrInfo &TII;
547 const AArch64RegisterInfo &TRI;
548 const AArch64RegisterBankInfo &RBI;
549
550 bool ProduceNonFlagSettingCondBr = false;
551
552 // Some cached values used during selection.
553 // We use LR as a live-in register, and we keep track of it here as it can be
554 // clobbered by calls.
555 Register MFReturnAddr;
556
557 MachineIRBuilder MIB;
558
559#define GET_GLOBALISEL_PREDICATES_DECL
560#include "AArch64GenGlobalISel.inc"
561#undef GET_GLOBALISEL_PREDICATES_DECL
562
563// We declare the temporaries used by selectImpl() in the class to minimize the
564// cost of constructing placeholder values.
565#define GET_GLOBALISEL_TEMPORARIES_DECL
566#include "AArch64GenGlobalISel.inc"
567#undef GET_GLOBALISEL_TEMPORARIES_DECL
568};
569
570} // end anonymous namespace
571
572#define GET_GLOBALISEL_IMPL
573#include "AArch64GenGlobalISel.inc"
574#undef GET_GLOBALISEL_IMPL
575
576AArch64InstructionSelector::AArch64InstructionSelector(
577 const AArch64TargetMachine &TM, const AArch64Subtarget &STI,
578 const AArch64RegisterBankInfo &RBI)
579 : TM(TM), STI(STI), TII(*STI.getInstrInfo()), TRI(*STI.getRegisterInfo()),
580 RBI(RBI),
581#define GET_GLOBALISEL_PREDICATES_INIT
582#include "AArch64GenGlobalISel.inc"
583#undef GET_GLOBALISEL_PREDICATES_INIT
584#define GET_GLOBALISEL_TEMPORARIES_INIT
585#include "AArch64GenGlobalISel.inc"
586#undef GET_GLOBALISEL_TEMPORARIES_INIT
587{
588}
589
590// FIXME: This should be target-independent, inferred from the types declared
591// for each class in the bank.
592//
593/// Given a register bank, and a type, return the smallest register class that
594/// can represent that combination.
595static const TargetRegisterClass *
596getRegClassForTypeOnBank(LLT Ty, const RegisterBank &RB,
597 bool GetAllRegSet = false) {
598 if (RB.getID() == AArch64::GPRRegBankID) {
599 if (Ty.getSizeInBits() <= 32)
600 return GetAllRegSet ? &AArch64::GPR32allRegClass
601 : &AArch64::GPR32RegClass;
602 if (Ty.getSizeInBits() == 64)
603 return GetAllRegSet ? &AArch64::GPR64allRegClass
604 : &AArch64::GPR64RegClass;
605 if (Ty.getSizeInBits() == 128)
606 return &AArch64::XSeqPairsClassRegClass;
607 return nullptr;
608 }
609
610 if (RB.getID() == AArch64::FPRRegBankID) {
611 switch (Ty.getSizeInBits()) {
612 case 8:
613 return &AArch64::FPR8RegClass;
614 case 16:
615 return &AArch64::FPR16RegClass;
616 case 32:
617 return &AArch64::FPR32RegClass;
618 case 64:
619 return &AArch64::FPR64RegClass;
620 case 128:
621 return &AArch64::FPR128RegClass;
622 }
623 return nullptr;
624 }
625
626 return nullptr;
627}
628
629/// Given a register bank, and size in bits, return the smallest register class
630/// that can represent that combination.
631static const TargetRegisterClass *
632getMinClassForRegBank(const RegisterBank &RB, TypeSize SizeInBits,
633 bool GetAllRegSet = false) {
634 if (SizeInBits.isScalable()) {
635 assert(RB.getID() == AArch64::FPRRegBankID &&
636 "Expected FPR regbank for scalable type size");
637 return &AArch64::ZPRRegClass;
638 }
639
640 unsigned RegBankID = RB.getID();
641
642 if (RegBankID == AArch64::GPRRegBankID) {
643 assert(!SizeInBits.isScalable() && "Unexpected scalable register size");
644 if (SizeInBits <= 32)
645 return GetAllRegSet ? &AArch64::GPR32allRegClass
646 : &AArch64::GPR32RegClass;
647 if (SizeInBits == 64)
648 return GetAllRegSet ? &AArch64::GPR64allRegClass
649 : &AArch64::GPR64RegClass;
650 if (SizeInBits == 128)
651 return &AArch64::XSeqPairsClassRegClass;
652 }
653
654 if (RegBankID == AArch64::FPRRegBankID) {
655 if (SizeInBits.isScalable()) {
656 assert(SizeInBits == TypeSize::getScalable(128) &&
657 "Unexpected scalable register size");
658 return &AArch64::ZPRRegClass;
659 }
660
661 switch (SizeInBits) {
662 default:
663 return nullptr;
664 case 8:
665 return &AArch64::FPR8RegClass;
666 case 16:
667 return &AArch64::FPR16RegClass;
668 case 32:
669 return &AArch64::FPR32RegClass;
670 case 64:
671 return &AArch64::FPR64RegClass;
672 case 128:
673 return &AArch64::FPR128RegClass;
674 }
675 }
676
677 return nullptr;
678}
679
680/// Returns the correct subregister to use for a given register class.
681static bool getSubRegForClass(const TargetRegisterClass *RC,
682 const TargetRegisterInfo &TRI, unsigned &SubReg) {
683 switch (TRI.getRegSizeInBits(RC: *RC)) {
684 case 8:
685 SubReg = AArch64::bsub;
686 break;
687 case 16:
688 SubReg = AArch64::hsub;
689 break;
690 case 32:
691 if (RC != &AArch64::FPR32RegClass)
692 SubReg = AArch64::sub_32;
693 else
694 SubReg = AArch64::ssub;
695 break;
696 case 64:
697 SubReg = AArch64::dsub;
698 break;
699 default:
700 LLVM_DEBUG(
701 dbgs() << "Couldn't find appropriate subregister for register class.");
702 return false;
703 }
704
705 return true;
706}
707
708/// Returns the minimum size the given register bank can hold.
709static unsigned getMinSizeForRegBank(const RegisterBank &RB) {
710 switch (RB.getID()) {
711 case AArch64::GPRRegBankID:
712 return 32;
713 case AArch64::FPRRegBankID:
714 return 8;
715 default:
716 llvm_unreachable("Tried to get minimum size for unknown register bank.");
717 }
718}
719
720/// Create a REG_SEQUENCE instruction using the registers in \p Regs.
721/// Helper function for functions like createDTuple and createQTuple.
722///
723/// \p RegClassIDs - The list of register class IDs available for some tuple of
724/// a scalar class. E.g. QQRegClassID, QQQRegClassID, QQQQRegClassID. This is
725/// expected to contain between 2 and 4 tuple classes.
726///
727/// \p SubRegs - The list of subregister classes associated with each register
728/// class ID in \p RegClassIDs. E.g., QQRegClassID should use the qsub0
729/// subregister class. The index of each subregister class is expected to
730/// correspond with the index of each register class.
731///
732/// \returns Either the destination register of REG_SEQUENCE instruction that
733/// was created, or the 0th element of \p Regs if \p Regs contains a single
734/// element.
735static Register createTuple(ArrayRef<Register> Regs,
736 const unsigned RegClassIDs[],
737 const unsigned SubRegs[], MachineIRBuilder &MIB) {
738 unsigned NumRegs = Regs.size();
739 if (NumRegs == 1)
740 return Regs[0];
741 assert(NumRegs >= 2 && NumRegs <= 4 &&
742 "Only support between two and 4 registers in a tuple!");
743 const TargetRegisterInfo *TRI = MIB.getMF().getSubtarget().getRegisterInfo();
744 auto *DesiredClass = TRI->getRegClass(i: RegClassIDs[NumRegs - 2]);
745 auto RegSequence =
746 MIB.buildInstr(Opc: TargetOpcode::REG_SEQUENCE, DstOps: {DesiredClass}, SrcOps: {});
747 for (unsigned I = 0, E = Regs.size(); I < E; ++I) {
748 RegSequence.addUse(RegNo: Regs[I]);
749 RegSequence.addImm(Val: SubRegs[I]);
750 }
751 return RegSequence.getReg(Idx: 0);
752}
753
754/// Create a tuple of D-registers using the registers in \p Regs.
755static Register createDTuple(ArrayRef<Register> Regs, MachineIRBuilder &MIB) {
756 static const unsigned RegClassIDs[] = {
757 AArch64::DDRegClassID, AArch64::DDDRegClassID, AArch64::DDDDRegClassID};
758 static const unsigned SubRegs[] = {AArch64::dsub0, AArch64::dsub1,
759 AArch64::dsub2, AArch64::dsub3};
760 return createTuple(Regs, RegClassIDs, SubRegs, MIB);
761}
762
763/// Create a tuple of Q-registers using the registers in \p Regs.
764static Register createQTuple(ArrayRef<Register> Regs, MachineIRBuilder &MIB) {
765 static const unsigned RegClassIDs[] = {
766 AArch64::QQRegClassID, AArch64::QQQRegClassID, AArch64::QQQQRegClassID};
767 static const unsigned SubRegs[] = {AArch64::qsub0, AArch64::qsub1,
768 AArch64::qsub2, AArch64::qsub3};
769 return createTuple(Regs, RegClassIDs, SubRegs, MIB);
770}
771
772static std::optional<uint64_t> getImmedFromMO(const MachineOperand &Root) {
773 auto &MI = *Root.getParent();
774 auto &MBB = *MI.getParent();
775 auto &MF = *MBB.getParent();
776 auto &MRI = MF.getRegInfo();
777 uint64_t Immed;
778 if (Root.isImm())
779 Immed = Root.getImm();
780 else if (Root.isCImm())
781 Immed = Root.getCImm()->getZExtValue();
782 else if (Root.isReg()) {
783 auto ValAndVReg =
784 getIConstantVRegValWithLookThrough(VReg: Root.getReg(), MRI, LookThroughInstrs: true);
785 if (!ValAndVReg)
786 return std::nullopt;
787 Immed = ValAndVReg->Value.getSExtValue();
788 } else
789 return std::nullopt;
790 return Immed;
791}
792
793/// Select the AArch64 opcode for the basic binary operation \p GenericOpc,
794/// appropriate for the register bank \p RegBankID and of size \p OpSize.
795/// \returns \p GenericOpc if the combination is unsupported.
796static unsigned selectBinaryOp(unsigned GenericOpc, unsigned RegBankID,
797 unsigned OpSize) {
798 if (RegBankID == AArch64::GPRRegBankID) {
799 if (OpSize == 32) {
800 switch (GenericOpc) {
801 case TargetOpcode::G_SHL:
802 return AArch64::LSLVWr;
803 case TargetOpcode::G_LSHR:
804 return AArch64::LSRVWr;
805 case TargetOpcode::G_ASHR:
806 return AArch64::ASRVWr;
807 default:
808 return GenericOpc;
809 }
810 } else if (OpSize == 64) {
811 switch (GenericOpc) {
812 case TargetOpcode::G_SHL:
813 return AArch64::LSLVXr;
814 case TargetOpcode::G_LSHR:
815 return AArch64::LSRVXr;
816 case TargetOpcode::G_ASHR:
817 return AArch64::ASRVXr;
818 default:
819 return GenericOpc;
820 }
821 }
822 }
823 return GenericOpc;
824}
825
826/// Select the AArch64 opcode for the G_LOAD or G_STORE operation \p GenericOpc,
827/// appropriate for the (value) register bank \p RegBankID and of memory access
828/// size \p OpSize. This returns the variant with the base+unsigned-immediate
829/// addressing mode (e.g., LDRXui).
830/// \returns \p GenericOpc if the combination is unsupported.
831static unsigned selectLoadStoreUIOp(unsigned GenericOpc, unsigned RegBankID,
832 unsigned OpSize) {
833 const bool isStore = GenericOpc == TargetOpcode::G_STORE;
834 switch (RegBankID) {
835 case AArch64::GPRRegBankID:
836 switch (OpSize) {
837 case 8:
838 return isStore ? AArch64::STRBBui : AArch64::LDRBBui;
839 case 16:
840 return isStore ? AArch64::STRHHui : AArch64::LDRHHui;
841 case 32:
842 return isStore ? AArch64::STRWui : AArch64::LDRWui;
843 case 64:
844 return isStore ? AArch64::STRXui : AArch64::LDRXui;
845 }
846 break;
847 case AArch64::FPRRegBankID:
848 switch (OpSize) {
849 case 8:
850 return isStore ? AArch64::STRBui : AArch64::LDRBui;
851 case 16:
852 return isStore ? AArch64::STRHui : AArch64::LDRHui;
853 case 32:
854 return isStore ? AArch64::STRSui : AArch64::LDRSui;
855 case 64:
856 return isStore ? AArch64::STRDui : AArch64::LDRDui;
857 case 128:
858 return isStore ? AArch64::STRQui : AArch64::LDRQui;
859 }
860 break;
861 }
862 return GenericOpc;
863}
864
865/// Helper function for selectCopy. Inserts a subregister copy from \p SrcReg
866/// to \p *To.
867///
868/// E.g "To = COPY SrcReg:SubReg"
869static bool copySubReg(MachineInstr &I, MachineRegisterInfo &MRI,
870 const RegisterBankInfo &RBI, Register SrcReg,
871 const TargetRegisterClass *To, unsigned SubReg) {
872 assert(SrcReg.isValid() && "Expected a valid source register?");
873 assert(To && "Destination register class cannot be null");
874 assert(SubReg && "Expected a valid subregister");
875
876 MachineIRBuilder MIB(I);
877 auto SubRegCopy =
878 MIB.buildInstr(Opc: TargetOpcode::COPY, DstOps: {To}, SrcOps: {}).addReg(RegNo: SrcReg, Flags: {}, SubReg);
879 MachineOperand &RegOp = I.getOperand(i: 1);
880 RegOp.setReg(SubRegCopy.getReg(Idx: 0));
881
882 // It's possible that the destination register won't be constrained. Make
883 // sure that happens.
884 if (!I.getOperand(i: 0).getReg().isPhysical())
885 RBI.constrainGenericRegister(Reg: I.getOperand(i: 0).getReg(), RC: *To, MRI);
886
887 return true;
888}
889
890// FIXME: We need some sort of API in RBI/TRI to allow generic code to
891// constrain operands of simple instructions given a TargetRegisterClass
892// and LLT
893static bool selectDebugInstr(MachineInstr &I, MachineRegisterInfo &MRI,
894 const RegisterBankInfo &RBI) {
895 for (MachineOperand &MO : I.operands()) {
896 if (!MO.isReg())
897 continue;
898 Register Reg = MO.getReg();
899 if (!Reg)
900 continue;
901 if (Reg.isPhysical())
902 continue;
903 LLT Ty = MRI.getType(Reg);
904 const RegClassOrRegBank &RegClassOrBank = MRI.getRegClassOrRegBank(Reg);
905 const TargetRegisterClass *RC =
906 dyn_cast<const TargetRegisterClass *>(Val: RegClassOrBank);
907 if (!RC) {
908 const RegisterBank &RB = *cast<const RegisterBank *>(Val: RegClassOrBank);
909 RC = getRegClassForTypeOnBank(Ty, RB);
910 if (!RC) {
911 LLVM_DEBUG(
912 dbgs() << "Warning: DBG_VALUE operand has unexpected size/bank\n");
913 break;
914 }
915 }
916 RBI.constrainGenericRegister(Reg, RC: *RC, MRI);
917 }
918
919 return true;
920}
921
922static bool selectCopy(MachineInstr &I, const TargetInstrInfo &TII,
923 MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI,
924 const RegisterBankInfo &RBI) {
925 Register DstReg = I.getOperand(i: 0).getReg();
926 Register SrcReg = I.getOperand(i: 1).getReg();
927 const RegisterBank &DstRegBank = *RBI.getRegBank(Reg: DstReg, MRI, TRI);
928 const RegisterBank &SrcRegBank = *RBI.getRegBank(Reg: SrcReg, MRI, TRI);
929
930 TypeSize DstRegSize = RBI.getSizeInBits(Reg: DstReg, MRI, TRI);
931 TypeSize SrcRegSize = RBI.getSizeInBits(Reg: SrcReg, MRI, TRI);
932
933 // Special casing for cross-bank copies of s1s. We can technically represent
934 // a 1-bit value with any size of register. The minimum size for a GPR is 32
935 // bits. So, we need to put the FPR on 32 bits as well.
936 //
937 // FIXME: I'm not sure if this case holds true outside of copies. If it does,
938 // then we can pull it into the helpers that get the appropriate class for a
939 // register bank. Or make a new helper that carries along some constraint
940 // information.
941 if (SrcRegBank != DstRegBank && (DstRegSize == TypeSize::getFixed(ExactSize: 1) &&
942 SrcRegSize == TypeSize::getFixed(ExactSize: 1)))
943 SrcRegSize = DstRegSize = TypeSize::getFixed(ExactSize: 32);
944
945 // Find the correct register classes for the source and destination registers.
946 const TargetRegisterClass *SrcRC =
947 getMinClassForRegBank(RB: SrcRegBank, SizeInBits: SrcRegSize, GetAllRegSet: true);
948 const TargetRegisterClass *DstRC =
949 getMinClassForRegBank(RB: DstRegBank, SizeInBits: DstRegSize, GetAllRegSet: true);
950
951 if (!DstRC) {
952 LLVM_DEBUG(dbgs() << "Unexpected dest size "
953 << RBI.getSizeInBits(DstReg, MRI, TRI) << '\n');
954 return false;
955 }
956
957 if (I.getOpcode() == TargetOpcode::G_BITCAST &&
958 RBI.getSizeInBits(Reg: DstReg, MRI, TRI) == TypeSize::getFixed(ExactSize: 16)) {
959 if (DstRegBank.getID() == AArch64::FPRRegBankID &&
960 SrcRegBank.getID() == AArch64::GPRRegBankID) {
961 if (!SrcReg.isPhysical() &&
962 !RBI.constrainGenericRegister(Reg: SrcReg, RC: AArch64::GPR32RegClass, MRI))
963 return false;
964 if (!DstReg.isPhysical() &&
965 !RBI.constrainGenericRegister(Reg: DstReg, RC: AArch64::FPR16RegClass, MRI))
966 return false;
967
968 Register FPR32 = MRI.createVirtualRegister(RegClass: &AArch64::FPR32RegClass);
969 BuildMI(BB&: *I.getParent(), I, MIMD: I.getDebugLoc(), MCID: TII.get(Opcode: AArch64::FMOVWSr))
970 .addDef(RegNo: FPR32)
971 .addUse(RegNo: SrcReg);
972 I.setDesc(TII.get(Opcode: TargetOpcode::COPY));
973 I.getOperand(i: 1).setReg(FPR32);
974 I.getOperand(i: 1).setSubReg(AArch64::hsub);
975 return true;
976 }
977
978 if (DstRegBank.getID() == AArch64::GPRRegBankID &&
979 SrcRegBank.getID() == AArch64::FPRRegBankID) {
980 if (!SrcReg.isPhysical() &&
981 !RBI.constrainGenericRegister(Reg: SrcReg, RC: AArch64::FPR16RegClass, MRI))
982 return false;
983 if (!DstReg.isPhysical() &&
984 !RBI.constrainGenericRegister(Reg: DstReg, RC: AArch64::GPR32RegClass, MRI))
985 return false;
986
987 Register FPR32 = MRI.createVirtualRegister(RegClass: &AArch64::FPR32RegClass);
988 BuildMI(BB&: *I.getParent(), I, MIMD: I.getDebugLoc(),
989 MCID: TII.get(Opcode: TargetOpcode::SUBREG_TO_REG))
990 .addDef(RegNo: FPR32)
991 .addUse(RegNo: SrcReg)
992 .addImm(Val: AArch64::hsub);
993 I.setDesc(TII.get(Opcode: AArch64::FMOVSWr));
994 I.getOperand(i: 1).setReg(FPR32);
995 return true;
996 }
997 }
998
999 // Is this a copy? If so, then we may need to insert a subregister copy.
1000 if (I.isCopy()) {
1001 // Yes. Check if there's anything to fix up.
1002 if (!SrcRC) {
1003 LLVM_DEBUG(dbgs() << "Couldn't determine source register class\n");
1004 return false;
1005 }
1006
1007 const TypeSize SrcSize = TRI.getRegSizeInBits(RC: *SrcRC);
1008 const TypeSize DstSize = TRI.getRegSizeInBits(RC: *DstRC);
1009 unsigned SrcSubReg = I.getOperand(i: 1).getSubReg();
1010 unsigned SubReg;
1011
1012 if (SrcSubReg)
1013 return RBI.constrainGenericRegister(Reg: DstReg, RC: *DstRC, MRI);
1014
1015 // If the source bank doesn't support a subregister copy small enough,
1016 // then we first need to copy to the destination bank.
1017 if (getMinSizeForRegBank(RB: SrcRegBank) > DstSize) {
1018 const TargetRegisterClass *DstTempRC =
1019 getMinClassForRegBank(RB: DstRegBank, SizeInBits: SrcSize, /* GetAllRegSet */ true);
1020 getSubRegForClass(RC: DstRC, TRI, SubReg);
1021
1022 MachineIRBuilder MIB(I);
1023 auto Copy = MIB.buildCopy(Res: {DstTempRC}, Op: {SrcReg});
1024 copySubReg(I, MRI, RBI, SrcReg: Copy.getReg(Idx: 0), To: DstRC, SubReg);
1025 } else if (SrcSize > DstSize) {
1026 // If the source register is bigger than the destination we need to
1027 // perform a subregister copy.
1028 const TargetRegisterClass *SubRegRC =
1029 getMinClassForRegBank(RB: SrcRegBank, SizeInBits: DstSize, /* GetAllRegSet */ true);
1030 getSubRegForClass(RC: SubRegRC, TRI, SubReg);
1031 copySubReg(I, MRI, RBI, SrcReg, To: DstRC, SubReg);
1032 } else if (DstSize > SrcSize) {
1033 // If the destination register is bigger than the source we need to do
1034 // a promotion using SUBREG_TO_REG.
1035 const TargetRegisterClass *PromotionRC =
1036 getMinClassForRegBank(RB: SrcRegBank, SizeInBits: DstSize, /* GetAllRegSet */ true);
1037 getSubRegForClass(RC: SrcRC, TRI, SubReg);
1038
1039 Register PromoteReg = MRI.createVirtualRegister(RegClass: PromotionRC);
1040 BuildMI(BB&: *I.getParent(), I, MIMD: I.getDebugLoc(),
1041 MCID: TII.get(Opcode: AArch64::SUBREG_TO_REG), DestReg: PromoteReg)
1042 .addUse(RegNo: SrcReg)
1043 .addImm(Val: SubReg);
1044 MachineOperand &RegOp = I.getOperand(i: 1);
1045 RegOp.setReg(PromoteReg);
1046 }
1047
1048 // If the destination is a physical register, then there's nothing to
1049 // change, so we're done.
1050 if (DstReg.isPhysical())
1051 return true;
1052 }
1053
1054 // No need to constrain SrcReg. It will get constrained when we hit another
1055 // of its use or its defs. Copies do not have constraints.
1056 if (!RBI.constrainGenericRegister(Reg: DstReg, RC: *DstRC, MRI)) {
1057 LLVM_DEBUG(dbgs() << "Failed to constrain " << TII.getName(I.getOpcode())
1058 << " operand\n");
1059 return false;
1060 }
1061
1062 // If this a GPR ZEXT that we want to just reduce down into a copy.
1063 // The sizes will be mismatched with the source < 32b but that's ok.
1064 if (I.getOpcode() == TargetOpcode::G_ZEXT) {
1065 I.setDesc(TII.get(Opcode: AArch64::COPY));
1066 assert(SrcRegBank.getID() == AArch64::GPRRegBankID);
1067 return selectCopy(I, TII, MRI, TRI, RBI);
1068 }
1069
1070 I.setDesc(TII.get(Opcode: AArch64::COPY));
1071 return true;
1072}
1073
1074MachineInstr *
1075AArch64InstructionSelector::emitSelect(Register Dst, Register True,
1076 Register False, AArch64CC::CondCode CC,
1077 MachineIRBuilder &MIB) const {
1078 MachineRegisterInfo &MRI = *MIB.getMRI();
1079 assert(RBI.getRegBank(False, MRI, TRI)->getID() ==
1080 RBI.getRegBank(True, MRI, TRI)->getID() &&
1081 "Expected both select operands to have the same regbank?");
1082 LLT Ty = MRI.getType(Reg: True);
1083 if (Ty.isVector())
1084 return nullptr;
1085 const unsigned Size = Ty.getSizeInBits();
1086 assert((Size == 32 || Size == 64) &&
1087 "Expected 32 bit or 64 bit select only?");
1088 const bool Is32Bit = Size == 32;
1089 if (RBI.getRegBank(Reg: True, MRI, TRI)->getID() != AArch64::GPRRegBankID) {
1090 unsigned Opc = Is32Bit ? AArch64::FCSELSrrr : AArch64::FCSELDrrr;
1091 auto FCSel = MIB.buildInstr(Opc, DstOps: {Dst}, SrcOps: {True, False}).addImm(Val: CC);
1092 constrainSelectedInstRegOperands(I&: *FCSel, TII, TRI, RBI);
1093 return &*FCSel;
1094 }
1095
1096 // By default, we'll try and emit a CSEL.
1097 unsigned Opc = Is32Bit ? AArch64::CSELWr : AArch64::CSELXr;
1098 bool Optimized = false;
1099 auto TryFoldBinOpIntoSelect = [&Opc, Is32Bit, &CC, &MRI,
1100 &Optimized](Register &Reg, Register &OtherReg,
1101 bool Invert) {
1102 if (Optimized)
1103 return false;
1104
1105 // Attempt to fold:
1106 //
1107 // %sub = G_SUB 0, %x
1108 // %select = G_SELECT cc, %reg, %sub
1109 //
1110 // Into:
1111 // %select = CSNEG %reg, %x, cc
1112 Register MatchReg;
1113 if (mi_match(R: Reg, MRI, P: m_Neg(Src: m_Reg(R&: MatchReg)))) {
1114 Opc = Is32Bit ? AArch64::CSNEGWr : AArch64::CSNEGXr;
1115 Reg = MatchReg;
1116 if (Invert) {
1117 CC = AArch64CC::getInvertedCondCode(Code: CC);
1118 std::swap(a&: Reg, b&: OtherReg);
1119 }
1120 return true;
1121 }
1122
1123 // Attempt to fold:
1124 //
1125 // %xor = G_XOR %x, -1
1126 // %select = G_SELECT cc, %reg, %xor
1127 //
1128 // Into:
1129 // %select = CSINV %reg, %x, cc
1130 if (mi_match(R: Reg, MRI, P: m_Not(Src: m_Reg(R&: MatchReg)))) {
1131 Opc = Is32Bit ? AArch64::CSINVWr : AArch64::CSINVXr;
1132 Reg = MatchReg;
1133 if (Invert) {
1134 CC = AArch64CC::getInvertedCondCode(Code: CC);
1135 std::swap(a&: Reg, b&: OtherReg);
1136 }
1137 return true;
1138 }
1139
1140 // Attempt to fold:
1141 //
1142 // %add = G_ADD %x, 1
1143 // %select = G_SELECT cc, %reg, %add
1144 //
1145 // Into:
1146 // %select = CSINC %reg, %x, cc
1147 if (mi_match(R: Reg, MRI,
1148 P: m_any_of(preds: m_GAdd(L: m_Reg(R&: MatchReg), R: m_SpecificICst(RequestedValue: 1)),
1149 preds: m_GPtrAdd(L: m_Reg(R&: MatchReg), R: m_SpecificICst(RequestedValue: 1))))) {
1150 Opc = Is32Bit ? AArch64::CSINCWr : AArch64::CSINCXr;
1151 Reg = MatchReg;
1152 if (Invert) {
1153 CC = AArch64CC::getInvertedCondCode(Code: CC);
1154 std::swap(a&: Reg, b&: OtherReg);
1155 }
1156 return true;
1157 }
1158
1159 return false;
1160 };
1161
1162 // Helper lambda which tries to use CSINC/CSINV for the instruction when its
1163 // true/false values are constants.
1164 // FIXME: All of these patterns already exist in tablegen. We should be
1165 // able to import these.
1166 auto TryOptSelectCst = [&Opc, &True, &False, &CC, Is32Bit, &MRI,
1167 &Optimized]() {
1168 if (Optimized)
1169 return false;
1170 auto TrueCst = getIConstantVRegValWithLookThrough(VReg: True, MRI);
1171 auto FalseCst = getIConstantVRegValWithLookThrough(VReg: False, MRI);
1172 if (!TrueCst && !FalseCst)
1173 return false;
1174
1175 Register ZReg = Is32Bit ? AArch64::WZR : AArch64::XZR;
1176 if (TrueCst && FalseCst) {
1177 int64_t T = TrueCst->Value.getSExtValue();
1178 int64_t F = FalseCst->Value.getSExtValue();
1179
1180 if (T == 0 && F == 1) {
1181 // G_SELECT cc, 0, 1 -> CSINC zreg, zreg, cc
1182 Opc = Is32Bit ? AArch64::CSINCWr : AArch64::CSINCXr;
1183 True = ZReg;
1184 False = ZReg;
1185 return true;
1186 }
1187
1188 if (T == 0 && F == -1) {
1189 // G_SELECT cc 0, -1 -> CSINV zreg, zreg cc
1190 Opc = Is32Bit ? AArch64::CSINVWr : AArch64::CSINVXr;
1191 True = ZReg;
1192 False = ZReg;
1193 return true;
1194 }
1195 }
1196
1197 if (TrueCst) {
1198 int64_t T = TrueCst->Value.getSExtValue();
1199 if (T == 1) {
1200 // G_SELECT cc, 1, f -> CSINC f, zreg, inv_cc
1201 Opc = Is32Bit ? AArch64::CSINCWr : AArch64::CSINCXr;
1202 True = False;
1203 False = ZReg;
1204 CC = AArch64CC::getInvertedCondCode(Code: CC);
1205 return true;
1206 }
1207
1208 if (T == -1) {
1209 // G_SELECT cc, -1, f -> CSINV f, zreg, inv_cc
1210 Opc = Is32Bit ? AArch64::CSINVWr : AArch64::CSINVXr;
1211 True = False;
1212 False = ZReg;
1213 CC = AArch64CC::getInvertedCondCode(Code: CC);
1214 return true;
1215 }
1216 }
1217
1218 if (FalseCst) {
1219 int64_t F = FalseCst->Value.getSExtValue();
1220 if (F == 1) {
1221 // G_SELECT cc, t, 1 -> CSINC t, zreg, cc
1222 Opc = Is32Bit ? AArch64::CSINCWr : AArch64::CSINCXr;
1223 False = ZReg;
1224 return true;
1225 }
1226
1227 if (F == -1) {
1228 // G_SELECT cc, t, -1 -> CSINC t, zreg, cc
1229 Opc = Is32Bit ? AArch64::CSINVWr : AArch64::CSINVXr;
1230 False = ZReg;
1231 return true;
1232 }
1233 }
1234 return false;
1235 };
1236
1237 Optimized |= TryFoldBinOpIntoSelect(False, True, /*Invert = */ false);
1238 Optimized |= TryFoldBinOpIntoSelect(True, False, /*Invert = */ true);
1239 Optimized |= TryOptSelectCst();
1240 auto SelectInst = MIB.buildInstr(Opc, DstOps: {Dst}, SrcOps: {True, False}).addImm(Val: CC);
1241 constrainSelectedInstRegOperands(I&: *SelectInst, TII, TRI, RBI);
1242 return &*SelectInst;
1243}
1244
1245static AArch64CC::CondCode
1246changeICMPPredToAArch64CC(CmpInst::Predicate P, Register RHS = {},
1247 MachineRegisterInfo *MRI = nullptr) {
1248 switch (P) {
1249 default:
1250 llvm_unreachable("Unknown condition code!");
1251 case CmpInst::ICMP_NE:
1252 return AArch64CC::NE;
1253 case CmpInst::ICMP_EQ:
1254 return AArch64CC::EQ;
1255 case CmpInst::ICMP_SGT:
1256 return AArch64CC::GT;
1257 case CmpInst::ICMP_SGE:
1258 if (RHS && MRI) {
1259 auto ValAndVReg = getIConstantVRegValWithLookThrough(VReg: RHS, MRI: *MRI);
1260 if (ValAndVReg && ValAndVReg->Value == 0)
1261 return AArch64CC::PL;
1262 }
1263 return AArch64CC::GE;
1264 case CmpInst::ICMP_SLT:
1265 if (RHS && MRI) {
1266 auto ValAndVReg = getIConstantVRegValWithLookThrough(VReg: RHS, MRI: *MRI);
1267 if (ValAndVReg && ValAndVReg->Value == 0)
1268 return AArch64CC::MI;
1269 }
1270 return AArch64CC::LT;
1271 case CmpInst::ICMP_SLE:
1272 return AArch64CC::LE;
1273 case CmpInst::ICMP_UGT:
1274 return AArch64CC::HI;
1275 case CmpInst::ICMP_UGE:
1276 return AArch64CC::HS;
1277 case CmpInst::ICMP_ULT:
1278 return AArch64CC::LO;
1279 case CmpInst::ICMP_ULE:
1280 return AArch64CC::LS;
1281 }
1282}
1283
1284/// changeFPCCToORAArch64CC - Convert an IR fp condition code to an AArch64 CC.
1285static void changeFPCCToORAArch64CC(CmpInst::Predicate CC,
1286 AArch64CC::CondCode &CondCode,
1287 AArch64CC::CondCode &CondCode2) {
1288 CondCode2 = AArch64CC::AL;
1289 switch (CC) {
1290 default:
1291 llvm_unreachable("Unknown FP condition!");
1292 case CmpInst::FCMP_OEQ:
1293 CondCode = AArch64CC::EQ;
1294 break;
1295 case CmpInst::FCMP_OGT:
1296 CondCode = AArch64CC::GT;
1297 break;
1298 case CmpInst::FCMP_OGE:
1299 CondCode = AArch64CC::GE;
1300 break;
1301 case CmpInst::FCMP_OLT:
1302 CondCode = AArch64CC::MI;
1303 break;
1304 case CmpInst::FCMP_OLE:
1305 CondCode = AArch64CC::LS;
1306 break;
1307 case CmpInst::FCMP_ONE:
1308 CondCode = AArch64CC::MI;
1309 CondCode2 = AArch64CC::GT;
1310 break;
1311 case CmpInst::FCMP_ORD:
1312 CondCode = AArch64CC::VC;
1313 break;
1314 case CmpInst::FCMP_UNO:
1315 CondCode = AArch64CC::VS;
1316 break;
1317 case CmpInst::FCMP_UEQ:
1318 CondCode = AArch64CC::EQ;
1319 CondCode2 = AArch64CC::VS;
1320 break;
1321 case CmpInst::FCMP_UGT:
1322 CondCode = AArch64CC::HI;
1323 break;
1324 case CmpInst::FCMP_UGE:
1325 CondCode = AArch64CC::PL;
1326 break;
1327 case CmpInst::FCMP_ULT:
1328 CondCode = AArch64CC::LT;
1329 break;
1330 case CmpInst::FCMP_ULE:
1331 CondCode = AArch64CC::LE;
1332 break;
1333 case CmpInst::FCMP_UNE:
1334 CondCode = AArch64CC::NE;
1335 break;
1336 }
1337}
1338
1339/// Convert an IR fp condition code to an AArch64 CC.
1340/// This differs from changeFPCCToAArch64CC in that it returns cond codes that
1341/// should be AND'ed instead of OR'ed.
1342static void changeFPCCToANDAArch64CC(CmpInst::Predicate CC,
1343 AArch64CC::CondCode &CondCode,
1344 AArch64CC::CondCode &CondCode2) {
1345 CondCode2 = AArch64CC::AL;
1346 switch (CC) {
1347 default:
1348 changeFPCCToORAArch64CC(CC, CondCode, CondCode2);
1349 assert(CondCode2 == AArch64CC::AL);
1350 break;
1351 case CmpInst::FCMP_ONE:
1352 // (a one b)
1353 // == ((a olt b) || (a ogt b))
1354 // == ((a ord b) && (a une b))
1355 CondCode = AArch64CC::VC;
1356 CondCode2 = AArch64CC::NE;
1357 break;
1358 case CmpInst::FCMP_UEQ:
1359 // (a ueq b)
1360 // == ((a uno b) || (a oeq b))
1361 // == ((a ule b) && (a uge b))
1362 CondCode = AArch64CC::PL;
1363 CondCode2 = AArch64CC::LE;
1364 break;
1365 }
1366}
1367
1368/// Return a register which can be used as a bit to test in a TB(N)Z.
1369static Register getTestBitReg(Register Reg, uint64_t &Bit, bool &Invert,
1370 MachineRegisterInfo &MRI) {
1371 assert(Reg.isValid() && "Expected valid register!");
1372 bool HasZext = false;
1373 while (MachineInstr *MI = getDefIgnoringCopies(Reg, MRI)) {
1374 unsigned Opc = MI->getOpcode();
1375
1376 if (!MI->getOperand(i: 0).isReg() ||
1377 !MRI.hasOneNonDBGUse(RegNo: MI->getOperand(i: 0).getReg()))
1378 break;
1379
1380 // (tbz (any_ext x), b) -> (tbz x, b) and
1381 // (tbz (zext x), b) -> (tbz x, b) if we don't use the extended bits.
1382 //
1383 // (tbz (trunc x), b) -> (tbz x, b) is always safe, because the bit number
1384 // on the truncated x is the same as the bit number on x.
1385 if (Opc == TargetOpcode::G_ANYEXT || Opc == TargetOpcode::G_ZEXT ||
1386 Opc == TargetOpcode::G_TRUNC) {
1387 if (Opc == TargetOpcode::G_ZEXT)
1388 HasZext = true;
1389
1390 Register NextReg = MI->getOperand(i: 1).getReg();
1391 // Did we find something worth folding?
1392 if (!NextReg.isValid() || !MRI.hasOneNonDBGUse(RegNo: NextReg))
1393 break;
1394 TypeSize InSize = MRI.getType(Reg: NextReg).getSizeInBits();
1395 if (Bit >= InSize)
1396 break;
1397
1398 // NextReg is worth folding. Keep looking.
1399 Reg = NextReg;
1400 continue;
1401 }
1402
1403 // Attempt to find a suitable operation with a constant on one side.
1404 std::optional<uint64_t> C;
1405 Register TestReg;
1406 switch (Opc) {
1407 default:
1408 break;
1409 case TargetOpcode::G_AND:
1410 case TargetOpcode::G_XOR: {
1411 TestReg = MI->getOperand(i: 1).getReg();
1412 Register ConstantReg = MI->getOperand(i: 2).getReg();
1413 auto VRegAndVal = getIConstantVRegValWithLookThrough(VReg: ConstantReg, MRI);
1414 if (!VRegAndVal) {
1415 // AND commutes, check the other side for a constant.
1416 // FIXME: Can we canonicalize the constant so that it's always on the
1417 // same side at some point earlier?
1418 std::swap(a&: ConstantReg, b&: TestReg);
1419 VRegAndVal = getIConstantVRegValWithLookThrough(VReg: ConstantReg, MRI);
1420 }
1421 if (VRegAndVal) {
1422 if (HasZext)
1423 C = VRegAndVal->Value.getZExtValue();
1424 else
1425 C = VRegAndVal->Value.getSExtValue();
1426 }
1427 break;
1428 }
1429 case TargetOpcode::G_ASHR:
1430 case TargetOpcode::G_LSHR:
1431 case TargetOpcode::G_SHL: {
1432 TestReg = MI->getOperand(i: 1).getReg();
1433 auto VRegAndVal =
1434 getIConstantVRegValWithLookThrough(VReg: MI->getOperand(i: 2).getReg(), MRI);
1435 if (VRegAndVal)
1436 C = VRegAndVal->Value.getSExtValue();
1437 break;
1438 }
1439 }
1440
1441 // Didn't find a constant or viable register. Bail out of the loop.
1442 if (!C || !TestReg.isValid())
1443 break;
1444
1445 // We found a suitable instruction with a constant. Check to see if we can
1446 // walk through the instruction.
1447 Register NextReg;
1448 unsigned TestRegSize = MRI.getType(Reg: TestReg).getSizeInBits();
1449 switch (Opc) {
1450 default:
1451 break;
1452 case TargetOpcode::G_AND:
1453 // (tbz (and x, m), b) -> (tbz x, b) when the b-th bit of m is set.
1454 if ((*C >> Bit) & 1)
1455 NextReg = TestReg;
1456 break;
1457 case TargetOpcode::G_SHL:
1458 // (tbz (shl x, c), b) -> (tbz x, b-c) when b-c is positive and fits in
1459 // the type of the register.
1460 if (*C <= Bit && (Bit - *C) < TestRegSize) {
1461 NextReg = TestReg;
1462 Bit = Bit - *C;
1463 }
1464 break;
1465 case TargetOpcode::G_ASHR:
1466 // (tbz (ashr x, c), b) -> (tbz x, b+c) or (tbz x, msb) if b+c is > # bits
1467 // in x
1468 NextReg = TestReg;
1469 Bit = Bit + *C;
1470 if (Bit >= TestRegSize)
1471 Bit = TestRegSize - 1;
1472 break;
1473 case TargetOpcode::G_LSHR:
1474 // (tbz (lshr x, c), b) -> (tbz x, b+c) when b + c is < # bits in x
1475 if ((Bit + *C) < TestRegSize) {
1476 NextReg = TestReg;
1477 Bit = Bit + *C;
1478 }
1479 break;
1480 case TargetOpcode::G_XOR:
1481 // We can walk through a G_XOR by inverting whether we use tbz/tbnz when
1482 // appropriate.
1483 //
1484 // e.g. If x' = xor x, c, and the b-th bit is set in c then
1485 //
1486 // tbz x', b -> tbnz x, b
1487 //
1488 // Because x' only has the b-th bit set if x does not.
1489 if ((*C >> Bit) & 1)
1490 Invert = !Invert;
1491 NextReg = TestReg;
1492 break;
1493 }
1494
1495 // Check if we found anything worth folding.
1496 if (!NextReg.isValid())
1497 return Reg;
1498 Reg = NextReg;
1499 }
1500
1501 return Reg;
1502}
1503
1504MachineInstr *AArch64InstructionSelector::emitTestBit(
1505 Register TestReg, uint64_t Bit, bool IsNegative, MachineBasicBlock *DstMBB,
1506 MachineIRBuilder &MIB) const {
1507 assert(TestReg.isValid());
1508 assert(ProduceNonFlagSettingCondBr &&
1509 "Cannot emit TB(N)Z with speculation tracking!");
1510 MachineRegisterInfo &MRI = *MIB.getMRI();
1511
1512 // Attempt to optimize the test bit by walking over instructions.
1513 TestReg = getTestBitReg(Reg: TestReg, Bit, Invert&: IsNegative, MRI);
1514 LLT Ty = MRI.getType(Reg: TestReg);
1515 unsigned Size = Ty.getSizeInBits();
1516 assert(!Ty.isVector() && "Expected a scalar!");
1517 assert(Bit < 64 && "Bit is too large!");
1518
1519 // When the test register is a 64-bit register, we have to narrow to make
1520 // TBNZW work.
1521 bool UseWReg = Bit < 32;
1522 unsigned NecessarySize = UseWReg ? 32 : 64;
1523 if (Size != NecessarySize)
1524 TestReg = moveScalarRegClass(
1525 Reg: TestReg, RC: UseWReg ? AArch64::GPR32RegClass : AArch64::GPR64RegClass,
1526 MIB);
1527
1528 static const unsigned OpcTable[2][2] = {{AArch64::TBZX, AArch64::TBNZX},
1529 {AArch64::TBZW, AArch64::TBNZW}};
1530 unsigned Opc = OpcTable[UseWReg][IsNegative];
1531 auto TestBitMI =
1532 MIB.buildInstr(Opcode: Opc).addReg(RegNo: TestReg).addImm(Val: Bit).addMBB(MBB: DstMBB);
1533 constrainSelectedInstRegOperands(I&: *TestBitMI, TII, TRI, RBI);
1534 return &*TestBitMI;
1535}
1536
1537bool AArch64InstructionSelector::tryOptAndIntoCompareBranch(
1538 MachineInstr &AndInst, bool Invert, MachineBasicBlock *DstMBB,
1539 MachineIRBuilder &MIB) const {
1540 assert(AndInst.getOpcode() == TargetOpcode::G_AND && "Expected G_AND only?");
1541 // Given something like this:
1542 //
1543 // %x = ...Something...
1544 // %one = G_CONSTANT i64 1
1545 // %zero = G_CONSTANT i64 0
1546 // %and = G_AND %x, %one
1547 // %cmp = G_ICMP intpred(ne), %and, %zero
1548 // %cmp_trunc = G_TRUNC %cmp
1549 // G_BRCOND %cmp_trunc, %bb.3
1550 //
1551 // We want to try and fold the AND into the G_BRCOND and produce either a
1552 // TBNZ (when we have intpred(ne)) or a TBZ (when we have intpred(eq)).
1553 //
1554 // In this case, we'd get
1555 //
1556 // TBNZ %x %bb.3
1557 //
1558
1559 // Check if the AND has a constant on its RHS which we can use as a mask.
1560 // If it's a power of 2, then it's the same as checking a specific bit.
1561 // (e.g, ANDing with 8 == ANDing with 000...100 == testing if bit 3 is set)
1562 auto MaybeBit = getIConstantVRegValWithLookThrough(
1563 VReg: AndInst.getOperand(i: 2).getReg(), MRI: *MIB.getMRI());
1564 if (!MaybeBit)
1565 return false;
1566
1567 int32_t Bit = MaybeBit->Value.exactLogBase2();
1568 if (Bit < 0)
1569 return false;
1570
1571 Register TestReg = AndInst.getOperand(i: 1).getReg();
1572
1573 // Emit a TB(N)Z.
1574 emitTestBit(TestReg, Bit, IsNegative: Invert, DstMBB, MIB);
1575 return true;
1576}
1577
1578MachineInstr *AArch64InstructionSelector::emitCBZ(Register CompareReg,
1579 bool IsNegative,
1580 MachineBasicBlock *DestMBB,
1581 MachineIRBuilder &MIB) const {
1582 assert(ProduceNonFlagSettingCondBr && "CBZ does not set flags!");
1583 MachineRegisterInfo &MRI = *MIB.getMRI();
1584 assert(RBI.getRegBank(CompareReg, MRI, TRI)->getID() ==
1585 AArch64::GPRRegBankID &&
1586 "Expected GPRs only?");
1587 auto Ty = MRI.getType(Reg: CompareReg);
1588 unsigned Width = Ty.getSizeInBits();
1589 assert(!Ty.isVector() && "Expected scalar only?");
1590 assert(Width <= 64 && "Expected width to be at most 64?");
1591 static const unsigned OpcTable[2][2] = {{AArch64::CBZW, AArch64::CBZX},
1592 {AArch64::CBNZW, AArch64::CBNZX}};
1593 unsigned Opc = OpcTable[IsNegative][Width == 64];
1594 auto BranchMI = MIB.buildInstr(Opc, DstOps: {}, SrcOps: {CompareReg}).addMBB(MBB: DestMBB);
1595 constrainSelectedInstRegOperands(I&: *BranchMI, TII, TRI, RBI);
1596 return &*BranchMI;
1597}
1598
1599bool AArch64InstructionSelector::selectCompareBranchFedByFCmp(
1600 MachineInstr &I, MachineInstr &FCmp, MachineIRBuilder &MIB) const {
1601 assert(FCmp.getOpcode() == TargetOpcode::G_FCMP);
1602 assert(I.getOpcode() == TargetOpcode::G_BRCOND);
1603 // Unfortunately, the mapping of LLVM FP CC's onto AArch64 CC's isn't
1604 // totally clean. Some of them require two branches to implement.
1605 auto Pred = (CmpInst::Predicate)FCmp.getOperand(i: 1).getPredicate();
1606 emitFPCompare(LHS: FCmp.getOperand(i: 2).getReg(), RHS: FCmp.getOperand(i: 3).getReg(), MIRBuilder&: MIB,
1607 Pred);
1608 AArch64CC::CondCode CC1, CC2;
1609 changeFCMPPredToAArch64CC(P: Pred, CondCode&: CC1, CondCode2&: CC2);
1610 MachineBasicBlock *DestMBB = I.getOperand(i: 1).getMBB();
1611 MIB.buildInstr(Opc: AArch64::Bcc, DstOps: {}, SrcOps: {}).addImm(Val: CC1).addMBB(MBB: DestMBB);
1612 if (CC2 != AArch64CC::AL)
1613 MIB.buildInstr(Opc: AArch64::Bcc, DstOps: {}, SrcOps: {}).addImm(Val: CC2).addMBB(MBB: DestMBB);
1614 I.eraseFromParent();
1615 return true;
1616}
1617
1618bool AArch64InstructionSelector::tryOptCompareBranchFedByICmp(
1619 MachineInstr &I, MachineInstr &ICmp, MachineIRBuilder &MIB) const {
1620 assert(ICmp.getOpcode() == TargetOpcode::G_ICMP);
1621 assert(I.getOpcode() == TargetOpcode::G_BRCOND);
1622 // Attempt to optimize the G_BRCOND + G_ICMP into a TB(N)Z/CB(N)Z.
1623 //
1624 // Speculation tracking/SLH assumes that optimized TB(N)Z/CB(N)Z
1625 // instructions will not be produced, as they are conditional branch
1626 // instructions that do not set flags.
1627 if (!ProduceNonFlagSettingCondBr)
1628 return false;
1629
1630 MachineRegisterInfo &MRI = *MIB.getMRI();
1631 MachineBasicBlock *DestMBB = I.getOperand(i: 1).getMBB();
1632 auto Pred =
1633 static_cast<CmpInst::Predicate>(ICmp.getOperand(i: 1).getPredicate());
1634 Register LHS = ICmp.getOperand(i: 2).getReg();
1635 Register RHS = ICmp.getOperand(i: 3).getReg();
1636
1637 // We're allowed to emit a TB(N)Z/CB(N)Z. Try to do that.
1638 auto VRegAndVal = getIConstantVRegValWithLookThrough(VReg: RHS, MRI);
1639 MachineInstr *AndInst = getOpcodeDef(Opcode: TargetOpcode::G_AND, Reg: LHS, MRI);
1640
1641 // When we can emit a TB(N)Z, prefer that.
1642 //
1643 // Handle non-commutative condition codes first.
1644 // Note that we don't want to do this when we have a G_AND because it can
1645 // become a tst. The tst will make the test bit in the TB(N)Z redundant.
1646 if (VRegAndVal && !AndInst) {
1647 int64_t C = VRegAndVal->Value.getSExtValue();
1648
1649 // When we have a greater-than comparison, we can just test if the msb is
1650 // zero.
1651 if (C == -1 && Pred == CmpInst::ICMP_SGT) {
1652 uint64_t Bit = MRI.getType(Reg: LHS).getSizeInBits() - 1;
1653 emitTestBit(TestReg: LHS, Bit, /*IsNegative = */ false, DstMBB: DestMBB, MIB);
1654 I.eraseFromParent();
1655 return true;
1656 }
1657
1658 // When we have a less than comparison, we can just test if the msb is not
1659 // zero.
1660 if (C == 0 && Pred == CmpInst::ICMP_SLT) {
1661 uint64_t Bit = MRI.getType(Reg: LHS).getSizeInBits() - 1;
1662 emitTestBit(TestReg: LHS, Bit, /*IsNegative = */ true, DstMBB: DestMBB, MIB);
1663 I.eraseFromParent();
1664 return true;
1665 }
1666
1667 // Inversely, if we have a signed greater-than-or-equal comparison to zero,
1668 // we can test if the msb is zero.
1669 if (C == 0 && Pred == CmpInst::ICMP_SGE) {
1670 uint64_t Bit = MRI.getType(Reg: LHS).getSizeInBits() - 1;
1671 emitTestBit(TestReg: LHS, Bit, /*IsNegative = */ false, DstMBB: DestMBB, MIB);
1672 I.eraseFromParent();
1673 return true;
1674 }
1675 }
1676
1677 // Attempt to handle commutative condition codes. Right now, that's only
1678 // eq/ne.
1679 if (ICmpInst::isEquality(P: Pred)) {
1680 if (!VRegAndVal) {
1681 std::swap(a&: RHS, b&: LHS);
1682 VRegAndVal = getIConstantVRegValWithLookThrough(VReg: RHS, MRI);
1683 AndInst = getOpcodeDef(Opcode: TargetOpcode::G_AND, Reg: LHS, MRI);
1684 }
1685
1686 if (VRegAndVal && VRegAndVal->Value == 0) {
1687 // If there's a G_AND feeding into this branch, try to fold it away by
1688 // emitting a TB(N)Z instead.
1689 //
1690 // Note: If we have LT, then it *is* possible to fold, but it wouldn't be
1691 // beneficial. When we have an AND and LT, we need a TST/ANDS, so folding
1692 // would be redundant.
1693 if (AndInst &&
1694 tryOptAndIntoCompareBranch(
1695 AndInst&: *AndInst, /*Invert = */ Pred == CmpInst::ICMP_NE, DstMBB: DestMBB, MIB)) {
1696 I.eraseFromParent();
1697 return true;
1698 }
1699
1700 // Otherwise, try to emit a CB(N)Z instead.
1701 auto LHSTy = MRI.getType(Reg: LHS);
1702 if (!LHSTy.isVector() && LHSTy.getSizeInBits() <= 64) {
1703 emitCBZ(CompareReg: LHS, /*IsNegative = */ Pred == CmpInst::ICMP_NE, DestMBB, MIB);
1704 I.eraseFromParent();
1705 return true;
1706 }
1707 }
1708 }
1709
1710 return false;
1711}
1712
1713bool AArch64InstructionSelector::selectCompareBranchFedByICmp(
1714 MachineInstr &I, MachineInstr &ICmp, MachineIRBuilder &MIB) const {
1715 assert(ICmp.getOpcode() == TargetOpcode::G_ICMP);
1716 assert(I.getOpcode() == TargetOpcode::G_BRCOND);
1717 if (tryOptCompareBranchFedByICmp(I, ICmp, MIB))
1718 return true;
1719
1720 // Couldn't optimize. Emit a compare + a Bcc.
1721 MachineBasicBlock *DestMBB = I.getOperand(i: 1).getMBB();
1722 auto &PredOp = ICmp.getOperand(i: 1);
1723 emitIntegerCompare(LHS&: ICmp.getOperand(i: 2), RHS&: ICmp.getOperand(i: 3), Predicate&: PredOp, MIRBuilder&: MIB);
1724 const AArch64CC::CondCode CC = changeICMPPredToAArch64CC(
1725 P: static_cast<CmpInst::Predicate>(PredOp.getPredicate()),
1726 RHS: ICmp.getOperand(i: 3).getReg(), MRI: MIB.getMRI());
1727 MIB.buildInstr(Opc: AArch64::Bcc, DstOps: {}, SrcOps: {}).addImm(Val: CC).addMBB(MBB: DestMBB);
1728 I.eraseFromParent();
1729 return true;
1730}
1731
1732bool AArch64InstructionSelector::selectCompareBranch(
1733 MachineInstr &I, MachineFunction &MF, MachineRegisterInfo &MRI) {
1734 Register CondReg = I.getOperand(i: 0).getReg();
1735 MachineInstr *CCMI = MRI.getVRegDef(Reg: CondReg);
1736 // Try to select the G_BRCOND using whatever is feeding the condition if
1737 // possible.
1738 unsigned CCMIOpc = CCMI->getOpcode();
1739 if (CCMIOpc == TargetOpcode::G_FCMP)
1740 return selectCompareBranchFedByFCmp(I, FCmp&: *CCMI, MIB);
1741 if (CCMIOpc == TargetOpcode::G_ICMP)
1742 return selectCompareBranchFedByICmp(I, ICmp&: *CCMI, MIB);
1743
1744 // Speculation tracking/SLH assumes that optimized TB(N)Z/CB(N)Z
1745 // instructions will not be produced, as they are conditional branch
1746 // instructions that do not set flags.
1747 if (ProduceNonFlagSettingCondBr) {
1748 emitTestBit(TestReg: CondReg, /*Bit = */ 0, /*IsNegative = */ true,
1749 DstMBB: I.getOperand(i: 1).getMBB(), MIB);
1750 I.eraseFromParent();
1751 return true;
1752 }
1753
1754 // Can't emit TB(N)Z/CB(N)Z. Emit a tst + bcc instead.
1755 auto TstMI =
1756 MIB.buildInstr(Opc: AArch64::ANDSWri, DstOps: {LLT::scalar(SizeInBits: 32)}, SrcOps: {CondReg}).addImm(Val: 1);
1757 constrainSelectedInstRegOperands(I&: *TstMI, TII, TRI, RBI);
1758 auto Bcc = MIB.buildInstr(Opcode: AArch64::Bcc)
1759 .addImm(Val: AArch64CC::NE)
1760 .addMBB(MBB: I.getOperand(i: 1).getMBB());
1761 I.eraseFromParent();
1762 constrainSelectedInstRegOperands(I&: *Bcc, TII, TRI, RBI);
1763 return true;
1764}
1765
1766/// Returns the element immediate value of a vector shift operand if found.
1767/// This needs to detect a splat-like operation, e.g. a G_BUILD_VECTOR.
1768static std::optional<int64_t> getVectorShiftImm(Register Reg,
1769 MachineRegisterInfo &MRI) {
1770 assert(MRI.getType(Reg).isVector() && "Expected a *vector* shift operand");
1771 MachineInstr *OpMI = MRI.getVRegDef(Reg);
1772 return getAArch64VectorSplatScalar(MI: *OpMI, MRI);
1773}
1774
1775/// Matches and returns the shift immediate value for a SHL instruction given
1776/// a shift operand.
1777static std::optional<int64_t> getVectorSHLImm(LLT SrcTy, Register Reg,
1778 MachineRegisterInfo &MRI) {
1779 std::optional<int64_t> ShiftImm = getVectorShiftImm(Reg, MRI);
1780 if (!ShiftImm)
1781 return std::nullopt;
1782 // Check the immediate is in range for a SHL.
1783 int64_t Imm = *ShiftImm;
1784 if (Imm < 0)
1785 return std::nullopt;
1786 switch (SrcTy.getElementType().getSizeInBits()) {
1787 default:
1788 LLVM_DEBUG(dbgs() << "Unhandled element type for vector shift");
1789 return std::nullopt;
1790 case 8:
1791 if (Imm > 7)
1792 return std::nullopt;
1793 break;
1794 case 16:
1795 if (Imm > 15)
1796 return std::nullopt;
1797 break;
1798 case 32:
1799 if (Imm > 31)
1800 return std::nullopt;
1801 break;
1802 case 64:
1803 if (Imm > 63)
1804 return std::nullopt;
1805 break;
1806 }
1807 return Imm;
1808}
1809
1810bool AArch64InstructionSelector::selectVectorSHL(MachineInstr &I,
1811 MachineRegisterInfo &MRI) {
1812 assert(I.getOpcode() == TargetOpcode::G_SHL);
1813 Register DstReg = I.getOperand(i: 0).getReg();
1814 const LLT Ty = MRI.getType(Reg: DstReg);
1815 Register Src1Reg = I.getOperand(i: 1).getReg();
1816 Register Src2Reg = I.getOperand(i: 2).getReg();
1817
1818 if (!Ty.isVector())
1819 return false;
1820
1821 // Check if we have a vector of constants on RHS that we can select as the
1822 // immediate form.
1823 std::optional<int64_t> ImmVal = getVectorSHLImm(SrcTy: Ty, Reg: Src2Reg, MRI);
1824
1825 unsigned Opc = 0;
1826 if (Ty == LLT::fixed_vector(NumElements: 2, ScalarSizeInBits: 64)) {
1827 Opc = ImmVal ? AArch64::SHLv2i64_shift : AArch64::USHLv2i64;
1828 } else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarSizeInBits: 32)) {
1829 Opc = ImmVal ? AArch64::SHLv4i32_shift : AArch64::USHLv4i32;
1830 } else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarSizeInBits: 32)) {
1831 Opc = ImmVal ? AArch64::SHLv2i32_shift : AArch64::USHLv2i32;
1832 } else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarSizeInBits: 16)) {
1833 Opc = ImmVal ? AArch64::SHLv4i16_shift : AArch64::USHLv4i16;
1834 } else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarSizeInBits: 16)) {
1835 Opc = ImmVal ? AArch64::SHLv8i16_shift : AArch64::USHLv8i16;
1836 } else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarSizeInBits: 8)) {
1837 Opc = ImmVal ? AArch64::SHLv16i8_shift : AArch64::USHLv16i8;
1838 } else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarSizeInBits: 8)) {
1839 Opc = ImmVal ? AArch64::SHLv8i8_shift : AArch64::USHLv8i8;
1840 } else {
1841 LLVM_DEBUG(dbgs() << "Unhandled G_SHL type");
1842 return false;
1843 }
1844
1845 auto Shl = MIB.buildInstr(Opc, DstOps: {DstReg}, SrcOps: {Src1Reg});
1846 if (ImmVal)
1847 Shl.addImm(Val: *ImmVal);
1848 else
1849 Shl.addUse(RegNo: Src2Reg);
1850 constrainSelectedInstRegOperands(I&: *Shl, TII, TRI, RBI);
1851 I.eraseFromParent();
1852 return true;
1853}
1854
1855bool AArch64InstructionSelector::selectVectorAshrLshr(
1856 MachineInstr &I, MachineRegisterInfo &MRI) {
1857 assert(I.getOpcode() == TargetOpcode::G_ASHR ||
1858 I.getOpcode() == TargetOpcode::G_LSHR);
1859 Register DstReg = I.getOperand(i: 0).getReg();
1860 const LLT Ty = MRI.getType(Reg: DstReg);
1861 Register Src1Reg = I.getOperand(i: 1).getReg();
1862 Register Src2Reg = I.getOperand(i: 2).getReg();
1863
1864 if (!Ty.isVector())
1865 return false;
1866
1867 bool IsASHR = I.getOpcode() == TargetOpcode::G_ASHR;
1868
1869 // We expect the immediate case to be lowered in the PostLegalCombiner to
1870 // AArch64ISD::VASHR or AArch64ISD::VLSHR equivalents.
1871
1872 // There is not a shift right register instruction, but the shift left
1873 // register instruction takes a signed value, where negative numbers specify a
1874 // right shift.
1875
1876 unsigned Opc = 0;
1877 unsigned NegOpc = 0;
1878 const TargetRegisterClass *RC =
1879 getRegClassForTypeOnBank(Ty, RB: RBI.getRegBank(ID: AArch64::FPRRegBankID));
1880 if (Ty == LLT::fixed_vector(NumElements: 2, ScalarSizeInBits: 64)) {
1881 Opc = IsASHR ? AArch64::SSHLv2i64 : AArch64::USHLv2i64;
1882 NegOpc = AArch64::NEGv2i64;
1883 } else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarSizeInBits: 32)) {
1884 Opc = IsASHR ? AArch64::SSHLv4i32 : AArch64::USHLv4i32;
1885 NegOpc = AArch64::NEGv4i32;
1886 } else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarSizeInBits: 32)) {
1887 Opc = IsASHR ? AArch64::SSHLv2i32 : AArch64::USHLv2i32;
1888 NegOpc = AArch64::NEGv2i32;
1889 } else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarSizeInBits: 16)) {
1890 Opc = IsASHR ? AArch64::SSHLv4i16 : AArch64::USHLv4i16;
1891 NegOpc = AArch64::NEGv4i16;
1892 } else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarSizeInBits: 16)) {
1893 Opc = IsASHR ? AArch64::SSHLv8i16 : AArch64::USHLv8i16;
1894 NegOpc = AArch64::NEGv8i16;
1895 } else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarSizeInBits: 8)) {
1896 Opc = IsASHR ? AArch64::SSHLv16i8 : AArch64::USHLv16i8;
1897 NegOpc = AArch64::NEGv16i8;
1898 } else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarSizeInBits: 8)) {
1899 Opc = IsASHR ? AArch64::SSHLv8i8 : AArch64::USHLv8i8;
1900 NegOpc = AArch64::NEGv8i8;
1901 } else {
1902 LLVM_DEBUG(dbgs() << "Unhandled G_ASHR type");
1903 return false;
1904 }
1905
1906 auto Neg = MIB.buildInstr(Opc: NegOpc, DstOps: {RC}, SrcOps: {Src2Reg});
1907 constrainSelectedInstRegOperands(I&: *Neg, TII, TRI, RBI);
1908 auto SShl = MIB.buildInstr(Opc, DstOps: {DstReg}, SrcOps: {Src1Reg, Neg});
1909 constrainSelectedInstRegOperands(I&: *SShl, TII, TRI, RBI);
1910 I.eraseFromParent();
1911 return true;
1912}
1913
1914bool AArch64InstructionSelector::selectVaStartAAPCS(
1915 MachineInstr &I, MachineFunction &MF, MachineRegisterInfo &MRI) const {
1916
1917 if (STI.isCallingConvWin64(CC: MF.getFunction().getCallingConv(),
1918 IsVarArg: MF.getFunction().isVarArg()))
1919 return false;
1920
1921 // The layout of the va_list struct is specified in the AArch64 Procedure Call
1922 // Standard, section 10.1.5.
1923
1924 const AArch64FunctionInfo *FuncInfo = MF.getInfo<AArch64FunctionInfo>();
1925 const unsigned PtrSize = STI.isTargetILP32() ? 4 : 8;
1926 const auto *PtrRegClass =
1927 STI.isTargetILP32() ? &AArch64::GPR32RegClass : &AArch64::GPR64RegClass;
1928
1929 const MCInstrDesc &MCIDAddAddr =
1930 TII.get(Opcode: STI.isTargetILP32() ? AArch64::ADDWri : AArch64::ADDXri);
1931 const MCInstrDesc &MCIDStoreAddr =
1932 TII.get(Opcode: STI.isTargetILP32() ? AArch64::STRWui : AArch64::STRXui);
1933
1934 /*
1935 * typedef struct va_list {
1936 * void * stack; // next stack param
1937 * void * gr_top; // end of GP arg reg save area
1938 * void * vr_top; // end of FP/SIMD arg reg save area
1939 * int gr_offs; // offset from gr_top to next GP register arg
1940 * int vr_offs; // offset from vr_top to next FP/SIMD register arg
1941 * } va_list;
1942 */
1943 const auto VAList = I.getOperand(i: 0).getReg();
1944
1945 // Our current offset in bytes from the va_list struct (VAList).
1946 unsigned OffsetBytes = 0;
1947
1948 // Helper function to store (FrameIndex + Imm) to VAList at offset OffsetBytes
1949 // and increment OffsetBytes by PtrSize.
1950 const auto PushAddress = [&](const int FrameIndex, const int64_t Imm) {
1951 const Register Top = MRI.createVirtualRegister(RegClass: PtrRegClass);
1952 auto MIB = BuildMI(BB&: *I.getParent(), I, MIMD: I.getDebugLoc(), MCID: MCIDAddAddr)
1953 .addDef(RegNo: Top)
1954 .addFrameIndex(Idx: FrameIndex)
1955 .addImm(Val: Imm)
1956 .addImm(Val: 0);
1957 constrainSelectedInstRegOperands(I&: *MIB, TII, TRI, RBI);
1958
1959 const auto *MMO = *I.memoperands_begin();
1960 MIB = BuildMI(BB&: *I.getParent(), I, MIMD: I.getDebugLoc(), MCID: MCIDStoreAddr)
1961 .addUse(RegNo: Top)
1962 .addUse(RegNo: VAList)
1963 .addImm(Val: OffsetBytes / PtrSize)
1964 .addMemOperand(MMO: MF.getMachineMemOperand(
1965 PtrInfo: MMO->getPointerInfo().getWithOffset(O: OffsetBytes),
1966 F: MachineMemOperand::MOStore, Size: PtrSize, BaseAlignment: MMO->getBaseAlign()));
1967 constrainSelectedInstRegOperands(I&: *MIB, TII, TRI, RBI);
1968
1969 OffsetBytes += PtrSize;
1970 };
1971
1972 // void* stack at offset 0
1973 PushAddress(FuncInfo->getVarArgsStackIndex(), 0);
1974
1975 // void* gr_top at offset 8 (4 on ILP32)
1976 const unsigned GPRSize = FuncInfo->getVarArgsGPRSize();
1977 PushAddress(FuncInfo->getVarArgsGPRIndex(), GPRSize);
1978
1979 // void* vr_top at offset 16 (8 on ILP32)
1980 const unsigned FPRSize = FuncInfo->getVarArgsFPRSize();
1981 PushAddress(FuncInfo->getVarArgsFPRIndex(), FPRSize);
1982
1983 // Helper function to store a 4-byte integer constant to VAList at offset
1984 // OffsetBytes, and increment OffsetBytes by 4.
1985 const auto PushIntConstant = [&](const int32_t Value) {
1986 constexpr int IntSize = 4;
1987 const Register Temp = MRI.createVirtualRegister(RegClass: &AArch64::GPR32RegClass);
1988 auto MIB =
1989 BuildMI(BB&: *I.getParent(), I, MIMD: I.getDebugLoc(), MCID: TII.get(Opcode: AArch64::MOVi32imm))
1990 .addDef(RegNo: Temp)
1991 .addImm(Val: Value);
1992 constrainSelectedInstRegOperands(I&: *MIB, TII, TRI, RBI);
1993
1994 const auto *MMO = *I.memoperands_begin();
1995 MIB = BuildMI(BB&: *I.getParent(), I, MIMD: I.getDebugLoc(), MCID: TII.get(Opcode: AArch64::STRWui))
1996 .addUse(RegNo: Temp)
1997 .addUse(RegNo: VAList)
1998 .addImm(Val: OffsetBytes / IntSize)
1999 .addMemOperand(MMO: MF.getMachineMemOperand(
2000 PtrInfo: MMO->getPointerInfo().getWithOffset(O: OffsetBytes),
2001 F: MachineMemOperand::MOStore, Size: IntSize, BaseAlignment: MMO->getBaseAlign()));
2002 constrainSelectedInstRegOperands(I&: *MIB, TII, TRI, RBI);
2003 OffsetBytes += IntSize;
2004 };
2005
2006 // int gr_offs at offset 24 (12 on ILP32)
2007 PushIntConstant(-static_cast<int32_t>(GPRSize));
2008
2009 // int vr_offs at offset 28 (16 on ILP32)
2010 PushIntConstant(-static_cast<int32_t>(FPRSize));
2011
2012 assert(OffsetBytes == (STI.isTargetILP32() ? 20 : 32) && "Unexpected offset");
2013
2014 I.eraseFromParent();
2015 return true;
2016}
2017
2018bool AArch64InstructionSelector::selectVaStartDarwin(
2019 MachineInstr &I, MachineFunction &MF, MachineRegisterInfo &MRI) const {
2020 AArch64FunctionInfo *FuncInfo = MF.getInfo<AArch64FunctionInfo>();
2021 Register ListReg = I.getOperand(i: 0).getReg();
2022
2023 Register ArgsAddrReg = MRI.createVirtualRegister(RegClass: &AArch64::GPR64RegClass);
2024
2025 int FrameIdx = FuncInfo->getVarArgsStackIndex();
2026 if (MF.getSubtarget<AArch64Subtarget>().isCallingConvWin64(
2027 CC: MF.getFunction().getCallingConv(), IsVarArg: MF.getFunction().isVarArg())) {
2028 FrameIdx = FuncInfo->getVarArgsGPRSize() > 0
2029 ? FuncInfo->getVarArgsGPRIndex()
2030 : FuncInfo->getVarArgsStackIndex();
2031 }
2032
2033 auto MIB =
2034 BuildMI(BB&: *I.getParent(), I, MIMD: I.getDebugLoc(), MCID: TII.get(Opcode: AArch64::ADDXri))
2035 .addDef(RegNo: ArgsAddrReg)
2036 .addFrameIndex(Idx: FrameIdx)
2037 .addImm(Val: 0)
2038 .addImm(Val: 0);
2039
2040 constrainSelectedInstRegOperands(I&: *MIB, TII, TRI, RBI);
2041
2042 MIB = BuildMI(BB&: *I.getParent(), I, MIMD: I.getDebugLoc(), MCID: TII.get(Opcode: AArch64::STRXui))
2043 .addUse(RegNo: ArgsAddrReg)
2044 .addUse(RegNo: ListReg)
2045 .addImm(Val: 0)
2046 .addMemOperand(MMO: *I.memoperands_begin());
2047
2048 constrainSelectedInstRegOperands(I&: *MIB, TII, TRI, RBI);
2049 I.eraseFromParent();
2050 return true;
2051}
2052
2053void AArch64InstructionSelector::materializeLargeCMVal(
2054 MachineInstr &I, const Value *V, unsigned OpFlags) {
2055 MachineBasicBlock &MBB = *I.getParent();
2056 MachineFunction &MF = *MBB.getParent();
2057 MachineRegisterInfo &MRI = MF.getRegInfo();
2058
2059 auto MovZ = MIB.buildInstr(Opc: AArch64::MOVZXi, DstOps: {&AArch64::GPR64RegClass}, SrcOps: {});
2060 MovZ->addOperand(MF, Op: I.getOperand(i: 1));
2061 MovZ->getOperand(i: 1).setTargetFlags(OpFlags | AArch64II::MO_G0 |
2062 AArch64II::MO_NC);
2063 MovZ->addOperand(MF, Op: MachineOperand::CreateImm(Val: 0));
2064 constrainSelectedInstRegOperands(I&: *MovZ, TII, TRI, RBI);
2065
2066 auto BuildMovK = [&](Register SrcReg, unsigned char Flags, unsigned Offset,
2067 Register ForceDstReg) {
2068 Register DstReg = ForceDstReg
2069 ? ForceDstReg
2070 : MRI.createVirtualRegister(RegClass: &AArch64::GPR64RegClass);
2071 auto MovI = MIB.buildInstr(Opcode: AArch64::MOVKXi).addDef(RegNo: DstReg).addUse(RegNo: SrcReg);
2072 if (auto *GV = dyn_cast<GlobalValue>(Val: V)) {
2073 MovI->addOperand(MF, Op: MachineOperand::CreateGA(
2074 GV, Offset: MovZ->getOperand(i: 1).getOffset(), TargetFlags: Flags));
2075 } else {
2076 MovI->addOperand(
2077 MF, Op: MachineOperand::CreateBA(BA: cast<BlockAddress>(Val: V),
2078 Offset: MovZ->getOperand(i: 1).getOffset(), TargetFlags: Flags));
2079 }
2080 MovI->addOperand(MF, Op: MachineOperand::CreateImm(Val: Offset));
2081 constrainSelectedInstRegOperands(I&: *MovI, TII, TRI, RBI);
2082 return DstReg;
2083 };
2084 Register DstReg = BuildMovK(MovZ.getReg(Idx: 0),
2085 AArch64II::MO_G1 | AArch64II::MO_NC, 16, 0);
2086 DstReg = BuildMovK(DstReg, AArch64II::MO_G2 | AArch64II::MO_NC, 32, 0);
2087 BuildMovK(DstReg, AArch64II::MO_G3, 48, I.getOperand(i: 0).getReg());
2088}
2089
2090bool AArch64InstructionSelector::preISelLower(MachineInstr &I) {
2091 MachineBasicBlock &MBB = *I.getParent();
2092 MachineFunction &MF = *MBB.getParent();
2093 MachineRegisterInfo &MRI = MF.getRegInfo();
2094
2095 switch (I.getOpcode()) {
2096 case TargetOpcode::G_CONSTANT: {
2097 Register DefReg = I.getOperand(i: 0).getReg();
2098 const LLT DefTy = MRI.getType(Reg: DefReg);
2099 if (!DefTy.isPointer()) {
2100 if (DefTy.getSizeInBits() >= 32 ||
2101 RBI.getRegBank(Reg: DefReg, MRI, TRI)->getID() != AArch64::GPRRegBankID)
2102 return false;
2103 // Widen narrow GPR constants to s32 so imported patterns can match.
2104 APInt Val = I.getOperand(i: 1).getCImm()->getValue().zext(width: 32);
2105 I.getOperand(i: 1).setCImm(
2106 ConstantInt::get(Context&: MF.getFunction().getContext(), V: Val));
2107
2108 Register WideReg = MRI.createGenericVirtualRegister(Ty: LLT::scalar(SizeInBits: 32));
2109 MRI.setRegBank(Reg: WideReg, RegBank: RBI.getRegBank(ID: AArch64::GPRRegBankID));
2110 I.getOperand(i: 0).setReg(WideReg);
2111
2112 MIB.setInsertPt(MBB, II: std::next(x: I.getIterator()));
2113 auto Copy = MIB.buildCopy(Res: DefReg, Op: WideReg);
2114 selectCopy(I&: *Copy, TII, MRI, TRI, RBI);
2115 MIB.setInstr(I);
2116 return true;
2117 }
2118 const unsigned PtrSize = DefTy.getSizeInBits();
2119 if (PtrSize != 32 && PtrSize != 64)
2120 return false;
2121 // Convert pointer typed constants to integers so TableGen can select.
2122 MRI.setType(VReg: DefReg, Ty: LLT::integer(SizeInBits: PtrSize));
2123 return true;
2124 }
2125 case TargetOpcode::G_STORE: {
2126 bool Changed = contractCrossBankCopyIntoStore(I, MRI);
2127 MachineOperand &SrcOp = I.getOperand(i: 0);
2128 if (MRI.getType(Reg: SrcOp.getReg()).isPointer()) {
2129 // Allow matching with imported patterns for stores of pointers. Unlike
2130 // G_LOAD/G_PTR_ADD, we may not have selected all users. So, emit a copy
2131 // and constrain.
2132 auto Copy = MIB.buildCopy(Res: LLT::scalar(SizeInBits: 64), Op: SrcOp);
2133 Register NewSrc = Copy.getReg(Idx: 0);
2134 SrcOp.setReg(NewSrc);
2135 RBI.constrainGenericRegister(Reg: NewSrc, RC: AArch64::GPR64RegClass, MRI);
2136 Changed = true;
2137 }
2138 return Changed;
2139 }
2140 case TargetOpcode::G_PTR_ADD: {
2141 // If Checked Pointer Arithmetic (FEAT_CPA) is present, preserve the pointer
2142 // arithmetic semantics instead of falling back to regular arithmetic.
2143 const auto &TL = STI.getTargetLowering();
2144 if (TL->shouldPreservePtrArith(F: MF.getFunction(), PtrVT: EVT()))
2145 return false;
2146 return convertPtrAddToAdd(I, MRI);
2147 }
2148 case TargetOpcode::G_LOAD: {
2149 // For scalar loads of pointers, we try to convert the dest type from p0
2150 // to s64 so that our imported patterns can match. Like with the G_PTR_ADD
2151 // conversion, this should be ok because all users should have been
2152 // selected already, so the type doesn't matter for them.
2153 Register DstReg = I.getOperand(i: 0).getReg();
2154 const LLT DstTy = MRI.getType(Reg: DstReg);
2155 if (!DstTy.isPointer())
2156 return false;
2157 MRI.setType(VReg: DstReg, Ty: LLT::scalar(SizeInBits: 64));
2158 return true;
2159 }
2160 case TargetOpcode::G_VECREDUCE_ADD:
2161 case TargetOpcode::G_VECREDUCE_SMAX:
2162 case TargetOpcode::G_VECREDUCE_SMIN:
2163 case TargetOpcode::G_VECREDUCE_UMAX:
2164 case TargetOpcode::G_VECREDUCE_UMIN: {
2165 // Imported patterns require an FPR result. For a GPR, use a temporary FPR
2166 // and insert a cross-bank copy.
2167 Register DstReg = I.getOperand(i: 0).getReg();
2168 const RegisterBank &DstRB = *RBI.getRegBank(Reg: DstReg, MRI, TRI);
2169 if (DstRB.getID() != AArch64::GPRRegBankID)
2170 return false;
2171
2172 LLT DstTy = MRI.getType(Reg: DstReg);
2173 const TargetRegisterClass *DstRC =
2174 getRegClassForTypeOnBank(Ty: DstTy, RB: DstRB, /*GetAllRegSet=*/true);
2175 if (!DstRC || !RBI.constrainGenericRegister(Reg: DstReg, RC: *DstRC, MRI))
2176 return false;
2177
2178 Register FPRDst = MRI.createGenericVirtualRegister(Ty: DstTy);
2179 MRI.setRegBank(Reg: FPRDst, RegBank: RBI.getRegBank(ID: AArch64::FPRRegBankID));
2180 I.getOperand(i: 0).setReg(FPRDst);
2181
2182 BuildMI(BB&: MBB, I: std::next(x: I.getIterator()), MIMD: MIMetadata(I),
2183 MCID: TII.get(Opcode: TargetOpcode::COPY), DestReg: DstReg)
2184 .addReg(RegNo: FPRDst);
2185 return true;
2186 }
2187 case AArch64::G_DUP: {
2188 // Convert the type from p0 to s64 to help selection.
2189 LLT DstTy = MRI.getType(Reg: I.getOperand(i: 0).getReg());
2190 if (!DstTy.isPointerVector())
2191 return false;
2192 auto NewSrc = MIB.buildCopy(Res: LLT::scalar(SizeInBits: 64), Op: I.getOperand(i: 1).getReg());
2193 MRI.setType(VReg: I.getOperand(i: 0).getReg(),
2194 Ty: DstTy.changeElementType(NewEltTy: LLT::scalar(SizeInBits: 64)));
2195 MRI.setRegClass(Reg: NewSrc.getReg(Idx: 0), RC: &AArch64::GPR64RegClass);
2196 I.getOperand(i: 1).setReg(NewSrc.getReg(Idx: 0));
2197 return true;
2198 }
2199 case AArch64::G_INSERT_VECTOR_ELT: {
2200 LLT DstTy = MRI.getType(Reg: I.getOperand(i: 0).getReg());
2201 LLT SrcVecTy = MRI.getType(Reg: I.getOperand(i: 1).getReg());
2202 if (SrcVecTy.isPointerVector()) {
2203 // Convert the type from p0 to s64 to help selection.
2204 auto NewSrc = MIB.buildCopy(Res: LLT::scalar(SizeInBits: 64), Op: I.getOperand(i: 2).getReg());
2205 MRI.setType(VReg: I.getOperand(i: 1).getReg(),
2206 Ty: DstTy.changeElementType(NewEltTy: LLT::scalar(SizeInBits: 64)));
2207 MRI.setType(VReg: I.getOperand(i: 0).getReg(),
2208 Ty: DstTy.changeElementType(NewEltTy: LLT::scalar(SizeInBits: 64)));
2209 MRI.setRegClass(Reg: NewSrc.getReg(Idx: 0), RC: &AArch64::GPR64RegClass);
2210 I.getOperand(i: 2).setReg(NewSrc.getReg(Idx: 0));
2211 return true;
2212 }
2213
2214 Register EltReg = I.getOperand(i: 2).getReg();
2215 LLT EltTy = MRI.getType(Reg: EltReg);
2216 if (EltTy.isScalar() &&
2217 (EltTy.getSizeInBits() == 8 || EltTy.getSizeInBits() == 16) &&
2218 RBI.getRegBank(Reg: EltReg, MRI, TRI)->getID() == AArch64::GPRRegBankID) {
2219 // Convert the type from s8/s16 to s32 to help selection.
2220 auto NewElt = MIB.buildCopy(Res: LLT::scalar(SizeInBits: 32), Op: EltReg);
2221 MRI.setRegClass(Reg: NewElt.getReg(Idx: 0), RC: &AArch64::GPR32RegClass);
2222 I.getOperand(i: 2).setReg(NewElt.getReg(Idx: 0));
2223 return true;
2224 }
2225 return false;
2226 }
2227 case TargetOpcode::G_UITOFP:
2228 case TargetOpcode::G_SITOFP: {
2229 // If both source and destination regbanks are FPR, then convert the opcode
2230 // to G_SITOF so that the importer can select it to an fpr variant.
2231 // Otherwise, it ends up matching an fpr/gpr variant and adding a cross-bank
2232 // copy.
2233 Register SrcReg = I.getOperand(i: 1).getReg();
2234 LLT SrcTy = MRI.getType(Reg: SrcReg);
2235 LLT DstTy = MRI.getType(Reg: I.getOperand(i: 0).getReg());
2236 if (SrcTy.isVector() || SrcTy.getSizeInBits() != DstTy.getSizeInBits())
2237 return false;
2238
2239 if (RBI.getRegBank(Reg: SrcReg, MRI, TRI)->getID() == AArch64::FPRRegBankID) {
2240 // Need to add a copy to change the type so that the existing patterns can
2241 // match when there is an integer on an FPR bank.
2242 if (SrcTy.getScalarType().isInteger()) {
2243 auto Copy = MIB.buildCopy(Res: DstTy, Op: SrcReg);
2244 I.getOperand(i: 1).setReg(Copy.getReg(Idx: 0));
2245 MRI.setRegClass(Reg: Copy.getReg(Idx: 0),
2246 RC: getRegClassForTypeOnBank(
2247 Ty: SrcTy, RB: RBI.getRegBank(ID: AArch64::FPRRegBankID)));
2248 }
2249 if (I.getOpcode() == TargetOpcode::G_SITOFP)
2250 I.setDesc(TII.get(Opcode: AArch64::G_SITOF));
2251 else
2252 I.setDesc(TII.get(Opcode: AArch64::G_UITOF));
2253 return true;
2254 }
2255 return false;
2256 }
2257 default:
2258 return false;
2259 }
2260}
2261
2262/// This lowering tries to look for G_PTR_ADD instructions and then converts
2263/// them to a standard G_ADD with a COPY on the source.
2264///
2265/// The motivation behind this is to expose the add semantics to the imported
2266/// tablegen patterns. We shouldn't need to check for uses being loads/stores,
2267/// because the selector works bottom up, uses before defs. By the time we
2268/// end up trying to select a G_PTR_ADD, we should have already attempted to
2269/// fold this into addressing modes and were therefore unsuccessful.
2270bool AArch64InstructionSelector::convertPtrAddToAdd(
2271 MachineInstr &I, MachineRegisterInfo &MRI) {
2272 assert(I.getOpcode() == TargetOpcode::G_PTR_ADD && "Expected G_PTR_ADD");
2273 Register DstReg = I.getOperand(i: 0).getReg();
2274 Register AddOp1Reg = I.getOperand(i: 1).getReg();
2275 const LLT PtrTy = MRI.getType(Reg: DstReg);
2276 if (PtrTy.getAddressSpace() != 0)
2277 return false;
2278
2279 const LLT CastPtrTy = PtrTy.isVector()
2280 ? LLT::fixed_vector(NumElements: 2, ScalarTy: LLT::integer(SizeInBits: 64))
2281 : LLT::integer(SizeInBits: 64);
2282 auto PtrToInt = MIB.buildPtrToInt(Dst: CastPtrTy, Src: AddOp1Reg);
2283 // Set regbanks on the registers.
2284 if (PtrTy.isVector())
2285 MRI.setRegBank(Reg: PtrToInt.getReg(Idx: 0), RegBank: RBI.getRegBank(ID: AArch64::FPRRegBankID));
2286 else
2287 MRI.setRegBank(Reg: PtrToInt.getReg(Idx: 0), RegBank: RBI.getRegBank(ID: AArch64::GPRRegBankID));
2288
2289 // Now turn the %dst(p0) = G_PTR_ADD %base, off into:
2290 // %dst(intty) = G_ADD %intbase, off
2291 I.setDesc(TII.get(Opcode: TargetOpcode::G_ADD));
2292 MRI.setType(VReg: DstReg, Ty: CastPtrTy);
2293 I.getOperand(i: 1).setReg(PtrToInt.getReg(Idx: 0));
2294 if (!select(I&: *PtrToInt)) {
2295 LLVM_DEBUG(dbgs() << "Failed to select G_PTRTOINT in convertPtrAddToAdd");
2296 return false;
2297 }
2298
2299 // Also take the opportunity here to try to do some optimization.
2300 // Try to convert this into a G_SUB if the offset is a 0-x negate idiom.
2301 Register NegatedReg;
2302 if (!mi_match(R: I.getOperand(i: 2).getReg(), MRI, P: m_Neg(Src: m_Reg(R&: NegatedReg))))
2303 return true;
2304 I.getOperand(i: 2).setReg(NegatedReg);
2305 I.setDesc(TII.get(Opcode: TargetOpcode::G_SUB));
2306 return true;
2307}
2308
2309bool AArch64InstructionSelector::earlySelectSHL(MachineInstr &I,
2310 MachineRegisterInfo &MRI) {
2311 // We try to match the immediate variant of LSL, which is actually an alias
2312 // for a special case of UBFM. Otherwise, we fall back to the imported
2313 // selector which will match the register variant.
2314 assert(I.getOpcode() == TargetOpcode::G_SHL && "unexpected op");
2315 const auto &MO = I.getOperand(i: 2);
2316 auto VRegAndVal = getIConstantVRegVal(VReg: MO.getReg(), MRI);
2317 if (!VRegAndVal)
2318 return false;
2319
2320 const LLT DstTy = MRI.getType(Reg: I.getOperand(i: 0).getReg());
2321 if (DstTy.isVector())
2322 return false;
2323 bool Is64Bit = DstTy.getSizeInBits() == 64;
2324 auto Imm1Fn = Is64Bit ? selectShiftA_64(Root: MO) : selectShiftA_32(Root: MO);
2325 auto Imm2Fn = Is64Bit ? selectShiftB_64(Root: MO) : selectShiftB_32(Root: MO);
2326
2327 if (!Imm1Fn || !Imm2Fn)
2328 return false;
2329
2330 auto NewI =
2331 MIB.buildInstr(Opc: Is64Bit ? AArch64::UBFMXri : AArch64::UBFMWri,
2332 DstOps: {I.getOperand(i: 0).getReg()}, SrcOps: {I.getOperand(i: 1).getReg()});
2333
2334 for (auto &RenderFn : *Imm1Fn)
2335 RenderFn(NewI);
2336 for (auto &RenderFn : *Imm2Fn)
2337 RenderFn(NewI);
2338
2339 I.eraseFromParent();
2340 constrainSelectedInstRegOperands(I&: *NewI, TII, TRI, RBI);
2341 return true;
2342}
2343
2344bool AArch64InstructionSelector::contractCrossBankCopyIntoStore(
2345 MachineInstr &I, MachineRegisterInfo &MRI) {
2346 assert(I.getOpcode() == TargetOpcode::G_STORE && "Expected G_STORE");
2347 // If we're storing a scalar, it doesn't matter what register bank that
2348 // scalar is on. All that matters is the size.
2349 //
2350 // So, if we see something like this (with a 32-bit scalar as an example):
2351 //
2352 // %x:gpr(s32) = ... something ...
2353 // %y:fpr(s32) = COPY %x:gpr(s32)
2354 // G_STORE %y:fpr(s32)
2355 //
2356 // We can fix this up into something like this:
2357 //
2358 // G_STORE %x:gpr(s32)
2359 //
2360 // And then continue the selection process normally.
2361 Register DefDstReg = getSrcRegIgnoringCopies(Reg: I.getOperand(i: 0).getReg(), MRI);
2362 if (!DefDstReg.isValid())
2363 return false;
2364 LLT DefDstTy = MRI.getType(Reg: DefDstReg);
2365 Register StoreSrcReg = I.getOperand(i: 0).getReg();
2366 LLT StoreSrcTy = MRI.getType(Reg: StoreSrcReg);
2367
2368 // If we get something strange like a physical register, then we shouldn't
2369 // go any further.
2370 if (!DefDstTy.isValid())
2371 return false;
2372
2373 // Are the source and dst types the same size?
2374 if (DefDstTy.getSizeInBits() != StoreSrcTy.getSizeInBits())
2375 return false;
2376
2377 if (RBI.getRegBank(Reg: StoreSrcReg, MRI, TRI) ==
2378 RBI.getRegBank(Reg: DefDstReg, MRI, TRI))
2379 return false;
2380
2381 // We have a cross-bank copy, which is entering a store. Let's fold it.
2382 I.getOperand(i: 0).setReg(DefDstReg);
2383 return true;
2384}
2385
2386bool AArch64InstructionSelector::earlySelect(MachineInstr &I) {
2387 assert(I.getParent() && "Instruction should be in a basic block!");
2388 assert(I.getParent()->getParent() && "Instruction should be in a function!");
2389
2390 MachineBasicBlock &MBB = *I.getParent();
2391 MachineFunction &MF = *MBB.getParent();
2392 MachineRegisterInfo &MRI = MF.getRegInfo();
2393
2394 switch (I.getOpcode()) {
2395 case AArch64::G_DUP: {
2396 // Before selecting a DUP instruction, check if it is better selected as a
2397 // MOV or load from a constant pool.
2398 Register Src = I.getOperand(i: 1).getReg();
2399 auto ValAndVReg = getAnyConstantVRegValWithLookThrough(
2400 VReg: Src, MRI, /*LookThroughInstrs=*/true, /*LookThroughAnyExt=*/true);
2401 if (!ValAndVReg)
2402 return false;
2403 LLVMContext &Ctx = MF.getFunction().getContext();
2404 Register Dst = I.getOperand(i: 0).getReg();
2405 auto *CV = ConstantDataVector::getSplat(
2406 NumElts: MRI.getType(Reg: Dst).getNumElements(),
2407 Elt: ConstantInt::get(
2408 Ty: Type::getIntNTy(C&: Ctx, N: MRI.getType(Reg: Dst).getScalarSizeInBits()),
2409 V: ValAndVReg->Value.trunc(width: MRI.getType(Reg: Dst).getScalarSizeInBits())));
2410 if (!emitConstantVector(Dst, CV, MIRBuilder&: MIB, MRI))
2411 return false;
2412 I.eraseFromParent();
2413 return true;
2414 }
2415 case TargetOpcode::G_SEXT:
2416 // Check for i64 sext(i32 vector_extract) prior to tablegen to select SMOV
2417 // over a normal extend.
2418 if (selectUSMovFromExtend(I, MRI))
2419 return true;
2420 return false;
2421 case TargetOpcode::G_BR:
2422 return false;
2423 case TargetOpcode::G_SHL:
2424 return earlySelectSHL(I, MRI);
2425 case TargetOpcode::G_CONSTANT: {
2426 bool IsZero = false;
2427 if (I.getOperand(i: 1).isCImm())
2428 IsZero = I.getOperand(i: 1).getCImm()->isZero();
2429 else if (I.getOperand(i: 1).isImm())
2430 IsZero = I.getOperand(i: 1).getImm() == 0;
2431
2432 if (!IsZero)
2433 return false;
2434
2435 Register DefReg = I.getOperand(i: 0).getReg();
2436 LLT Ty = MRI.getType(Reg: DefReg);
2437 if (Ty.getSizeInBits() == 64) {
2438 I.getOperand(i: 1).ChangeToRegister(Reg: AArch64::XZR, isDef: false);
2439 RBI.constrainGenericRegister(Reg: DefReg, RC: AArch64::GPR64RegClass, MRI);
2440 } else if (Ty.getSizeInBits() <= 32) {
2441 I.getOperand(i: 1).ChangeToRegister(Reg: AArch64::WZR, isDef: false);
2442 RBI.constrainGenericRegister(Reg: DefReg, RC: AArch64::GPR32RegClass, MRI);
2443 } else
2444 return false;
2445
2446 I.setDesc(TII.get(Opcode: TargetOpcode::COPY));
2447 return true;
2448 }
2449
2450 case TargetOpcode::G_ADD: {
2451 // Check if this is being fed by a G_ICMP on either side.
2452 //
2453 // (cmp pred, x, y) + z
2454 //
2455 // In the above case, when the cmp is true, we increment z by 1. So, we can
2456 // fold the add into the cset for the cmp by using cinc.
2457 //
2458 // FIXME: This would probably be a lot nicer in PostLegalizerLowering.
2459 Register AddDst = I.getOperand(i: 0).getReg();
2460 Register AddLHS = I.getOperand(i: 1).getReg();
2461 Register AddRHS = I.getOperand(i: 2).getReg();
2462 // Only handle scalars.
2463 LLT Ty = MRI.getType(Reg: AddLHS);
2464 if (Ty.isVector())
2465 return false;
2466 // Since G_ICMP is modeled as ADDS/SUBS/ANDS, we can handle 32 bits or 64
2467 // bits.
2468 unsigned Size = Ty.getSizeInBits();
2469 if (Size != 32 && Size != 64)
2470 return false;
2471 auto MatchCmp = [&](Register Reg) -> MachineInstr * {
2472 if (!MRI.hasOneNonDBGUse(RegNo: Reg))
2473 return nullptr;
2474 // If the LHS of the add is 32 bits, then we want to fold a 32-bit
2475 // compare.
2476 if (Size == 32)
2477 return getOpcodeDef(Opcode: TargetOpcode::G_ICMP, Reg, MRI);
2478 // We model scalar compares using 32-bit destinations right now.
2479 // If it's a 64-bit compare, it'll have 64-bit sources.
2480 Register ZExt;
2481 if (!mi_match(R: Reg, MRI,
2482 P: m_OneNonDBGUse(SP: m_GZExt(Src: m_OneNonDBGUse(SP: m_Reg(R&: ZExt))))))
2483 return nullptr;
2484 auto *Cmp = getOpcodeDef(Opcode: TargetOpcode::G_ICMP, Reg: ZExt, MRI);
2485 if (!Cmp ||
2486 MRI.getType(Reg: Cmp->getOperand(i: 2).getReg()).getSizeInBits() != 64)
2487 return nullptr;
2488 return Cmp;
2489 };
2490 // Try to match
2491 // z + (cmp pred, x, y)
2492 MachineInstr *Cmp = MatchCmp(AddRHS);
2493 if (!Cmp) {
2494 // (cmp pred, x, y) + z
2495 std::swap(a&: AddLHS, b&: AddRHS);
2496 Cmp = MatchCmp(AddRHS);
2497 if (!Cmp)
2498 return false;
2499 }
2500 auto &PredOp = Cmp->getOperand(i: 1);
2501 MIB.setInstrAndDebugLoc(I);
2502 emitIntegerCompare(/*LHS=*/Cmp->getOperand(i: 2),
2503 /*RHS=*/Cmp->getOperand(i: 3), Predicate&: PredOp, MIRBuilder&: MIB);
2504 auto Pred = static_cast<CmpInst::Predicate>(PredOp.getPredicate());
2505 const AArch64CC::CondCode InvCC = changeICMPPredToAArch64CC(
2506 P: CmpInst::getInversePredicate(pred: Pred), RHS: Cmp->getOperand(i: 3).getReg(), MRI: &MRI);
2507 emitCSINC(/*Dst=*/AddDst, /*Src =*/Src1: AddLHS, /*Src2=*/AddLHS, Pred: InvCC, MIRBuilder&: MIB);
2508 I.eraseFromParent();
2509 return true;
2510 }
2511 case TargetOpcode::G_OR: {
2512 // Look for operations that take the lower `Width=Size-ShiftImm` bits of
2513 // `ShiftSrc` and insert them into the upper `Width` bits of `MaskSrc` via
2514 // shifting and masking that we can replace with a BFI (encoded as a BFM).
2515 Register Dst = I.getOperand(i: 0).getReg();
2516 LLT Ty = MRI.getType(Reg: Dst);
2517
2518 if (!Ty.isScalar())
2519 return false;
2520
2521 unsigned Size = Ty.getSizeInBits();
2522 if (Size != 32 && Size != 64)
2523 return false;
2524
2525 Register ShiftSrc;
2526 int64_t ShiftImm;
2527 Register MaskSrc;
2528 int64_t MaskImm;
2529 if (!mi_match(
2530 R: Dst, MRI,
2531 P: m_GOr(L: m_OneNonDBGUse(SP: m_GShl(L: m_Reg(R&: ShiftSrc), R: m_ICst(Cst&: ShiftImm))),
2532 R: m_OneNonDBGUse(SP: m_GAnd(L: m_Reg(R&: MaskSrc), R: m_ICst(Cst&: MaskImm))))))
2533 return false;
2534
2535 if (ShiftImm > Size || ((1ULL << ShiftImm) - 1ULL) != uint64_t(MaskImm))
2536 return false;
2537
2538 int64_t Immr = Size - ShiftImm;
2539 int64_t Imms = Size - ShiftImm - 1;
2540 unsigned Opc = Size == 32 ? AArch64::BFMWri : AArch64::BFMXri;
2541 emitInstr(Opcode: Opc, DstOps: {Dst}, SrcOps: {MaskSrc, ShiftSrc, Immr, Imms}, MIRBuilder&: MIB);
2542 I.eraseFromParent();
2543 return true;
2544 }
2545 case TargetOpcode::G_FENCE: {
2546 if (I.getOperand(i: 1).getImm() == 0)
2547 BuildMI(BB&: MBB, I, MIMD: MIMetadata(I), MCID: TII.get(Opcode: TargetOpcode::MEMBARRIER));
2548 else
2549 BuildMI(BB&: MBB, I, MIMD: MIMetadata(I), MCID: TII.get(Opcode: AArch64::DMB))
2550 .addImm(Val: I.getOperand(i: 0).getImm() == 4 ? 0x9 : 0xb);
2551 I.eraseFromParent();
2552 return true;
2553 }
2554 case TargetOpcode::G_BRINDIRECT: {
2555 const Function &Fn = MF.getFunction();
2556 if (std::optional<uint16_t> BADisc =
2557 STI.getPtrAuthBlockAddressDiscriminatorIfEnabled(ParentFn: Fn)) {
2558 auto MI = MIB.buildInstr(Opc: AArch64::BRA, DstOps: {}, SrcOps: {I.getOperand(i: 0).getReg()});
2559 MI.addImm(Val: AArch64PACKey::IA);
2560 MI.addImm(Val: *BADisc);
2561 MI.addReg(/*AddrDisc=*/RegNo: AArch64::XZR);
2562 I.eraseFromParent();
2563 constrainSelectedInstRegOperands(I&: *MI, TII, TRI, RBI);
2564 return true;
2565 }
2566 // Use table-based selection.
2567 return false;
2568 }
2569
2570 default:
2571 return false;
2572 }
2573}
2574
2575bool AArch64InstructionSelector::select(MachineInstr &I) {
2576 assert(I.getParent() && "Instruction should be in a basic block!");
2577 assert(I.getParent()->getParent() && "Instruction should be in a function!");
2578
2579 MachineBasicBlock &MBB = *I.getParent();
2580 MachineFunction &MF = *MBB.getParent();
2581 MachineRegisterInfo &MRI = MF.getRegInfo();
2582
2583 const AArch64Subtarget *Subtarget = &MF.getSubtarget<AArch64Subtarget>();
2584 if (Subtarget->requiresStrictAlign()) {
2585 // We don't support this feature yet.
2586 LLVM_DEBUG(dbgs() << "AArch64 GISel does not support strict-align yet\n");
2587 return false;
2588 }
2589
2590 MIB.setInstrAndDebugLoc(I);
2591
2592 unsigned Opcode = I.getOpcode();
2593 // G_PHI requires same handling as PHI
2594 if (!I.isPreISelOpcode() || Opcode == TargetOpcode::G_PHI) {
2595 // Certain non-generic instructions also need some special handling.
2596
2597 if (Opcode == TargetOpcode::LOAD_STACK_GUARD) {
2598 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
2599 return true;
2600 }
2601
2602 if (Opcode == TargetOpcode::PHI || Opcode == TargetOpcode::G_PHI) {
2603 const Register DefReg = I.getOperand(i: 0).getReg();
2604 const LLT DefTy = MRI.getType(Reg: DefReg);
2605
2606 const RegClassOrRegBank &RegClassOrBank =
2607 MRI.getRegClassOrRegBank(Reg: DefReg);
2608
2609 const TargetRegisterClass *DefRC =
2610 dyn_cast<const TargetRegisterClass *>(Val: RegClassOrBank);
2611 if (!DefRC) {
2612 if (!DefTy.isValid()) {
2613 LLVM_DEBUG(dbgs() << "PHI operand has no type, not a gvreg?\n");
2614 return false;
2615 }
2616 const RegisterBank &RB = *cast<const RegisterBank *>(Val: RegClassOrBank);
2617 DefRC = getRegClassForTypeOnBank(Ty: DefTy, RB);
2618 if (!DefRC) {
2619 LLVM_DEBUG(dbgs() << "PHI operand has unexpected size/bank\n");
2620 return false;
2621 }
2622 }
2623
2624 I.setDesc(TII.get(Opcode: TargetOpcode::PHI));
2625
2626 return RBI.constrainGenericRegister(Reg: DefReg, RC: *DefRC, MRI);
2627 }
2628
2629 if (I.isCopy())
2630 return selectCopy(I, TII, MRI, TRI, RBI);
2631
2632 if (I.isDebugInstr())
2633 return selectDebugInstr(I, MRI, RBI);
2634
2635 return true;
2636 }
2637
2638
2639 if (I.getNumOperands() != I.getNumExplicitOperands()) {
2640 LLVM_DEBUG(
2641 dbgs() << "Generic instruction has unexpected implicit operands\n");
2642 return false;
2643 }
2644
2645 // Try to do some lowering before we start instruction selecting. These
2646 // lowerings are purely transformations on the input G_MIR and so selection
2647 // must continue after any modification of the instruction.
2648 if (preISelLower(I)) {
2649 Opcode = I.getOpcode(); // The opcode may have been modified, refresh it.
2650 }
2651
2652 // There may be patterns where the importer can't deal with them optimally,
2653 // but does select it to a suboptimal sequence so our custom C++ selection
2654 // code later never has a chance to work on it. Therefore, we have an early
2655 // selection attempt here to give priority to certain selection routines
2656 // over the imported ones.
2657 if (earlySelect(I))
2658 return true;
2659
2660 if (selectImpl(I, CoverageInfo&: *CoverageInfo))
2661 return true;
2662
2663 LLT Ty =
2664 I.getOperand(i: 0).isReg() ? MRI.getType(Reg: I.getOperand(i: 0).getReg()) : LLT{};
2665
2666 switch (Opcode) {
2667 case TargetOpcode::G_SBFX:
2668 case TargetOpcode::G_UBFX: {
2669 static const unsigned OpcTable[2][2] = {
2670 {AArch64::UBFMWri, AArch64::UBFMXri},
2671 {AArch64::SBFMWri, AArch64::SBFMXri}};
2672 bool IsSigned = Opcode == TargetOpcode::G_SBFX;
2673 unsigned Size = Ty.getSizeInBits();
2674 unsigned Opc = OpcTable[IsSigned][Size == 64];
2675 auto Cst1 =
2676 getIConstantVRegValWithLookThrough(VReg: I.getOperand(i: 2).getReg(), MRI);
2677 assert(Cst1 && "Should have gotten a constant for src 1?");
2678 auto Cst2 =
2679 getIConstantVRegValWithLookThrough(VReg: I.getOperand(i: 3).getReg(), MRI);
2680 assert(Cst2 && "Should have gotten a constant for src 2?");
2681 auto LSB = Cst1->Value.getZExtValue();
2682 auto Width = Cst2->Value.getZExtValue();
2683 auto BitfieldInst =
2684 MIB.buildInstr(Opc, DstOps: {I.getOperand(i: 0)}, SrcOps: {I.getOperand(i: 1)})
2685 .addImm(Val: LSB)
2686 .addImm(Val: LSB + Width - 1);
2687 I.eraseFromParent();
2688 constrainSelectedInstRegOperands(I&: *BitfieldInst, TII, TRI, RBI);
2689 return true;
2690 }
2691 case TargetOpcode::G_BRCOND:
2692 return selectCompareBranch(I, MF, MRI);
2693
2694 case TargetOpcode::G_BRJT:
2695 return selectBrJT(I, MRI);
2696
2697 case AArch64::G_ADD_LOW: {
2698 // This op may have been separated from it's ADRP companion by the localizer
2699 // or some other code motion pass. Given that many CPUs will try to
2700 // macro fuse these operations anyway, select this into a MOVaddr pseudo
2701 // which will later be expanded into an ADRP+ADD pair after scheduling.
2702 MachineInstr *BaseMI = MRI.getVRegDef(Reg: I.getOperand(i: 1).getReg());
2703 if (BaseMI->getOpcode() != AArch64::ADRP) {
2704 I.setDesc(TII.get(Opcode: AArch64::ADDXri));
2705 I.addOperand(Op: MachineOperand::CreateImm(Val: 0));
2706 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
2707 return true;
2708 }
2709 assert(TM.getCodeModel() == CodeModel::Small &&
2710 "Expected small code model");
2711 auto Op1 = BaseMI->getOperand(i: 1);
2712 auto Op2 = I.getOperand(i: 2);
2713 auto MovAddr = MIB.buildInstr(Opc: AArch64::MOVaddr, DstOps: {I.getOperand(i: 0)}, SrcOps: {})
2714 .addGlobalAddress(GV: Op1.getGlobal(), Offset: Op1.getOffset(),
2715 TargetFlags: Op1.getTargetFlags())
2716 .addGlobalAddress(GV: Op2.getGlobal(), Offset: Op2.getOffset(),
2717 TargetFlags: Op2.getTargetFlags());
2718 I.eraseFromParent();
2719 constrainSelectedInstRegOperands(I&: *MovAddr, TII, TRI, RBI);
2720 return true;
2721 }
2722
2723 case TargetOpcode::G_FCONSTANT: {
2724 const Register DefReg = I.getOperand(i: 0).getReg();
2725 const LLT DefTy = MRI.getType(Reg: DefReg);
2726 const unsigned DefSize = DefTy.getSizeInBits();
2727 const RegisterBank &RB = *RBI.getRegBank(Reg: DefReg, MRI, TRI);
2728
2729 const TargetRegisterClass &FPRRC = *getRegClassForTypeOnBank(Ty: DefTy, RB);
2730 // For 16, 64, and 128b values, emit a constant pool load.
2731 switch (DefSize) {
2732 default:
2733 llvm_unreachable("Unexpected destination size for G_FCONSTANT?");
2734 case 32:
2735 case 64: {
2736 bool OptForSize = shouldOptForSize(MF: &MF);
2737 const auto &TLI = MF.getSubtarget().getTargetLowering();
2738 // If TLI says that this fpimm is illegal, then we'll expand to a
2739 // constant pool load.
2740 if (TLI->isFPImmLegal(I.getOperand(i: 1).getFPImm()->getValueAPF(),
2741 EVT::getFloatingPointVT(BitWidth: DefSize), ForCodeSize: OptForSize))
2742 break;
2743 [[fallthrough]];
2744 }
2745 case 16:
2746 case 128: {
2747 auto *FPImm = I.getOperand(i: 1).getFPImm();
2748 auto *LoadMI = emitLoadFromConstantPool(CPVal: FPImm, MIRBuilder&: MIB);
2749 if (!LoadMI) {
2750 LLVM_DEBUG(dbgs() << "Failed to load double constant pool entry\n");
2751 return false;
2752 }
2753 MIB.buildCopy(Res: {DefReg}, Op: {LoadMI->getOperand(i: 0).getReg()});
2754 I.eraseFromParent();
2755 return RBI.constrainGenericRegister(Reg: DefReg, RC: FPRRC, MRI);
2756 }
2757 }
2758
2759 assert((DefSize == 32 || DefSize == 64) && "Unexpected const def size");
2760 // Either emit a FMOV, or emit a copy to emit a normal mov.
2761 const Register DefGPRReg = MRI.createVirtualRegister(
2762 RegClass: DefSize == 32 ? &AArch64::GPR32RegClass : &AArch64::GPR64RegClass);
2763 MachineOperand &RegOp = I.getOperand(i: 0);
2764 RegOp.setReg(DefGPRReg);
2765 MIB.setInsertPt(MBB&: MIB.getMBB(), II: std::next(x: I.getIterator()));
2766 MIB.buildCopy(Res: {DefReg}, Op: {DefGPRReg});
2767
2768 if (!RBI.constrainGenericRegister(Reg: DefReg, RC: FPRRC, MRI)) {
2769 LLVM_DEBUG(dbgs() << "Failed to constrain G_FCONSTANT def operand\n");
2770 return false;
2771 }
2772
2773 MachineOperand &ImmOp = I.getOperand(i: 1);
2774 ImmOp.ChangeToImmediate(
2775 ImmVal: ImmOp.getFPImm()->getValueAPF().bitcastToAPInt().getZExtValue());
2776
2777 const unsigned MovOpc =
2778 DefSize == 64 ? AArch64::MOVi64imm : AArch64::MOVi32imm;
2779 I.setDesc(TII.get(Opcode: MovOpc));
2780 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
2781 return true;
2782 }
2783 case TargetOpcode::G_EXTRACT: {
2784 Register DstReg = I.getOperand(i: 0).getReg();
2785 Register SrcReg = I.getOperand(i: 1).getReg();
2786 LLT SrcTy = MRI.getType(Reg: SrcReg);
2787 LLT DstTy = MRI.getType(Reg: DstReg);
2788 (void)DstTy;
2789 unsigned SrcSize = SrcTy.getSizeInBits();
2790
2791 if (SrcTy.getSizeInBits() > 64) {
2792 // This should be an extract of an s128, which is like a vector extract.
2793 if (SrcTy.getSizeInBits() != 128)
2794 return false;
2795 // Only support extracting 64 bits from an s128 at the moment.
2796 if (DstTy.getSizeInBits() != 64)
2797 return false;
2798
2799 unsigned Offset = I.getOperand(i: 2).getImm();
2800 if (Offset % 64 != 0)
2801 return false;
2802
2803 // Check we have the right regbank always.
2804 const RegisterBank &SrcRB = *RBI.getRegBank(Reg: SrcReg, MRI, TRI);
2805 const RegisterBank &DstRB = *RBI.getRegBank(Reg: DstReg, MRI, TRI);
2806 assert(SrcRB.getID() == DstRB.getID() && "Wrong extract regbank!");
2807
2808 if (SrcRB.getID() == AArch64::GPRRegBankID) {
2809 auto NewI =
2810 MIB.buildInstr(Opc: TargetOpcode::COPY, DstOps: {DstReg}, SrcOps: {})
2811 .addUse(RegNo: SrcReg, Flags: {},
2812 SubReg: Offset == 0 ? AArch64::sube64 : AArch64::subo64);
2813 constrainOperandRegClass(MF, TRI, MRI, TII, RBI, InsertPt&: *NewI,
2814 RegClass: AArch64::GPR64RegClass, RegMO&: NewI->getOperand(i: 0));
2815 I.eraseFromParent();
2816 return true;
2817 }
2818
2819 // Emit the same code as a vector extract.
2820 // Offset must be a multiple of 64.
2821 unsigned LaneIdx = Offset / 64;
2822 MachineInstr *Extract = emitExtractVectorElt(
2823 DstReg, DstRB, ScalarTy: LLT::scalar(SizeInBits: 64), VecReg: SrcReg, LaneIdx, MIRBuilder&: MIB);
2824 if (!Extract)
2825 return false;
2826 I.eraseFromParent();
2827 return true;
2828 }
2829
2830 I.setDesc(TII.get(Opcode: SrcSize == 64 ? AArch64::UBFMXri : AArch64::UBFMWri));
2831 MachineInstrBuilder(MF, I).addImm(Val: I.getOperand(i: 2).getImm() +
2832 Ty.getSizeInBits() - 1);
2833
2834 if (SrcSize < 64) {
2835 assert(SrcSize == 32 && DstTy.getSizeInBits() == 16 &&
2836 "unexpected G_EXTRACT types");
2837 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
2838 return true;
2839 }
2840
2841 DstReg = MRI.createGenericVirtualRegister(Ty: LLT::scalar(SizeInBits: 64));
2842 MIB.setInsertPt(MBB&: MIB.getMBB(), II: std::next(x: I.getIterator()));
2843 MIB.buildInstr(Opc: TargetOpcode::COPY, DstOps: {I.getOperand(i: 0).getReg()}, SrcOps: {})
2844 .addReg(RegNo: DstReg, Flags: {}, SubReg: AArch64::sub_32);
2845 RBI.constrainGenericRegister(Reg: I.getOperand(i: 0).getReg(),
2846 RC: AArch64::GPR32RegClass, MRI);
2847 I.getOperand(i: 0).setReg(DstReg);
2848
2849 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
2850 return true;
2851 }
2852
2853 case TargetOpcode::G_INSERT: {
2854 LLT SrcTy = MRI.getType(Reg: I.getOperand(i: 2).getReg());
2855 LLT DstTy = MRI.getType(Reg: I.getOperand(i: 0).getReg());
2856 unsigned DstSize = DstTy.getSizeInBits();
2857 // Larger inserts are vectors, same-size ones should be something else by
2858 // now (split up or turned into COPYs).
2859 if (Ty.getSizeInBits() > 64 || SrcTy.getSizeInBits() > 32)
2860 return false;
2861
2862 I.setDesc(TII.get(Opcode: DstSize == 64 ? AArch64::BFMXri : AArch64::BFMWri));
2863 unsigned LSB = I.getOperand(i: 3).getImm();
2864 unsigned Width = MRI.getType(Reg: I.getOperand(i: 2).getReg()).getSizeInBits();
2865 I.getOperand(i: 3).setImm((DstSize - LSB) % DstSize);
2866 MachineInstrBuilder(MF, I).addImm(Val: Width - 1);
2867
2868 if (DstSize < 64) {
2869 assert(DstSize == 32 && SrcTy.getSizeInBits() == 16 &&
2870 "unexpected G_INSERT types");
2871 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
2872 return true;
2873 }
2874
2875 Register SrcReg = MRI.createGenericVirtualRegister(Ty: LLT::scalar(SizeInBits: 64));
2876 BuildMI(BB&: MBB, I: I.getIterator(), MIMD: I.getDebugLoc(),
2877 MCID: TII.get(Opcode: AArch64::SUBREG_TO_REG))
2878 .addDef(RegNo: SrcReg)
2879 .addUse(RegNo: I.getOperand(i: 2).getReg())
2880 .addImm(Val: AArch64::sub_32);
2881 RBI.constrainGenericRegister(Reg: I.getOperand(i: 2).getReg(),
2882 RC: AArch64::GPR32RegClass, MRI);
2883 I.getOperand(i: 2).setReg(SrcReg);
2884
2885 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
2886 return true;
2887 }
2888 case TargetOpcode::G_FRAME_INDEX: {
2889 // allocas and G_FRAME_INDEX are only supported in addrspace(0).
2890 if (Ty != LLT::pointer(AddressSpace: 0, SizeInBits: 64)) {
2891 LLVM_DEBUG(dbgs() << "G_FRAME_INDEX pointer has type: " << Ty
2892 << ", expected: " << LLT::pointer(0, 64) << '\n');
2893 return false;
2894 }
2895 I.setDesc(TII.get(Opcode: AArch64::ADDXri));
2896
2897 // MOs for a #0 shifted immediate.
2898 I.addOperand(Op: MachineOperand::CreateImm(Val: 0));
2899 I.addOperand(Op: MachineOperand::CreateImm(Val: 0));
2900
2901 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
2902 return true;
2903 }
2904
2905 case TargetOpcode::G_GLOBAL_VALUE: {
2906 const GlobalValue *GV = nullptr;
2907 unsigned OpFlags;
2908 if (I.getOperand(i: 1).isSymbol()) {
2909 OpFlags = I.getOperand(i: 1).getTargetFlags();
2910 // Currently only used by "RtLibUseGOT".
2911 assert(OpFlags == AArch64II::MO_GOT);
2912 } else {
2913 GV = I.getOperand(i: 1).getGlobal();
2914 if (GV->isThreadLocal())
2915 return selectTLSGlobalValue(I, MRI);
2916
2917 OpFlags = STI.ClassifyGlobalReference(GV, TM);
2918 }
2919
2920 if (OpFlags & AArch64II::MO_GOT) {
2921 bool IsGOTSigned = MF.getInfo<AArch64FunctionInfo>()->hasELFSignedGOT();
2922 I.setDesc(TII.get(Opcode: IsGOTSigned ? AArch64::LOADgotAUTH : AArch64::LOADgot));
2923 I.getOperand(i: 1).setTargetFlags(OpFlags);
2924 I.addImplicitDefUseOperands(MF);
2925 I.setImplicitPhysRegDefsDead();
2926 } else if (TM.getCodeModel() == CodeModel::Large &&
2927 !TM.isPositionIndependent()) {
2928 // Materialize the global using movz/movk instructions.
2929 materializeLargeCMVal(I, V: GV, OpFlags);
2930 I.eraseFromParent();
2931 return true;
2932 } else if (TM.getCodeModel() == CodeModel::Tiny) {
2933 I.setDesc(TII.get(Opcode: AArch64::ADR));
2934 I.getOperand(i: 1).setTargetFlags(OpFlags);
2935 } else {
2936 I.setDesc(TII.get(Opcode: AArch64::MOVaddr));
2937 I.getOperand(i: 1).setTargetFlags(OpFlags | AArch64II::MO_PAGE);
2938 MachineInstrBuilder MIB(MF, I);
2939 MIB.addGlobalAddress(GV, Offset: I.getOperand(i: 1).getOffset(),
2940 TargetFlags: OpFlags | AArch64II::MO_PAGEOFF | AArch64II::MO_NC);
2941 }
2942 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
2943 return true;
2944 }
2945
2946 case TargetOpcode::G_PTRAUTH_GLOBAL_VALUE:
2947 return selectPtrAuthGlobalValue(I, MRI);
2948
2949 case TargetOpcode::G_ZEXTLOAD:
2950 case TargetOpcode::G_LOAD:
2951 case TargetOpcode::G_STORE: {
2952 GLoadStore &LdSt = cast<GLoadStore>(Val&: I);
2953 bool IsZExtLoad = I.getOpcode() == TargetOpcode::G_ZEXTLOAD;
2954 LLT PtrTy = MRI.getType(Reg: LdSt.getPointerReg());
2955
2956 // Can only handle AddressSpace 0, 64-bit pointers.
2957 if (PtrTy != LLT::pointer(AddressSpace: 0, SizeInBits: 64)) {
2958 return false;
2959 }
2960
2961 uint64_t MemSizeInBytes = LdSt.getMemSize().getValue();
2962 unsigned MemSizeInBits = LdSt.getMemSizeInBits().getValue();
2963 AtomicOrdering Order = LdSt.getMMO().getSuccessOrdering();
2964
2965 // Need special instructions for atomics that affect ordering.
2966 if (isStrongerThanMonotonic(AO: Order)) {
2967 assert(!isa<GZExtLoad>(LdSt));
2968 assert(MemSizeInBytes <= 8 &&
2969 "128-bit atomics should already be custom-legalized");
2970
2971 if (isa<GLoad>(Val: LdSt)) {
2972 static constexpr unsigned LDAPROpcodes[] = {
2973 AArch64::LDAPRB, AArch64::LDAPRH, AArch64::LDAPRW, AArch64::LDAPRX};
2974 static constexpr unsigned LDAROpcodes[] = {
2975 AArch64::LDARB, AArch64::LDARH, AArch64::LDARW, AArch64::LDARX};
2976 ArrayRef<unsigned> Opcodes =
2977 STI.hasRCPC() && Order != AtomicOrdering::SequentiallyConsistent
2978 ? LDAPROpcodes
2979 : LDAROpcodes;
2980 I.setDesc(TII.get(Opcode: Opcodes[Log2_32(Value: MemSizeInBytes)]));
2981 } else {
2982 static constexpr unsigned Opcodes[] = {AArch64::STLRB, AArch64::STLRH,
2983 AArch64::STLRW, AArch64::STLRX};
2984 Register ValReg = LdSt.getReg(Idx: 0);
2985 if (MRI.getType(Reg: ValReg).getSizeInBits() == 64 && MemSizeInBits != 64) {
2986 // Emit a subreg copy of 32 bits.
2987 Register NewVal = MRI.createVirtualRegister(RegClass: &AArch64::GPR32RegClass);
2988 MIB.buildInstr(Opc: TargetOpcode::COPY, DstOps: {NewVal}, SrcOps: {})
2989 .addReg(RegNo: I.getOperand(i: 0).getReg(), Flags: {}, SubReg: AArch64::sub_32);
2990 I.getOperand(i: 0).setReg(NewVal);
2991 }
2992 I.setDesc(TII.get(Opcode: Opcodes[Log2_32(Value: MemSizeInBytes)]));
2993 }
2994 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
2995 return true;
2996 }
2997
2998#ifndef NDEBUG
2999 const Register PtrReg = LdSt.getPointerReg();
3000 const RegisterBank &PtrRB = *RBI.getRegBank(PtrReg, MRI, TRI);
3001 // Check that the pointer register is valid.
3002 assert(PtrRB.getID() == AArch64::GPRRegBankID &&
3003 "Load/Store pointer operand isn't a GPR");
3004 assert(MRI.getType(PtrReg).isPointer() &&
3005 "Load/Store pointer operand isn't a pointer");
3006#endif
3007
3008 const Register ValReg = LdSt.getReg(Idx: 0);
3009 const RegisterBank &RB = *RBI.getRegBank(Reg: ValReg, MRI, TRI);
3010 LLT ValTy = MRI.getType(Reg: ValReg);
3011
3012 // The code below doesn't support truncating stores, so we need to split it
3013 // again.
3014 if (isa<GStore>(Val: LdSt) && ValTy.getSizeInBits() > MemSizeInBits &&
3015 RB.getID() == AArch64::FPRRegBankID) {
3016 unsigned SubReg;
3017 LLT MemTy = LdSt.getMMO().getMemoryType();
3018 auto *RC = getRegClassForTypeOnBank(Ty: MemTy, RB);
3019 if (!getSubRegForClass(RC, TRI, SubReg))
3020 return false;
3021
3022 // Generate a subreg copy.
3023 auto Copy = MIB.buildInstr(Opc: TargetOpcode::COPY, DstOps: {MemTy}, SrcOps: {})
3024 .addReg(RegNo: ValReg, Flags: {}, SubReg)
3025 .getReg(Idx: 0);
3026 RBI.constrainGenericRegister(Reg: Copy, RC: *RC, MRI);
3027 LdSt.getOperand(i: 0).setReg(Copy);
3028 } else if (isa<GLoad>(Val: LdSt) && ValTy.getSizeInBits() > MemSizeInBits) {
3029 // If this is an any-extending load from the FPR bank, split it into a regular
3030 // load + extend.
3031 if (RB.getID() == AArch64::FPRRegBankID) {
3032 unsigned SubReg;
3033 LLT MemTy = LdSt.getMMO().getMemoryType();
3034 auto *RC = getRegClassForTypeOnBank(Ty: MemTy, RB);
3035 if (!getSubRegForClass(RC, TRI, SubReg))
3036 return false;
3037 Register OldDst = LdSt.getReg(Idx: 0);
3038 Register NewDst =
3039 MRI.createGenericVirtualRegister(Ty: LdSt.getMMO().getMemoryType());
3040 LdSt.getOperand(i: 0).setReg(NewDst);
3041 MRI.setRegBank(Reg: NewDst, RegBank: RB);
3042 // Generate a SUBREG_TO_REG to extend it.
3043 MIB.setInsertPt(MBB&: MIB.getMBB(), II: std::next(x: LdSt.getIterator()));
3044 MIB.buildInstr(Opc: AArch64::SUBREG_TO_REG, DstOps: {OldDst}, SrcOps: {})
3045 .addUse(RegNo: NewDst)
3046 .addImm(Val: SubReg);
3047 auto SubRegRC = getRegClassForTypeOnBank(Ty: MRI.getType(Reg: OldDst), RB);
3048 RBI.constrainGenericRegister(Reg: OldDst, RC: *SubRegRC, MRI);
3049 MIB.setInstr(LdSt);
3050 ValTy = MemTy; // This is no longer an extending load.
3051 }
3052 }
3053
3054 // Helper lambda for partially selecting I. Either returns the original
3055 // instruction with an updated opcode, or a new instruction.
3056 auto SelectLoadStoreAddressingMode = [&]() -> MachineInstr * {
3057 bool IsStore = isa<GStore>(Val: I);
3058 const unsigned NewOpc =
3059 selectLoadStoreUIOp(GenericOpc: I.getOpcode(), RegBankID: RB.getID(), OpSize: MemSizeInBits);
3060 if (NewOpc == I.getOpcode())
3061 return nullptr;
3062 // Check if we can fold anything into the addressing mode.
3063 auto AddrModeFns =
3064 selectAddrModeIndexed(Root&: I.getOperand(i: 1), Size: MemSizeInBytes);
3065 if (!AddrModeFns) {
3066 // Can't fold anything. Use the original instruction.
3067 I.setDesc(TII.get(Opcode: NewOpc));
3068 I.addOperand(Op: MachineOperand::CreateImm(Val: 0));
3069 return &I;
3070 }
3071
3072 // Folded something. Create a new instruction and return it.
3073 auto NewInst = MIB.buildInstr(Opc: NewOpc, DstOps: {}, SrcOps: {}, Flags: I.getFlags());
3074 Register CurValReg = I.getOperand(i: 0).getReg();
3075 IsStore ? NewInst.addUse(RegNo: CurValReg) : NewInst.addDef(RegNo: CurValReg);
3076 NewInst.cloneMemRefs(OtherMI: I);
3077 for (auto &Fn : *AddrModeFns)
3078 Fn(NewInst);
3079 I.eraseFromParent();
3080 return &*NewInst;
3081 };
3082
3083 MachineInstr *LoadStore = SelectLoadStoreAddressingMode();
3084 if (!LoadStore)
3085 return false;
3086
3087 // If we're storing a 0, use WZR/XZR.
3088 if (Opcode == TargetOpcode::G_STORE) {
3089 auto CVal = getIConstantVRegValWithLookThrough(
3090 VReg: LoadStore->getOperand(i: 0).getReg(), MRI);
3091 if (CVal && CVal->Value == 0) {
3092 switch (LoadStore->getOpcode()) {
3093 case AArch64::STRWui:
3094 case AArch64::STRHHui:
3095 case AArch64::STRBBui:
3096 LoadStore->getOperand(i: 0).setReg(AArch64::WZR);
3097 break;
3098 case AArch64::STRXui:
3099 LoadStore->getOperand(i: 0).setReg(AArch64::XZR);
3100 break;
3101 }
3102 }
3103 }
3104
3105 if (IsZExtLoad || (Opcode == TargetOpcode::G_LOAD &&
3106 ValTy == LLT::scalar(SizeInBits: 64) && MemSizeInBits == 32)) {
3107 // The any/zextload from a smaller type to i32 should be handled by the
3108 // importer.
3109 if (MRI.getType(Reg: LoadStore->getOperand(i: 0).getReg()).getSizeInBits() != 64)
3110 return false;
3111 // If we have an extending load then change the load's type to be a
3112 // narrower reg and zero_extend with SUBREG_TO_REG.
3113 Register LdReg = MRI.createVirtualRegister(RegClass: &AArch64::GPR32RegClass);
3114 Register DstReg = LoadStore->getOperand(i: 0).getReg();
3115 LoadStore->getOperand(i: 0).setReg(LdReg);
3116
3117 MIB.setInsertPt(MBB&: MIB.getMBB(), II: std::next(x: LoadStore->getIterator()));
3118 MIB.buildInstr(Opc: AArch64::SUBREG_TO_REG, DstOps: {DstReg}, SrcOps: {})
3119 .addUse(RegNo: LdReg)
3120 .addImm(Val: AArch64::sub_32);
3121 constrainSelectedInstRegOperands(I&: *LoadStore, TII, TRI, RBI);
3122 return RBI.constrainGenericRegister(Reg: DstReg, RC: AArch64::GPR64allRegClass,
3123 MRI);
3124 }
3125 constrainSelectedInstRegOperands(I&: *LoadStore, TII, TRI, RBI);
3126 return true;
3127 }
3128
3129 case TargetOpcode::G_INDEXED_ZEXTLOAD:
3130 case TargetOpcode::G_INDEXED_SEXTLOAD:
3131 return selectIndexedExtLoad(I, MRI);
3132 case TargetOpcode::G_INDEXED_LOAD:
3133 return selectIndexedLoad(I, MRI);
3134 case TargetOpcode::G_INDEXED_STORE:
3135 return selectIndexedStore(I&: cast<GIndexedStore>(Val&: I), MRI);
3136
3137 case TargetOpcode::G_LSHR:
3138 case TargetOpcode::G_ASHR:
3139 if (MRI.getType(Reg: I.getOperand(i: 0).getReg()).isVector())
3140 return selectVectorAshrLshr(I, MRI);
3141 [[fallthrough]];
3142 case TargetOpcode::G_SHL: {
3143 if (Opcode == TargetOpcode::G_SHL &&
3144 MRI.getType(Reg: I.getOperand(i: 0).getReg()).isVector())
3145 return selectVectorSHL(I, MRI);
3146
3147 // These shifts were legalized to have 64 bit shift amounts because we
3148 // want to take advantage of the selection patterns that assume the
3149 // immediates are s64s, however, selectBinaryOp will assume both operands
3150 // will have the same bit size.
3151 {
3152 Register SrcReg = I.getOperand(i: 1).getReg();
3153 Register ShiftReg = I.getOperand(i: 2).getReg();
3154 const LLT ShiftTy = MRI.getType(Reg: ShiftReg);
3155 const LLT SrcTy = MRI.getType(Reg: SrcReg);
3156 if (!SrcTy.isVector() && SrcTy.getSizeInBits() == 32 &&
3157 ShiftTy.getSizeInBits() == 64) {
3158 assert(!ShiftTy.isVector() && "unexpected vector shift ty");
3159 // Insert a subregister copy to implement a 64->32 trunc
3160 auto Trunc = MIB.buildInstr(Opc: TargetOpcode::COPY, DstOps: {SrcTy}, SrcOps: {})
3161 .addReg(RegNo: ShiftReg, Flags: {}, SubReg: AArch64::sub_32);
3162 MRI.setRegBank(Reg: Trunc.getReg(Idx: 0), RegBank: RBI.getRegBank(ID: AArch64::GPRRegBankID));
3163 I.getOperand(i: 2).setReg(Trunc.getReg(Idx: 0));
3164 }
3165 }
3166
3167 const unsigned OpSize = Ty.getSizeInBits();
3168 const Register DefReg = I.getOperand(i: 0).getReg();
3169 const RegisterBank &RB = *RBI.getRegBank(Reg: DefReg, MRI, TRI);
3170
3171 const unsigned NewOpc = selectBinaryOp(GenericOpc: I.getOpcode(), RegBankID: RB.getID(), OpSize);
3172 if (NewOpc == I.getOpcode())
3173 return false;
3174
3175 I.setDesc(TII.get(Opcode: NewOpc));
3176 // FIXME: Should the type be always reset in setDesc?
3177
3178 // Now that we selected an opcode, we need to constrain the register
3179 // operands to use appropriate classes.
3180 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
3181 return true;
3182 }
3183 case TargetOpcode::G_PTR_ADD: {
3184 emitADD(DefReg: I.getOperand(i: 0).getReg(), LHS&: I.getOperand(i: 1), RHS&: I.getOperand(i: 2), MIRBuilder&: MIB);
3185 I.eraseFromParent();
3186 return true;
3187 }
3188
3189 case TargetOpcode::G_SADDE:
3190 case TargetOpcode::G_UADDE:
3191 case TargetOpcode::G_SSUBE:
3192 case TargetOpcode::G_USUBE:
3193 case TargetOpcode::G_SADDO:
3194 case TargetOpcode::G_UADDO:
3195 case TargetOpcode::G_SSUBO:
3196 case TargetOpcode::G_USUBO:
3197 return selectOverflowOp(I, MRI);
3198
3199 case TargetOpcode::G_PTRMASK: {
3200 Register MaskReg = I.getOperand(i: 2).getReg();
3201 std::optional<int64_t> MaskVal = getIConstantVRegSExtVal(VReg: MaskReg, MRI);
3202 // TODO: Implement arbitrary cases
3203 if (!MaskVal || !isShiftedMask_64(Value: *MaskVal))
3204 return false;
3205
3206 uint64_t Mask = *MaskVal;
3207 I.setDesc(TII.get(Opcode: AArch64::ANDXri));
3208 I.getOperand(i: 2).ChangeToImmediate(
3209 ImmVal: AArch64_AM::encodeLogicalImmediate(imm: Mask, regSize: 64));
3210
3211 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
3212 return true;
3213 }
3214 case TargetOpcode::G_PTRTOINT:
3215 case TargetOpcode::G_TRUNC: {
3216 const LLT DstTy = MRI.getType(Reg: I.getOperand(i: 0).getReg());
3217 const LLT SrcTy = MRI.getType(Reg: I.getOperand(i: 1).getReg());
3218
3219 const Register DstReg = I.getOperand(i: 0).getReg();
3220 const Register SrcReg = I.getOperand(i: 1).getReg();
3221
3222 const RegisterBank &DstRB = *RBI.getRegBank(Reg: DstReg, MRI, TRI);
3223 const RegisterBank &SrcRB = *RBI.getRegBank(Reg: SrcReg, MRI, TRI);
3224
3225 if (DstRB.getID() != SrcRB.getID()) {
3226 LLVM_DEBUG(
3227 dbgs() << "G_TRUNC/G_PTRTOINT input/output on different banks\n");
3228 return false;
3229 }
3230
3231 if (DstRB.getID() == AArch64::GPRRegBankID) {
3232 const TargetRegisterClass *DstRC = getRegClassForTypeOnBank(Ty: DstTy, RB: DstRB);
3233 if (!DstRC)
3234 return false;
3235
3236 const TargetRegisterClass *SrcRC = getRegClassForTypeOnBank(Ty: SrcTy, RB: SrcRB);
3237 if (!SrcRC)
3238 return false;
3239
3240 if (!RBI.constrainGenericRegister(Reg: SrcReg, RC: *SrcRC, MRI) ||
3241 !RBI.constrainGenericRegister(Reg: DstReg, RC: *DstRC, MRI)) {
3242 LLVM_DEBUG(dbgs() << "Failed to constrain G_TRUNC/G_PTRTOINT\n");
3243 return false;
3244 }
3245
3246 if (DstRC == SrcRC) {
3247 // Nothing to be done
3248 } else if (Opcode == TargetOpcode::G_TRUNC && DstTy == LLT::scalar(SizeInBits: 32) &&
3249 SrcTy == LLT::scalar(SizeInBits: 64)) {
3250 llvm_unreachable("TableGen can import this case");
3251 return false;
3252 } else if (DstRC == &AArch64::GPR32RegClass &&
3253 SrcRC == &AArch64::GPR64RegClass) {
3254 I.getOperand(i: 1).setSubReg(AArch64::sub_32);
3255 } else {
3256 LLVM_DEBUG(
3257 dbgs() << "Unhandled mismatched classes in G_TRUNC/G_PTRTOINT\n");
3258 return false;
3259 }
3260
3261 I.setDesc(TII.get(Opcode: TargetOpcode::COPY));
3262 return true;
3263 } else if (DstRB.getID() == AArch64::FPRRegBankID) {
3264 if (DstTy == LLT::fixed_vector(NumElements: 4, ScalarSizeInBits: 16) &&
3265 SrcTy == LLT::fixed_vector(NumElements: 4, ScalarSizeInBits: 32)) {
3266 I.setDesc(TII.get(Opcode: AArch64::XTNv4i16));
3267 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
3268 return true;
3269 }
3270
3271 if (!SrcTy.isVector() && SrcTy.getSizeInBits() == 128) {
3272 MachineInstr *Extract = emitExtractVectorElt(
3273 DstReg, DstRB, ScalarTy: LLT::scalar(SizeInBits: DstTy.getSizeInBits()), VecReg: SrcReg, LaneIdx: 0, MIRBuilder&: MIB);
3274 if (!Extract)
3275 return false;
3276 I.eraseFromParent();
3277 return true;
3278 }
3279
3280 // We might have a vector G_PTRTOINT, in which case just emit a COPY.
3281 if (Opcode == TargetOpcode::G_PTRTOINT) {
3282 assert(DstTy.isVector() && "Expected an FPR ptrtoint to be a vector");
3283 I.setDesc(TII.get(Opcode: TargetOpcode::COPY));
3284 return selectCopy(I, TII, MRI, TRI, RBI);
3285 }
3286 }
3287
3288 return false;
3289 }
3290
3291 case TargetOpcode::G_ANYEXT: {
3292 if (selectUSMovFromExtend(I, MRI))
3293 return true;
3294
3295 const Register DstReg = I.getOperand(i: 0).getReg();
3296 const Register SrcReg = I.getOperand(i: 1).getReg();
3297
3298 const RegisterBank &RBDst = *RBI.getRegBank(Reg: DstReg, MRI, TRI);
3299 if (RBDst.getID() != AArch64::GPRRegBankID) {
3300 LLVM_DEBUG(dbgs() << "G_ANYEXT on bank: " << RBDst
3301 << ", expected: GPR\n");
3302 return false;
3303 }
3304
3305 const RegisterBank &RBSrc = *RBI.getRegBank(Reg: SrcReg, MRI, TRI);
3306 if (RBSrc.getID() != AArch64::GPRRegBankID) {
3307 LLVM_DEBUG(dbgs() << "G_ANYEXT on bank: " << RBSrc
3308 << ", expected: GPR\n");
3309 return false;
3310 }
3311
3312 const unsigned DstSize = MRI.getType(Reg: DstReg).getSizeInBits();
3313
3314 if (DstSize == 0) {
3315 LLVM_DEBUG(dbgs() << "G_ANYEXT operand has no size, not a gvreg?\n");
3316 return false;
3317 }
3318
3319 if (DstSize != 64 && DstSize > 32) {
3320 LLVM_DEBUG(dbgs() << "G_ANYEXT to size: " << DstSize
3321 << ", expected: 32 or 64\n");
3322 return false;
3323 }
3324 // At this point G_ANYEXT is just like a plain COPY, but we need
3325 // to explicitly form the 64-bit value if any.
3326 if (DstSize > 32) {
3327 Register ExtSrc = MRI.createVirtualRegister(RegClass: &AArch64::GPR64allRegClass);
3328 BuildMI(BB&: MBB, I, MIMD: I.getDebugLoc(), MCID: TII.get(Opcode: AArch64::SUBREG_TO_REG))
3329 .addDef(RegNo: ExtSrc)
3330 .addUse(RegNo: SrcReg)
3331 .addImm(Val: AArch64::sub_32);
3332 I.getOperand(i: 1).setReg(ExtSrc);
3333 }
3334 return selectCopy(I, TII, MRI, TRI, RBI);
3335 }
3336
3337 case TargetOpcode::G_ZEXT:
3338 case TargetOpcode::G_SEXT_INREG:
3339 case TargetOpcode::G_SEXT: {
3340 if (selectUSMovFromExtend(I, MRI))
3341 return true;
3342
3343 unsigned Opcode = I.getOpcode();
3344 const bool IsSigned = Opcode != TargetOpcode::G_ZEXT;
3345 const Register DefReg = I.getOperand(i: 0).getReg();
3346 Register SrcReg = I.getOperand(i: 1).getReg();
3347 const LLT DstTy = MRI.getType(Reg: DefReg);
3348 const LLT SrcTy = MRI.getType(Reg: SrcReg);
3349 unsigned DstSize = DstTy.getSizeInBits();
3350 unsigned SrcSize = SrcTy.getSizeInBits();
3351
3352 // SEXT_INREG has the same src reg size as dst, the size of the value to be
3353 // extended is encoded in the imm.
3354 if (Opcode == TargetOpcode::G_SEXT_INREG)
3355 SrcSize = I.getOperand(i: 2).getImm();
3356
3357 if (DstTy.isVector())
3358 return false; // Should be handled by imported patterns.
3359
3360 assert((*RBI.getRegBank(DefReg, MRI, TRI)).getID() ==
3361 AArch64::GPRRegBankID &&
3362 "Unexpected ext regbank");
3363
3364 MachineInstr *ExtI;
3365
3366 // First check if we're extending the result of a load which has a dest type
3367 // smaller than 32 bits, then this zext is redundant. GPR32 is the smallest
3368 // GPR register on AArch64 and all loads which are smaller automatically
3369 // zero-extend the upper bits. E.g.
3370 // %v(s8) = G_LOAD %p, :: (load 1)
3371 // %v2(s32) = G_ZEXT %v(s8)
3372 if (!IsSigned) {
3373 auto *LoadMI = getOpcodeDef(Opcode: TargetOpcode::G_LOAD, Reg: SrcReg, MRI);
3374 bool IsGPR =
3375 RBI.getRegBank(Reg: SrcReg, MRI, TRI)->getID() == AArch64::GPRRegBankID;
3376 if (LoadMI && IsGPR) {
3377 const MachineMemOperand *MemOp = *LoadMI->memoperands_begin();
3378 unsigned BytesLoaded = MemOp->getSize().getValue();
3379 if (BytesLoaded < 4 && SrcTy.getSizeInBytes() == BytesLoaded)
3380 return selectCopy(I, TII, MRI, TRI, RBI);
3381 }
3382
3383 // For the 32-bit -> 64-bit case, we can emit a mov (ORRWrs)
3384 // + SUBREG_TO_REG.
3385 if (IsGPR && SrcSize == 32 && DstSize == 64) {
3386 Register SubregToRegSrc =
3387 MRI.createVirtualRegister(RegClass: &AArch64::GPR32RegClass);
3388 const Register ZReg = AArch64::WZR;
3389 MIB.buildInstr(Opc: AArch64::ORRWrs, DstOps: {SubregToRegSrc}, SrcOps: {ZReg, SrcReg})
3390 .addImm(Val: 0);
3391
3392 MIB.buildInstr(Opc: AArch64::SUBREG_TO_REG, DstOps: {DefReg}, SrcOps: {})
3393 .addUse(RegNo: SubregToRegSrc)
3394 .addImm(Val: AArch64::sub_32);
3395
3396 if (!RBI.constrainGenericRegister(Reg: DefReg, RC: AArch64::GPR64RegClass,
3397 MRI)) {
3398 LLVM_DEBUG(dbgs() << "Failed to constrain G_ZEXT destination\n");
3399 return false;
3400 }
3401
3402 if (!RBI.constrainGenericRegister(Reg: SrcReg, RC: AArch64::GPR32RegClass,
3403 MRI)) {
3404 LLVM_DEBUG(dbgs() << "Failed to constrain G_ZEXT source\n");
3405 return false;
3406 }
3407
3408 I.eraseFromParent();
3409 return true;
3410 }
3411 }
3412
3413 if (DstSize == 64) {
3414 if (Opcode != TargetOpcode::G_SEXT_INREG) {
3415 // FIXME: Can we avoid manually doing this?
3416 if (!RBI.constrainGenericRegister(Reg: SrcReg, RC: AArch64::GPR32RegClass,
3417 MRI)) {
3418 LLVM_DEBUG(dbgs() << "Failed to constrain " << TII.getName(Opcode)
3419 << " operand\n");
3420 return false;
3421 }
3422 SrcReg = MIB.buildInstr(Opc: AArch64::SUBREG_TO_REG,
3423 DstOps: {&AArch64::GPR64RegClass}, SrcOps: {})
3424 .addUse(RegNo: SrcReg)
3425 .addImm(Val: AArch64::sub_32)
3426 .getReg(Idx: 0);
3427 }
3428
3429 ExtI = MIB.buildInstr(Opc: IsSigned ? AArch64::SBFMXri : AArch64::UBFMXri,
3430 DstOps: {DefReg}, SrcOps: {SrcReg})
3431 .addImm(Val: 0)
3432 .addImm(Val: SrcSize - 1);
3433 } else if (DstSize <= 32) {
3434 ExtI = MIB.buildInstr(Opc: IsSigned ? AArch64::SBFMWri : AArch64::UBFMWri,
3435 DstOps: {DefReg}, SrcOps: {SrcReg})
3436 .addImm(Val: 0)
3437 .addImm(Val: SrcSize - 1);
3438 } else {
3439 return false;
3440 }
3441
3442 constrainSelectedInstRegOperands(I&: *ExtI, TII, TRI, RBI);
3443 I.eraseFromParent();
3444 return true;
3445 }
3446
3447 case TargetOpcode::G_FREEZE:
3448 return selectCopy(I, TII, MRI, TRI, RBI);
3449
3450 case TargetOpcode::G_INTTOPTR:
3451 // The importer is currently unable to import pointer types since they
3452 // didn't exist in SelectionDAG.
3453 return selectCopy(I, TII, MRI, TRI, RBI);
3454
3455 case TargetOpcode::G_BITCAST:
3456 // Imported SelectionDAG rules can handle every bitcast except those that
3457 // bitcast from a type to the same type. Ideally, these shouldn't occur
3458 // but we might not run an optimizer that deletes them. The other exception
3459 // is bitcasts involving pointer types, as SelectionDAG has no knowledge
3460 // of them.
3461 return selectCopy(I, TII, MRI, TRI, RBI);
3462
3463 case TargetOpcode::G_SELECT: {
3464 auto &Sel = cast<GSelect>(Val&: I);
3465 const Register CondReg = Sel.getCondReg();
3466 const Register TReg = Sel.getTrueReg();
3467 const Register FReg = Sel.getFalseReg();
3468
3469 if (tryOptSelect(Sel))
3470 return true;
3471
3472 // Make sure to use an unused vreg instead of wzr, so that the peephole
3473 // optimizations will be able to optimize these.
3474 Register DeadVReg = MRI.createVirtualRegister(RegClass: &AArch64::GPR32RegClass);
3475 auto TstMI = MIB.buildInstr(Opc: AArch64::ANDSWri, DstOps: {DeadVReg}, SrcOps: {CondReg})
3476 .addImm(Val: AArch64_AM::encodeLogicalImmediate(imm: 1, regSize: 32));
3477 constrainSelectedInstRegOperands(I&: *TstMI, TII, TRI, RBI);
3478 if (!emitSelect(Dst: Sel.getReg(Idx: 0), True: TReg, False: FReg, CC: AArch64CC::NE, MIB))
3479 return false;
3480 Sel.eraseFromParent();
3481 return true;
3482 }
3483 case TargetOpcode::G_ICMP: {
3484 if (Ty.isVector())
3485 return false;
3486
3487 if (Ty != LLT::scalar(SizeInBits: 32)) {
3488 LLVM_DEBUG(dbgs() << "G_ICMP result has type: " << Ty
3489 << ", expected: " << LLT::scalar(32) << '\n');
3490 return false;
3491 }
3492
3493 auto &PredOp = I.getOperand(i: 1);
3494 emitIntegerCompare(LHS&: I.getOperand(i: 2), RHS&: I.getOperand(i: 3), Predicate&: PredOp, MIRBuilder&: MIB);
3495 auto Pred = static_cast<CmpInst::Predicate>(PredOp.getPredicate());
3496 const AArch64CC::CondCode InvCC = changeICMPPredToAArch64CC(
3497 P: CmpInst::getInversePredicate(pred: Pred), RHS: I.getOperand(i: 3).getReg(), MRI: &MRI);
3498 emitCSINC(/*Dst=*/I.getOperand(i: 0).getReg(), /*Src1=*/AArch64::WZR,
3499 /*Src2=*/AArch64::WZR, Pred: InvCC, MIRBuilder&: MIB);
3500 I.eraseFromParent();
3501 return true;
3502 }
3503
3504 case TargetOpcode::G_FCMP: {
3505 CmpInst::Predicate Pred =
3506 static_cast<CmpInst::Predicate>(I.getOperand(i: 1).getPredicate());
3507 if (!emitFPCompare(LHS: I.getOperand(i: 2).getReg(), RHS: I.getOperand(i: 3).getReg(), MIRBuilder&: MIB,
3508 Pred) ||
3509 !emitCSetForFCmp(Dst: I.getOperand(i: 0).getReg(), Pred, MIRBuilder&: MIB))
3510 return false;
3511 I.eraseFromParent();
3512 return true;
3513 }
3514 case TargetOpcode::G_VASTART:
3515 return STI.isTargetDarwin() ? selectVaStartDarwin(I, MF, MRI)
3516 : selectVaStartAAPCS(I, MF, MRI);
3517 case TargetOpcode::G_INTRINSIC:
3518 return selectIntrinsic(I, MRI);
3519 case TargetOpcode::G_INTRINSIC_W_SIDE_EFFECTS:
3520 return selectIntrinsicWithSideEffects(I, MRI);
3521 case TargetOpcode::G_IMPLICIT_DEF: {
3522 I.setDesc(TII.get(Opcode: TargetOpcode::IMPLICIT_DEF));
3523 const LLT DstTy = MRI.getType(Reg: I.getOperand(i: 0).getReg());
3524 const Register DstReg = I.getOperand(i: 0).getReg();
3525 const RegisterBank &DstRB = *RBI.getRegBank(Reg: DstReg, MRI, TRI);
3526 const TargetRegisterClass *DstRC = getRegClassForTypeOnBank(Ty: DstTy, RB: DstRB);
3527 RBI.constrainGenericRegister(Reg: DstReg, RC: *DstRC, MRI);
3528 return true;
3529 }
3530 case TargetOpcode::G_BLOCK_ADDR: {
3531 Function *BAFn = I.getOperand(i: 1).getBlockAddress()->getFunction();
3532 if (std::optional<uint16_t> BADisc =
3533 STI.getPtrAuthBlockAddressDiscriminatorIfEnabled(ParentFn: *BAFn)) {
3534 MIB.buildInstr(Opcode: AArch64::MOVaddrPAC)
3535 .addBlockAddress(BA: I.getOperand(i: 1).getBlockAddress())
3536 .addImm(Val: AArch64PACKey::IA)
3537 .addReg(/*AddrDisc=*/RegNo: AArch64::XZR)
3538 .addImm(Val: *BADisc)
3539 .constrainAllUses(TII, TRI, RBI);
3540 MIB.buildCopy(Res: I.getOperand(i: 0).getReg(), Op: Register(AArch64::X16));
3541 RBI.constrainGenericRegister(Reg: I.getOperand(i: 0).getReg(),
3542 RC: AArch64::GPR64RegClass, MRI);
3543 I.eraseFromParent();
3544 return true;
3545 }
3546 if (TM.getCodeModel() == CodeModel::Large && !TM.isPositionIndependent()) {
3547 materializeLargeCMVal(I, V: I.getOperand(i: 1).getBlockAddress(), OpFlags: 0);
3548 I.eraseFromParent();
3549 return true;
3550 } else {
3551 I.setDesc(TII.get(Opcode: AArch64::MOVaddrBA));
3552 auto MovMI = BuildMI(BB&: MBB, I, MIMD: I.getDebugLoc(), MCID: TII.get(Opcode: AArch64::MOVaddrBA),
3553 DestReg: I.getOperand(i: 0).getReg())
3554 .addBlockAddress(BA: I.getOperand(i: 1).getBlockAddress(),
3555 /* Offset */ 0, TargetFlags: AArch64II::MO_PAGE)
3556 .addBlockAddress(
3557 BA: I.getOperand(i: 1).getBlockAddress(), /* Offset */ 0,
3558 TargetFlags: AArch64II::MO_NC | AArch64II::MO_PAGEOFF);
3559 I.eraseFromParent();
3560 constrainSelectedInstRegOperands(I&: *MovMI, TII, TRI, RBI);
3561 return true;
3562 }
3563 }
3564 case AArch64::G_DUP: {
3565 // When the scalar of G_DUP is an s8/s16 gpr, they can't be selected by
3566 // imported patterns. Do it manually here. Avoiding generating s16 gpr is
3567 // difficult because at RBS we may end up pessimizing the fpr case if we
3568 // decided to add an anyextend to fix this. Manual selection is the most
3569 // robust solution for now.
3570 if (RBI.getRegBank(Reg: I.getOperand(i: 1).getReg(), MRI, TRI)->getID() !=
3571 AArch64::GPRRegBankID)
3572 return false; // We expect the fpr regbank case to be imported.
3573 LLT VecTy = MRI.getType(Reg: I.getOperand(i: 0).getReg());
3574 if (VecTy == LLT::fixed_vector(NumElements: 8, ScalarSizeInBits: 8))
3575 I.setDesc(TII.get(Opcode: AArch64::DUPv8i8gpr));
3576 else if (VecTy == LLT::fixed_vector(NumElements: 16, ScalarSizeInBits: 8))
3577 I.setDesc(TII.get(Opcode: AArch64::DUPv16i8gpr));
3578 else if (VecTy == LLT::fixed_vector(NumElements: 4, ScalarSizeInBits: 16))
3579 I.setDesc(TII.get(Opcode: AArch64::DUPv4i16gpr));
3580 else if (VecTy == LLT::fixed_vector(NumElements: 8, ScalarSizeInBits: 16))
3581 I.setDesc(TII.get(Opcode: AArch64::DUPv8i16gpr));
3582 else
3583 return false;
3584 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
3585 return true;
3586 }
3587 case TargetOpcode::G_BUILD_VECTOR:
3588 return selectBuildVector(I, MRI);
3589 case TargetOpcode::G_MERGE_VALUES:
3590 return selectMergeValues(I, MRI);
3591 case TargetOpcode::G_UNMERGE_VALUES:
3592 return selectUnmergeValues(I, MRI);
3593 case TargetOpcode::G_SHUFFLE_VECTOR:
3594 return selectShuffleVector(I, MRI);
3595 case TargetOpcode::G_EXTRACT_VECTOR_ELT:
3596 return selectExtractElt(I, MRI);
3597 case TargetOpcode::G_CONCAT_VECTORS:
3598 return selectConcatVectors(I, MRI);
3599 case TargetOpcode::G_JUMP_TABLE:
3600 return selectJumpTable(I, MRI);
3601 case TargetOpcode::G_MEMCPY:
3602 case TargetOpcode::G_MEMCPY_INLINE:
3603 case TargetOpcode::G_MEMMOVE:
3604 case TargetOpcode::G_MEMSET:
3605 case TargetOpcode::G_MEMSET_INLINE:
3606 assert(STI.hasMOPS() && "Shouldn't get here without +mops feature");
3607 return selectMOPS(I, MRI);
3608 }
3609
3610 return false;
3611}
3612
3613bool AArch64InstructionSelector::selectAndRestoreState(MachineInstr &I) {
3614 MachineIRBuilderState OldMIBState = MIB.getState();
3615 bool Success = select(I);
3616 MIB.setState(OldMIBState);
3617 return Success;
3618}
3619
3620bool AArch64InstructionSelector::selectMOPS(MachineInstr &GI,
3621 MachineRegisterInfo &MRI) {
3622 unsigned Mopcode;
3623 switch (GI.getOpcode()) {
3624 case TargetOpcode::G_MEMCPY:
3625 case TargetOpcode::G_MEMCPY_INLINE:
3626 Mopcode = AArch64::MOPSMemoryCopyPseudo;
3627 break;
3628 case TargetOpcode::G_MEMMOVE:
3629 Mopcode = AArch64::MOPSMemoryMovePseudo;
3630 break;
3631 case TargetOpcode::G_MEMSET:
3632 case TargetOpcode::G_MEMSET_INLINE:
3633 // For tagged memset see llvm.aarch64.mops.memset.tag
3634 Mopcode = AArch64::MOPSMemorySetPseudo;
3635 break;
3636 }
3637
3638 auto &DstPtr = GI.getOperand(i: 0);
3639 auto &SrcOrVal = GI.getOperand(i: 1);
3640 auto &Size = GI.getOperand(i: 2);
3641
3642 // Create copies of the registers that can be clobbered.
3643 const Register DstPtrCopy = MRI.cloneVirtualRegister(VReg: DstPtr.getReg());
3644 const Register SrcValCopy = MRI.cloneVirtualRegister(VReg: SrcOrVal.getReg());
3645 const Register SizeCopy = MRI.cloneVirtualRegister(VReg: Size.getReg());
3646
3647 const bool IsSet = Mopcode == AArch64::MOPSMemorySetPseudo;
3648 const auto &SrcValRegClass =
3649 IsSet ? AArch64::GPR64RegClass : AArch64::GPR64commonRegClass;
3650
3651 // Constrain to specific registers
3652 RBI.constrainGenericRegister(Reg: DstPtrCopy, RC: AArch64::GPR64commonRegClass, MRI);
3653 RBI.constrainGenericRegister(Reg: SrcValCopy, RC: SrcValRegClass, MRI);
3654 RBI.constrainGenericRegister(Reg: SizeCopy, RC: AArch64::GPR64RegClass, MRI);
3655
3656 MIB.buildCopy(Res: DstPtrCopy, Op: DstPtr);
3657 MIB.buildCopy(Res: SrcValCopy, Op: SrcOrVal);
3658 MIB.buildCopy(Res: SizeCopy, Op: Size);
3659
3660 // New instruction uses the copied registers because it must update them.
3661 // The defs are not used since they don't exist in G_MEM*. They are still
3662 // tied.
3663 // Note: order of operands is different from G_MEMSET, G_MEMCPY, G_MEMMOVE
3664 Register DefDstPtr = MRI.createVirtualRegister(RegClass: &AArch64::GPR64commonRegClass);
3665 Register DefSize = MRI.createVirtualRegister(RegClass: &AArch64::GPR64RegClass);
3666 if (IsSet) {
3667 MIB.buildInstr(Opc: Mopcode, DstOps: {DefDstPtr, DefSize},
3668 SrcOps: {DstPtrCopy, SizeCopy, SrcValCopy})
3669 .setOperandDead(5); // implicit-def $nzcv
3670 } else {
3671 Register DefSrcPtr = MRI.createVirtualRegister(RegClass: &SrcValRegClass);
3672 MIB.buildInstr(Opc: Mopcode, DstOps: {DefDstPtr, DefSrcPtr, DefSize},
3673 SrcOps: {DstPtrCopy, SrcValCopy, SizeCopy})
3674 .setOperandDead(6); // implicit-def $nzcv
3675 }
3676
3677 GI.eraseFromParent();
3678 return true;
3679}
3680
3681bool AArch64InstructionSelector::selectBrJT(MachineInstr &I,
3682 MachineRegisterInfo &MRI) {
3683 assert(I.getOpcode() == TargetOpcode::G_BRJT && "Expected G_BRJT");
3684 Register JTAddr = I.getOperand(i: 0).getReg();
3685 unsigned JTI = I.getOperand(i: 1).getIndex();
3686 Register Index = I.getOperand(i: 2).getReg();
3687
3688 MF->getInfo<AArch64FunctionInfo>()->setJumpTableEntryInfo(Idx: JTI, Size: 4, PCRelSym: nullptr);
3689
3690 // With aarch64-jump-table-hardening, we only expand the jump table dispatch
3691 // sequence later, to guarantee the integrity of the intermediate values.
3692 if (MF->getFunction().hasFnAttribute(Kind: "aarch64-jump-table-hardening")) {
3693 CodeModel::Model CM = TM.getCodeModel();
3694 if (STI.isTargetMachO()) {
3695 if (CM != CodeModel::Small && CM != CodeModel::Large)
3696 report_fatal_error(reason: "Unsupported code-model for hardened jump-table");
3697 } else {
3698 // Note that COFF support would likely also need JUMP_TABLE_DEBUG_INFO.
3699 assert(STI.isTargetELF() &&
3700 "jump table hardening only supported on MachO/ELF");
3701 if (CM != CodeModel::Small)
3702 report_fatal_error(reason: "Unsupported code-model for hardened jump-table");
3703 }
3704
3705 MIB.buildCopy(Res: {AArch64::X16}, Op: I.getOperand(i: 2).getReg());
3706 MIB.buildInstr(Opcode: AArch64::BR_JumpTable)
3707 .addJumpTableIndex(Idx: I.getOperand(i: 1).getIndex());
3708 I.eraseFromParent();
3709 return true;
3710 }
3711
3712 Register TargetReg = MRI.createVirtualRegister(RegClass: &AArch64::GPR64RegClass);
3713 Register ScratchReg = MRI.createVirtualRegister(RegClass: &AArch64::GPR64spRegClass);
3714
3715 auto JumpTableInst = MIB.buildInstr(Opc: AArch64::JumpTableDest32,
3716 DstOps: {TargetReg, ScratchReg}, SrcOps: {JTAddr, Index})
3717 .addJumpTableIndex(Idx: JTI);
3718 // Save the jump table info.
3719 MIB.buildInstr(Opc: TargetOpcode::JUMP_TABLE_DEBUG_INFO, DstOps: {},
3720 SrcOps: {static_cast<int64_t>(JTI)});
3721 // Build the indirect branch.
3722 MIB.buildInstr(Opc: AArch64::BR, DstOps: {}, SrcOps: {TargetReg});
3723 I.eraseFromParent();
3724 constrainSelectedInstRegOperands(I&: *JumpTableInst, TII, TRI, RBI);
3725 return true;
3726}
3727
3728bool AArch64InstructionSelector::selectJumpTable(MachineInstr &I,
3729 MachineRegisterInfo &MRI) {
3730 assert(I.getOpcode() == TargetOpcode::G_JUMP_TABLE && "Expected jump table");
3731 assert(I.getOperand(1).isJTI() && "Jump table op should have a JTI!");
3732
3733 Register DstReg = I.getOperand(i: 0).getReg();
3734 unsigned JTI = I.getOperand(i: 1).getIndex();
3735 // We generate a MOVaddrJT which will get expanded to an ADRP + ADD later.
3736 auto MovMI =
3737 MIB.buildInstr(Opc: AArch64::MOVaddrJT, DstOps: {DstReg}, SrcOps: {})
3738 .addJumpTableIndex(Idx: JTI, TargetFlags: AArch64II::MO_PAGE)
3739 .addJumpTableIndex(Idx: JTI, TargetFlags: AArch64II::MO_NC | AArch64II::MO_PAGEOFF);
3740 I.eraseFromParent();
3741 constrainSelectedInstRegOperands(I&: *MovMI, TII, TRI, RBI);
3742 return true;
3743}
3744
3745bool AArch64InstructionSelector::selectTLSLocalExecELF(
3746 const GlobalValue *GV, MachineInstr &I, MachineRegisterInfo &MRI) {
3747 auto ConstrainRegOps = [&](MachineInstrBuilder MIB) {
3748 constrainSelectedInstRegOperands(I&: *MIB, TII, TRI, RBI);
3749 };
3750 Register ThreadBase = MRI.createGenericVirtualRegister(Ty: LLT::pointer(AddressSpace: 0, SizeInBits: 64));
3751 ConstrainRegOps(MIB.buildInstr(Opc: AArch64::MOVbaseTLS, DstOps: {ThreadBase}, SrcOps: {}));
3752
3753 switch (MF->getTarget().Options.TLSSize) {
3754 default:
3755 llvm_unreachable("Unexpected TLS size");
3756 case 12: {
3757 // add x0, x0, :tprel_lo12:a
3758 ConstrainRegOps(
3759 MIB.buildInstr(Opc: AArch64::ADDXri, DstOps: {I.getOperand(i: 0).getReg()},
3760 SrcOps: {ThreadBase})
3761 .addGlobalAddress(GV, Offset: 0, TargetFlags: AArch64II::MO_TLS | AArch64II::MO_PAGEOFF)
3762 .addImm(Val: 0));
3763 break;
3764 }
3765 case 24: {
3766 // add x0, x0, :tprel_hi12:a
3767 // add x0, x0, :tprel_lo12_nc:a
3768 Register Addr = MRI.createVirtualRegister(RegClass: &AArch64::GPR64RegClass);
3769 ConstrainRegOps(
3770 MIB.buildInstr(Opc: AArch64::ADDXri, DstOps: {Addr}, SrcOps: {ThreadBase})
3771 .addGlobalAddress(GV, Offset: 0, TargetFlags: AArch64II::MO_TLS | AArch64II::MO_HI12)
3772 .addImm(Val: 0));
3773 ConstrainRegOps(
3774 MIB.buildInstr(Opc: AArch64::ADDXri, DstOps: {I.getOperand(i: 0).getReg()}, SrcOps: {Addr})
3775 .addGlobalAddress(GV, Offset: 0,
3776 TargetFlags: AArch64II::MO_TLS | AArch64II::MO_PAGEOFF |
3777 AArch64II::MO_NC)
3778 .addImm(Val: 0));
3779 break;
3780 }
3781 case 32: {
3782 // movz x0, #:tprel_g1:a
3783 // movk x0, #:tprel_g0_nc:a
3784 // add x0, x1, x0
3785 Register Addr = MRI.createVirtualRegister(RegClass: &AArch64::GPR64RegClass);
3786 ConstrainRegOps(
3787 MIB.buildInstr(Opc: AArch64::MOVZXi, DstOps: {Addr}, SrcOps: {})
3788 .addGlobalAddress(GV, Offset: 0, TargetFlags: AArch64II::MO_TLS | AArch64II::MO_G1)
3789 .addImm(Val: 16));
3790 Register Addr2 = MRI.createVirtualRegister(RegClass: &AArch64::GPR64RegClass);
3791 ConstrainRegOps(MIB.buildInstr(Opc: AArch64::MOVKXi, DstOps: {Addr2}, SrcOps: {Addr})
3792 .addGlobalAddress(GV, Offset: 0,
3793 TargetFlags: AArch64II::MO_TLS | AArch64II::MO_G0 |
3794 AArch64II::MO_NC)
3795 .addImm(Val: 0));
3796 ConstrainRegOps(MIB.buildInstr(Opc: AArch64::ADDXrr, DstOps: {I.getOperand(i: 0).getReg()},
3797 SrcOps: {ThreadBase, Addr2}));
3798 break;
3799 }
3800 case 48: {
3801 // movz x0, #:tprel_g2:a
3802 // movk x0, #:tprel_g1_nc:a
3803 // movk x0, #:tprel_g0_nc:a
3804 // add x0, x1, x0
3805 Register Addr = MRI.createVirtualRegister(RegClass: &AArch64::GPR64RegClass);
3806 ConstrainRegOps(
3807 MIB.buildInstr(Opc: AArch64::MOVZXi, DstOps: {Addr}, SrcOps: {})
3808 .addGlobalAddress(GV, Offset: 0, TargetFlags: AArch64II::MO_TLS | AArch64II::MO_G2)
3809 .addImm(Val: 32));
3810 Register Addr2 = MRI.createVirtualRegister(RegClass: &AArch64::GPR64RegClass);
3811 ConstrainRegOps(MIB.buildInstr(Opc: AArch64::MOVKXi, DstOps: {Addr2}, SrcOps: {Addr})
3812 .addGlobalAddress(GV, Offset: 0,
3813 TargetFlags: AArch64II::MO_TLS | AArch64II::MO_G1 |
3814 AArch64II::MO_NC)
3815 .addImm(Val: 16));
3816 Register Addr3 = MRI.createVirtualRegister(RegClass: &AArch64::GPR64RegClass);
3817 ConstrainRegOps(MIB.buildInstr(Opc: AArch64::MOVKXi, DstOps: {Addr3}, SrcOps: {Addr2})
3818 .addGlobalAddress(GV, Offset: 0,
3819 TargetFlags: AArch64II::MO_TLS | AArch64II::MO_G0 |
3820 AArch64II::MO_NC)
3821 .addImm(Val: 0));
3822 ConstrainRegOps(MIB.buildInstr(Opc: AArch64::ADDXrr, DstOps: {I.getOperand(i: 0).getReg()},
3823 SrcOps: {ThreadBase, Addr3}));
3824 break;
3825 }
3826 }
3827 I.eraseFromParent();
3828 return true;
3829}
3830
3831// TLS lowering below mirrors the corresponding DAGISel implementation.
3832// See LowerELFTLSModel() for details.
3833bool AArch64InstructionSelector::selectTLSGlobalValueELF(
3834 MachineInstr &I, MachineRegisterInfo &MRI) {
3835 const GlobalValue *GV = I.getOperand(i: 1).getGlobal();
3836 auto *FuncInfo = MF->getInfo<AArch64FunctionInfo>();
3837 TLSModel::Model Model =
3838 AArch64::getELFTLSModel(GV, TM, HasELFSignedGOT: FuncInfo->hasELFSignedGOT());
3839
3840 Register TPOff = MRI.createVirtualRegister(RegClass: &AArch64::GPR64commonRegClass);
3841 switch (Model) {
3842 case TLSModel::LocalExec:
3843 return selectTLSLocalExecELF(GV, I, MRI);
3844 case TLSModel::InitialExec:
3845 MIB.buildInstr(Opc: AArch64::LOADgot, DstOps: {TPOff}, SrcOps: {})
3846 .addGlobalAddress(GV, Offset: 0, TargetFlags: AArch64II::MO_TLS);
3847 break;
3848 case TLSModel::LocalDynamic:
3849 case TLSModel::GeneralDynamic: {
3850#ifndef NDEBUG
3851 SMEAttrs Attrs = MF->getInfo<AArch64FunctionInfo>()->getSMEFnAttrs();
3852 assert(!Attrs.hasZAState() && !Attrs.hasStreamingInterfaceOrBody() &&
3853 !Attrs.hasStreamingCompatibleInterface() &&
3854 "unsupported SME features reached GlobalISel TLS lowering");
3855#endif
3856 unsigned Opcode = FuncInfo->hasELFSignedGOT()
3857 ? AArch64::TLSDESC_AUTH_CALLSEQ
3858 : AArch64::TLSDESC_CALLSEQ;
3859
3860 if (TLSModel::GeneralDynamic == Model) {
3861 MIB.buildInstr(Opc: Opcode, DstOps: {}, SrcOps: {}).addGlobalAddress(GV, Offset: 0, TargetFlags: AArch64II::MO_TLS);
3862 MIB.buildCopy(Res: TPOff, Op: Register(AArch64::X0));
3863 break;
3864 }
3865 assert(TLSModel::LocalDynamic == Model);
3866 // These accesses will need deduplicating if there's more than one.
3867 FuncInfo->incNumLocalDynamicTLSAccesses();
3868
3869 MIB.buildInstr(Opc: Opcode, DstOps: {}, SrcOps: {})
3870 .addExternalSymbol(FnName: "_TLS_MODULE_BASE_", TargetFlags: AArch64II::MO_TLS);
3871 auto Copy = MIB.buildCopy(Res: LLT::scalar(SizeInBits: 64), Op: Register(AArch64::X0));
3872 auto Add1 =
3873 MIB.buildInstr(Opc: AArch64::ADDXri, DstOps: {LLT::scalar(SizeInBits: 64)}, SrcOps: {Copy.getReg(Idx: 0)})
3874 .addGlobalAddress(GV, Offset: 0, TargetFlags: AArch64II::MO_TLS | AArch64II::MO_HI12)
3875 .addImm(Val: 0);
3876 auto Add2 =
3877 MIB.buildInstr(Opc: AArch64::ADDXri, DstOps: {TPOff}, SrcOps: {Add1.getReg(Idx: 0)})
3878 .addGlobalAddress(GV, Offset: 0,
3879 TargetFlags: AArch64II::MO_TLS | AArch64II::MO_PAGEOFF |
3880 AArch64II::MO_NC)
3881 .addImm(Val: 0);
3882 constrainSelectedInstRegOperands(I&: *Add1, TII, TRI, RBI);
3883 constrainSelectedInstRegOperands(I&: *Add2, TII, TRI, RBI);
3884 }
3885 }
3886 Register ThreadBase = MRI.createGenericVirtualRegister(Ty: LLT::pointer(AddressSpace: 0, SizeInBits: 64));
3887 MIB.buildInstr(Opc: AArch64::MOVbaseTLS, DstOps: {ThreadBase}, SrcOps: {});
3888 auto Add = MIB.buildInstr(Opc: AArch64::ADDXrr, DstOps: {I.getOperand(i: 0).getReg()},
3889 SrcOps: {ThreadBase, TPOff});
3890 constrainSelectedInstRegOperands(I&: *Add, TII, TRI, RBI);
3891
3892 I.eraseFromParent();
3893 return true;
3894}
3895
3896bool AArch64InstructionSelector::selectTLSGlobalValueMachO(
3897 MachineInstr &I, MachineRegisterInfo &MRI) {
3898 const auto &GlobalOp = I.getOperand(i: 1);
3899 assert(GlobalOp.getOffset() == 0 &&
3900 "Shouldn't have an offset on TLS globals!");
3901
3902 const GlobalValue &GV = *GlobalOp.getGlobal();
3903 MF->getFrameInfo().setAdjustsStack(true);
3904 auto LoadGOT =
3905 MIB.buildInstr(Opc: AArch64::LOADgot, DstOps: {&AArch64::GPR64commonRegClass}, SrcOps: {})
3906 .addGlobalAddress(GV: &GV, Offset: 0, TargetFlags: AArch64II::MO_TLS);
3907
3908 auto Load = MIB.buildInstr(Opc: AArch64::LDRXui, DstOps: {&AArch64::GPR64commonRegClass},
3909 SrcOps: {LoadGOT.getReg(Idx: 0)})
3910 .addImm(Val: 0);
3911
3912 MIB.buildCopy(Res: Register(AArch64::X0), Op: LoadGOT.getReg(Idx: 0));
3913 // TLS calls preserve all registers except those that absolutely must be
3914 // trashed: X0 (it takes an argument), LR (it's a call) and NZCV (let's not be
3915 // silly).
3916 unsigned Opcode = getBLRCallOpcode(MF: *MF);
3917
3918 // With ptrauth-calls, the tlv access thunk pointer is authenticated (IA, 0).
3919 if (MF->getFunction().hasFnAttribute(Kind: "ptrauth-calls")) {
3920 assert(Opcode == AArch64::BLR);
3921 Opcode = AArch64::BLRAAZ;
3922 }
3923
3924 MIB.buildInstr(Opc: Opcode, DstOps: {}, SrcOps: {Load})
3925 .setOperandDead(1) // implicit-def $lr
3926 .addUse(RegNo: AArch64::X0, Flags: RegState::Implicit)
3927 .addDef(RegNo: AArch64::X0, Flags: RegState::Implicit)
3928 .addRegMask(Mask: TRI.getTLSCallPreservedMask());
3929
3930 MIB.buildCopy(Res: I.getOperand(i: 0).getReg(), Op: Register(AArch64::X0));
3931 RBI.constrainGenericRegister(Reg: I.getOperand(i: 0).getReg(), RC: AArch64::GPR64RegClass,
3932 MRI);
3933 I.eraseFromParent();
3934 return true;
3935}
3936
3937bool AArch64InstructionSelector::selectTLSGlobalValue(
3938 MachineInstr &I, MachineRegisterInfo &MRI) {
3939 // We don't support instructions with emulated TLS variables yet.
3940 if (TM.useEmulatedTLS())
3941 return false;
3942
3943 if (STI.isTargetELF())
3944 return selectTLSGlobalValueELF(I, MRI);
3945
3946 if (STI.isTargetMachO())
3947 return selectTLSGlobalValueMachO(I, MRI);
3948
3949 return false;
3950}
3951
3952MachineInstr *AArch64InstructionSelector::emitScalarToVector(
3953 unsigned EltSize, const TargetRegisterClass *DstRC, Register Scalar,
3954 MachineIRBuilder &MIRBuilder) const {
3955 auto Undef = MIRBuilder.buildInstr(Opc: TargetOpcode::IMPLICIT_DEF, DstOps: {DstRC}, SrcOps: {});
3956
3957 auto BuildFn = [&](unsigned SubregIndex) {
3958 auto Ins =
3959 MIRBuilder
3960 .buildInstr(Opc: TargetOpcode::INSERT_SUBREG, DstOps: {DstRC}, SrcOps: {Undef, Scalar})
3961 .addImm(Val: SubregIndex);
3962 constrainSelectedInstRegOperands(I&: *Undef, TII, TRI, RBI);
3963 constrainSelectedInstRegOperands(I&: *Ins, TII, TRI, RBI);
3964 return &*Ins;
3965 };
3966
3967 switch (EltSize) {
3968 case 8:
3969 return BuildFn(AArch64::bsub);
3970 case 16:
3971 return BuildFn(AArch64::hsub);
3972 case 32:
3973 return BuildFn(AArch64::ssub);
3974 case 64:
3975 return BuildFn(AArch64::dsub);
3976 default:
3977 return nullptr;
3978 }
3979}
3980
3981MachineInstr *
3982AArch64InstructionSelector::emitNarrowVector(Register DstReg, Register SrcReg,
3983 MachineIRBuilder &MIB,
3984 MachineRegisterInfo &MRI) const {
3985 LLT DstTy = MRI.getType(Reg: DstReg);
3986 const TargetRegisterClass *RC =
3987 getRegClassForTypeOnBank(Ty: DstTy, RB: *RBI.getRegBank(Reg: SrcReg, MRI, TRI));
3988 if (RC != &AArch64::FPR32RegClass && RC != &AArch64::FPR64RegClass) {
3989 LLVM_DEBUG(dbgs() << "Unsupported register class!\n");
3990 return nullptr;
3991 }
3992 unsigned SubReg = 0;
3993 if (!getSubRegForClass(RC, TRI, SubReg))
3994 return nullptr;
3995 if (SubReg != AArch64::ssub && SubReg != AArch64::dsub) {
3996 LLVM_DEBUG(dbgs() << "Unsupported destination size! ("
3997 << DstTy.getSizeInBits() << "\n");
3998 return nullptr;
3999 }
4000 auto Copy = MIB.buildInstr(Opc: TargetOpcode::COPY, DstOps: {DstReg}, SrcOps: {})
4001 .addReg(RegNo: SrcReg, Flags: {}, SubReg);
4002 RBI.constrainGenericRegister(Reg: DstReg, RC: *RC, MRI);
4003 return Copy;
4004}
4005
4006bool AArch64InstructionSelector::selectMergeValues(
4007 MachineInstr &I, MachineRegisterInfo &MRI) {
4008 assert(I.getOpcode() == TargetOpcode::G_MERGE_VALUES && "unexpected opcode");
4009 const LLT DstTy = MRI.getType(Reg: I.getOperand(i: 0).getReg());
4010 const LLT SrcTy = MRI.getType(Reg: I.getOperand(i: 1).getReg());
4011 assert(!DstTy.isVector() && !SrcTy.isVector() && "invalid merge operation");
4012 const RegisterBank &RB = *RBI.getRegBank(Reg: I.getOperand(i: 1).getReg(), MRI, TRI);
4013
4014 if (I.getNumOperands() != 3)
4015 return false;
4016
4017 // Merging 2 s64s into an s128.
4018 if (DstTy == LLT::scalar(SizeInBits: 128)) {
4019 if (SrcTy.getSizeInBits() != 64)
4020 return false;
4021 Register DstReg = I.getOperand(i: 0).getReg();
4022 Register Src1Reg = I.getOperand(i: 1).getReg();
4023 Register Src2Reg = I.getOperand(i: 2).getReg();
4024 auto Tmp = MIB.buildInstr(Opc: TargetOpcode::IMPLICIT_DEF, DstOps: {DstTy}, SrcOps: {});
4025 MachineInstr *InsMI = emitLaneInsert(DstReg: std::nullopt, SrcReg: Tmp.getReg(Idx: 0), EltReg: Src1Reg,
4026 /* LaneIdx */ 0, RB, MIRBuilder&: MIB);
4027 if (!InsMI)
4028 return false;
4029 MachineInstr *Ins2MI = emitLaneInsert(DstReg, SrcReg: InsMI->getOperand(i: 0).getReg(),
4030 EltReg: Src2Reg, /* LaneIdx */ 1, RB, MIRBuilder&: MIB);
4031 if (!Ins2MI)
4032 return false;
4033 constrainSelectedInstRegOperands(I&: *InsMI, TII, TRI, RBI);
4034 constrainSelectedInstRegOperands(I&: *Ins2MI, TII, TRI, RBI);
4035 I.eraseFromParent();
4036 return true;
4037 }
4038
4039 if (RB.getID() != AArch64::GPRRegBankID)
4040 return false;
4041
4042 if (DstTy.getSizeInBits() != 64 || SrcTy.getSizeInBits() != 32)
4043 return false;
4044
4045 auto *DstRC = &AArch64::GPR64RegClass;
4046 Register SubToRegDef = MRI.createVirtualRegister(RegClass: DstRC);
4047 MachineInstr &SubRegMI = *BuildMI(BB&: *I.getParent(), I, MIMD: I.getDebugLoc(),
4048 MCID: TII.get(Opcode: TargetOpcode::SUBREG_TO_REG))
4049 .addDef(RegNo: SubToRegDef)
4050 .addUse(RegNo: I.getOperand(i: 1).getReg())
4051 .addImm(Val: AArch64::sub_32);
4052 Register SubToRegDef2 = MRI.createVirtualRegister(RegClass: DstRC);
4053 // Need to anyext the second scalar before we can use bfm
4054 MachineInstr &SubRegMI2 = *BuildMI(BB&: *I.getParent(), I, MIMD: I.getDebugLoc(),
4055 MCID: TII.get(Opcode: TargetOpcode::SUBREG_TO_REG))
4056 .addDef(RegNo: SubToRegDef2)
4057 .addUse(RegNo: I.getOperand(i: 2).getReg())
4058 .addImm(Val: AArch64::sub_32);
4059 MachineInstr &BFM =
4060 *BuildMI(BB&: *I.getParent(), I, MIMD: I.getDebugLoc(), MCID: TII.get(Opcode: AArch64::BFMXri))
4061 .addDef(RegNo: I.getOperand(i: 0).getReg())
4062 .addUse(RegNo: SubToRegDef)
4063 .addUse(RegNo: SubToRegDef2)
4064 .addImm(Val: 32)
4065 .addImm(Val: 31);
4066 constrainSelectedInstRegOperands(I&: SubRegMI, TII, TRI, RBI);
4067 constrainSelectedInstRegOperands(I&: SubRegMI2, TII, TRI, RBI);
4068 constrainSelectedInstRegOperands(I&: BFM, TII, TRI, RBI);
4069 I.eraseFromParent();
4070 return true;
4071}
4072
4073static bool getLaneCopyOpcode(unsigned &CopyOpc, unsigned &ExtractSubReg,
4074 const unsigned EltSize) {
4075 // Choose a lane copy opcode and subregister based off of the size of the
4076 // vector's elements.
4077 switch (EltSize) {
4078 case 8:
4079 CopyOpc = AArch64::DUPi8;
4080 ExtractSubReg = AArch64::bsub;
4081 break;
4082 case 16:
4083 CopyOpc = AArch64::DUPi16;
4084 ExtractSubReg = AArch64::hsub;
4085 break;
4086 case 32:
4087 CopyOpc = AArch64::DUPi32;
4088 ExtractSubReg = AArch64::ssub;
4089 break;
4090 case 64:
4091 CopyOpc = AArch64::DUPi64;
4092 ExtractSubReg = AArch64::dsub;
4093 break;
4094 default:
4095 // Unknown size, bail out.
4096 LLVM_DEBUG(dbgs() << "Elt size '" << EltSize << "' unsupported.\n");
4097 return false;
4098 }
4099 return true;
4100}
4101
4102MachineInstr *AArch64InstructionSelector::emitExtractVectorElt(
4103 std::optional<Register> DstReg, const RegisterBank &DstRB, LLT ScalarTy,
4104 Register VecReg, unsigned LaneIdx, MachineIRBuilder &MIRBuilder) const {
4105 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
4106 unsigned CopyOpc = 0;
4107 unsigned ExtractSubReg = 0;
4108 if (!getLaneCopyOpcode(CopyOpc, ExtractSubReg, EltSize: ScalarTy.getSizeInBits())) {
4109 LLVM_DEBUG(
4110 dbgs() << "Couldn't determine lane copy opcode for instruction.\n");
4111 return nullptr;
4112 }
4113
4114 const TargetRegisterClass *DstRC =
4115 getRegClassForTypeOnBank(Ty: ScalarTy, RB: DstRB, GetAllRegSet: true);
4116 if (!DstRC) {
4117 LLVM_DEBUG(dbgs() << "Could not determine destination register class.\n");
4118 return nullptr;
4119 }
4120
4121 const RegisterBank &VecRB = *RBI.getRegBank(Reg: VecReg, MRI, TRI);
4122 const LLT &VecTy = MRI.getType(Reg: VecReg);
4123 const TargetRegisterClass *VecRC =
4124 getRegClassForTypeOnBank(Ty: VecTy, RB: VecRB, GetAllRegSet: true);
4125 if (!VecRC) {
4126 LLVM_DEBUG(dbgs() << "Could not determine source register class.\n");
4127 return nullptr;
4128 }
4129
4130 // The register that we're going to copy into.
4131 Register InsertReg = VecReg;
4132 if (!DstReg)
4133 DstReg = MRI.createVirtualRegister(RegClass: DstRC);
4134 // If the lane index is 0, we just use a subregister COPY.
4135 if (LaneIdx == 0) {
4136 auto Copy = MIRBuilder.buildInstr(Opc: TargetOpcode::COPY, DstOps: {*DstReg}, SrcOps: {})
4137 .addReg(RegNo: VecReg, Flags: {}, SubReg: ExtractSubReg);
4138 RBI.constrainGenericRegister(Reg: *DstReg, RC: *DstRC, MRI);
4139 return &*Copy;
4140 }
4141
4142 // Lane copies require 128-bit wide registers. If we're dealing with an
4143 // unpacked vector, then we need to move up to that width. Insert an implicit
4144 // def and a subregister insert to get us there.
4145 if (VecTy.getSizeInBits() != 128) {
4146 MachineInstr *ScalarToVector = emitScalarToVector(
4147 EltSize: VecTy.getSizeInBits(), DstRC: &AArch64::FPR128RegClass, Scalar: VecReg, MIRBuilder);
4148 if (!ScalarToVector)
4149 return nullptr;
4150 InsertReg = ScalarToVector->getOperand(i: 0).getReg();
4151 }
4152
4153 MachineInstr *LaneCopyMI =
4154 MIRBuilder.buildInstr(Opc: CopyOpc, DstOps: {*DstReg}, SrcOps: {InsertReg}).addImm(Val: LaneIdx);
4155 constrainSelectedInstRegOperands(I&: *LaneCopyMI, TII, TRI, RBI);
4156
4157 // Make sure that we actually constrain the initial copy.
4158 RBI.constrainGenericRegister(Reg: *DstReg, RC: *DstRC, MRI);
4159 return LaneCopyMI;
4160}
4161
4162bool AArch64InstructionSelector::selectExtractElt(
4163 MachineInstr &I, MachineRegisterInfo &MRI) {
4164 assert(I.getOpcode() == TargetOpcode::G_EXTRACT_VECTOR_ELT &&
4165 "unexpected opcode!");
4166 Register DstReg = I.getOperand(i: 0).getReg();
4167 const LLT NarrowTy = MRI.getType(Reg: DstReg);
4168 const Register SrcReg = I.getOperand(i: 1).getReg();
4169 const LLT WideTy = MRI.getType(Reg: SrcReg);
4170 assert(WideTy.getSizeInBits() >= NarrowTy.getSizeInBits() &&
4171 "source register size too small!");
4172 assert(!NarrowTy.isVector() && "cannot extract vector into vector!");
4173
4174 // Need the lane index to determine the correct copy opcode.
4175 MachineOperand &LaneIdxOp = I.getOperand(i: 2);
4176 assert(LaneIdxOp.isReg() && "Lane index operand was not a register?");
4177
4178 // Find the index to extract from.
4179 auto VRegAndVal = getIConstantVRegValWithLookThrough(VReg: LaneIdxOp.getReg(), MRI);
4180 if (!VRegAndVal)
4181 return false;
4182 unsigned LaneIdx = VRegAndVal->Value.getSExtValue();
4183
4184 const RegisterBank &DstRB = *RBI.getRegBank(Reg: DstReg, MRI, TRI);
4185 if (DstRB.getID() == AArch64::GPRRegBankID) {
4186 unsigned Opcode;
4187 switch (WideTy.getScalarSizeInBits()) {
4188 case 8:
4189 Opcode = AArch64::UMOVvi8;
4190 break;
4191 case 16:
4192 Opcode = AArch64::UMOVvi16;
4193 break;
4194 case 32:
4195 Opcode = AArch64::UMOVvi32;
4196 break;
4197 default:
4198 return false;
4199 }
4200
4201 if (WideTy.getSizeInBits() != 128) {
4202 MachineInstr *ScalarToVector = emitScalarToVector(
4203 EltSize: WideTy.getSizeInBits(), DstRC: &AArch64::FPR128RegClass, Scalar: SrcReg, MIRBuilder&: MIB);
4204 assert(ScalarToVector && "Didn't expect emitScalarToVector to fail!");
4205 I.getOperand(i: 1).setReg(ScalarToVector->getOperand(i: 0).getReg());
4206 }
4207
4208 I.setDesc(TII.get(Opcode));
4209 I.getOperand(i: 2).ChangeToImmediate(ImmVal: LaneIdx);
4210 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
4211 return true;
4212 }
4213
4214 MachineInstr *Extract = emitExtractVectorElt(DstReg, DstRB, ScalarTy: NarrowTy, VecReg: SrcReg,
4215 LaneIdx, MIRBuilder&: MIB);
4216 if (!Extract)
4217 return false;
4218
4219 I.eraseFromParent();
4220 return true;
4221}
4222
4223bool AArch64InstructionSelector::selectSplitVectorUnmerge(
4224 MachineInstr &I, MachineRegisterInfo &MRI) {
4225 unsigned NumElts = I.getNumOperands() - 1;
4226 Register SrcReg = I.getOperand(i: NumElts).getReg();
4227 const LLT NarrowTy = MRI.getType(Reg: I.getOperand(i: 0).getReg());
4228 const LLT SrcTy = MRI.getType(Reg: SrcReg);
4229
4230 assert(NarrowTy.isVector() && "Expected an unmerge into vectors");
4231 if (SrcTy.getSizeInBits() > 128) {
4232 LLVM_DEBUG(dbgs() << "Unexpected vector type for vec split unmerge");
4233 return false;
4234 }
4235
4236 // We implement a split vector operation by treating the sub-vectors as
4237 // scalars and extracting them.
4238 const RegisterBank &DstRB =
4239 *RBI.getRegBank(Reg: I.getOperand(i: 0).getReg(), MRI, TRI);
4240 for (unsigned OpIdx = 0; OpIdx < NumElts; ++OpIdx) {
4241 Register Dst = I.getOperand(i: OpIdx).getReg();
4242 MachineInstr *Extract =
4243 emitExtractVectorElt(DstReg: Dst, DstRB, ScalarTy: NarrowTy, VecReg: SrcReg, LaneIdx: OpIdx, MIRBuilder&: MIB);
4244 if (!Extract)
4245 return false;
4246 }
4247 I.eraseFromParent();
4248 return true;
4249}
4250
4251bool AArch64InstructionSelector::selectUnmergeValues(MachineInstr &I,
4252 MachineRegisterInfo &MRI) {
4253 assert(I.getOpcode() == TargetOpcode::G_UNMERGE_VALUES &&
4254 "unexpected opcode");
4255
4256 // The last operand is the vector source register, and every other operand is
4257 // a register to unpack into.
4258 unsigned NumElts = I.getNumOperands() - 1;
4259 Register SrcReg = I.getOperand(i: NumElts).getReg();
4260 Register LoReg = I.getOperand(i: 0).getReg();
4261 Register HiReg = I.getOperand(i: 1).getReg();
4262 const LLT NarrowTy = MRI.getType(Reg: LoReg);
4263 const LLT WideTy = MRI.getType(Reg: SrcReg);
4264 const RegisterBank &LoRB = *RBI.getRegBank(Reg: LoReg, MRI, TRI);
4265 const RegisterBank &HiRB = *RBI.getRegBank(Reg: HiReg, MRI, TRI);
4266 const RegisterBank &SrcRB = *RBI.getRegBank(Reg: SrcReg, MRI, TRI);
4267
4268 // Handle unmerging a 128-bit FPR value into two 64-bit GPR values.
4269 if (NarrowTy == LLT::scalar(SizeInBits: 64) && WideTy == LLT::scalar(SizeInBits: 128) &&
4270 LoRB.getID() == AArch64::GPRRegBankID &&
4271 HiRB.getID() == AArch64::GPRRegBankID &&
4272 SrcRB.getID() == AArch64::FPRRegBankID) {
4273 MachineInstr &Lo = *BuildMI(BB&: *I.getParent(), I, MIMD: I.getDebugLoc(),
4274 MCID: TII.get(Opcode: AArch64::UMOVvi64), DestReg: LoReg)
4275 .addUse(RegNo: SrcReg)
4276 .addImm(Val: 0);
4277 MachineInstr &Hi = *BuildMI(BB&: *I.getParent(), I, MIMD: I.getDebugLoc(),
4278 MCID: TII.get(Opcode: AArch64::UMOVvi64), DestReg: HiReg)
4279 .addUse(RegNo: SrcReg)
4280 .addImm(Val: 1);
4281 constrainSelectedInstRegOperands(I&: Lo, TII, TRI, RBI);
4282 constrainSelectedInstRegOperands(I&: Hi, TII, TRI, RBI);
4283 I.eraseFromParent();
4284 return true;
4285 }
4286
4287 // TODO: Handle other unmerges into GPRs and from scalars to scalars.
4288 if (LoRB.getID() != AArch64::FPRRegBankID ||
4289 HiRB.getID() != AArch64::FPRRegBankID) {
4290 LLVM_DEBUG(dbgs() << "Unmerging vector-to-gpr and scalar-to-scalar "
4291 "currently unsupported.\n");
4292 return false;
4293 }
4294
4295 assert(WideTy.getSizeInBits() > NarrowTy.getSizeInBits() &&
4296 "source register size too small!");
4297
4298 if (!NarrowTy.isScalar())
4299 return selectSplitVectorUnmerge(I, MRI);
4300
4301 // Choose a lane copy opcode and subregister based off of the size of the
4302 // vector's elements.
4303 unsigned CopyOpc = 0;
4304 unsigned ExtractSubReg = 0;
4305 if (!getLaneCopyOpcode(CopyOpc, ExtractSubReg, EltSize: NarrowTy.getSizeInBits()))
4306 return false;
4307
4308 // Set up for the lane copies.
4309 MachineBasicBlock &MBB = *I.getParent();
4310
4311 // Stores the registers we'll be copying from.
4312 SmallVector<Register, 4> InsertRegs;
4313
4314 // We'll use the first register twice, so we only need NumElts-1 registers.
4315 unsigned NumInsertRegs = NumElts - 1;
4316
4317 // If our elements fit into exactly 128 bits, then we can copy from the source
4318 // directly. Otherwise, we need to do a bit of setup with some subregister
4319 // inserts.
4320 if (NarrowTy.getSizeInBits() * NumElts == 128) {
4321 InsertRegs.assign(NumElts: NumInsertRegs, Elt: SrcReg);
4322 } else {
4323 // No. We have to perform subregister inserts. For each insert, create an
4324 // implicit def and a subregister insert, and save the register we create.
4325 // For scalar sources, treat as a pseudo-vector of NarrowTy elements.
4326 unsigned EltSize = WideTy.isVector() ? WideTy.getScalarSizeInBits()
4327 : NarrowTy.getSizeInBits();
4328 const TargetRegisterClass *RC = getRegClassForTypeOnBank(
4329 Ty: LLT::fixed_vector(NumElements: NumElts, ScalarSizeInBits: EltSize), RB: *RBI.getRegBank(Reg: SrcReg, MRI, TRI));
4330 unsigned SubReg = 0;
4331 bool Found = getSubRegForClass(RC, TRI, SubReg);
4332 (void)Found;
4333 assert(Found && "expected to find last operand's subeg idx");
4334 for (unsigned Idx = 0; Idx < NumInsertRegs; ++Idx) {
4335 Register ImpDefReg = MRI.createVirtualRegister(RegClass: &AArch64::FPR128RegClass);
4336 MachineInstr &ImpDefMI =
4337 *BuildMI(BB&: MBB, I, MIMD: I.getDebugLoc(), MCID: TII.get(Opcode: TargetOpcode::IMPLICIT_DEF),
4338 DestReg: ImpDefReg);
4339
4340 // Now, create the subregister insert from SrcReg.
4341 Register InsertReg = MRI.createVirtualRegister(RegClass: &AArch64::FPR128RegClass);
4342 MachineInstr &InsMI =
4343 *BuildMI(BB&: MBB, I, MIMD: I.getDebugLoc(),
4344 MCID: TII.get(Opcode: TargetOpcode::INSERT_SUBREG), DestReg: InsertReg)
4345 .addUse(RegNo: ImpDefReg)
4346 .addUse(RegNo: SrcReg)
4347 .addImm(Val: SubReg);
4348
4349 constrainSelectedInstRegOperands(I&: ImpDefMI, TII, TRI, RBI);
4350 constrainSelectedInstRegOperands(I&: InsMI, TII, TRI, RBI);
4351
4352 // Save the register so that we can copy from it after.
4353 InsertRegs.push_back(Elt: InsertReg);
4354 }
4355 }
4356
4357 // Now that we've created any necessary subregister inserts, we can
4358 // create the copies.
4359 //
4360 // Perform the first copy separately as a subregister copy.
4361 Register CopyTo = I.getOperand(i: 0).getReg();
4362 auto FirstCopy = MIB.buildInstr(Opc: TargetOpcode::COPY, DstOps: {CopyTo}, SrcOps: {})
4363 .addReg(RegNo: InsertRegs[0], Flags: {}, SubReg: ExtractSubReg);
4364 constrainSelectedInstRegOperands(I&: *FirstCopy, TII, TRI, RBI);
4365
4366 // Now, perform the remaining copies as vector lane copies.
4367 unsigned LaneIdx = 1;
4368 for (Register InsReg : InsertRegs) {
4369 Register CopyTo = I.getOperand(i: LaneIdx).getReg();
4370 MachineInstr &CopyInst =
4371 *BuildMI(BB&: MBB, I, MIMD: I.getDebugLoc(), MCID: TII.get(Opcode: CopyOpc), DestReg: CopyTo)
4372 .addUse(RegNo: InsReg)
4373 .addImm(Val: LaneIdx);
4374 constrainSelectedInstRegOperands(I&: CopyInst, TII, TRI, RBI);
4375 ++LaneIdx;
4376 }
4377
4378 // Separately constrain the first copy's destination. Because of the
4379 // limitation in constrainOperandRegClass, we can't guarantee that this will
4380 // actually be constrained. So, do it ourselves using the second operand.
4381 const TargetRegisterClass *RC =
4382 MRI.getRegClassOrNull(Reg: I.getOperand(i: 1).getReg());
4383 if (!RC) {
4384 LLVM_DEBUG(dbgs() << "Couldn't constrain copy destination.\n");
4385 return false;
4386 }
4387
4388 RBI.constrainGenericRegister(Reg: CopyTo, RC: *RC, MRI);
4389 I.eraseFromParent();
4390 return true;
4391}
4392
4393bool AArch64InstructionSelector::selectConcatVectors(
4394 MachineInstr &I, MachineRegisterInfo &MRI) {
4395 assert(I.getOpcode() == TargetOpcode::G_CONCAT_VECTORS &&
4396 "Unexpected opcode");
4397 Register Dst = I.getOperand(i: 0).getReg();
4398 Register Op1 = I.getOperand(i: 1).getReg();
4399 Register Op2 = I.getOperand(i: 2).getReg();
4400 MachineInstr *ConcatMI = emitVectorConcat(Dst, Op1, Op2, MIRBuilder&: MIB);
4401 if (!ConcatMI)
4402 return false;
4403 I.eraseFromParent();
4404 return true;
4405}
4406
4407unsigned
4408AArch64InstructionSelector::emitConstantPoolEntry(const Constant *CPVal,
4409 MachineFunction &MF) const {
4410 Type *CPTy = CPVal->getType();
4411 Align Alignment = MF.getDataLayout().getPrefTypeAlign(Ty: CPTy);
4412
4413 MachineConstantPool *MCP = MF.getConstantPool();
4414 return MCP->getConstantPoolIndex(C: CPVal, Alignment);
4415}
4416
4417MachineInstr *AArch64InstructionSelector::emitLoadFromConstantPool(
4418 const Constant *CPVal, MachineIRBuilder &MIRBuilder) const {
4419 const TargetRegisterClass *RC;
4420 unsigned Opc;
4421 bool IsTiny = TM.getCodeModel() == CodeModel::Tiny;
4422 unsigned Size = MIRBuilder.getDataLayout().getTypeStoreSize(Ty: CPVal->getType());
4423 switch (Size) {
4424 case 16:
4425 RC = &AArch64::FPR128RegClass;
4426 Opc = IsTiny ? AArch64::LDRQl : AArch64::LDRQui;
4427 break;
4428 case 8:
4429 RC = &AArch64::FPR64RegClass;
4430 Opc = IsTiny ? AArch64::LDRDl : AArch64::LDRDui;
4431 break;
4432 case 4:
4433 RC = &AArch64::FPR32RegClass;
4434 Opc = IsTiny ? AArch64::LDRSl : AArch64::LDRSui;
4435 break;
4436 case 2:
4437 RC = &AArch64::FPR16RegClass;
4438 Opc = AArch64::LDRHui;
4439 break;
4440 default:
4441 LLVM_DEBUG(dbgs() << "Could not load from constant pool of type "
4442 << *CPVal->getType());
4443 return nullptr;
4444 }
4445
4446 MachineInstr *LoadMI = nullptr;
4447 auto &MF = MIRBuilder.getMF();
4448 unsigned CPIdx = emitConstantPoolEntry(CPVal, MF);
4449 if (IsTiny && (Size == 16 || Size == 8 || Size == 4)) {
4450 // Use load(literal) for tiny code model.
4451 LoadMI = &*MIRBuilder.buildInstr(Opc, DstOps: {RC}, SrcOps: {}).addConstantPoolIndex(Idx: CPIdx);
4452 } else {
4453 auto Adrp =
4454 MIRBuilder.buildInstr(Opc: AArch64::ADRP, DstOps: {&AArch64::GPR64RegClass}, SrcOps: {})
4455 .addConstantPoolIndex(Idx: CPIdx, Offset: 0, TargetFlags: AArch64II::MO_PAGE);
4456
4457 LoadMI = &*MIRBuilder.buildInstr(Opc, DstOps: {RC}, SrcOps: {Adrp})
4458 .addConstantPoolIndex(
4459 Idx: CPIdx, Offset: 0, TargetFlags: AArch64II::MO_PAGEOFF | AArch64II::MO_NC);
4460
4461 constrainSelectedInstRegOperands(I&: *Adrp, TII, TRI, RBI);
4462 }
4463
4464 MachinePointerInfo PtrInfo = MachinePointerInfo::getConstantPool(MF);
4465 LoadMI->addMemOperand(MF, MO: MF.getMachineMemOperand(PtrInfo,
4466 F: MachineMemOperand::MOLoad,
4467 Size, BaseAlignment: Align(Size)));
4468 constrainSelectedInstRegOperands(I&: *LoadMI, TII, TRI, RBI);
4469 return LoadMI;
4470}
4471
4472/// Return an <Opcode, SubregIndex> pair to do an vector elt insert of a given
4473/// size and RB.
4474static std::pair<unsigned, unsigned>
4475getInsertVecEltOpInfo(const RegisterBank &RB, unsigned EltSize) {
4476 unsigned Opc, SubregIdx;
4477 if (RB.getID() == AArch64::GPRRegBankID) {
4478 if (EltSize == 8) {
4479 Opc = AArch64::INSvi8gpr;
4480 SubregIdx = AArch64::bsub;
4481 } else if (EltSize == 16) {
4482 Opc = AArch64::INSvi16gpr;
4483 SubregIdx = AArch64::ssub;
4484 } else if (EltSize == 32) {
4485 Opc = AArch64::INSvi32gpr;
4486 SubregIdx = AArch64::ssub;
4487 } else if (EltSize == 64) {
4488 Opc = AArch64::INSvi64gpr;
4489 SubregIdx = AArch64::dsub;
4490 } else {
4491 llvm_unreachable("invalid elt size!");
4492 }
4493 } else {
4494 if (EltSize == 8) {
4495 Opc = AArch64::INSvi8lane;
4496 SubregIdx = AArch64::bsub;
4497 } else if (EltSize == 16) {
4498 Opc = AArch64::INSvi16lane;
4499 SubregIdx = AArch64::hsub;
4500 } else if (EltSize == 32) {
4501 Opc = AArch64::INSvi32lane;
4502 SubregIdx = AArch64::ssub;
4503 } else if (EltSize == 64) {
4504 Opc = AArch64::INSvi64lane;
4505 SubregIdx = AArch64::dsub;
4506 } else {
4507 llvm_unreachable("invalid elt size!");
4508 }
4509 }
4510 return std::make_pair(x&: Opc, y&: SubregIdx);
4511}
4512
4513MachineInstr *AArch64InstructionSelector::emitInstr(
4514 unsigned Opcode, std::initializer_list<llvm::DstOp> DstOps,
4515 std::initializer_list<llvm::SrcOp> SrcOps, MachineIRBuilder &MIRBuilder,
4516 const ComplexRendererFns &RenderFns) const {
4517 assert(Opcode && "Expected an opcode?");
4518 assert(!isPreISelGenericOpcode(Opcode) &&
4519 "Function should only be used to produce selected instructions!");
4520 auto MI = MIRBuilder.buildInstr(Opc: Opcode, DstOps, SrcOps);
4521 if (RenderFns)
4522 for (auto &Fn : *RenderFns)
4523 Fn(MI);
4524 constrainSelectedInstRegOperands(I&: *MI, TII, TRI, RBI);
4525 return &*MI;
4526}
4527
4528MachineInstr *AArch64InstructionSelector::emitAddSub(
4529 const std::array<std::array<unsigned, 2>, 5> &AddrModeAndSizeToOpcode,
4530 Register Dst, MachineOperand &LHS, MachineOperand &RHS,
4531 MachineIRBuilder &MIRBuilder) const {
4532 MachineRegisterInfo &MRI = MIRBuilder.getMF().getRegInfo();
4533 assert(LHS.isReg() && RHS.isReg() && "Expected register operands?");
4534 auto Ty = MRI.getType(Reg: LHS.getReg());
4535 assert(!Ty.isVector() && "Expected a scalar or pointer?");
4536 unsigned Size = Ty.getSizeInBits();
4537 assert((Size == 32 || Size == 64) && "Expected a 32-bit or 64-bit type only");
4538 bool Is32Bit = Size == 32;
4539
4540 // INSTRri form with positive arithmetic immediate.
4541 if (auto Fns = selectArithImmed(Root&: RHS))
4542 return emitInstr(Opcode: AddrModeAndSizeToOpcode[0][Is32Bit], DstOps: {Dst}, SrcOps: {LHS},
4543 MIRBuilder, RenderFns: Fns);
4544
4545 // INSTRri form with negative arithmetic immediate.
4546 if (auto Fns = selectNegArithImmed(Root&: RHS))
4547 return emitInstr(Opcode: AddrModeAndSizeToOpcode[3][Is32Bit], DstOps: {Dst}, SrcOps: {LHS},
4548 MIRBuilder, RenderFns: Fns);
4549
4550 // INSTRrx form.
4551 if (auto Fns = selectArithExtendedRegister(Root&: RHS))
4552 return emitInstr(Opcode: AddrModeAndSizeToOpcode[4][Is32Bit], DstOps: {Dst}, SrcOps: {LHS},
4553 MIRBuilder, RenderFns: Fns);
4554
4555 // INSTRrs form.
4556 if (auto Fns = selectShiftedRegister(Root&: RHS))
4557 return emitInstr(Opcode: AddrModeAndSizeToOpcode[1][Is32Bit], DstOps: {Dst}, SrcOps: {LHS},
4558 MIRBuilder, RenderFns: Fns);
4559 return emitInstr(Opcode: AddrModeAndSizeToOpcode[2][Is32Bit], DstOps: {Dst}, SrcOps: {LHS, RHS},
4560 MIRBuilder);
4561}
4562
4563MachineInstr *
4564AArch64InstructionSelector::emitADD(Register DefReg, MachineOperand &LHS,
4565 MachineOperand &RHS,
4566 MachineIRBuilder &MIRBuilder) const {
4567 const std::array<std::array<unsigned, 2>, 5> OpcTable{
4568 ._M_elems: {{AArch64::ADDXri, AArch64::ADDWri},
4569 {AArch64::ADDXrs, AArch64::ADDWrs},
4570 {AArch64::ADDXrr, AArch64::ADDWrr},
4571 {AArch64::SUBXri, AArch64::SUBWri},
4572 {AArch64::ADDXrx, AArch64::ADDWrx}}};
4573 return emitAddSub(AddrModeAndSizeToOpcode: OpcTable, Dst: DefReg, LHS, RHS, MIRBuilder);
4574}
4575
4576MachineInstr *
4577AArch64InstructionSelector::emitADDS(Register Dst, MachineOperand &LHS,
4578 MachineOperand &RHS,
4579 MachineIRBuilder &MIRBuilder) const {
4580 const std::array<std::array<unsigned, 2>, 5> OpcTable{
4581 ._M_elems: {{AArch64::ADDSXri, AArch64::ADDSWri},
4582 {AArch64::ADDSXrs, AArch64::ADDSWrs},
4583 {AArch64::ADDSXrr, AArch64::ADDSWrr},
4584 {AArch64::SUBSXri, AArch64::SUBSWri},
4585 {AArch64::ADDSXrx, AArch64::ADDSWrx}}};
4586 return emitAddSub(AddrModeAndSizeToOpcode: OpcTable, Dst, LHS, RHS, MIRBuilder);
4587}
4588
4589MachineInstr *
4590AArch64InstructionSelector::emitSUBS(Register Dst, MachineOperand &LHS,
4591 MachineOperand &RHS,
4592 MachineIRBuilder &MIRBuilder) const {
4593 const std::array<std::array<unsigned, 2>, 5> OpcTable{
4594 ._M_elems: {{AArch64::SUBSXri, AArch64::SUBSWri},
4595 {AArch64::SUBSXrs, AArch64::SUBSWrs},
4596 {AArch64::SUBSXrr, AArch64::SUBSWrr},
4597 {AArch64::ADDSXri, AArch64::ADDSWri},
4598 {AArch64::SUBSXrx, AArch64::SUBSWrx}}};
4599 return emitAddSub(AddrModeAndSizeToOpcode: OpcTable, Dst, LHS, RHS, MIRBuilder);
4600}
4601
4602MachineInstr *
4603AArch64InstructionSelector::emitADCS(Register Dst, MachineOperand &LHS,
4604 MachineOperand &RHS,
4605 MachineIRBuilder &MIRBuilder) const {
4606 assert(LHS.isReg() && RHS.isReg() && "Expected register operands?");
4607 MachineRegisterInfo *MRI = MIRBuilder.getMRI();
4608 bool Is32Bit = (MRI->getType(Reg: LHS.getReg()).getSizeInBits() == 32);
4609 static const unsigned OpcTable[2] = {AArch64::ADCSXr, AArch64::ADCSWr};
4610 return emitInstr(Opcode: OpcTable[Is32Bit], DstOps: {Dst}, SrcOps: {LHS, RHS}, MIRBuilder);
4611}
4612
4613MachineInstr *
4614AArch64InstructionSelector::emitSBCS(Register Dst, MachineOperand &LHS,
4615 MachineOperand &RHS,
4616 MachineIRBuilder &MIRBuilder) const {
4617 assert(LHS.isReg() && RHS.isReg() && "Expected register operands?");
4618 MachineRegisterInfo *MRI = MIRBuilder.getMRI();
4619 bool Is32Bit = (MRI->getType(Reg: LHS.getReg()).getSizeInBits() == 32);
4620 static const unsigned OpcTable[2] = {AArch64::SBCSXr, AArch64::SBCSWr};
4621 return emitInstr(Opcode: OpcTable[Is32Bit], DstOps: {Dst}, SrcOps: {LHS, RHS}, MIRBuilder);
4622}
4623
4624MachineInstr *
4625AArch64InstructionSelector::emitCMP(MachineOperand &LHS, MachineOperand &RHS,
4626 MachineIRBuilder &MIRBuilder) const {
4627 MachineRegisterInfo &MRI = MIRBuilder.getMF().getRegInfo();
4628 bool Is32Bit = MRI.getType(Reg: LHS.getReg()).getSizeInBits() == 32;
4629 auto RC = Is32Bit ? &AArch64::GPR32RegClass : &AArch64::GPR64RegClass;
4630 return emitSUBS(Dst: MRI.createVirtualRegister(RegClass: RC), LHS, RHS, MIRBuilder);
4631}
4632
4633MachineInstr *
4634AArch64InstructionSelector::emitCMN(MachineOperand &LHS, MachineOperand &RHS,
4635 MachineIRBuilder &MIRBuilder) const {
4636 MachineRegisterInfo &MRI = MIRBuilder.getMF().getRegInfo();
4637 bool Is32Bit = (MRI.getType(Reg: LHS.getReg()).getSizeInBits() == 32);
4638 auto RC = Is32Bit ? &AArch64::GPR32RegClass : &AArch64::GPR64RegClass;
4639 return emitADDS(Dst: MRI.createVirtualRegister(RegClass: RC), LHS, RHS, MIRBuilder);
4640}
4641
4642MachineInstr *
4643AArch64InstructionSelector::emitTST(MachineOperand &LHS, MachineOperand &RHS,
4644 MachineIRBuilder &MIRBuilder) const {
4645 assert(LHS.isReg() && RHS.isReg() && "Expected register operands?");
4646 MachineRegisterInfo &MRI = MIRBuilder.getMF().getRegInfo();
4647 LLT Ty = MRI.getType(Reg: LHS.getReg());
4648 unsigned RegSize = Ty.getSizeInBits();
4649 bool Is32Bit = (RegSize == 32);
4650 const unsigned OpcTable[3][2] = {{AArch64::ANDSXri, AArch64::ANDSWri},
4651 {AArch64::ANDSXrs, AArch64::ANDSWrs},
4652 {AArch64::ANDSXrr, AArch64::ANDSWrr}};
4653 // ANDS needs a logical immediate for its immediate form. Check if we can
4654 // fold one in.
4655 if (auto ValAndVReg = getIConstantVRegValWithLookThrough(VReg: RHS.getReg(), MRI)) {
4656 int64_t Imm = ValAndVReg->Value.getSExtValue();
4657
4658 if (AArch64_AM::isLogicalImmediate(imm: Imm, regSize: RegSize)) {
4659 auto TstMI = MIRBuilder.buildInstr(Opc: OpcTable[0][Is32Bit], DstOps: {Ty}, SrcOps: {LHS});
4660 TstMI.addImm(Val: AArch64_AM::encodeLogicalImmediate(imm: Imm, regSize: RegSize));
4661 constrainSelectedInstRegOperands(I&: *TstMI, TII, TRI, RBI);
4662 return &*TstMI;
4663 }
4664 }
4665
4666 if (auto Fns = selectLogicalShiftedRegister(Root&: RHS))
4667 return emitInstr(Opcode: OpcTable[1][Is32Bit], DstOps: {Ty}, SrcOps: {LHS}, MIRBuilder, RenderFns: Fns);
4668 return emitInstr(Opcode: OpcTable[2][Is32Bit], DstOps: {Ty}, SrcOps: {LHS, RHS}, MIRBuilder);
4669}
4670
4671MachineInstr *AArch64InstructionSelector::emitIntegerCompare(
4672 MachineOperand &LHS, MachineOperand &RHS, MachineOperand &Predicate,
4673 MachineIRBuilder &MIRBuilder) const {
4674 assert(LHS.isReg() && RHS.isReg() && "Expected LHS and RHS to be registers!");
4675 assert(Predicate.isPredicate() && "Expected predicate?");
4676 MachineRegisterInfo &MRI = MIRBuilder.getMF().getRegInfo();
4677 LLT CmpTy = MRI.getType(Reg: LHS.getReg());
4678 assert(!CmpTy.isVector() && "Expected scalar or pointer");
4679 unsigned Size = CmpTy.getSizeInBits();
4680 (void)Size;
4681 assert((Size == 32 || Size == 64) && "Expected a 32-bit or 64-bit LHS/RHS?");
4682 // Fold the compare into a cmn or tst if possible.
4683 if (auto FoldCmp = tryFoldIntegerCompare(LHS, RHS, Predicate, MIRBuilder))
4684 return FoldCmp;
4685 return emitCMP(LHS, RHS, MIRBuilder);
4686}
4687
4688MachineInstr *AArch64InstructionSelector::emitCSetForFCmp(
4689 Register Dst, CmpInst::Predicate Pred, MachineIRBuilder &MIRBuilder) const {
4690 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
4691#ifndef NDEBUG
4692 LLT Ty = MRI.getType(Dst);
4693 assert(!Ty.isVector() && Ty.getSizeInBits() == 32 &&
4694 "Expected a 32-bit scalar register?");
4695#endif
4696 const Register ZReg = AArch64::WZR;
4697 AArch64CC::CondCode CC1, CC2;
4698 changeFCMPPredToAArch64CC(P: Pred, CondCode&: CC1, CondCode2&: CC2);
4699 auto InvCC1 = AArch64CC::getInvertedCondCode(Code: CC1);
4700 if (CC2 == AArch64CC::AL)
4701 return emitCSINC(/*Dst=*/Dst, /*Src1=*/ZReg, /*Src2=*/ZReg, Pred: InvCC1,
4702 MIRBuilder);
4703 const TargetRegisterClass *RC = &AArch64::GPR32RegClass;
4704 Register Def1Reg = MRI.createVirtualRegister(RegClass: RC);
4705 Register Def2Reg = MRI.createVirtualRegister(RegClass: RC);
4706 auto InvCC2 = AArch64CC::getInvertedCondCode(Code: CC2);
4707 emitCSINC(/*Dst=*/Def1Reg, /*Src1=*/ZReg, /*Src2=*/ZReg, Pred: InvCC1, MIRBuilder);
4708 emitCSINC(/*Dst=*/Def2Reg, /*Src1=*/ZReg, /*Src2=*/ZReg, Pred: InvCC2, MIRBuilder);
4709 auto OrMI = MIRBuilder.buildInstr(Opc: AArch64::ORRWrr, DstOps: {Dst}, SrcOps: {Def1Reg, Def2Reg});
4710 constrainSelectedInstRegOperands(I&: *OrMI, TII, TRI, RBI);
4711 return &*OrMI;
4712}
4713
4714MachineInstr *AArch64InstructionSelector::emitFPCompare(
4715 Register LHS, Register RHS, MachineIRBuilder &MIRBuilder,
4716 std::optional<CmpInst::Predicate> Pred) const {
4717 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
4718 LLT Ty = MRI.getType(Reg: LHS);
4719 if (Ty.isVector())
4720 return nullptr;
4721 unsigned OpSize = Ty.getSizeInBits();
4722 assert(OpSize == 16 || OpSize == 32 || OpSize == 64);
4723
4724 // If this is a compare against +0.0, then we don't have
4725 // to explicitly materialize a constant.
4726 bool ShouldUseImm = mi_match(R: RHS, MRI, P: m_PosZeroFP());
4727
4728 auto IsEqualityPred = [](CmpInst::Predicate P) {
4729 return P == CmpInst::FCMP_OEQ || P == CmpInst::FCMP_ONE ||
4730 P == CmpInst::FCMP_UEQ || P == CmpInst::FCMP_UNE;
4731 };
4732 if (!ShouldUseImm && Pred && IsEqualityPred(*Pred)) {
4733 // Try commuting the operands.
4734 if (mi_match(R: LHS, MRI, P: m_PosZeroFP())) {
4735 ShouldUseImm = true;
4736 std::swap(a&: LHS, b&: RHS);
4737 }
4738 }
4739 unsigned CmpOpcTbl[2][3] = {
4740 {AArch64::FCMPHrr, AArch64::FCMPSrr, AArch64::FCMPDrr},
4741 {AArch64::FCMPHri, AArch64::FCMPSri, AArch64::FCMPDri}};
4742 unsigned CmpOpc =
4743 CmpOpcTbl[ShouldUseImm][OpSize == 16 ? 0 : (OpSize == 32 ? 1 : 2)];
4744
4745 // Partially build the compare. Decide if we need to add a use for the
4746 // third operand based off whether or not we're comparing against 0.0.
4747 auto CmpMI = MIRBuilder.buildInstr(Opcode: CmpOpc).addUse(RegNo: LHS);
4748 CmpMI.setMIFlags(MachineInstr::NoFPExcept);
4749 if (!ShouldUseImm)
4750 CmpMI.addUse(RegNo: RHS);
4751 constrainSelectedInstRegOperands(I&: *CmpMI, TII, TRI, RBI);
4752 return &*CmpMI;
4753}
4754
4755MachineInstr *AArch64InstructionSelector::emitVectorConcat(
4756 std::optional<Register> Dst, Register Op1, Register Op2,
4757 MachineIRBuilder &MIRBuilder) const {
4758 // We implement a vector concat by:
4759 // 1. Use scalar_to_vector to insert the lower vector into the larger dest
4760 // 2. Insert the upper vector into the destination's upper element
4761 // TODO: some of this code is common with G_BUILD_VECTOR handling.
4762 MachineRegisterInfo &MRI = MIRBuilder.getMF().getRegInfo();
4763
4764 const LLT Op1Ty = MRI.getType(Reg: Op1);
4765 const LLT Op2Ty = MRI.getType(Reg: Op2);
4766
4767 if (Op1Ty != Op2Ty) {
4768 LLVM_DEBUG(dbgs() << "Could not do vector concat of differing vector tys");
4769 return nullptr;
4770 }
4771 assert(Op1Ty.isVector() && "Expected a vector for vector concat");
4772
4773 if (Op1Ty.getSizeInBits() >= 128) {
4774 LLVM_DEBUG(dbgs() << "Vector concat not supported for full size vectors");
4775 return nullptr;
4776 }
4777
4778 // At the moment we just support 64 bit vector concats.
4779 if (Op1Ty.getSizeInBits() != 64) {
4780 LLVM_DEBUG(dbgs() << "Vector concat supported for 64b vectors");
4781 return nullptr;
4782 }
4783
4784 const LLT ScalarTy = LLT::scalar(SizeInBits: Op1Ty.getSizeInBits());
4785 const RegisterBank &FPRBank = *RBI.getRegBank(Reg: Op1, MRI, TRI);
4786 const TargetRegisterClass *DstRC =
4787 getRegClassForTypeOnBank(Ty: Op1Ty.multiplyElements(Factor: 2), RB: FPRBank);
4788
4789 MachineInstr *WidenedOp1 =
4790 emitScalarToVector(EltSize: ScalarTy.getSizeInBits(), DstRC, Scalar: Op1, MIRBuilder);
4791 MachineInstr *WidenedOp2 =
4792 emitScalarToVector(EltSize: ScalarTy.getSizeInBits(), DstRC, Scalar: Op2, MIRBuilder);
4793 if (!WidenedOp1 || !WidenedOp2) {
4794 LLVM_DEBUG(dbgs() << "Could not emit a vector from scalar value");
4795 return nullptr;
4796 }
4797
4798 // Now do the insert of the upper element.
4799 unsigned InsertOpc, InsSubRegIdx;
4800 std::tie(args&: InsertOpc, args&: InsSubRegIdx) =
4801 getInsertVecEltOpInfo(RB: FPRBank, EltSize: ScalarTy.getSizeInBits());
4802
4803 if (!Dst)
4804 Dst = MRI.createVirtualRegister(RegClass: DstRC);
4805 auto InsElt =
4806 MIRBuilder
4807 .buildInstr(Opc: InsertOpc, DstOps: {*Dst}, SrcOps: {WidenedOp1->getOperand(i: 0).getReg()})
4808 .addImm(Val: 1) /* Lane index */
4809 .addUse(RegNo: WidenedOp2->getOperand(i: 0).getReg())
4810 .addImm(Val: 0);
4811 constrainSelectedInstRegOperands(I&: *InsElt, TII, TRI, RBI);
4812 return &*InsElt;
4813}
4814
4815MachineInstr *
4816AArch64InstructionSelector::emitCSINC(Register Dst, Register Src1,
4817 Register Src2, AArch64CC::CondCode Pred,
4818 MachineIRBuilder &MIRBuilder) const {
4819 auto &MRI = *MIRBuilder.getMRI();
4820 const RegClassOrRegBank &RegClassOrBank = MRI.getRegClassOrRegBank(Reg: Dst);
4821 // If we used a register class, then this won't necessarily have an LLT.
4822 // Compute the size based off whether or not we have a class or bank.
4823 unsigned Size;
4824 if (const auto *RC = dyn_cast<const TargetRegisterClass *>(Val: RegClassOrBank))
4825 Size = TRI.getRegSizeInBits(RC: *RC);
4826 else
4827 Size = MRI.getType(Reg: Dst).getSizeInBits();
4828 // Some opcodes use s1.
4829 assert(Size <= 64 && "Expected 64 bits or less only!");
4830 static const unsigned OpcTable[2] = {AArch64::CSINCWr, AArch64::CSINCXr};
4831 unsigned Opc = OpcTable[Size == 64];
4832 auto CSINC = MIRBuilder.buildInstr(Opc, DstOps: {Dst}, SrcOps: {Src1, Src2}).addImm(Val: Pred);
4833 constrainSelectedInstRegOperands(I&: *CSINC, TII, TRI, RBI);
4834 return &*CSINC;
4835}
4836
4837MachineInstr *AArch64InstructionSelector::emitCarryIn(MachineInstr &I,
4838 Register CarryReg) {
4839 MachineRegisterInfo *MRI = MIB.getMRI();
4840 unsigned Opcode = I.getOpcode();
4841
4842 // If the instruction is a SUB, we need to negate the carry,
4843 // because borrowing is indicated by carry-flag == 0.
4844 bool NeedsNegatedCarry =
4845 (Opcode == TargetOpcode::G_USUBE || Opcode == TargetOpcode::G_SSUBE);
4846
4847 // If the previous instruction will already produce the correct carry, do not
4848 // emit a carry generating instruction. E.g. for G_UADDE/G_USUBE sequences
4849 // generated during legalization of wide add/sub. This optimization depends on
4850 // these sequences not being interrupted by other instructions.
4851 // We have to select the previous instruction before the carry-using
4852 // instruction is deleted by the calling function, otherwise the previous
4853 // instruction might become dead and would get deleted.
4854 MachineInstr *SrcMI = MRI->getVRegDef(Reg: CarryReg);
4855 if (SrcMI == I.getPrevNode()) {
4856 if (auto *CarrySrcMI = dyn_cast<GAddSubCarryOut>(Val: SrcMI)) {
4857 bool ProducesNegatedCarry = CarrySrcMI->isSub();
4858 if (NeedsNegatedCarry == ProducesNegatedCarry &&
4859 CarrySrcMI->isUnsigned() &&
4860 CarrySrcMI->getCarryOutReg() == CarryReg &&
4861 selectAndRestoreState(I&: *SrcMI))
4862 return nullptr;
4863 }
4864 }
4865
4866 Register DeadReg = MRI->createVirtualRegister(RegClass: &AArch64::GPR32RegClass);
4867
4868 if (NeedsNegatedCarry) {
4869 // (0 - Carry) sets !C in NZCV when Carry == 1
4870 Register ZReg = AArch64::WZR;
4871 return emitInstr(Opcode: AArch64::SUBSWrr, DstOps: {DeadReg}, SrcOps: {ZReg, CarryReg}, MIRBuilder&: MIB);
4872 }
4873
4874 // (Carry - 1) sets !C in NZCV when Carry == 0
4875 auto Fns = select12BitValueWithLeftShift(Immed: 1);
4876 return emitInstr(Opcode: AArch64::SUBSWri, DstOps: {DeadReg}, SrcOps: {CarryReg}, MIRBuilder&: MIB, RenderFns: Fns);
4877}
4878
4879bool AArch64InstructionSelector::selectOverflowOp(MachineInstr &I,
4880 MachineRegisterInfo &MRI) {
4881 auto &CarryMI = cast<GAddSubCarryOut>(Val&: I);
4882
4883 if (auto *CarryInMI = dyn_cast<GAddSubCarryInOut>(Val: &I)) {
4884 // Set NZCV carry according to carry-in VReg
4885 emitCarryIn(I, CarryReg: CarryInMI->getCarryInReg());
4886 }
4887
4888 // Emit the operation and get the correct condition code.
4889 auto OpAndCC = emitOverflowOp(Opcode: I.getOpcode(), Dst: CarryMI.getDstReg(),
4890 LHS&: CarryMI.getLHS(), RHS&: CarryMI.getRHS(), MIRBuilder&: MIB);
4891
4892 Register CarryOutReg = CarryMI.getCarryOutReg();
4893
4894 // Don't convert carry-out to VReg if it is never used
4895 if (MRI.use_nodbg_empty(RegNo: CarryOutReg)) {
4896 OpAndCC.first->addRegisterDead(Reg: AArch64::NZCV, RegInfo: &TRI);
4897 } else {
4898 // Now, put the overflow result in the register given by the first operand
4899 // to the overflow op. CSINC increments the result when the predicate is
4900 // false, so to get the increment when it's true, we need to use the
4901 // inverse. In this case, we want to increment when carry is set.
4902 Register ZReg = AArch64::WZR;
4903 emitCSINC(/*Dst=*/CarryOutReg, /*Src1=*/ZReg, /*Src2=*/ZReg,
4904 Pred: getInvertedCondCode(Code: OpAndCC.second), MIRBuilder&: MIB);
4905 }
4906
4907 I.eraseFromParent();
4908 return true;
4909}
4910
4911std::pair<MachineInstr *, AArch64CC::CondCode>
4912AArch64InstructionSelector::emitOverflowOp(unsigned Opcode, Register Dst,
4913 MachineOperand &LHS,
4914 MachineOperand &RHS,
4915 MachineIRBuilder &MIRBuilder) const {
4916 switch (Opcode) {
4917 default:
4918 llvm_unreachable("Unexpected opcode!");
4919 case TargetOpcode::G_SADDO:
4920 return std::make_pair(x: emitADDS(Dst, LHS, RHS, MIRBuilder), y: AArch64CC::VS);
4921 case TargetOpcode::G_UADDO:
4922 return std::make_pair(x: emitADDS(Dst, LHS, RHS, MIRBuilder), y: AArch64CC::HS);
4923 case TargetOpcode::G_SSUBO:
4924 return std::make_pair(x: emitSUBS(Dst, LHS, RHS, MIRBuilder), y: AArch64CC::VS);
4925 case TargetOpcode::G_USUBO:
4926 return std::make_pair(x: emitSUBS(Dst, LHS, RHS, MIRBuilder), y: AArch64CC::LO);
4927 case TargetOpcode::G_SADDE:
4928 return std::make_pair(x: emitADCS(Dst, LHS, RHS, MIRBuilder), y: AArch64CC::VS);
4929 case TargetOpcode::G_UADDE:
4930 return std::make_pair(x: emitADCS(Dst, LHS, RHS, MIRBuilder), y: AArch64CC::HS);
4931 case TargetOpcode::G_SSUBE:
4932 return std::make_pair(x: emitSBCS(Dst, LHS, RHS, MIRBuilder), y: AArch64CC::VS);
4933 case TargetOpcode::G_USUBE:
4934 return std::make_pair(x: emitSBCS(Dst, LHS, RHS, MIRBuilder), y: AArch64CC::LO);
4935 }
4936}
4937
4938/// Returns true if @p Val is a tree of AND/OR/CMP operations that can be
4939/// expressed as a conjunction.
4940/// \param CanNegate Set to true if we can negate the whole sub-tree just by
4941/// changing the conditions on the CMP tests.
4942/// (this means we can call emitConjunctionRec() with
4943/// Negate==true on this sub-tree)
4944/// \param MustBeFirst Set to true if this subtree needs to be negated and we
4945/// cannot do the negation naturally. We are required to
4946/// emit the subtree first in this case.
4947/// \param WillNegate Is true if are called when the result of this
4948/// subexpression must be negated. This happens when the
4949/// outer expression is an OR. We can use this fact to know
4950/// that we have a double negation (or (or ...) ...) that
4951/// can be implemented for free.
4952static bool canEmitConjunction(Register Val, bool &CanNegate, bool &MustBeFirst,
4953 bool WillNegate, MachineRegisterInfo &MRI,
4954 unsigned Depth = 0) {
4955 if (!MRI.hasOneNonDBGUse(RegNo: Val))
4956 return false;
4957 MachineInstr *ValDef = MRI.getVRegDef(Reg: Val);
4958 unsigned Opcode = ValDef->getOpcode();
4959 if (isa<GAnyCmp>(Val: ValDef)) {
4960 CanNegate = true;
4961 MustBeFirst = false;
4962 return true;
4963 }
4964 // Protect against exponential runtime and stack overflow.
4965 if (Depth > 6)
4966 return false;
4967 if (Opcode == TargetOpcode::G_AND || Opcode == TargetOpcode::G_OR) {
4968 bool IsOR = Opcode == TargetOpcode::G_OR;
4969 Register O0 = ValDef->getOperand(i: 1).getReg();
4970 Register O1 = ValDef->getOperand(i: 2).getReg();
4971 bool CanNegateL;
4972 bool MustBeFirstL;
4973 if (!canEmitConjunction(Val: O0, CanNegate&: CanNegateL, MustBeFirst&: MustBeFirstL, WillNegate: IsOR, MRI, Depth: Depth + 1))
4974 return false;
4975 bool CanNegateR;
4976 bool MustBeFirstR;
4977 if (!canEmitConjunction(Val: O1, CanNegate&: CanNegateR, MustBeFirst&: MustBeFirstR, WillNegate: IsOR, MRI, Depth: Depth + 1))
4978 return false;
4979
4980 if (MustBeFirstL && MustBeFirstR)
4981 return false;
4982
4983 if (IsOR) {
4984 // For an OR expression we need to be able to naturally negate at least
4985 // one side or we cannot do the transformation at all.
4986 if (!CanNegateL && !CanNegateR)
4987 return false;
4988 // If we the result of the OR will be negated and we can naturally negate
4989 // the leaves, then this sub-tree as a whole negates naturally.
4990 CanNegate = WillNegate && CanNegateL && CanNegateR;
4991 // If we cannot naturally negate the whole sub-tree, then this must be
4992 // emitted first.
4993 MustBeFirst = !CanNegate;
4994 } else {
4995 assert(Opcode == TargetOpcode::G_AND && "Must be G_AND");
4996 // We cannot naturally negate an AND operation.
4997 CanNegate = false;
4998 MustBeFirst = MustBeFirstL || MustBeFirstR;
4999 }
5000 return true;
5001 }
5002 return false;
5003}
5004
5005MachineInstr *AArch64InstructionSelector::emitConditionalComparison(
5006 Register LHS, Register RHS, CmpInst::Predicate CC,
5007 AArch64CC::CondCode Predicate, AArch64CC::CondCode OutCC,
5008 MachineIRBuilder &MIB) const {
5009 auto &MRI = *MIB.getMRI();
5010 LLT OpTy = MRI.getType(Reg: LHS);
5011 unsigned CCmpOpc;
5012 std::optional<ValueAndVReg> C;
5013 if (CmpInst::isIntPredicate(P: CC)) {
5014 assert(OpTy.getSizeInBits() == 32 || OpTy.getSizeInBits() == 64);
5015 C = getIConstantVRegValWithLookThrough(VReg: RHS, MRI);
5016 if (!C || C->Value.sgt(RHS: 31) || C->Value.slt(RHS: -31))
5017 CCmpOpc = OpTy.getSizeInBits() == 32 ? AArch64::CCMPWr : AArch64::CCMPXr;
5018 else if (C->Value.ule(RHS: 31))
5019 CCmpOpc = OpTy.getSizeInBits() == 32 ? AArch64::CCMPWi : AArch64::CCMPXi;
5020 else
5021 CCmpOpc = OpTy.getSizeInBits() == 32 ? AArch64::CCMNWi : AArch64::CCMNXi;
5022 } else {
5023 assert(OpTy.getSizeInBits() == 16 || OpTy.getSizeInBits() == 32 ||
5024 OpTy.getSizeInBits() == 64);
5025 switch (OpTy.getSizeInBits()) {
5026 case 16:
5027 assert(STI.hasFullFP16() && "Expected Full FP16 for fp16 comparisons");
5028 CCmpOpc = AArch64::FCCMPHrr;
5029 break;
5030 case 32:
5031 CCmpOpc = AArch64::FCCMPSrr;
5032 break;
5033 case 64:
5034 CCmpOpc = AArch64::FCCMPDrr;
5035 break;
5036 default:
5037 return nullptr;
5038 }
5039 }
5040 AArch64CC::CondCode InvOutCC = AArch64CC::getInvertedCondCode(Code: OutCC);
5041 unsigned NZCV = AArch64CC::getNZCVToSatisfyCondCode(Code: InvOutCC);
5042 auto CCmp =
5043 MIB.buildInstr(Opc: CCmpOpc, DstOps: {}, SrcOps: {LHS});
5044 if (CCmpOpc == AArch64::CCMPWi || CCmpOpc == AArch64::CCMPXi)
5045 CCmp.addImm(Val: C->Value.getZExtValue());
5046 else if (CCmpOpc == AArch64::CCMNWi || CCmpOpc == AArch64::CCMNXi)
5047 CCmp.addImm(Val: C->Value.abs().getZExtValue());
5048 else
5049 CCmp.addReg(RegNo: RHS);
5050 CCmp.addImm(Val: NZCV).addImm(Val: Predicate);
5051 constrainSelectedInstRegOperands(I&: *CCmp, TII, TRI, RBI);
5052 return &*CCmp;
5053}
5054
5055MachineInstr *AArch64InstructionSelector::emitConjunctionRec(
5056 Register Val, AArch64CC::CondCode &OutCC, bool Negate, Register CCOp,
5057 AArch64CC::CondCode Predicate, MachineIRBuilder &MIB) const {
5058 // We're at a tree leaf, produce a conditional comparison operation.
5059 auto &MRI = *MIB.getMRI();
5060 MachineInstr *ValDef = MRI.getVRegDef(Reg: Val);
5061 unsigned Opcode = ValDef->getOpcode();
5062 if (auto *Cmp = dyn_cast<GAnyCmp>(Val: ValDef)) {
5063 Register LHS = Cmp->getLHSReg();
5064 Register RHS = Cmp->getRHSReg();
5065 CmpInst::Predicate CC = Cmp->getCond();
5066 if (Negate)
5067 CC = CmpInst::getInversePredicate(pred: CC);
5068 if (isa<GICmp>(Val: Cmp)) {
5069 OutCC = changeICMPPredToAArch64CC(P: CC, RHS, MRI: MIB.getMRI());
5070 } else {
5071 // Handle special FP cases.
5072 AArch64CC::CondCode ExtraCC;
5073 changeFPCCToANDAArch64CC(CC, CondCode&: OutCC, CondCode2&: ExtraCC);
5074 // Some floating point conditions can't be tested with a single condition
5075 // code. Construct an additional comparison in this case.
5076 if (ExtraCC != AArch64CC::AL) {
5077 MachineInstr *ExtraCmp;
5078 if (!CCOp)
5079 ExtraCmp = emitFPCompare(LHS, RHS, MIRBuilder&: MIB, Pred: CC);
5080 else
5081 ExtraCmp =
5082 emitConditionalComparison(LHS, RHS, CC, Predicate, OutCC: ExtraCC, MIB);
5083 CCOp = ExtraCmp->getOperand(i: 0).getReg();
5084 Predicate = ExtraCC;
5085 }
5086 }
5087
5088 // Produce a normal comparison if we are first in the chain
5089 if (!CCOp) {
5090 if (isa<GICmp>(Val: Cmp))
5091 return emitCMP(LHS&: Cmp->getOperand(i: 2), RHS&: Cmp->getOperand(i: 3), MIRBuilder&: MIB);
5092 return emitFPCompare(LHS: Cmp->getOperand(i: 2).getReg(),
5093 RHS: Cmp->getOperand(i: 3).getReg(), MIRBuilder&: MIB);
5094 }
5095 // Otherwise produce a ccmp.
5096 return emitConditionalComparison(LHS, RHS, CC, Predicate, OutCC, MIB);
5097 }
5098 assert(MRI.hasOneNonDBGUse(Val) && "Valid conjunction/disjunction tree");
5099
5100 bool IsOR = Opcode == TargetOpcode::G_OR;
5101
5102 Register LHS = ValDef->getOperand(i: 1).getReg();
5103 bool CanNegateL;
5104 bool MustBeFirstL;
5105 bool ValidL = canEmitConjunction(Val: LHS, CanNegate&: CanNegateL, MustBeFirst&: MustBeFirstL, WillNegate: IsOR, MRI);
5106 assert(ValidL && "Valid conjunction/disjunction tree");
5107 (void)ValidL;
5108
5109 Register RHS = ValDef->getOperand(i: 2).getReg();
5110 bool CanNegateR;
5111 bool MustBeFirstR;
5112 bool ValidR = canEmitConjunction(Val: RHS, CanNegate&: CanNegateR, MustBeFirst&: MustBeFirstR, WillNegate: IsOR, MRI);
5113 assert(ValidR && "Valid conjunction/disjunction tree");
5114 (void)ValidR;
5115
5116 // Swap sub-tree that must come first to the right side.
5117 if (MustBeFirstL) {
5118 assert(!MustBeFirstR && "Valid conjunction/disjunction tree");
5119 std::swap(a&: LHS, b&: RHS);
5120 std::swap(a&: CanNegateL, b&: CanNegateR);
5121 std::swap(a&: MustBeFirstL, b&: MustBeFirstR);
5122 }
5123
5124 bool NegateR;
5125 bool NegateAfterR;
5126 bool NegateL;
5127 bool NegateAfterAll;
5128 if (Opcode == TargetOpcode::G_OR) {
5129 // Swap the sub-tree that we can negate naturally to the left.
5130 if (!CanNegateL) {
5131 assert(CanNegateR && "at least one side must be negatable");
5132 assert(!MustBeFirstR && "invalid conjunction/disjunction tree");
5133 assert(!Negate);
5134 std::swap(a&: LHS, b&: RHS);
5135 NegateR = false;
5136 NegateAfterR = true;
5137 } else {
5138 // Negate the left sub-tree if possible, otherwise negate the result.
5139 NegateR = CanNegateR;
5140 NegateAfterR = !CanNegateR;
5141 }
5142 NegateL = true;
5143 NegateAfterAll = !Negate;
5144 } else {
5145 assert(Opcode == TargetOpcode::G_AND &&
5146 "Valid conjunction/disjunction tree");
5147 assert(!Negate && "Valid conjunction/disjunction tree");
5148
5149 NegateL = false;
5150 NegateR = false;
5151 NegateAfterR = false;
5152 NegateAfterAll = false;
5153 }
5154
5155 // Emit sub-trees.
5156 AArch64CC::CondCode RHSCC;
5157 MachineInstr *CmpR =
5158 emitConjunctionRec(Val: RHS, OutCC&: RHSCC, Negate: NegateR, CCOp, Predicate, MIB);
5159 if (NegateAfterR)
5160 RHSCC = AArch64CC::getInvertedCondCode(Code: RHSCC);
5161 MachineInstr *CmpL = emitConjunctionRec(
5162 Val: LHS, OutCC, Negate: NegateL, CCOp: CmpR->getOperand(i: 0).getReg(), Predicate: RHSCC, MIB);
5163 if (NegateAfterAll)
5164 OutCC = AArch64CC::getInvertedCondCode(Code: OutCC);
5165 return CmpL;
5166}
5167
5168MachineInstr *AArch64InstructionSelector::emitConjunction(
5169 Register Val, AArch64CC::CondCode &OutCC, MachineIRBuilder &MIB) const {
5170 bool DummyCanNegate;
5171 bool DummyMustBeFirst;
5172 if (!canEmitConjunction(Val, CanNegate&: DummyCanNegate, MustBeFirst&: DummyMustBeFirst, WillNegate: false,
5173 MRI&: *MIB.getMRI()))
5174 return nullptr;
5175 return emitConjunctionRec(Val, OutCC, Negate: false, CCOp: Register(), Predicate: AArch64CC::AL, MIB);
5176}
5177
5178bool AArch64InstructionSelector::tryOptSelectConjunction(GSelect &SelI,
5179 MachineInstr &CondMI) {
5180 AArch64CC::CondCode AArch64CC;
5181 MachineInstr *ConjMI = emitConjunction(Val: SelI.getCondReg(), OutCC&: AArch64CC, MIB);
5182 if (!ConjMI)
5183 return false;
5184
5185 emitSelect(Dst: SelI.getReg(Idx: 0), True: SelI.getTrueReg(), False: SelI.getFalseReg(), CC: AArch64CC, MIB);
5186 SelI.eraseFromParent();
5187 return true;
5188}
5189
5190bool AArch64InstructionSelector::tryOptSelect(GSelect &I) {
5191 MachineRegisterInfo &MRI = *MIB.getMRI();
5192 // We want to recognize this pattern:
5193 //
5194 // $z = G_FCMP pred, $x, $y
5195 // ...
5196 // $w = G_SELECT $z, $a, $b
5197 //
5198 // Where the value of $z is *only* ever used by the G_SELECT (possibly with
5199 // some copies/truncs in between.)
5200 //
5201 // If we see this, then we can emit something like this:
5202 //
5203 // fcmp $x, $y
5204 // fcsel $w, $a, $b, pred
5205 //
5206 // Rather than emitting both of the rather long sequences in the standard
5207 // G_FCMP/G_SELECT select methods.
5208
5209 // First, check if the condition is defined by a compare.
5210 MachineInstr *CondDef = MRI.getVRegDef(Reg: I.getOperand(i: 1).getReg());
5211
5212 // We can only fold if all of the defs have one use.
5213 Register CondDefReg = CondDef->getOperand(i: 0).getReg();
5214 if (!MRI.hasOneNonDBGUse(RegNo: CondDefReg)) {
5215 // Unless it's another select.
5216 for (const MachineInstr &UI : MRI.use_nodbg_instructions(Reg: CondDefReg)) {
5217 if (CondDef == &UI)
5218 continue;
5219 if (UI.getOpcode() != TargetOpcode::G_SELECT)
5220 return false;
5221 }
5222 }
5223
5224 // Is the condition defined by a compare?
5225 unsigned CondOpc = CondDef->getOpcode();
5226 if (CondOpc != TargetOpcode::G_ICMP && CondOpc != TargetOpcode::G_FCMP) {
5227 if (tryOptSelectConjunction(SelI&: I, CondMI&: *CondDef))
5228 return true;
5229 return false;
5230 }
5231
5232 AArch64CC::CondCode CondCode;
5233 if (CondOpc == TargetOpcode::G_ICMP) {
5234 auto &PredOp = CondDef->getOperand(i: 1);
5235 emitIntegerCompare(LHS&: CondDef->getOperand(i: 2), RHS&: CondDef->getOperand(i: 3), Predicate&: PredOp,
5236 MIRBuilder&: MIB);
5237 auto Pred = static_cast<CmpInst::Predicate>(PredOp.getPredicate());
5238 CondCode =
5239 changeICMPPredToAArch64CC(P: Pred, RHS: CondDef->getOperand(i: 3).getReg(), MRI: &MRI);
5240 } else {
5241 // Get the condition code for the select.
5242 auto Pred =
5243 static_cast<CmpInst::Predicate>(CondDef->getOperand(i: 1).getPredicate());
5244 AArch64CC::CondCode CondCode2;
5245 changeFCMPPredToAArch64CC(P: Pred, CondCode, CondCode2);
5246
5247 // changeFCMPPredToAArch64CC sets CondCode2 to AL when we require two
5248 // instructions to emit the comparison.
5249 // TODO: Handle FCMP_UEQ and FCMP_ONE. After that, this check will be
5250 // unnecessary.
5251 if (CondCode2 != AArch64CC::AL)
5252 return false;
5253
5254 if (!emitFPCompare(LHS: CondDef->getOperand(i: 2).getReg(),
5255 RHS: CondDef->getOperand(i: 3).getReg(), MIRBuilder&: MIB)) {
5256 LLVM_DEBUG(dbgs() << "Couldn't emit compare for select!\n");
5257 return false;
5258 }
5259 }
5260
5261 // Emit the select.
5262 emitSelect(Dst: I.getOperand(i: 0).getReg(), True: I.getOperand(i: 2).getReg(),
5263 False: I.getOperand(i: 3).getReg(), CC: CondCode, MIB);
5264 I.eraseFromParent();
5265 return true;
5266}
5267
5268MachineInstr *AArch64InstructionSelector::tryFoldIntegerCompare(
5269 MachineOperand &LHS, MachineOperand &RHS, MachineOperand &Predicate,
5270 MachineIRBuilder &MIRBuilder) const {
5271 assert(LHS.isReg() && RHS.isReg() && Predicate.isPredicate() &&
5272 "Unexpected MachineOperand");
5273 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
5274 // We want to find this sort of thing:
5275 // x = G_SUB 0, y
5276 // G_ICMP z, x
5277 //
5278 // In this case, we can fold the G_SUB into the G_ICMP using a CMN instead.
5279 // e.g:
5280 //
5281 // cmn z, y
5282
5283 // Check if the RHS or LHS of the G_ICMP is defined by a SUB
5284 MachineInstr *LHSDef = getDefIgnoringCopies(Reg: LHS.getReg(), MRI);
5285 MachineInstr *RHSDef = getDefIgnoringCopies(Reg: RHS.getReg(), MRI);
5286 auto P = static_cast<CmpInst::Predicate>(Predicate.getPredicate());
5287
5288 // Given this:
5289 //
5290 // x = G_SUB 0, y
5291 // G_ICMP z, x
5292 //
5293 // Produce this:
5294 //
5295 // cmn z, y
5296 if (isCMN(MaybeSub: RHSDef, Pred: P, MRI))
5297 return emitCMN(LHS, RHS&: RHSDef->getOperand(i: 2), MIRBuilder);
5298
5299 // Same idea here, but with the LHS of the compare instead:
5300 //
5301 // Given this:
5302 //
5303 // x = G_SUB 0, y
5304 // G_ICMP x, z
5305 //
5306 // Produce this:
5307 //
5308 // cmn y, z
5309 //
5310 // But be careful! We need to swap the predicate!
5311 if (isCMN(MaybeSub: LHSDef, Pred: P, MRI)) {
5312 if (!CmpInst::isEquality(pred: P)) {
5313 P = CmpInst::getSwappedPredicate(pred: P);
5314 Predicate = MachineOperand::CreatePredicate(Pred: P);
5315 }
5316 return emitCMN(LHS&: LHSDef->getOperand(i: 2), RHS, MIRBuilder);
5317 }
5318
5319 // Given this:
5320 //
5321 // z = G_AND x, y
5322 // G_ICMP z, 0
5323 //
5324 // Produce this if the compare is signed:
5325 //
5326 // tst x, y
5327 if (!CmpInst::isUnsigned(Pred: P) && LHSDef &&
5328 LHSDef->getOpcode() == TargetOpcode::G_AND) {
5329 // Make sure that the RHS is 0.
5330 auto ValAndVReg = getIConstantVRegValWithLookThrough(VReg: RHS.getReg(), MRI);
5331 if (!ValAndVReg || ValAndVReg->Value != 0)
5332 return nullptr;
5333
5334 return emitTST(LHS&: LHSDef->getOperand(i: 1),
5335 RHS&: LHSDef->getOperand(i: 2), MIRBuilder);
5336 }
5337
5338 return nullptr;
5339}
5340
5341bool AArch64InstructionSelector::selectShuffleVector(
5342 MachineInstr &I, MachineRegisterInfo &MRI) {
5343 const LLT DstTy = MRI.getType(Reg: I.getOperand(i: 0).getReg());
5344 Register Src1Reg = I.getOperand(i: 1).getReg();
5345 Register Src2Reg = I.getOperand(i: 2).getReg();
5346 ArrayRef<int> Mask = I.getOperand(i: 3).getShuffleMask();
5347 assert(DstTy == MRI.getType(Src1Reg) &&
5348 "Expected equal shuffle types during selection");
5349
5350 MachineBasicBlock &MBB = *I.getParent();
5351 MachineFunction &MF = *MBB.getParent();
5352 LLVMContext &Ctx = MF.getFunction().getContext();
5353
5354 unsigned BytesPerElt = DstTy.getElementType().getSizeInBits() / 8;
5355 int NumElts = DstTy.getNumElements();
5356
5357 SmallVector<int> NewMask;
5358 bool FirstUsed = false;
5359 bool SecondUsed = false;
5360 for (int M : Mask) {
5361 // Map any undef or zero lanes to 255.
5362 if (M < 0 || VT->getKnownBits(R: M < NumElts ? Src1Reg : Src2Reg,
5363 DemandedElts: APInt::getOneBitSet(numBits: NumElts, BitNo: M % NumElts))
5364 .isZero()) {
5365 for (unsigned Byte = 0; Byte < BytesPerElt; ++Byte)
5366 NewMask.push_back(Elt: 255);
5367 continue;
5368 }
5369
5370 FirstUsed |= M < NumElts;
5371 SecondUsed |= M >= NumElts;
5372 for (unsigned Byte = 0; Byte < BytesPerElt; ++Byte) {
5373 unsigned Offset = Byte + M * BytesPerElt;
5374 NewMask.push_back(Elt: Offset);
5375 }
5376 }
5377
5378 // If the first is unused or all zeros, use the second src in a tbl1.
5379 if (!FirstUsed) {
5380 int ByteLanes = DstTy.getSizeInBits() == 128 ? 16 : 8;
5381 for (int &M : NewMask) {
5382 if (M != 255) {
5383 assert(M >= ByteLanes && M < 2 * ByteLanes);
5384 M -= ByteLanes;
5385 }
5386 }
5387 std::swap(a&: Src1Reg, b&: Src2Reg);
5388 std::swap(a&: FirstUsed, b&: SecondUsed);
5389 }
5390
5391 // Use a constant pool to load the index vector for TBL.
5392 SmallVector<Constant *> CstIdxs;
5393 transform(Range&: NewMask, d_first: std::back_inserter(x&: CstIdxs), F: [&Ctx](int M) {
5394 return ConstantInt::get(Ty: Type::getInt8Ty(C&: Ctx), V: M);
5395 });
5396 Constant *CPVal = ConstantVector::get(V: CstIdxs);
5397 MachineInstr *IndexLoad = emitLoadFromConstantPool(CPVal, MIRBuilder&: MIB);
5398 if (!IndexLoad) {
5399 LLVM_DEBUG(dbgs() << "Could not load from a constant pool");
5400 return false;
5401 }
5402
5403 if (DstTy.getSizeInBits() != 128) {
5404 assert(DstTy.getSizeInBits() == 64 && "Unexpected shuffle result ty");
5405 // This case can be done with TBL1.
5406 MachineInstr *Concat =
5407 emitVectorConcat(Dst: std::nullopt, Op1: Src1Reg, Op2: Src2Reg, MIRBuilder&: MIB);
5408 if (!Concat) {
5409 LLVM_DEBUG(dbgs() << "Could not do vector concat for tbl1");
5410 return false;
5411 }
5412
5413 // The constant pool load will be 64 bits, so need to convert to FPR128 reg.
5414 IndexLoad = emitScalarToVector(EltSize: 64, DstRC: &AArch64::FPR128RegClass,
5415 Scalar: IndexLoad->getOperand(i: 0).getReg(), MIRBuilder&: MIB);
5416
5417 auto TBL1 = MIB.buildInstr(
5418 Opc: AArch64::TBLv16i8One, DstOps: {&AArch64::FPR128RegClass},
5419 SrcOps: {Concat->getOperand(i: 0).getReg(), IndexLoad->getOperand(i: 0).getReg()});
5420 constrainSelectedInstRegOperands(I&: *TBL1, TII, TRI, RBI);
5421
5422 auto Copy =
5423 MIB.buildInstr(Opc: TargetOpcode::COPY, DstOps: {I.getOperand(i: 0).getReg()}, SrcOps: {})
5424 .addReg(RegNo: TBL1.getReg(Idx: 0), Flags: {}, SubReg: AArch64::dsub);
5425 RBI.constrainGenericRegister(Reg: Copy.getReg(Idx: 0), RC: AArch64::FPR64RegClass, MRI);
5426 I.eraseFromParent();
5427 return true;
5428 }
5429
5430 if (!SecondUsed) {
5431 auto TBL1 = MIB.buildInstr(Opc: AArch64::TBLv16i8One, DstOps: {I.getOperand(i: 0)},
5432 SrcOps: {Src1Reg, IndexLoad->getOperand(i: 0)});
5433 constrainSelectedInstRegOperands(I&: *TBL1, TII, TRI, RBI);
5434 I.eraseFromParent();
5435 return true;
5436 }
5437
5438 // For TBL2 we need to emit a REG_SEQUENCE to tie together two consecutive
5439 // Q registers for regalloc.
5440 SmallVector<Register, 2> Regs = {Src1Reg, Src2Reg};
5441 auto RegSeq = createQTuple(Regs, MIB);
5442 auto TBL2 = MIB.buildInstr(Opc: AArch64::TBLv16i8Two, DstOps: {I.getOperand(i: 0)},
5443 SrcOps: {RegSeq, IndexLoad->getOperand(i: 0)});
5444 constrainSelectedInstRegOperands(I&: *TBL2, TII, TRI, RBI);
5445 I.eraseFromParent();
5446 return true;
5447}
5448
5449MachineInstr *AArch64InstructionSelector::emitLaneInsert(
5450 std::optional<Register> DstReg, Register SrcReg, Register EltReg,
5451 unsigned LaneIdx, const RegisterBank &RB,
5452 MachineIRBuilder &MIRBuilder) const {
5453 MachineInstr *InsElt = nullptr;
5454 const TargetRegisterClass *DstRC = &AArch64::FPR128RegClass;
5455 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
5456
5457 // Create a register to define with the insert if one wasn't passed in.
5458 if (!DstReg)
5459 DstReg = MRI.createVirtualRegister(RegClass: DstRC);
5460
5461 unsigned EltSize = MRI.getType(Reg: EltReg).getSizeInBits();
5462 unsigned Opc = getInsertVecEltOpInfo(RB, EltSize).first;
5463
5464 if (RB.getID() == AArch64::FPRRegBankID) {
5465 auto InsSub = emitScalarToVector(EltSize, DstRC, Scalar: EltReg, MIRBuilder);
5466 InsElt = MIRBuilder.buildInstr(Opc, DstOps: {*DstReg}, SrcOps: {SrcReg})
5467 .addImm(Val: LaneIdx)
5468 .addUse(RegNo: InsSub->getOperand(i: 0).getReg())
5469 .addImm(Val: 0);
5470 } else {
5471 InsElt = MIRBuilder.buildInstr(Opc, DstOps: {*DstReg}, SrcOps: {SrcReg})
5472 .addImm(Val: LaneIdx)
5473 .addUse(RegNo: EltReg);
5474 }
5475
5476 constrainSelectedInstRegOperands(I&: *InsElt, TII, TRI, RBI);
5477 return InsElt;
5478}
5479
5480bool AArch64InstructionSelector::selectUSMovFromExtend(
5481 MachineInstr &MI, MachineRegisterInfo &MRI) {
5482 if (MI.getOpcode() != TargetOpcode::G_SEXT &&
5483 MI.getOpcode() != TargetOpcode::G_ZEXT &&
5484 MI.getOpcode() != TargetOpcode::G_ANYEXT)
5485 return false;
5486 bool IsSigned = MI.getOpcode() == TargetOpcode::G_SEXT;
5487 const Register DefReg = MI.getOperand(i: 0).getReg();
5488 const LLT DstTy = MRI.getType(Reg: DefReg);
5489 unsigned DstSize = DstTy.getSizeInBits();
5490
5491 if (DstSize != 32 && DstSize != 64)
5492 return false;
5493
5494 MachineInstr *Extract = getOpcodeDef(Opcode: TargetOpcode::G_EXTRACT_VECTOR_ELT,
5495 Reg: MI.getOperand(i: 1).getReg(), MRI);
5496 int64_t Lane;
5497 if (!Extract || !mi_match(R: Extract->getOperand(i: 2).getReg(), MRI, P: m_ICst(Cst&: Lane)))
5498 return false;
5499 Register Src0 = Extract->getOperand(i: 1).getReg();
5500
5501 const LLT VecTy = MRI.getType(Reg: Src0);
5502 if (VecTy.isScalableVector())
5503 return false;
5504
5505 if (VecTy.getSizeInBits() != 128) {
5506 const MachineInstr *ScalarToVector = emitScalarToVector(
5507 EltSize: VecTy.getSizeInBits(), DstRC: &AArch64::FPR128RegClass, Scalar: Src0, MIRBuilder&: MIB);
5508 assert(ScalarToVector && "Didn't expect emitScalarToVector to fail!");
5509 Src0 = ScalarToVector->getOperand(i: 0).getReg();
5510 }
5511
5512 unsigned Opcode;
5513 if (DstSize == 64 && VecTy.getScalarSizeInBits() == 32)
5514 Opcode = IsSigned ? AArch64::SMOVvi32to64 : AArch64::UMOVvi32;
5515 else if (DstSize == 64 && VecTy.getScalarSizeInBits() == 16)
5516 Opcode = IsSigned ? AArch64::SMOVvi16to64 : AArch64::UMOVvi16;
5517 else if (DstSize == 64 && VecTy.getScalarSizeInBits() == 8)
5518 Opcode = IsSigned ? AArch64::SMOVvi8to64 : AArch64::UMOVvi8;
5519 else if (DstSize == 32 && VecTy.getScalarSizeInBits() == 16)
5520 Opcode = IsSigned ? AArch64::SMOVvi16to32 : AArch64::UMOVvi16;
5521 else if (DstSize == 32 && VecTy.getScalarSizeInBits() == 8)
5522 Opcode = IsSigned ? AArch64::SMOVvi8to32 : AArch64::UMOVvi8;
5523 else
5524 llvm_unreachable("Unexpected type combo for S/UMov!");
5525
5526 // We may need to generate one of these, depending on the type and sign of the
5527 // input:
5528 // DstReg = SMOV Src0, Lane;
5529 // NewReg = UMOV Src0, Lane; DstReg = SUBREG_TO_REG NewReg, sub_32;
5530 MachineInstr *ExtI = nullptr;
5531 if (DstSize == 64 && !IsSigned) {
5532 Register NewReg = MRI.createVirtualRegister(RegClass: &AArch64::GPR32RegClass);
5533 MIB.buildInstr(Opc: Opcode, DstOps: {NewReg}, SrcOps: {Src0}).addImm(Val: Lane);
5534 ExtI = MIB.buildInstr(Opc: AArch64::SUBREG_TO_REG, DstOps: {DefReg}, SrcOps: {})
5535 .addUse(RegNo: NewReg)
5536 .addImm(Val: AArch64::sub_32);
5537 RBI.constrainGenericRegister(Reg: DefReg, RC: AArch64::GPR64RegClass, MRI);
5538 } else
5539 ExtI = MIB.buildInstr(Opc: Opcode, DstOps: {DefReg}, SrcOps: {Src0}).addImm(Val: Lane);
5540
5541 constrainSelectedInstRegOperands(I&: *ExtI, TII, TRI, RBI);
5542 MI.eraseFromParent();
5543 return true;
5544}
5545
5546MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImm8(
5547 Register Dst, unsigned DstSize, APInt Bits, MachineIRBuilder &Builder) {
5548 unsigned int Op;
5549 if (DstSize == 128) {
5550 if (Bits.getHiBits(numBits: 64) != Bits.getLoBits(numBits: 64))
5551 return nullptr;
5552 Op = AArch64::MOVIv16b_ns;
5553 } else {
5554 Op = AArch64::MOVIv8b_ns;
5555 }
5556
5557 uint64_t Val = Bits.zextOrTrunc(width: 64).getZExtValue();
5558
5559 if (AArch64_AM::isAdvSIMDModImmType9(Imm: Val)) {
5560 Val = AArch64_AM::encodeAdvSIMDModImmType9(Imm: Val);
5561 auto Mov = Builder.buildInstr(Opc: Op, DstOps: {Dst}, SrcOps: {}).addImm(Val);
5562 constrainSelectedInstRegOperands(I&: *Mov, TII, TRI, RBI);
5563 return &*Mov;
5564 }
5565 return nullptr;
5566}
5567
5568MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImm16(
5569 Register Dst, unsigned DstSize, APInt Bits, MachineIRBuilder &Builder,
5570 bool Inv) {
5571
5572 unsigned int Op;
5573 if (DstSize == 128) {
5574 if (Bits.getHiBits(numBits: 64) != Bits.getLoBits(numBits: 64))
5575 return nullptr;
5576 Op = Inv ? AArch64::MVNIv8i16 : AArch64::MOVIv8i16;
5577 } else {
5578 Op = Inv ? AArch64::MVNIv4i16 : AArch64::MOVIv4i16;
5579 }
5580
5581 uint64_t Val = Bits.zextOrTrunc(width: 64).getZExtValue();
5582 uint64_t Shift;
5583
5584 if (AArch64_AM::isAdvSIMDModImmType5(Imm: Val)) {
5585 Val = AArch64_AM::encodeAdvSIMDModImmType5(Imm: Val);
5586 Shift = 0;
5587 } else if (AArch64_AM::isAdvSIMDModImmType6(Imm: Val)) {
5588 Val = AArch64_AM::encodeAdvSIMDModImmType6(Imm: Val);
5589 Shift = 8;
5590 } else
5591 return nullptr;
5592
5593 auto Mov = Builder.buildInstr(Opc: Op, DstOps: {Dst}, SrcOps: {}).addImm(Val).addImm(Val: Shift);
5594 constrainSelectedInstRegOperands(I&: *Mov, TII, TRI, RBI);
5595 return &*Mov;
5596}
5597
5598MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImm32(
5599 Register Dst, unsigned DstSize, APInt Bits, MachineIRBuilder &Builder,
5600 bool Inv) {
5601
5602 unsigned int Op;
5603 if (DstSize == 128) {
5604 if (Bits.getHiBits(numBits: 64) != Bits.getLoBits(numBits: 64))
5605 return nullptr;
5606 Op = Inv ? AArch64::MVNIv4i32 : AArch64::MOVIv4i32;
5607 } else {
5608 Op = Inv ? AArch64::MVNIv2i32 : AArch64::MOVIv2i32;
5609 }
5610
5611 uint64_t Val = Bits.zextOrTrunc(width: 64).getZExtValue();
5612 uint64_t Shift;
5613
5614 if ((AArch64_AM::isAdvSIMDModImmType1(Imm: Val))) {
5615 Val = AArch64_AM::encodeAdvSIMDModImmType1(Imm: Val);
5616 Shift = 0;
5617 } else if ((AArch64_AM::isAdvSIMDModImmType2(Imm: Val))) {
5618 Val = AArch64_AM::encodeAdvSIMDModImmType2(Imm: Val);
5619 Shift = 8;
5620 } else if ((AArch64_AM::isAdvSIMDModImmType3(Imm: Val))) {
5621 Val = AArch64_AM::encodeAdvSIMDModImmType3(Imm: Val);
5622 Shift = 16;
5623 } else if ((AArch64_AM::isAdvSIMDModImmType4(Imm: Val))) {
5624 Val = AArch64_AM::encodeAdvSIMDModImmType4(Imm: Val);
5625 Shift = 24;
5626 } else
5627 return nullptr;
5628
5629 auto Mov = Builder.buildInstr(Opc: Op, DstOps: {Dst}, SrcOps: {}).addImm(Val).addImm(Val: Shift);
5630 constrainSelectedInstRegOperands(I&: *Mov, TII, TRI, RBI);
5631 return &*Mov;
5632}
5633
5634MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImm64(
5635 Register Dst, unsigned DstSize, APInt Bits, MachineIRBuilder &Builder) {
5636
5637 unsigned int Op;
5638 if (DstSize == 128) {
5639 if (Bits.getHiBits(numBits: 64) != Bits.getLoBits(numBits: 64))
5640 return nullptr;
5641 Op = AArch64::MOVIv2d_ns;
5642 } else {
5643 Op = AArch64::MOVID;
5644 }
5645
5646 uint64_t Val = Bits.zextOrTrunc(width: 64).getZExtValue();
5647 if (AArch64_AM::isAdvSIMDModImmType10(Imm: Val)) {
5648 Val = AArch64_AM::encodeAdvSIMDModImmType10(Imm: Val);
5649 auto Mov = Builder.buildInstr(Opc: Op, DstOps: {Dst}, SrcOps: {}).addImm(Val);
5650 constrainSelectedInstRegOperands(I&: *Mov, TII, TRI, RBI);
5651 return &*Mov;
5652 }
5653 return nullptr;
5654}
5655
5656MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImm321s(
5657 Register Dst, unsigned DstSize, APInt Bits, MachineIRBuilder &Builder,
5658 bool Inv) {
5659
5660 unsigned int Op;
5661 if (DstSize == 128) {
5662 if (Bits.getHiBits(numBits: 64) != Bits.getLoBits(numBits: 64))
5663 return nullptr;
5664 Op = Inv ? AArch64::MVNIv4s_msl : AArch64::MOVIv4s_msl;
5665 } else {
5666 Op = Inv ? AArch64::MVNIv2s_msl : AArch64::MOVIv2s_msl;
5667 }
5668
5669 uint64_t Val = Bits.zextOrTrunc(width: 64).getZExtValue();
5670 uint64_t Shift;
5671
5672 if (AArch64_AM::isAdvSIMDModImmType7(Imm: Val)) {
5673 Val = AArch64_AM::encodeAdvSIMDModImmType7(Imm: Val);
5674 Shift = 264;
5675 } else if (AArch64_AM::isAdvSIMDModImmType8(Imm: Val)) {
5676 Val = AArch64_AM::encodeAdvSIMDModImmType8(Imm: Val);
5677 Shift = 272;
5678 } else
5679 return nullptr;
5680
5681 auto Mov = Builder.buildInstr(Opc: Op, DstOps: {Dst}, SrcOps: {}).addImm(Val).addImm(Val: Shift);
5682 constrainSelectedInstRegOperands(I&: *Mov, TII, TRI, RBI);
5683 return &*Mov;
5684}
5685
5686MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImmFP(
5687 Register Dst, unsigned DstSize, APInt Bits, MachineIRBuilder &Builder) {
5688
5689 unsigned int Op;
5690 bool IsWide = false;
5691 if (DstSize == 128) {
5692 if (Bits.getHiBits(numBits: 64) != Bits.getLoBits(numBits: 64))
5693 return nullptr;
5694 Op = AArch64::FMOVv4f32_ns;
5695 IsWide = true;
5696 } else {
5697 Op = AArch64::FMOVv2f32_ns;
5698 }
5699
5700 uint64_t Val = Bits.zextOrTrunc(width: 64).getZExtValue();
5701
5702 if (AArch64_AM::isAdvSIMDModImmType11(Imm: Val)) {
5703 Val = AArch64_AM::encodeAdvSIMDModImmType11(Imm: Val);
5704 } else if (IsWide && AArch64_AM::isAdvSIMDModImmType12(Imm: Val)) {
5705 Val = AArch64_AM::encodeAdvSIMDModImmType12(Imm: Val);
5706 Op = AArch64::FMOVv2f64_ns;
5707 } else
5708 return nullptr;
5709
5710 auto Mov = Builder.buildInstr(Opc: Op, DstOps: {Dst}, SrcOps: {}).addImm(Val);
5711 constrainSelectedInstRegOperands(I&: *Mov, TII, TRI, RBI);
5712 return &*Mov;
5713}
5714
5715bool AArch64InstructionSelector::selectIndexedExtLoad(
5716 MachineInstr &MI, MachineRegisterInfo &MRI) {
5717 auto &ExtLd = cast<GIndexedAnyExtLoad>(Val&: MI);
5718 Register Dst = ExtLd.getDstReg();
5719 Register WriteBack = ExtLd.getWritebackReg();
5720 Register Base = ExtLd.getBaseReg();
5721 Register Offset = ExtLd.getOffsetReg();
5722 LLT Ty = MRI.getType(Reg: Dst);
5723 assert(Ty.getSizeInBits() <= 64); // Only for scalar GPRs.
5724 unsigned MemSizeBits = ExtLd.getMMO().getMemoryType().getSizeInBits();
5725 bool IsPre = ExtLd.isPre();
5726 bool IsSExt = isa<GIndexedSExtLoad>(Val: ExtLd);
5727 unsigned InsertIntoSubReg = 0;
5728 bool IsDst64 = Ty.getSizeInBits() == 64;
5729
5730 // ZExt/SExt should be on gpr but can handle extload and zextload of fpr, so
5731 // long as they are scalar.
5732 bool IsFPR = RBI.getRegBank(Reg: Dst, MRI, TRI)->getID() == AArch64::FPRRegBankID;
5733 if ((IsSExt && IsFPR) || Ty.isVector())
5734 return false;
5735
5736 unsigned Opc = 0;
5737 LLT NewLdDstTy;
5738 LLT s32 = LLT::scalar(SizeInBits: 32);
5739 LLT s64 = LLT::scalar(SizeInBits: 64);
5740
5741 if (MemSizeBits == 8) {
5742 if (IsSExt) {
5743 if (IsDst64)
5744 Opc = IsPre ? AArch64::LDRSBXpre : AArch64::LDRSBXpost;
5745 else
5746 Opc = IsPre ? AArch64::LDRSBWpre : AArch64::LDRSBWpost;
5747 NewLdDstTy = IsDst64 ? s64 : s32;
5748 } else if (IsFPR) {
5749 Opc = IsPre ? AArch64::LDRBpre : AArch64::LDRBpost;
5750 InsertIntoSubReg = AArch64::bsub;
5751 NewLdDstTy = LLT::scalar(SizeInBits: MemSizeBits);
5752 } else {
5753 Opc = IsPre ? AArch64::LDRBBpre : AArch64::LDRBBpost;
5754 InsertIntoSubReg = IsDst64 ? AArch64::sub_32 : 0;
5755 NewLdDstTy = s32;
5756 }
5757 } else if (MemSizeBits == 16) {
5758 if (IsSExt) {
5759 if (IsDst64)
5760 Opc = IsPre ? AArch64::LDRSHXpre : AArch64::LDRSHXpost;
5761 else
5762 Opc = IsPre ? AArch64::LDRSHWpre : AArch64::LDRSHWpost;
5763 NewLdDstTy = IsDst64 ? s64 : s32;
5764 } else if (IsFPR) {
5765 Opc = IsPre ? AArch64::LDRHpre : AArch64::LDRHpost;
5766 InsertIntoSubReg = AArch64::hsub;
5767 NewLdDstTy = LLT::scalar(SizeInBits: MemSizeBits);
5768 } else {
5769 Opc = IsPre ? AArch64::LDRHHpre : AArch64::LDRHHpost;
5770 InsertIntoSubReg = IsDst64 ? AArch64::sub_32 : 0;
5771 NewLdDstTy = s32;
5772 }
5773 } else if (MemSizeBits == 32) {
5774 if (IsSExt) {
5775 Opc = IsPre ? AArch64::LDRSWpre : AArch64::LDRSWpost;
5776 NewLdDstTy = s64;
5777 } else if (IsFPR) {
5778 Opc = IsPre ? AArch64::LDRSpre : AArch64::LDRSpost;
5779 InsertIntoSubReg = AArch64::ssub;
5780 NewLdDstTy = LLT::scalar(SizeInBits: MemSizeBits);
5781 } else {
5782 Opc = IsPre ? AArch64::LDRWpre : AArch64::LDRWpost;
5783 InsertIntoSubReg = IsDst64 ? AArch64::sub_32 : 0;
5784 NewLdDstTy = s32;
5785 }
5786 } else {
5787 llvm_unreachable("Unexpected size for indexed load");
5788 }
5789
5790 auto Cst = getIConstantVRegVal(VReg: Offset, MRI);
5791 if (!Cst)
5792 return false; // Shouldn't happen, but just in case.
5793
5794 auto LdMI = MIB.buildInstr(Opc, DstOps: {WriteBack, NewLdDstTy}, SrcOps: {Base})
5795 .addImm(Val: Cst->getSExtValue());
5796 LdMI.cloneMemRefs(OtherMI: ExtLd);
5797 constrainSelectedInstRegOperands(I&: *LdMI, TII, TRI, RBI);
5798 // Make sure to select the load with the MemTy as the dest type, and then
5799 // insert into a larger reg if needed.
5800 if (InsertIntoSubReg) {
5801 // Generate a SUBREG_TO_REG.
5802 auto SubToReg = MIB.buildInstr(Opc: TargetOpcode::SUBREG_TO_REG, DstOps: {Dst}, SrcOps: {})
5803 .addUse(RegNo: LdMI.getReg(Idx: 1))
5804 .addImm(Val: InsertIntoSubReg);
5805 RBI.constrainGenericRegister(
5806 Reg: SubToReg.getReg(Idx: 0),
5807 RC: *getRegClassForTypeOnBank(Ty: MRI.getType(Reg: Dst),
5808 RB: *RBI.getRegBank(Reg: Dst, MRI, TRI)),
5809 MRI);
5810 } else {
5811 auto Copy = MIB.buildCopy(Res: Dst, Op: LdMI.getReg(Idx: 1));
5812 selectCopy(I&: *Copy, TII, MRI, TRI, RBI);
5813 }
5814 MI.eraseFromParent();
5815
5816 return true;
5817}
5818
5819bool AArch64InstructionSelector::selectIndexedLoad(MachineInstr &MI,
5820 MachineRegisterInfo &MRI) {
5821 auto &Ld = cast<GIndexedLoad>(Val&: MI);
5822 Register Dst = Ld.getDstReg();
5823 Register WriteBack = Ld.getWritebackReg();
5824 Register Base = Ld.getBaseReg();
5825 Register Offset = Ld.getOffsetReg();
5826 assert(MRI.getType(Dst).getSizeInBits() <= 128 &&
5827 "Unexpected type for indexed load");
5828 unsigned MemSize = Ld.getMMO().getMemoryType().getSizeInBytes();
5829
5830 if (MemSize < MRI.getType(Reg: Dst).getSizeInBytes())
5831 return selectIndexedExtLoad(MI, MRI);
5832
5833 unsigned Opc = 0;
5834 if (Ld.isPre()) {
5835 static constexpr unsigned GPROpcodes[] = {
5836 AArch64::LDRBBpre, AArch64::LDRHHpre, AArch64::LDRWpre,
5837 AArch64::LDRXpre};
5838 static constexpr unsigned FPROpcodes[] = {
5839 AArch64::LDRBpre, AArch64::LDRHpre, AArch64::LDRSpre, AArch64::LDRDpre,
5840 AArch64::LDRQpre};
5841 Opc = (RBI.getRegBank(Reg: Dst, MRI, TRI)->getID() == AArch64::FPRRegBankID)
5842 ? FPROpcodes[Log2_32(Value: MemSize)]
5843 : GPROpcodes[Log2_32(Value: MemSize)];
5844 ;
5845 } else {
5846 static constexpr unsigned GPROpcodes[] = {
5847 AArch64::LDRBBpost, AArch64::LDRHHpost, AArch64::LDRWpost,
5848 AArch64::LDRXpost};
5849 static constexpr unsigned FPROpcodes[] = {
5850 AArch64::LDRBpost, AArch64::LDRHpost, AArch64::LDRSpost,
5851 AArch64::LDRDpost, AArch64::LDRQpost};
5852 Opc = (RBI.getRegBank(Reg: Dst, MRI, TRI)->getID() == AArch64::FPRRegBankID)
5853 ? FPROpcodes[Log2_32(Value: MemSize)]
5854 : GPROpcodes[Log2_32(Value: MemSize)];
5855 ;
5856 }
5857 auto Cst = getIConstantVRegVal(VReg: Offset, MRI);
5858 if (!Cst)
5859 return false; // Shouldn't happen, but just in case.
5860 auto LdMI =
5861 MIB.buildInstr(Opc, DstOps: {WriteBack, Dst}, SrcOps: {Base}).addImm(Val: Cst->getSExtValue());
5862 LdMI.cloneMemRefs(OtherMI: Ld);
5863 constrainSelectedInstRegOperands(I&: *LdMI, TII, TRI, RBI);
5864 MI.eraseFromParent();
5865 return true;
5866}
5867
5868bool AArch64InstructionSelector::selectIndexedStore(GIndexedStore &I,
5869 MachineRegisterInfo &MRI) {
5870 Register Dst = I.getWritebackReg();
5871 Register Val = I.getValueReg();
5872 Register Base = I.getBaseReg();
5873 Register Offset = I.getOffsetReg();
5874 assert(MRI.getType(Val).getSizeInBits() <= 128 &&
5875 "Unexpected type for indexed store");
5876
5877 LocationSize MemSize = I.getMMO().getSize();
5878 unsigned MemSizeInBytes = MemSize.getValue();
5879
5880 assert(MemSizeInBytes && MemSizeInBytes <= 16 &&
5881 "Unexpected indexed store size");
5882 unsigned MemSizeLog2 = Log2_32(Value: MemSizeInBytes);
5883
5884 unsigned Opc = 0;
5885 if (I.isPre()) {
5886 static constexpr unsigned GPROpcodes[] = {
5887 AArch64::STRBBpre, AArch64::STRHHpre, AArch64::STRWpre,
5888 AArch64::STRXpre};
5889 static constexpr unsigned FPROpcodes[] = {
5890 AArch64::STRBpre, AArch64::STRHpre, AArch64::STRSpre, AArch64::STRDpre,
5891 AArch64::STRQpre};
5892
5893 if (RBI.getRegBank(Reg: Val, MRI, TRI)->getID() == AArch64::FPRRegBankID)
5894 Opc = FPROpcodes[MemSizeLog2];
5895 else
5896 Opc = GPROpcodes[MemSizeLog2];
5897 } else {
5898 static constexpr unsigned GPROpcodes[] = {
5899 AArch64::STRBBpost, AArch64::STRHHpost, AArch64::STRWpost,
5900 AArch64::STRXpost};
5901 static constexpr unsigned FPROpcodes[] = {
5902 AArch64::STRBpost, AArch64::STRHpost, AArch64::STRSpost,
5903 AArch64::STRDpost, AArch64::STRQpost};
5904
5905 if (RBI.getRegBank(Reg: Val, MRI, TRI)->getID() == AArch64::FPRRegBankID)
5906 Opc = FPROpcodes[MemSizeLog2];
5907 else
5908 Opc = GPROpcodes[MemSizeLog2];
5909 }
5910
5911 auto Cst = getIConstantVRegVal(VReg: Offset, MRI);
5912 if (!Cst)
5913 return false; // Shouldn't happen, but just in case.
5914 auto Str =
5915 MIB.buildInstr(Opc, DstOps: {Dst}, SrcOps: {Val, Base}).addImm(Val: Cst->getSExtValue());
5916 Str.cloneMemRefs(OtherMI: I);
5917 constrainSelectedInstRegOperands(I&: *Str, TII, TRI, RBI);
5918 I.eraseFromParent();
5919 return true;
5920}
5921
5922MachineInstr *
5923AArch64InstructionSelector::emitConstantVector(Register Dst, Constant *CV,
5924 MachineIRBuilder &MIRBuilder,
5925 MachineRegisterInfo &MRI) {
5926 LLT DstTy = MRI.getType(Reg: Dst);
5927 unsigned DstSize = DstTy.getSizeInBits();
5928 assert((DstSize == 64 || DstSize == 128) &&
5929 "Unexpected vector constant size");
5930
5931 if (CV->isNullValue()) {
5932 if (DstSize == 128) {
5933 auto Mov =
5934 MIRBuilder.buildInstr(Opc: AArch64::MOVIv2d_ns, DstOps: {Dst}, SrcOps: {}).addImm(Val: 0);
5935 constrainSelectedInstRegOperands(I&: *Mov, TII, TRI, RBI);
5936 return &*Mov;
5937 }
5938
5939 if (DstSize == 64) {
5940 auto Mov =
5941 MIRBuilder
5942 .buildInstr(Opc: AArch64::MOVIv2d_ns, DstOps: {&AArch64::FPR128RegClass}, SrcOps: {})
5943 .addImm(Val: 0);
5944 auto Copy = MIRBuilder.buildInstr(Opc: TargetOpcode::COPY, DstOps: {Dst}, SrcOps: {})
5945 .addReg(RegNo: Mov.getReg(Idx: 0), Flags: {}, SubReg: AArch64::dsub);
5946 RBI.constrainGenericRegister(Reg: Dst, RC: AArch64::FPR64RegClass, MRI);
5947 return &*Copy;
5948 }
5949 }
5950
5951 if (Constant *SplatValue = CV->getSplatValue()) {
5952 APInt SplatValueAsInt =
5953 isa<ConstantFP>(Val: SplatValue)
5954 ? cast<ConstantFP>(Val: SplatValue)->getValueAPF().bitcastToAPInt()
5955 : SplatValue->getUniqueInteger();
5956 APInt DefBits = APInt::getSplat(
5957 NewLen: DstSize, V: SplatValueAsInt.trunc(width: DstTy.getScalarSizeInBits()));
5958 auto TryMOVIWithBits = [&](APInt DefBits) -> MachineInstr * {
5959 MachineInstr *NewOp;
5960 bool Inv = false;
5961 if ((NewOp = tryAdvSIMDModImm64(Dst, DstSize, Bits: DefBits, Builder&: MIRBuilder)) ||
5962 (NewOp =
5963 tryAdvSIMDModImm32(Dst, DstSize, Bits: DefBits, Builder&: MIRBuilder, Inv)) ||
5964 (NewOp =
5965 tryAdvSIMDModImm321s(Dst, DstSize, Bits: DefBits, Builder&: MIRBuilder, Inv)) ||
5966 (NewOp =
5967 tryAdvSIMDModImm16(Dst, DstSize, Bits: DefBits, Builder&: MIRBuilder, Inv)) ||
5968 (NewOp = tryAdvSIMDModImm8(Dst, DstSize, Bits: DefBits, Builder&: MIRBuilder)) ||
5969 (NewOp = tryAdvSIMDModImmFP(Dst, DstSize, Bits: DefBits, Builder&: MIRBuilder)))
5970 return NewOp;
5971
5972 DefBits = ~DefBits;
5973 Inv = true;
5974 if ((NewOp =
5975 tryAdvSIMDModImm32(Dst, DstSize, Bits: DefBits, Builder&: MIRBuilder, Inv)) ||
5976 (NewOp =
5977 tryAdvSIMDModImm321s(Dst, DstSize, Bits: DefBits, Builder&: MIRBuilder, Inv)) ||
5978 (NewOp = tryAdvSIMDModImm16(Dst, DstSize, Bits: DefBits, Builder&: MIRBuilder, Inv)))
5979 return NewOp;
5980 return nullptr;
5981 };
5982
5983 if (auto *NewOp = TryMOVIWithBits(DefBits))
5984 return NewOp;
5985
5986 // See if a fneg of the constant can be materialized with a MOVI, etc
5987 auto TryWithFNeg = [&](APInt DefBits, int NumBits,
5988 unsigned NegOpc) -> MachineInstr * {
5989 // FNegate each sub-element of the constant
5990 APInt Neg = APInt::getHighBitsSet(numBits: NumBits, hiBitsSet: 1).zext(width: DstSize);
5991 APInt NegBits(DstSize, 0);
5992 unsigned NumElts = DstSize / NumBits;
5993 for (unsigned i = 0; i < NumElts; i++)
5994 NegBits |= Neg << (NumBits * i);
5995 NegBits = DefBits ^ NegBits;
5996
5997 // Try to create the new constants with MOVI, and if so generate a fneg
5998 // for it.
5999 if (auto *NewOp = TryMOVIWithBits(NegBits)) {
6000 Register NewDst = MRI.createVirtualRegister(
6001 RegClass: DstSize == 64 ? &AArch64::FPR64RegClass : &AArch64::FPR128RegClass);
6002 NewOp->getOperand(i: 0).setReg(NewDst);
6003 return MIRBuilder.buildInstr(Opc: NegOpc, DstOps: {Dst}, SrcOps: {NewDst});
6004 }
6005 return nullptr;
6006 };
6007 MachineInstr *R;
6008 if ((R = TryWithFNeg(DefBits, 32,
6009 DstSize == 64 ? AArch64::FNEGv2f32
6010 : AArch64::FNEGv4f32)) ||
6011 (R = TryWithFNeg(DefBits, 64,
6012 DstSize == 64 ? AArch64::FNEGDr
6013 : AArch64::FNEGv2f64)) ||
6014 (STI.hasFullFP16() &&
6015 (R = TryWithFNeg(DefBits, 16,
6016 DstSize == 64 ? AArch64::FNEGv4f16
6017 : AArch64::FNEGv8f16))))
6018 return R;
6019 }
6020
6021 auto *CPLoad = emitLoadFromConstantPool(CPVal: CV, MIRBuilder);
6022 if (!CPLoad) {
6023 LLVM_DEBUG(dbgs() << "Could not generate cp load for constant vector!");
6024 return nullptr;
6025 }
6026
6027 auto Copy = MIRBuilder.buildCopy(Res: Dst, Op: CPLoad->getOperand(i: 0));
6028 RBI.constrainGenericRegister(
6029 Reg: Dst, RC: *MRI.getRegClass(Reg: CPLoad->getOperand(i: 0).getReg()), MRI);
6030 return &*Copy;
6031}
6032
6033bool AArch64InstructionSelector::tryOptConstantBuildVec(
6034 MachineInstr &I, LLT DstTy, MachineRegisterInfo &MRI) {
6035 assert(I.getOpcode() == TargetOpcode::G_BUILD_VECTOR);
6036 unsigned DstSize = DstTy.getSizeInBits();
6037 assert(DstSize <= 128 && "Unexpected build_vec type!");
6038 if (DstSize < 32)
6039 return false;
6040 // Check if we're building a constant vector, in which case we want to
6041 // generate a constant pool load instead of a vector insert sequence.
6042 SmallVector<Constant *, 16> Csts;
6043 for (unsigned Idx = 1; Idx < I.getNumOperands(); ++Idx) {
6044 Register OpReg = I.getOperand(i: Idx).getReg();
6045 if (auto AnyConst = getAnyConstantVRegValWithLookThrough(
6046 VReg: OpReg, MRI, /*LookThroughInstrs=*/true,
6047 /*LookThroughAnyExt=*/true)) {
6048 MachineInstr *DefMI = MRI.getVRegDef(Reg: AnyConst->VReg);
6049
6050 if (DefMI->getOpcode() == TargetOpcode::G_CONSTANT) {
6051 Csts.emplace_back(
6052 Args: ConstantInt::get(Context&: MIB.getMF().getFunction().getContext(),
6053 V: std::move(AnyConst->Value)));
6054 continue;
6055 }
6056
6057 if (DefMI->getOpcode() == TargetOpcode::G_FCONSTANT) {
6058 Csts.emplace_back(
6059 Args: const_cast<ConstantFP *>(DefMI->getOperand(i: 1).getFPImm()));
6060 continue;
6061 }
6062 }
6063 return false;
6064 }
6065 Constant *CV = ConstantVector::get(V: Csts);
6066 if (!emitConstantVector(Dst: I.getOperand(i: 0).getReg(), CV, MIRBuilder&: MIB, MRI))
6067 return false;
6068 I.eraseFromParent();
6069 return true;
6070}
6071
6072bool AArch64InstructionSelector::tryOptBuildVecToSubregToReg(
6073 MachineInstr &I, MachineRegisterInfo &MRI) {
6074 // Given:
6075 // %vec = G_BUILD_VECTOR %elt, %undef, %undef, ... %undef
6076 //
6077 // Select the G_BUILD_VECTOR as a SUBREG_TO_REG from %elt.
6078 Register Dst = I.getOperand(i: 0).getReg();
6079 Register EltReg = I.getOperand(i: 1).getReg();
6080 LLT EltTy = MRI.getType(Reg: EltReg);
6081 // If the index isn't on the same bank as its elements, then this can't be a
6082 // SUBREG_TO_REG.
6083 const RegisterBank &EltRB = *RBI.getRegBank(Reg: EltReg, MRI, TRI);
6084 const RegisterBank &DstRB = *RBI.getRegBank(Reg: Dst, MRI, TRI);
6085 if (EltRB != DstRB)
6086 return false;
6087 if (any_of(Range: drop_begin(RangeOrContainer: I.operands(), N: 2), P: [&MRI](const MachineOperand &Op) {
6088 return !getOpcodeDef(Opcode: TargetOpcode::G_IMPLICIT_DEF, Reg: Op.getReg(), MRI);
6089 }))
6090 return false;
6091 unsigned SubReg;
6092 const TargetRegisterClass *EltRC = getRegClassForTypeOnBank(Ty: EltTy, RB: EltRB);
6093 if (!EltRC)
6094 return false;
6095 const TargetRegisterClass *DstRC =
6096 getRegClassForTypeOnBank(Ty: MRI.getType(Reg: Dst), RB: DstRB);
6097 if (!DstRC)
6098 return false;
6099 if (!getSubRegForClass(RC: EltRC, TRI, SubReg))
6100 return false;
6101 auto SubregToReg = MIB.buildInstr(Opc: AArch64::SUBREG_TO_REG, DstOps: {Dst}, SrcOps: {})
6102 .addUse(RegNo: EltReg)
6103 .addImm(Val: SubReg);
6104 I.eraseFromParent();
6105 constrainSelectedInstRegOperands(I&: *SubregToReg, TII, TRI, RBI);
6106 return RBI.constrainGenericRegister(Reg: Dst, RC: *DstRC, MRI);
6107}
6108
6109bool AArch64InstructionSelector::selectBuildVector(MachineInstr &I,
6110 MachineRegisterInfo &MRI) {
6111 assert(I.getOpcode() == TargetOpcode::G_BUILD_VECTOR);
6112 // Until we port more of the optimized selections, for now just use a vector
6113 // insert sequence.
6114 const LLT DstTy = MRI.getType(Reg: I.getOperand(i: 0).getReg());
6115 const LLT EltTy = MRI.getType(Reg: I.getOperand(i: 1).getReg());
6116 unsigned EltSize = EltTy.getSizeInBits();
6117
6118 if (tryOptConstantBuildVec(I, DstTy, MRI))
6119 return true;
6120 if (tryOptBuildVecToSubregToReg(I, MRI))
6121 return true;
6122
6123 if (EltSize != 8 && EltSize != 16 && EltSize != 32 && EltSize != 64)
6124 return false; // Don't support all element types yet.
6125 const RegisterBank &RB = *RBI.getRegBank(Reg: I.getOperand(i: 1).getReg(), MRI, TRI);
6126
6127 const TargetRegisterClass *DstRC = &AArch64::FPR128RegClass;
6128 MachineInstr *ScalarToVec =
6129 emitScalarToVector(EltSize: DstTy.getElementType().getSizeInBits(), DstRC,
6130 Scalar: I.getOperand(i: 1).getReg(), MIRBuilder&: MIB);
6131 if (!ScalarToVec)
6132 return false;
6133
6134 Register DstVec = ScalarToVec->getOperand(i: 0).getReg();
6135 unsigned DstSize = DstTy.getSizeInBits();
6136
6137 // Keep track of the last MI we inserted. Later on, we might be able to save
6138 // a copy using it.
6139 MachineInstr *PrevMI = ScalarToVec;
6140 for (unsigned i = 2, e = DstSize / EltSize + 1; i < e; ++i) {
6141 // Note that if we don't do a subregister copy, we can end up making an
6142 // extra register.
6143 Register OpReg = I.getOperand(i).getReg();
6144 // Do not emit inserts for undefs
6145 if (!getOpcodeDef<GImplicitDef>(Reg: OpReg, MRI)) {
6146 PrevMI = &*emitLaneInsert(DstReg: std::nullopt, SrcReg: DstVec, EltReg: OpReg, LaneIdx: i - 1, RB, MIRBuilder&: MIB);
6147 DstVec = PrevMI->getOperand(i: 0).getReg();
6148 }
6149 }
6150
6151 // If DstTy's size in bits is less than 128, then emit a subregister copy
6152 // from DstVec to the last register we've defined.
6153 if (DstSize < 128) {
6154 // Force this to be FPR using the destination vector.
6155 const TargetRegisterClass *RC =
6156 getRegClassForTypeOnBank(Ty: DstTy, RB: *RBI.getRegBank(Reg: DstVec, MRI, TRI));
6157 if (!RC)
6158 return false;
6159 if (RC != &AArch64::FPR32RegClass && RC != &AArch64::FPR64RegClass) {
6160 LLVM_DEBUG(dbgs() << "Unsupported register class!\n");
6161 return false;
6162 }
6163
6164 unsigned SubReg = 0;
6165 if (!getSubRegForClass(RC, TRI, SubReg))
6166 return false;
6167 if (SubReg != AArch64::ssub && SubReg != AArch64::dsub) {
6168 LLVM_DEBUG(dbgs() << "Unsupported destination size! (" << DstSize
6169 << "\n");
6170 return false;
6171 }
6172
6173 Register Reg = MRI.createVirtualRegister(RegClass: RC);
6174 Register DstReg = I.getOperand(i: 0).getReg();
6175
6176 MIB.buildInstr(Opc: TargetOpcode::COPY, DstOps: {DstReg}, SrcOps: {}).addReg(RegNo: DstVec, Flags: {}, SubReg);
6177 MachineOperand &RegOp = I.getOperand(i: 1);
6178 RegOp.setReg(Reg);
6179 RBI.constrainGenericRegister(Reg: DstReg, RC: *RC, MRI);
6180 } else {
6181 // We either have a vector with all elements (except the first one) undef or
6182 // at least one non-undef non-first element. In the first case, we need to
6183 // constrain the output register ourselves as we may have generated an
6184 // INSERT_SUBREG operation which is a generic operation for which the
6185 // output regclass cannot be automatically chosen.
6186 //
6187 // In the second case, there is no need to do this as it may generate an
6188 // instruction like INSvi32gpr where the regclass can be automatically
6189 // chosen.
6190 //
6191 // Also, we save a copy by re-using the destination register on the final
6192 // insert.
6193 PrevMI->getOperand(i: 0).setReg(I.getOperand(i: 0).getReg());
6194 constrainSelectedInstRegOperands(I&: *PrevMI, TII, TRI, RBI);
6195
6196 Register DstReg = PrevMI->getOperand(i: 0).getReg();
6197 if (PrevMI == ScalarToVec && DstReg.isVirtual()) {
6198 const TargetRegisterClass *RC =
6199 getRegClassForTypeOnBank(Ty: DstTy, RB: *RBI.getRegBank(Reg: DstVec, MRI, TRI));
6200 RBI.constrainGenericRegister(Reg: DstReg, RC: *RC, MRI);
6201 }
6202 }
6203
6204 I.eraseFromParent();
6205 return true;
6206}
6207
6208bool AArch64InstructionSelector::selectVectorLoadIntrinsic(unsigned Opc,
6209 unsigned NumVecs,
6210 MachineInstr &I) {
6211 assert(I.getOpcode() == TargetOpcode::G_INTRINSIC_W_SIDE_EFFECTS);
6212 assert(Opc && "Expected an opcode?");
6213 assert(NumVecs > 1 && NumVecs < 5 && "Only support 2, 3, or 4 vectors");
6214 auto &MRI = *MIB.getMRI();
6215 LLT Ty = MRI.getType(Reg: I.getOperand(i: 0).getReg());
6216 unsigned Size = Ty.getSizeInBits();
6217 assert((Size == 64 || Size == 128) &&
6218 "Destination must be 64 bits or 128 bits?");
6219 unsigned SubReg = Size == 64 ? AArch64::dsub0 : AArch64::qsub0;
6220 auto Ptr = I.getOperand(i: I.getNumOperands() - 1).getReg();
6221 assert(MRI.getType(Ptr).isPointer() && "Expected a pointer type?");
6222 auto Load = MIB.buildInstr(Opc, DstOps: {Ty}, SrcOps: {Ptr});
6223 Load.cloneMemRefs(OtherMI: I);
6224 constrainSelectedInstRegOperands(I&: *Load, TII, TRI, RBI);
6225 Register SelectedLoadDst = Load->getOperand(i: 0).getReg();
6226 for (unsigned Idx = 0; Idx < NumVecs; ++Idx) {
6227 auto Vec = MIB.buildInstr(Opc: TargetOpcode::COPY, DstOps: {I.getOperand(i: Idx)}, SrcOps: {})
6228 .addReg(RegNo: SelectedLoadDst, Flags: {}, SubReg: SubReg + Idx);
6229 // Emit the subreg copies and immediately select them.
6230 // FIXME: We should refactor our copy code into an emitCopy helper and
6231 // clean up uses of this pattern elsewhere in the selector.
6232 selectCopy(I&: *Vec, TII, MRI, TRI, RBI);
6233 }
6234 return true;
6235}
6236
6237bool AArch64InstructionSelector::selectVectorLoadLaneIntrinsic(
6238 unsigned Opc, unsigned NumVecs, MachineInstr &I) {
6239 assert(I.getOpcode() == TargetOpcode::G_INTRINSIC_W_SIDE_EFFECTS);
6240 assert(Opc && "Expected an opcode?");
6241 assert(NumVecs > 1 && NumVecs < 5 && "Only support 2, 3, or 4 vectors");
6242 auto &MRI = *MIB.getMRI();
6243 LLT Ty = MRI.getType(Reg: I.getOperand(i: 0).getReg());
6244 bool Narrow = Ty.getSizeInBits() == 64;
6245
6246 auto FirstSrcRegIt = I.operands_begin() + NumVecs + 1;
6247 SmallVector<Register, 4> Regs(NumVecs);
6248 std::transform(first: FirstSrcRegIt, last: FirstSrcRegIt + NumVecs, result: Regs.begin(),
6249 unary_op: [](auto MO) { return MO.getReg(); });
6250
6251 if (Narrow) {
6252 transform(Range&: Regs, d_first: Regs.begin(), F: [this](Register Reg) {
6253 return emitScalarToVector(EltSize: 64, DstRC: &AArch64::FPR128RegClass, Scalar: Reg, MIRBuilder&: MIB)
6254 ->getOperand(i: 0)
6255 .getReg();
6256 });
6257 Ty = Ty.multiplyElements(Factor: 2);
6258 }
6259
6260 Register Tuple = createQTuple(Regs, MIB);
6261 auto LaneNo = getIConstantVRegVal(VReg: (FirstSrcRegIt + NumVecs)->getReg(), MRI);
6262 if (!LaneNo)
6263 return false;
6264
6265 Register Ptr = (FirstSrcRegIt + NumVecs + 1)->getReg();
6266 auto Load = MIB.buildInstr(Opc, DstOps: {Ty}, SrcOps: {})
6267 .addReg(RegNo: Tuple)
6268 .addImm(Val: LaneNo->getZExtValue())
6269 .addReg(RegNo: Ptr);
6270 Load.cloneMemRefs(OtherMI: I);
6271 constrainSelectedInstRegOperands(I&: *Load, TII, TRI, RBI);
6272 Register SelectedLoadDst = Load->getOperand(i: 0).getReg();
6273 unsigned SubReg = AArch64::qsub0;
6274 for (unsigned Idx = 0; Idx < NumVecs; ++Idx) {
6275 auto Vec = MIB.buildInstr(Opc: TargetOpcode::COPY,
6276 DstOps: {Narrow ? DstOp(&AArch64::FPR128RegClass)
6277 : DstOp(I.getOperand(i: Idx).getReg())},
6278 SrcOps: {})
6279 .addReg(RegNo: SelectedLoadDst, Flags: {}, SubReg: SubReg + Idx);
6280 Register WideReg = Vec.getReg(Idx: 0);
6281 // Emit the subreg copies and immediately select them.
6282 selectCopy(I&: *Vec, TII, MRI, TRI, RBI);
6283 if (Narrow &&
6284 !emitNarrowVector(DstReg: I.getOperand(i: Idx).getReg(), SrcReg: WideReg, MIB, MRI))
6285 return false;
6286 }
6287 return true;
6288}
6289
6290void AArch64InstructionSelector::selectVectorStoreIntrinsic(MachineInstr &I,
6291 unsigned NumVecs,
6292 unsigned Opc) {
6293 MachineRegisterInfo &MRI = I.getParent()->getParent()->getRegInfo();
6294 LLT Ty = MRI.getType(Reg: I.getOperand(i: 1).getReg());
6295 Register Ptr = I.getOperand(i: 1 + NumVecs).getReg();
6296
6297 SmallVector<Register, 2> Regs(NumVecs);
6298 std::transform(first: I.operands_begin() + 1, last: I.operands_begin() + 1 + NumVecs,
6299 result: Regs.begin(), unary_op: [](auto MO) { return MO.getReg(); });
6300
6301 Register Tuple = Ty.getSizeInBits() == 128 ? createQTuple(Regs, MIB)
6302 : createDTuple(Regs, MIB);
6303 auto Store = MIB.buildInstr(Opc, DstOps: {}, SrcOps: {Tuple, Ptr});
6304 Store.cloneMemRefs(OtherMI: I);
6305 constrainSelectedInstRegOperands(I&: *Store, TII, TRI, RBI);
6306}
6307
6308bool AArch64InstructionSelector::selectVectorStoreLaneIntrinsic(
6309 MachineInstr &I, unsigned NumVecs, unsigned Opc) {
6310 MachineRegisterInfo &MRI = I.getParent()->getParent()->getRegInfo();
6311 LLT Ty = MRI.getType(Reg: I.getOperand(i: 1).getReg());
6312 bool Narrow = Ty.getSizeInBits() == 64;
6313
6314 SmallVector<Register, 2> Regs(NumVecs);
6315 std::transform(first: I.operands_begin() + 1, last: I.operands_begin() + 1 + NumVecs,
6316 result: Regs.begin(), unary_op: [](auto MO) { return MO.getReg(); });
6317
6318 if (Narrow)
6319 transform(Range&: Regs, d_first: Regs.begin(), F: [this](Register Reg) {
6320 return emitScalarToVector(EltSize: 64, DstRC: &AArch64::FPR128RegClass, Scalar: Reg, MIRBuilder&: MIB)
6321 ->getOperand(i: 0)
6322 .getReg();
6323 });
6324
6325 Register Tuple = createQTuple(Regs, MIB);
6326
6327 auto LaneNo = getIConstantVRegVal(VReg: I.getOperand(i: 1 + NumVecs).getReg(), MRI);
6328 if (!LaneNo)
6329 return false;
6330 Register Ptr = I.getOperand(i: 1 + NumVecs + 1).getReg();
6331 auto Store = MIB.buildInstr(Opc, DstOps: {}, SrcOps: {})
6332 .addReg(RegNo: Tuple)
6333 .addImm(Val: LaneNo->getZExtValue())
6334 .addReg(RegNo: Ptr);
6335 Store.cloneMemRefs(OtherMI: I);
6336 constrainSelectedInstRegOperands(I&: *Store, TII, TRI, RBI);
6337 return true;
6338}
6339
6340bool AArch64InstructionSelector::selectIntrinsicWithSideEffects(
6341 MachineInstr &I, MachineRegisterInfo &MRI) {
6342 // Find the intrinsic ID.
6343 unsigned IntrinID = cast<GIntrinsic>(Val&: I).getIntrinsicID();
6344
6345 const LLT S8 = LLT::scalar(SizeInBits: 8);
6346 const LLT S16 = LLT::scalar(SizeInBits: 16);
6347 const LLT S32 = LLT::scalar(SizeInBits: 32);
6348 const LLT S64 = LLT::scalar(SizeInBits: 64);
6349 const LLT P0 = LLT::pointer(AddressSpace: 0, SizeInBits: 64);
6350 // Select the instruction.
6351 switch (IntrinID) {
6352 default:
6353 return false;
6354 case Intrinsic::aarch64_ldxp:
6355 case Intrinsic::aarch64_ldaxp: {
6356 auto NewI = MIB.buildInstr(
6357 Opc: IntrinID == Intrinsic::aarch64_ldxp ? AArch64::LDXPX : AArch64::LDAXPX,
6358 DstOps: {I.getOperand(i: 0).getReg(), I.getOperand(i: 1).getReg()},
6359 SrcOps: {I.getOperand(i: 3)});
6360 NewI.cloneMemRefs(OtherMI: I);
6361 constrainSelectedInstRegOperands(I&: *NewI, TII, TRI, RBI);
6362 break;
6363 }
6364 case Intrinsic::aarch64_neon_ld1x2: {
6365 LLT Ty = MRI.getType(Reg: I.getOperand(i: 0).getReg());
6366 unsigned Opc = 0;
6367 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8))
6368 Opc = AArch64::LD1Twov8b;
6369 else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6370 Opc = AArch64::LD1Twov16b;
6371 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16))
6372 Opc = AArch64::LD1Twov4h;
6373 else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6374 Opc = AArch64::LD1Twov8h;
6375 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32))
6376 Opc = AArch64::LD1Twov2s;
6377 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6378 Opc = AArch64::LD1Twov4s;
6379 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) || Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0))
6380 Opc = AArch64::LD1Twov2d;
6381 else if (Ty == S64 || Ty == P0)
6382 Opc = AArch64::LD1Twov1d;
6383 else
6384 llvm_unreachable("Unexpected type for ld1x2!");
6385 selectVectorLoadIntrinsic(Opc, NumVecs: 2, I);
6386 break;
6387 }
6388 case Intrinsic::aarch64_neon_ld1x3: {
6389 LLT Ty = MRI.getType(Reg: I.getOperand(i: 0).getReg());
6390 unsigned Opc = 0;
6391 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8))
6392 Opc = AArch64::LD1Threev8b;
6393 else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6394 Opc = AArch64::LD1Threev16b;
6395 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16))
6396 Opc = AArch64::LD1Threev4h;
6397 else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6398 Opc = AArch64::LD1Threev8h;
6399 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32))
6400 Opc = AArch64::LD1Threev2s;
6401 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6402 Opc = AArch64::LD1Threev4s;
6403 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) || Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0))
6404 Opc = AArch64::LD1Threev2d;
6405 else if (Ty == S64 || Ty == P0)
6406 Opc = AArch64::LD1Threev1d;
6407 else
6408 llvm_unreachable("Unexpected type for ld1x3!");
6409 selectVectorLoadIntrinsic(Opc, NumVecs: 3, I);
6410 break;
6411 }
6412 case Intrinsic::aarch64_neon_ld1x4: {
6413 LLT Ty = MRI.getType(Reg: I.getOperand(i: 0).getReg());
6414 unsigned Opc = 0;
6415 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8))
6416 Opc = AArch64::LD1Fourv8b;
6417 else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6418 Opc = AArch64::LD1Fourv16b;
6419 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16))
6420 Opc = AArch64::LD1Fourv4h;
6421 else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6422 Opc = AArch64::LD1Fourv8h;
6423 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32))
6424 Opc = AArch64::LD1Fourv2s;
6425 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6426 Opc = AArch64::LD1Fourv4s;
6427 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) || Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0))
6428 Opc = AArch64::LD1Fourv2d;
6429 else if (Ty == S64 || Ty == P0)
6430 Opc = AArch64::LD1Fourv1d;
6431 else
6432 llvm_unreachable("Unexpected type for ld1x4!");
6433 selectVectorLoadIntrinsic(Opc, NumVecs: 4, I);
6434 break;
6435 }
6436 case Intrinsic::aarch64_neon_ld2: {
6437 LLT Ty = MRI.getType(Reg: I.getOperand(i: 0).getReg());
6438 unsigned Opc = 0;
6439 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8))
6440 Opc = AArch64::LD2Twov8b;
6441 else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6442 Opc = AArch64::LD2Twov16b;
6443 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16))
6444 Opc = AArch64::LD2Twov4h;
6445 else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6446 Opc = AArch64::LD2Twov8h;
6447 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32))
6448 Opc = AArch64::LD2Twov2s;
6449 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6450 Opc = AArch64::LD2Twov4s;
6451 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) || Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0))
6452 Opc = AArch64::LD2Twov2d;
6453 else if (Ty == S64 || Ty == P0)
6454 Opc = AArch64::LD1Twov1d;
6455 else
6456 llvm_unreachable("Unexpected type for ld2!");
6457 selectVectorLoadIntrinsic(Opc, NumVecs: 2, I);
6458 break;
6459 }
6460 case Intrinsic::aarch64_neon_ld2lane: {
6461 LLT Ty = MRI.getType(Reg: I.getOperand(i: 0).getReg());
6462 unsigned Opc;
6463 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8) || Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6464 Opc = AArch64::LD2i8;
6465 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16) || Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6466 Opc = AArch64::LD2i16;
6467 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32) || Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6468 Opc = AArch64::LD2i32;
6469 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) ||
6470 Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0) || Ty == S64 || Ty == P0)
6471 Opc = AArch64::LD2i64;
6472 else
6473 llvm_unreachable("Unexpected type for ld2lane!");
6474 if (!selectVectorLoadLaneIntrinsic(Opc, NumVecs: 2, I))
6475 return false;
6476 break;
6477 }
6478 case Intrinsic::aarch64_neon_ld2r: {
6479 LLT Ty = MRI.getType(Reg: I.getOperand(i: 0).getReg());
6480 unsigned Opc = 0;
6481 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8))
6482 Opc = AArch64::LD2Rv8b;
6483 else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6484 Opc = AArch64::LD2Rv16b;
6485 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16))
6486 Opc = AArch64::LD2Rv4h;
6487 else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6488 Opc = AArch64::LD2Rv8h;
6489 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32))
6490 Opc = AArch64::LD2Rv2s;
6491 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6492 Opc = AArch64::LD2Rv4s;
6493 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) || Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0))
6494 Opc = AArch64::LD2Rv2d;
6495 else if (Ty == S64 || Ty == P0)
6496 Opc = AArch64::LD2Rv1d;
6497 else
6498 llvm_unreachable("Unexpected type for ld2r!");
6499 selectVectorLoadIntrinsic(Opc, NumVecs: 2, I);
6500 break;
6501 }
6502 case Intrinsic::aarch64_neon_ld3: {
6503 LLT Ty = MRI.getType(Reg: I.getOperand(i: 0).getReg());
6504 unsigned Opc = 0;
6505 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8))
6506 Opc = AArch64::LD3Threev8b;
6507 else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6508 Opc = AArch64::LD3Threev16b;
6509 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16))
6510 Opc = AArch64::LD3Threev4h;
6511 else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6512 Opc = AArch64::LD3Threev8h;
6513 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32))
6514 Opc = AArch64::LD3Threev2s;
6515 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6516 Opc = AArch64::LD3Threev4s;
6517 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) || Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0))
6518 Opc = AArch64::LD3Threev2d;
6519 else if (Ty == S64 || Ty == P0)
6520 Opc = AArch64::LD1Threev1d;
6521 else
6522 llvm_unreachable("Unexpected type for ld3!");
6523 selectVectorLoadIntrinsic(Opc, NumVecs: 3, I);
6524 break;
6525 }
6526 case Intrinsic::aarch64_neon_ld3lane: {
6527 LLT Ty = MRI.getType(Reg: I.getOperand(i: 0).getReg());
6528 unsigned Opc;
6529 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8) || Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6530 Opc = AArch64::LD3i8;
6531 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16) || Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6532 Opc = AArch64::LD3i16;
6533 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32) || Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6534 Opc = AArch64::LD3i32;
6535 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) ||
6536 Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0) || Ty == S64 || Ty == P0)
6537 Opc = AArch64::LD3i64;
6538 else
6539 llvm_unreachable("Unexpected type for ld3lane!");
6540 if (!selectVectorLoadLaneIntrinsic(Opc, NumVecs: 3, I))
6541 return false;
6542 break;
6543 }
6544 case Intrinsic::aarch64_neon_ld3r: {
6545 LLT Ty = MRI.getType(Reg: I.getOperand(i: 0).getReg());
6546 unsigned Opc = 0;
6547 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8))
6548 Opc = AArch64::LD3Rv8b;
6549 else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6550 Opc = AArch64::LD3Rv16b;
6551 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16))
6552 Opc = AArch64::LD3Rv4h;
6553 else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6554 Opc = AArch64::LD3Rv8h;
6555 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32))
6556 Opc = AArch64::LD3Rv2s;
6557 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6558 Opc = AArch64::LD3Rv4s;
6559 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) || Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0))
6560 Opc = AArch64::LD3Rv2d;
6561 else if (Ty == S64 || Ty == P0)
6562 Opc = AArch64::LD3Rv1d;
6563 else
6564 llvm_unreachable("Unexpected type for ld3r!");
6565 selectVectorLoadIntrinsic(Opc, NumVecs: 3, I);
6566 break;
6567 }
6568 case Intrinsic::aarch64_neon_ld4: {
6569 LLT Ty = MRI.getType(Reg: I.getOperand(i: 0).getReg());
6570 unsigned Opc = 0;
6571 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8))
6572 Opc = AArch64::LD4Fourv8b;
6573 else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6574 Opc = AArch64::LD4Fourv16b;
6575 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16))
6576 Opc = AArch64::LD4Fourv4h;
6577 else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6578 Opc = AArch64::LD4Fourv8h;
6579 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32))
6580 Opc = AArch64::LD4Fourv2s;
6581 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6582 Opc = AArch64::LD4Fourv4s;
6583 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) || Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0))
6584 Opc = AArch64::LD4Fourv2d;
6585 else if (Ty == S64 || Ty == P0)
6586 Opc = AArch64::LD1Fourv1d;
6587 else
6588 llvm_unreachable("Unexpected type for ld4!");
6589 selectVectorLoadIntrinsic(Opc, NumVecs: 4, I);
6590 break;
6591 }
6592 case Intrinsic::aarch64_neon_ld4lane: {
6593 LLT Ty = MRI.getType(Reg: I.getOperand(i: 0).getReg());
6594 unsigned Opc;
6595 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8) || Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6596 Opc = AArch64::LD4i8;
6597 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16) || Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6598 Opc = AArch64::LD4i16;
6599 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32) || Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6600 Opc = AArch64::LD4i32;
6601 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) ||
6602 Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0) || Ty == S64 || Ty == P0)
6603 Opc = AArch64::LD4i64;
6604 else
6605 llvm_unreachable("Unexpected type for ld4lane!");
6606 if (!selectVectorLoadLaneIntrinsic(Opc, NumVecs: 4, I))
6607 return false;
6608 break;
6609 }
6610 case Intrinsic::aarch64_neon_ld4r: {
6611 LLT Ty = MRI.getType(Reg: I.getOperand(i: 0).getReg());
6612 unsigned Opc = 0;
6613 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8))
6614 Opc = AArch64::LD4Rv8b;
6615 else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6616 Opc = AArch64::LD4Rv16b;
6617 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16))
6618 Opc = AArch64::LD4Rv4h;
6619 else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6620 Opc = AArch64::LD4Rv8h;
6621 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32))
6622 Opc = AArch64::LD4Rv2s;
6623 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6624 Opc = AArch64::LD4Rv4s;
6625 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) || Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0))
6626 Opc = AArch64::LD4Rv2d;
6627 else if (Ty == S64 || Ty == P0)
6628 Opc = AArch64::LD4Rv1d;
6629 else
6630 llvm_unreachable("Unexpected type for ld4r!");
6631 selectVectorLoadIntrinsic(Opc, NumVecs: 4, I);
6632 break;
6633 }
6634 case Intrinsic::aarch64_neon_st1x2: {
6635 LLT Ty = MRI.getType(Reg: I.getOperand(i: 1).getReg());
6636 unsigned Opc;
6637 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8))
6638 Opc = AArch64::ST1Twov8b;
6639 else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6640 Opc = AArch64::ST1Twov16b;
6641 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16))
6642 Opc = AArch64::ST1Twov4h;
6643 else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6644 Opc = AArch64::ST1Twov8h;
6645 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32))
6646 Opc = AArch64::ST1Twov2s;
6647 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6648 Opc = AArch64::ST1Twov4s;
6649 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) || Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0))
6650 Opc = AArch64::ST1Twov2d;
6651 else if (Ty == S64 || Ty == P0)
6652 Opc = AArch64::ST1Twov1d;
6653 else
6654 llvm_unreachable("Unexpected type for st1x2!");
6655 selectVectorStoreIntrinsic(I, NumVecs: 2, Opc);
6656 break;
6657 }
6658 case Intrinsic::aarch64_neon_st1x3: {
6659 LLT Ty = MRI.getType(Reg: I.getOperand(i: 1).getReg());
6660 unsigned Opc;
6661 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8))
6662 Opc = AArch64::ST1Threev8b;
6663 else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6664 Opc = AArch64::ST1Threev16b;
6665 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16))
6666 Opc = AArch64::ST1Threev4h;
6667 else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6668 Opc = AArch64::ST1Threev8h;
6669 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32))
6670 Opc = AArch64::ST1Threev2s;
6671 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6672 Opc = AArch64::ST1Threev4s;
6673 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) || Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0))
6674 Opc = AArch64::ST1Threev2d;
6675 else if (Ty == S64 || Ty == P0)
6676 Opc = AArch64::ST1Threev1d;
6677 else
6678 llvm_unreachable("Unexpected type for st1x3!");
6679 selectVectorStoreIntrinsic(I, NumVecs: 3, Opc);
6680 break;
6681 }
6682 case Intrinsic::aarch64_neon_st1x4: {
6683 LLT Ty = MRI.getType(Reg: I.getOperand(i: 1).getReg());
6684 unsigned Opc;
6685 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8))
6686 Opc = AArch64::ST1Fourv8b;
6687 else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6688 Opc = AArch64::ST1Fourv16b;
6689 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16))
6690 Opc = AArch64::ST1Fourv4h;
6691 else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6692 Opc = AArch64::ST1Fourv8h;
6693 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32))
6694 Opc = AArch64::ST1Fourv2s;
6695 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6696 Opc = AArch64::ST1Fourv4s;
6697 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) || Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0))
6698 Opc = AArch64::ST1Fourv2d;
6699 else if (Ty == S64 || Ty == P0)
6700 Opc = AArch64::ST1Fourv1d;
6701 else
6702 llvm_unreachable("Unexpected type for st1x4!");
6703 selectVectorStoreIntrinsic(I, NumVecs: 4, Opc);
6704 break;
6705 }
6706 case Intrinsic::aarch64_neon_st2: {
6707 LLT Ty = MRI.getType(Reg: I.getOperand(i: 1).getReg());
6708 unsigned Opc;
6709 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8))
6710 Opc = AArch64::ST2Twov8b;
6711 else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6712 Opc = AArch64::ST2Twov16b;
6713 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16))
6714 Opc = AArch64::ST2Twov4h;
6715 else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6716 Opc = AArch64::ST2Twov8h;
6717 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32))
6718 Opc = AArch64::ST2Twov2s;
6719 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6720 Opc = AArch64::ST2Twov4s;
6721 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) || Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0))
6722 Opc = AArch64::ST2Twov2d;
6723 else if (Ty == S64 || Ty == P0)
6724 Opc = AArch64::ST1Twov1d;
6725 else
6726 llvm_unreachable("Unexpected type for st2!");
6727 selectVectorStoreIntrinsic(I, NumVecs: 2, Opc);
6728 break;
6729 }
6730 case Intrinsic::aarch64_neon_st3: {
6731 LLT Ty = MRI.getType(Reg: I.getOperand(i: 1).getReg());
6732 unsigned Opc;
6733 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8))
6734 Opc = AArch64::ST3Threev8b;
6735 else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6736 Opc = AArch64::ST3Threev16b;
6737 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16))
6738 Opc = AArch64::ST3Threev4h;
6739 else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6740 Opc = AArch64::ST3Threev8h;
6741 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32))
6742 Opc = AArch64::ST3Threev2s;
6743 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6744 Opc = AArch64::ST3Threev4s;
6745 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) || Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0))
6746 Opc = AArch64::ST3Threev2d;
6747 else if (Ty == S64 || Ty == P0)
6748 Opc = AArch64::ST1Threev1d;
6749 else
6750 llvm_unreachable("Unexpected type for st3!");
6751 selectVectorStoreIntrinsic(I, NumVecs: 3, Opc);
6752 break;
6753 }
6754 case Intrinsic::aarch64_neon_st4: {
6755 LLT Ty = MRI.getType(Reg: I.getOperand(i: 1).getReg());
6756 unsigned Opc;
6757 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8))
6758 Opc = AArch64::ST4Fourv8b;
6759 else if (Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6760 Opc = AArch64::ST4Fourv16b;
6761 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16))
6762 Opc = AArch64::ST4Fourv4h;
6763 else if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6764 Opc = AArch64::ST4Fourv8h;
6765 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32))
6766 Opc = AArch64::ST4Fourv2s;
6767 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6768 Opc = AArch64::ST4Fourv4s;
6769 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) || Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0))
6770 Opc = AArch64::ST4Fourv2d;
6771 else if (Ty == S64 || Ty == P0)
6772 Opc = AArch64::ST1Fourv1d;
6773 else
6774 llvm_unreachable("Unexpected type for st4!");
6775 selectVectorStoreIntrinsic(I, NumVecs: 4, Opc);
6776 break;
6777 }
6778 case Intrinsic::aarch64_neon_st2lane: {
6779 LLT Ty = MRI.getType(Reg: I.getOperand(i: 1).getReg());
6780 unsigned Opc;
6781 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8) || Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6782 Opc = AArch64::ST2i8;
6783 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16) || Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6784 Opc = AArch64::ST2i16;
6785 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32) || Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6786 Opc = AArch64::ST2i32;
6787 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) ||
6788 Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0) || Ty == S64 || Ty == P0)
6789 Opc = AArch64::ST2i64;
6790 else
6791 llvm_unreachable("Unexpected type for st2lane!");
6792 if (!selectVectorStoreLaneIntrinsic(I, NumVecs: 2, Opc))
6793 return false;
6794 break;
6795 }
6796 case Intrinsic::aarch64_neon_st3lane: {
6797 LLT Ty = MRI.getType(Reg: I.getOperand(i: 1).getReg());
6798 unsigned Opc;
6799 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8) || Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6800 Opc = AArch64::ST3i8;
6801 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16) || Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6802 Opc = AArch64::ST3i16;
6803 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32) || Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6804 Opc = AArch64::ST3i32;
6805 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) ||
6806 Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0) || Ty == S64 || Ty == P0)
6807 Opc = AArch64::ST3i64;
6808 else
6809 llvm_unreachable("Unexpected type for st3lane!");
6810 if (!selectVectorStoreLaneIntrinsic(I, NumVecs: 3, Opc))
6811 return false;
6812 break;
6813 }
6814 case Intrinsic::aarch64_neon_st4lane: {
6815 LLT Ty = MRI.getType(Reg: I.getOperand(i: 1).getReg());
6816 unsigned Opc;
6817 if (Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S8) || Ty == LLT::fixed_vector(NumElements: 16, ScalarTy: S8))
6818 Opc = AArch64::ST4i8;
6819 else if (Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S16) || Ty == LLT::fixed_vector(NumElements: 8, ScalarTy: S16))
6820 Opc = AArch64::ST4i16;
6821 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S32) || Ty == LLT::fixed_vector(NumElements: 4, ScalarTy: S32))
6822 Opc = AArch64::ST4i32;
6823 else if (Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: S64) ||
6824 Ty == LLT::fixed_vector(NumElements: 2, ScalarTy: P0) || Ty == S64 || Ty == P0)
6825 Opc = AArch64::ST4i64;
6826 else
6827 llvm_unreachable("Unexpected type for st4lane!");
6828 if (!selectVectorStoreLaneIntrinsic(I, NumVecs: 4, Opc))
6829 return false;
6830 break;
6831 }
6832 case Intrinsic::aarch64_mops_memset_tag: {
6833 // Transform
6834 // %dst:gpr(p0) = \
6835 // G_INTRINSIC_W_SIDE_EFFECTS intrinsic(@llvm.aarch64.mops.memset.tag),
6836 // \ %dst:gpr(p0), %val:gpr(s64), %n:gpr(s64)
6837 // where %dst is updated, into
6838 // %Rd:GPR64common, %Rn:GPR64) = \
6839 // MOPSMemorySetTaggingPseudo \
6840 // %Rd:GPR64common, %Rn:GPR64, %Rm:GPR64
6841 // where Rd and Rn are tied.
6842 // It is expected that %val has been extended to s64 in legalization.
6843 // Note that the order of the size/value operands are swapped.
6844
6845 Register DstDef = I.getOperand(i: 0).getReg();
6846 // I.getOperand(1) is the intrinsic function
6847 Register DstUse = I.getOperand(i: 2).getReg();
6848 Register ValUse = I.getOperand(i: 3).getReg();
6849 Register SizeUse = I.getOperand(i: 4).getReg();
6850
6851 // MOPSMemorySetTaggingPseudo has two defs; the intrinsic call has only one.
6852 // Therefore an additional virtual register is required for the updated size
6853 // operand. This value is not accessible via the semantics of the intrinsic.
6854 Register SizeDef = MRI.createGenericVirtualRegister(Ty: LLT::scalar(SizeInBits: 64));
6855
6856 auto Memset = MIB.buildInstr(Opc: AArch64::MOPSMemorySetTaggingPseudo,
6857 DstOps: {DstDef, SizeDef}, SrcOps: {DstUse, SizeUse, ValUse});
6858 Memset.cloneMemRefs(OtherMI: I);
6859 Memset.setOperandDead(5); // implicit-def $nzcv
6860 constrainSelectedInstRegOperands(I&: *Memset, TII, TRI, RBI);
6861 break;
6862 }
6863 case Intrinsic::ptrauth_resign_load_relative: {
6864 Register DstReg = I.getOperand(i: 0).getReg();
6865 Register ValReg = I.getOperand(i: 2).getReg();
6866 uint64_t AUTKey = I.getOperand(i: 3).getImm();
6867 Register AUTDisc = I.getOperand(i: 4).getReg();
6868 uint64_t PACKey = I.getOperand(i: 5).getImm();
6869 Register PACDisc = I.getOperand(i: 6).getReg();
6870 int64_t Addend = I.getOperand(i: 7).getImm();
6871
6872 Register AUTAddrDisc = AUTDisc;
6873 uint16_t AUTConstDiscC = 0;
6874 std::tie(args&: AUTConstDiscC, args&: AUTAddrDisc) =
6875 extractPtrauthBlendDiscriminators(Disc: AUTDisc, MRI);
6876
6877 Register PACAddrDisc = PACDisc;
6878 uint16_t PACConstDiscC = 0;
6879 std::tie(args&: PACConstDiscC, args&: PACAddrDisc) =
6880 extractPtrauthBlendDiscriminators(Disc: PACDisc, MRI);
6881
6882 MIB.buildCopy(Res: {AArch64::X16}, Op: {ValReg});
6883
6884 MIB.buildInstr(Opcode: AArch64::AUTRELLOADPAC)
6885 .addImm(Val: AUTKey)
6886 .addImm(Val: AUTConstDiscC)
6887 .addUse(RegNo: AUTAddrDisc)
6888 .addImm(Val: PACKey)
6889 .addImm(Val: PACConstDiscC)
6890 .addUse(RegNo: PACAddrDisc)
6891 .addImm(Val: Addend)
6892 .setOperandDead(8) // implicit-def $x17
6893 .setOperandDead(9) // implicit-def $nzcv
6894 .constrainAllUses(TII, TRI, RBI);
6895 MIB.buildCopy(Res: {DstReg}, Op: Register(AArch64::X16));
6896
6897 RBI.constrainGenericRegister(Reg: DstReg, RC: AArch64::GPR64RegClass, MRI);
6898 I.eraseFromParent();
6899 return true;
6900 }
6901 }
6902
6903 I.eraseFromParent();
6904 return true;
6905}
6906
6907bool AArch64InstructionSelector::selectIntrinsic(MachineInstr &I,
6908 MachineRegisterInfo &MRI) {
6909 unsigned IntrinID = cast<GIntrinsic>(Val&: I).getIntrinsicID();
6910
6911 switch (IntrinID) {
6912 default:
6913 break;
6914 case Intrinsic::ptrauth_resign: {
6915 Register DstReg = I.getOperand(i: 0).getReg();
6916 Register ValReg = I.getOperand(i: 2).getReg();
6917 uint64_t AUTKey = I.getOperand(i: 3).getImm();
6918 Register AUTDisc = I.getOperand(i: 4).getReg();
6919 uint64_t PACKey = I.getOperand(i: 5).getImm();
6920 Register PACDisc = I.getOperand(i: 6).getReg();
6921
6922 Register AUTAddrDisc = AUTDisc;
6923 uint16_t AUTConstDiscC = 0;
6924 std::tie(args&: AUTConstDiscC, args&: AUTAddrDisc) =
6925 extractPtrauthBlendDiscriminators(Disc: AUTDisc, MRI);
6926
6927 Register PACAddrDisc = PACDisc;
6928 uint16_t PACConstDiscC = 0;
6929 std::tie(args&: PACConstDiscC, args&: PACAddrDisc) =
6930 extractPtrauthBlendDiscriminators(Disc: PACDisc, MRI);
6931
6932 MIB.buildCopy(Res: {AArch64::X16}, Op: {ValReg});
6933 MIB.buildInstr(Opcode: AArch64::AUTPAC)
6934 .addImm(Val: AUTKey)
6935 .addImm(Val: AUTConstDiscC)
6936 .addUse(RegNo: AUTAddrDisc)
6937 .addImm(Val: PACKey)
6938 .addImm(Val: PACConstDiscC)
6939 .addUse(RegNo: PACAddrDisc)
6940 .setOperandDead(7) // implicit-def $x17
6941 .setOperandDead(8) // implicit-def $nzcv
6942 .constrainAllUses(TII, TRI, RBI);
6943 MIB.buildCopy(Res: {DstReg}, Op: Register(AArch64::X16));
6944
6945 RBI.constrainGenericRegister(Reg: DstReg, RC: AArch64::GPR64RegClass, MRI);
6946 I.eraseFromParent();
6947 return true;
6948 }
6949 case Intrinsic::ptrauth_auth_with_pc_and_resign: {
6950 Register DstReg = I.getOperand(i: 0).getReg();
6951 Register ValReg = I.getOperand(i: 2).getReg();
6952 uint64_t AUTKey = I.getOperand(i: 3).getImm();
6953 Register AUTDisc = I.getOperand(i: 4).getReg();
6954 Register AUTPC = I.getOperand(i: 5).getReg();
6955 uint64_t PACKey = I.getOperand(i: 6).getImm();
6956 Register PACDisc = I.getOperand(i: 7).getReg();
6957
6958 assert((AUTKey == AArch64PACKey::IA || AUTKey == AArch64PACKey::IB) &&
6959 "auth_with_pc_and_resign only supports IA and IB keys");
6960
6961 uint16_t PACConstDiscC = 0;
6962 Register PACAddrDisc;
6963 std::tie(args&: PACConstDiscC, args&: PACAddrDisc) =
6964 extractPtrauthBlendDiscriminators(Disc: PACDisc, MRI);
6965
6966 if (!PACAddrDisc.isValid())
6967 PACAddrDisc = AArch64::XZR;
6968
6969 MIB.buildCopy(Res: {AArch64::X17}, Op: {ValReg});
6970 MIB.buildCopy(Res: {AArch64::X16}, Op: {AUTDisc});
6971 MIB.buildCopy(Res: {AArch64::X15}, Op: {AUTPC});
6972
6973 MIB.buildInstr(Opcode: AArch64::AUTPCPAC)
6974 .addImm(Val: AUTKey)
6975 .addImm(Val: PACKey)
6976 .addImm(Val: PACConstDiscC)
6977 .addUse(RegNo: PACAddrDisc)
6978 .setOperandDead(5) // implicit-def $x15
6979 .setOperandDead(6) // implicit-def $x16
6980 .setOperandDead(7) // implicit-def $nzcv
6981 .constrainAllUses(TII, TRI, RBI);
6982
6983 MIB.buildCopy(Res: {DstReg}, Op: Register(AArch64::X17));
6984 RBI.constrainGenericRegister(Reg: DstReg, RC: AArch64::GPR64RegClass, MRI);
6985 I.eraseFromParent();
6986 return true;
6987 }
6988 case Intrinsic::ptrauth_auth: {
6989 Register DstReg = I.getOperand(i: 0).getReg();
6990 Register ValReg = I.getOperand(i: 2).getReg();
6991 uint64_t AUTKey = I.getOperand(i: 3).getImm();
6992 Register AUTDisc = I.getOperand(i: 4).getReg();
6993
6994 Register AUTAddrDisc = AUTDisc;
6995 uint16_t AUTConstDiscC = 0;
6996 std::tie(args&: AUTConstDiscC, args&: AUTAddrDisc) =
6997 extractPtrauthBlendDiscriminators(Disc: AUTDisc, MRI);
6998
6999 if (STI.isX16X17Safer()) {
7000 MIB.buildCopy(Res: {AArch64::X16}, Op: {ValReg});
7001 MIB.buildInstr(Opcode: AArch64::AUTx16x17)
7002 .addImm(Val: AUTKey)
7003 .addImm(Val: AUTConstDiscC)
7004 .addUse(RegNo: AUTAddrDisc)
7005 .setOperandDead(4) // implicit-def $x17
7006 .setOperandDead(5) // implicit-def $nzcv
7007 .constrainAllUses(TII, TRI, RBI);
7008 MIB.buildCopy(Res: {DstReg}, Op: Register(AArch64::X16));
7009 } else {
7010 Register ScratchReg =
7011 MRI.createVirtualRegister(RegClass: &AArch64::GPR64commonRegClass);
7012 auto Auth = MIB.buildInstr(Opcode: AArch64::AUTxMxN)
7013 .addDef(RegNo: DstReg)
7014 .addDef(RegNo: ScratchReg)
7015 .addUse(RegNo: ValReg)
7016 .addImm(Val: AUTKey)
7017 .addImm(Val: AUTConstDiscC)
7018 .addUse(RegNo: AUTAddrDisc);
7019 Auth->setImplicitPhysRegDefsDead();
7020 Auth.constrainAllUses(TII, TRI, RBI);
7021 }
7022
7023 RBI.constrainGenericRegister(Reg: DstReg, RC: AArch64::GPR64RegClass, MRI);
7024 I.eraseFromParent();
7025 return true;
7026 }
7027 case Intrinsic::frameaddress:
7028 case Intrinsic::returnaddress: {
7029 MachineFunction &MF = *I.getParent()->getParent();
7030 MachineFrameInfo &MFI = MF.getFrameInfo();
7031
7032 unsigned Depth = I.getOperand(i: 2).getImm();
7033 Register DstReg = I.getOperand(i: 0).getReg();
7034 RBI.constrainGenericRegister(Reg: DstReg, RC: AArch64::GPR64RegClass, MRI);
7035
7036 if (Depth == 0 && IntrinID == Intrinsic::returnaddress) {
7037 if (!MFReturnAddr) {
7038 // Insert the copy from LR/X30 into the entry block, before it can be
7039 // clobbered by anything.
7040 MFI.setReturnAddressIsTaken(true);
7041 MFReturnAddr = getFunctionLiveInPhysReg(
7042 MF, TII, PhysReg: AArch64::LR, RC: AArch64::GPR64RegClass, DL: I.getDebugLoc());
7043 }
7044
7045 if (STI.hasPAuth()) {
7046 MIB.buildInstr(Opc: AArch64::XPACI, DstOps: {DstReg}, SrcOps: {MFReturnAddr});
7047 } else {
7048 MIB.buildCopy(Res: {Register(AArch64::LR)}, Op: {MFReturnAddr});
7049 MIB.buildInstr(Opcode: AArch64::XPACLRI);
7050 MIB.buildCopy(Res: {DstReg}, Op: {Register(AArch64::LR)});
7051 }
7052
7053 I.eraseFromParent();
7054 return true;
7055 }
7056
7057 MFI.setFrameAddressIsTaken(true);
7058 Register FrameAddr(AArch64::FP);
7059 while (Depth--) {
7060 Register NextFrame = MRI.createVirtualRegister(RegClass: &AArch64::GPR64spRegClass);
7061 auto Ldr =
7062 MIB.buildInstr(Opc: AArch64::LDRXui, DstOps: {NextFrame}, SrcOps: {FrameAddr}).addImm(Val: 0);
7063 constrainSelectedInstRegOperands(I&: *Ldr, TII, TRI, RBI);
7064 FrameAddr = NextFrame;
7065 }
7066
7067 if (IntrinID == Intrinsic::frameaddress)
7068 MIB.buildCopy(Res: {DstReg}, Op: {FrameAddr});
7069 else {
7070 MFI.setReturnAddressIsTaken(true);
7071
7072 if (STI.hasPAuth()) {
7073 Register TmpReg = MRI.createVirtualRegister(RegClass: &AArch64::GPR64RegClass);
7074 MIB.buildInstr(Opc: AArch64::LDRXui, DstOps: {TmpReg}, SrcOps: {FrameAddr}).addImm(Val: 1);
7075 MIB.buildInstr(Opc: AArch64::XPACI, DstOps: {DstReg}, SrcOps: {TmpReg});
7076 } else {
7077 MIB.buildInstr(Opc: AArch64::LDRXui, DstOps: {Register(AArch64::LR)}, SrcOps: {FrameAddr})
7078 .addImm(Val: 1);
7079 MIB.buildInstr(Opcode: AArch64::XPACLRI);
7080 MIB.buildCopy(Res: {DstReg}, Op: {Register(AArch64::LR)});
7081 }
7082 }
7083
7084 I.eraseFromParent();
7085 return true;
7086 }
7087 case Intrinsic::aarch64_neon_tbl2:
7088 SelectTable(I, MRI, NumVecs: 2, Opc1: AArch64::TBLv8i8Two, Opc2: AArch64::TBLv16i8Two, isExt: false);
7089 return true;
7090 case Intrinsic::aarch64_neon_tbl3:
7091 SelectTable(I, MRI, NumVecs: 3, Opc1: AArch64::TBLv8i8Three, Opc2: AArch64::TBLv16i8Three,
7092 isExt: false);
7093 return true;
7094 case Intrinsic::aarch64_neon_tbl4:
7095 SelectTable(I, MRI, NumVecs: 4, Opc1: AArch64::TBLv8i8Four, Opc2: AArch64::TBLv16i8Four, isExt: false);
7096 return true;
7097 case Intrinsic::aarch64_neon_tbx2:
7098 SelectTable(I, MRI, NumVecs: 2, Opc1: AArch64::TBXv8i8Two, Opc2: AArch64::TBXv16i8Two, isExt: true);
7099 return true;
7100 case Intrinsic::aarch64_neon_tbx3:
7101 SelectTable(I, MRI, NumVecs: 3, Opc1: AArch64::TBXv8i8Three, Opc2: AArch64::TBXv16i8Three, isExt: true);
7102 return true;
7103 case Intrinsic::aarch64_neon_tbx4:
7104 SelectTable(I, MRI, NumVecs: 4, Opc1: AArch64::TBXv8i8Four, Opc2: AArch64::TBXv16i8Four, isExt: true);
7105 return true;
7106 case Intrinsic::swift_async_context_addr:
7107 auto Sub = MIB.buildInstr(Opc: AArch64::SUBXri, DstOps: {I.getOperand(i: 0).getReg()},
7108 SrcOps: {Register(AArch64::FP)})
7109 .addImm(Val: 8)
7110 .addImm(Val: 0);
7111 constrainSelectedInstRegOperands(I&: *Sub, TII, TRI, RBI);
7112
7113 MF->getFrameInfo().setFrameAddressIsTaken(true);
7114 MF->getInfo<AArch64FunctionInfo>()->setHasSwiftAsyncContext(true);
7115 I.eraseFromParent();
7116 return true;
7117 }
7118 return false;
7119}
7120
7121// G_PTRAUTH_GLOBAL_VALUE lowering
7122//
7123// We have 3 lowering alternatives to choose from:
7124// - MOVaddrPAC: similar to MOVaddr, with added PAC.
7125// If the GV doesn't need a GOT load (i.e., is locally defined)
7126// materialize the pointer using adrp+add+pac. See LowerMOVaddrPAC.
7127//
7128// - LOADgotPAC: similar to LOADgot, with added PAC.
7129// If the GV needs a GOT load, materialize the pointer using the usual
7130// GOT adrp+ldr, +pac. Pointers in GOT are assumed to be not signed, the GOT
7131// section is assumed to be read-only (for example, via relro mechanism). See
7132// LowerMOVaddrPAC.
7133//
7134// - LOADauthptrstatic: similar to LOADgot, but use a
7135// special stub slot instead of a GOT slot.
7136// Load a signed pointer for symbol 'sym' from a stub slot named
7137// 'sym$auth_ptr$key$disc' filled by dynamic linker during relocation
7138// resolving. This usually lowers to adrp+ldr, but also emits an entry into
7139// .data with an
7140// @AUTH relocation. See LowerLOADauthptrstatic.
7141//
7142// All 3 are pseudos that are expand late to longer sequences: this lets us
7143// provide integrity guarantees on the to-be-signed intermediate values.
7144//
7145// LOADauthptrstatic is undesirable because it requires a large section filled
7146// with often similarly-signed pointers, making it a good harvesting target.
7147// Thus, it's only used for ptrauth references to extern_weak to avoid null
7148// checks.
7149
7150bool AArch64InstructionSelector::selectPtrAuthGlobalValue(
7151 MachineInstr &I, MachineRegisterInfo &MRI) const {
7152 Register DefReg = I.getOperand(i: 0).getReg();
7153 Register Addr = I.getOperand(i: 1).getReg();
7154 uint64_t Key = I.getOperand(i: 2).getImm();
7155 Register AddrDisc = I.getOperand(i: 3).getReg();
7156 uint64_t Disc = I.getOperand(i: 4).getImm();
7157 int64_t Offset = 0;
7158
7159 if (Key > AArch64PACKey::LAST)
7160 report_fatal_error(reason: "key in ptrauth global out of range [0, " +
7161 Twine((int)AArch64PACKey::LAST) + "]");
7162
7163 // Blend only works if the integer discriminator is 16-bit wide.
7164 if (!isUInt<16>(x: Disc))
7165 report_fatal_error(
7166 reason: "constant discriminator in ptrauth global out of range [0, 0xffff]");
7167
7168 // Choosing between 3 lowering alternatives is target-specific.
7169 if (!STI.isTargetELF() && !STI.isTargetMachO())
7170 report_fatal_error(reason: "ptrauth global lowering only supported on MachO/ELF");
7171
7172 if (!MRI.hasOneDef(RegNo: Addr))
7173 return false;
7174
7175 // First match any offset we take from the real global.
7176 const MachineInstr *DefMI = &*MRI.def_instr_begin(RegNo: Addr);
7177 if (DefMI->getOpcode() == TargetOpcode::G_PTR_ADD) {
7178 Register OffsetReg = DefMI->getOperand(i: 2).getReg();
7179 if (!MRI.hasOneDef(RegNo: OffsetReg))
7180 return false;
7181 const MachineInstr &OffsetMI = *MRI.def_instr_begin(RegNo: OffsetReg);
7182 if (OffsetMI.getOpcode() != TargetOpcode::G_CONSTANT)
7183 return false;
7184
7185 Addr = DefMI->getOperand(i: 1).getReg();
7186 if (!MRI.hasOneDef(RegNo: Addr))
7187 return false;
7188
7189 DefMI = &*MRI.def_instr_begin(RegNo: Addr);
7190 Offset = OffsetMI.getOperand(i: 1).getCImm()->getSExtValue();
7191 }
7192
7193 // We should be left with a genuine unauthenticated GlobalValue.
7194 const GlobalValue *GV;
7195 if (DefMI->getOpcode() == TargetOpcode::G_GLOBAL_VALUE) {
7196 GV = DefMI->getOperand(i: 1).getGlobal();
7197 Offset += DefMI->getOperand(i: 1).getOffset();
7198 } else if (DefMI->getOpcode() == AArch64::G_ADD_LOW) {
7199 GV = DefMI->getOperand(i: 2).getGlobal();
7200 Offset += DefMI->getOperand(i: 2).getOffset();
7201 } else {
7202 return false;
7203 }
7204
7205 MachineIRBuilder MIB(I);
7206
7207 // Classify the reference to determine whether it needs a GOT load.
7208 unsigned OpFlags = STI.ClassifyGlobalReference(GV, TM);
7209 const bool NeedsGOTLoad = ((OpFlags & AArch64II::MO_GOT) != 0);
7210 assert(((OpFlags & (~AArch64II::MO_GOT)) == 0) &&
7211 "unsupported non-GOT op flags on ptrauth global reference");
7212 assert((!GV->hasExternalWeakLinkage() || NeedsGOTLoad) &&
7213 "unsupported non-GOT reference to weak ptrauth global");
7214
7215 std::optional<APInt> AddrDiscVal = getIConstantVRegVal(VReg: AddrDisc, MRI);
7216 bool HasAddrDisc = !AddrDiscVal || *AddrDiscVal != 0;
7217
7218 // Non-extern_weak:
7219 // - No GOT load needed -> MOVaddrPAC
7220 // - GOT load for non-extern_weak -> LOADgotPAC
7221 // Note that we disallow extern_weak refs to avoid null checks later.
7222 if (!GV->hasExternalWeakLinkage()) {
7223 MIB.buildInstr(Opcode: NeedsGOTLoad ? AArch64::LOADgotPAC : AArch64::MOVaddrPAC)
7224 .addGlobalAddress(GV, Offset)
7225 .addImm(Val: Key)
7226 .addReg(RegNo: HasAddrDisc ? AddrDisc : AArch64::XZR)
7227 .addImm(Val: Disc)
7228 .constrainAllUses(TII, TRI, RBI);
7229 MIB.buildCopy(Res: DefReg, Op: Register(AArch64::X16));
7230 RBI.constrainGenericRegister(Reg: DefReg, RC: AArch64::GPR64RegClass, MRI);
7231 I.eraseFromParent();
7232 return true;
7233 }
7234
7235 // extern_weak -> LOADauthptrstatic
7236
7237 // Offsets and extern_weak don't mix well: ptrauth aside, you'd get the
7238 // offset alone as a pointer if the symbol wasn't available, which would
7239 // probably break null checks in users. Ptrauth complicates things further:
7240 // error out.
7241 if (Offset != 0)
7242 report_fatal_error(
7243 reason: "unsupported non-zero offset in weak ptrauth global reference");
7244
7245 if (HasAddrDisc)
7246 report_fatal_error(reason: "unsupported weak addr-div ptrauth global");
7247
7248 MIB.buildInstr(Opc: AArch64::LOADauthptrstatic, DstOps: {DefReg}, SrcOps: {})
7249 .addGlobalAddress(GV, Offset)
7250 .addImm(Val: Key)
7251 .addImm(Val: Disc);
7252 RBI.constrainGenericRegister(Reg: DefReg, RC: AArch64::GPR64RegClass, MRI);
7253
7254 I.eraseFromParent();
7255 return true;
7256}
7257
7258void AArch64InstructionSelector::SelectTable(MachineInstr &I,
7259 MachineRegisterInfo &MRI,
7260 unsigned NumVec, unsigned Opc1,
7261 unsigned Opc2, bool isExt) {
7262 Register DstReg = I.getOperand(i: 0).getReg();
7263 unsigned Opc = MRI.getType(Reg: DstReg) == LLT::fixed_vector(NumElements: 8, ScalarSizeInBits: 8) ? Opc1 : Opc2;
7264
7265 // Create the REG_SEQUENCE
7266 SmallVector<Register, 4> Regs;
7267 for (unsigned i = 0; i < NumVec; i++)
7268 Regs.push_back(Elt: I.getOperand(i: i + 2 + isExt).getReg());
7269 Register RegSeq = createQTuple(Regs, MIB);
7270
7271 Register IdxReg = I.getOperand(i: 2 + NumVec + isExt).getReg();
7272 MachineInstrBuilder Instr;
7273 if (isExt) {
7274 Register Reg = I.getOperand(i: 2).getReg();
7275 Instr = MIB.buildInstr(Opc, DstOps: {DstReg}, SrcOps: {Reg, RegSeq, IdxReg});
7276 } else
7277 Instr = MIB.buildInstr(Opc, DstOps: {DstReg}, SrcOps: {RegSeq, IdxReg});
7278 constrainSelectedInstRegOperands(I&: *Instr, TII, TRI, RBI);
7279 I.eraseFromParent();
7280}
7281
7282InstructionSelector::ComplexRendererFns
7283AArch64InstructionSelector::selectShiftA_32(const MachineOperand &Root) const {
7284 auto MaybeImmed = getImmedFromMO(Root);
7285 if (MaybeImmed == std::nullopt || *MaybeImmed > 31)
7286 return std::nullopt;
7287 uint64_t Enc = (32 - *MaybeImmed) & 0x1f;
7288 return {{[=](MachineInstrBuilder &MIB) { MIB.addImm(Val: Enc); }}};
7289}
7290
7291InstructionSelector::ComplexRendererFns
7292AArch64InstructionSelector::selectShiftB_32(const MachineOperand &Root) const {
7293 auto MaybeImmed = getImmedFromMO(Root);
7294 if (MaybeImmed == std::nullopt || *MaybeImmed > 31)
7295 return std::nullopt;
7296 uint64_t Enc = 31 - *MaybeImmed;
7297 return {{[=](MachineInstrBuilder &MIB) { MIB.addImm(Val: Enc); }}};
7298}
7299
7300InstructionSelector::ComplexRendererFns
7301AArch64InstructionSelector::selectShiftA_64(const MachineOperand &Root) const {
7302 auto MaybeImmed = getImmedFromMO(Root);
7303 if (MaybeImmed == std::nullopt || *MaybeImmed > 63)
7304 return std::nullopt;
7305 uint64_t Enc = (64 - *MaybeImmed) & 0x3f;
7306 return {{[=](MachineInstrBuilder &MIB) { MIB.addImm(Val: Enc); }}};
7307}
7308
7309InstructionSelector::ComplexRendererFns
7310AArch64InstructionSelector::selectShiftB_64(const MachineOperand &Root) const {
7311 auto MaybeImmed = getImmedFromMO(Root);
7312 if (MaybeImmed == std::nullopt || *MaybeImmed > 63)
7313 return std::nullopt;
7314 uint64_t Enc = 63 - *MaybeImmed;
7315 return {{[=](MachineInstrBuilder &MIB) { MIB.addImm(Val: Enc); }}};
7316}
7317
7318template <unsigned ShiftWidth>
7319InstructionSelector::ComplexRendererFns
7320AArch64InstructionSelector::selectShiftMask(MachineOperand &Root) const {
7321 if (!Root.isReg())
7322 return std::nullopt;
7323
7324 MachineRegisterInfo &MRI =
7325 Root.getParent()->getParent()->getParent()->getRegInfo();
7326
7327 Register ShAmtReg = Root.getReg();
7328
7329 // Peek through zext for i32 shifts only. For i64 shifts the zext case
7330 // is already handled by existing patterns in the Shift multiclass.
7331 if (ShiftWidth == 32) {
7332 Register ZExtSrcReg;
7333 if (mi_match(R: ShAmtReg, MRI, P: m_GZExt(Src: m_Reg(R&: ZExtSrcReg))))
7334 ShAmtReg = ZExtSrcReg;
7335 }
7336
7337 // Remove AND if the mask covers at least the low log2(ShiftWidth) bits.
7338 APInt AndMask;
7339 Register AndSrcReg;
7340 if (mi_match(R: ShAmtReg, MRI, P: m_GAnd(L: m_Reg(R&: AndSrcReg), R: m_ICst(Cst&: AndMask))) &&
7341 MRI.getType(Reg: ShAmtReg).getSizeInBits() == ShiftWidth) {
7342 if (AndMask.countr_one() >= Log2_32(Value: ShiftWidth))
7343 ShAmtReg = AndSrcReg;
7344 }
7345
7346 // If shifting by X+/-N where N == 0 mod ShiftWidth, then just shift by X
7347 // to avoid the ADD/SUB. The low log2(ShiftWidth) bits are unchanged, so the
7348 // shift can use X directly; the original ADD/SUB stays for any other users.
7349 Register AddSrcReg;
7350 int64_t AddImm;
7351 if ((mi_match(R: ShAmtReg, MRI,
7352 P: m_GAdd(L: m_Reg(R&: AddSrcReg), R: m_ICstOrSplat(Cst&: AddImm))) ||
7353 mi_match(R: ShAmtReg, MRI,
7354 P: m_GSub(L: m_Reg(R&: AddSrcReg), R: m_ICstOrSplat(Cst&: AddImm)))) &&
7355 (AddImm % ShiftWidth == 0)) {
7356 ShAmtReg = AddSrcReg;
7357 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(RegNo: ShAmtReg); }}};
7358 }
7359
7360 // If shifting by N-X where N == 0 mod ShiftWidth, then just shift by -X
7361 // to generate a NEG instead of a SUB from a constant.
7362 Register SubSrcReg;
7363 int64_t SubImm;
7364 if (MRI.hasOneUse(RegNo: ShAmtReg) &&
7365 mi_match(R: ShAmtReg, MRI, P: m_GSub(L: m_ICst(Cst&: SubImm), R: m_Reg(R&: SubSrcReg))) &&
7366 SubImm != 0 && (SubImm % ShiftWidth == 0)) {
7367 return {{[=](MachineInstrBuilder &MIB) {
7368 MachineInstr *I = MIB.getInstr();
7369 MachineRegisterInfo &MRI2 = I->getMF()->getRegInfo();
7370 const TargetRegisterClass &RC =
7371 ShiftWidth == 32 ? AArch64::GPR32RegClass : AArch64::GPR64RegClass;
7372 unsigned SubOpc = ShiftWidth == 32 ? AArch64::SUBWrr : AArch64::SUBXrr;
7373 Register ZeroReg = ShiftWidth == 32 ? AArch64::WZR : AArch64::XZR;
7374 Register NegReg = MRI2.createVirtualRegister(RegClass: &RC);
7375 auto NegMI = BuildMI(BB&: *I->getParent(), I&: *I, MIMD: I->getDebugLoc(),
7376 MCID: TII.get(Opcode: SubOpc), DestReg: NegReg)
7377 .addReg(RegNo: ZeroReg)
7378 .addReg(RegNo: SubSrcReg);
7379 constrainSelectedInstRegOperands(I&: *NegMI, TII, TRI, RBI);
7380 MIB.addReg(RegNo: NegReg);
7381 }}};
7382 }
7383
7384 // If shifting by N-X where N == -1 mod ShiftWidth, then just shift by ~X
7385 // to generate a NOT (MVN) instead of a SUB from a constant.
7386 Register NotSrcReg;
7387 int64_t NotImm;
7388 if (MRI.hasOneUse(RegNo: ShAmtReg) &&
7389 mi_match(R: ShAmtReg, MRI, P: m_GSub(L: m_ICst(Cst&: NotImm), R: m_Reg(R&: NotSrcReg))) &&
7390 (NotImm % ShiftWidth == ShiftWidth - 1)) {
7391 return {{[=](MachineInstrBuilder &MIB) {
7392 MachineInstr *I = MIB.getInstr();
7393 MachineRegisterInfo &MRI2 = I->getMF()->getRegInfo();
7394 const TargetRegisterClass &RC =
7395 ShiftWidth == 32 ? AArch64::GPR32RegClass : AArch64::GPR64RegClass;
7396 unsigned NotOpc = ShiftWidth == 32 ? AArch64::ORNWrr : AArch64::ORNXrr;
7397 Register ZeroReg = ShiftWidth == 32 ? AArch64::WZR : AArch64::XZR;
7398 Register NotReg = MRI2.createVirtualRegister(RegClass: &RC);
7399 auto NotMI = BuildMI(BB&: *I->getParent(), I&: *I, MIMD: I->getDebugLoc(),
7400 MCID: TII.get(Opcode: NotOpc), DestReg: NotReg)
7401 .addReg(RegNo: ZeroReg)
7402 .addReg(RegNo: NotSrcReg);
7403 constrainSelectedInstRegOperands(I&: *NotMI, TII, TRI, RBI);
7404 MIB.addReg(RegNo: NotReg);
7405 }}};
7406 }
7407
7408 // Only succeed if we changed the shift amount; otherwise let other
7409 // patterns (e.g. zext GPR32 -> SUBREG_TO_REG) match instead.
7410 if (ShAmtReg == Root.getReg())
7411 return std::nullopt;
7412
7413 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(RegNo: ShAmtReg); }}};
7414}
7415
7416/// Helper to select an immediate value that can be represented as a 12-bit
7417/// value shifted left by either 0 or 12. If it is possible to do so, return
7418/// the immediate and shift value. If not, return std::nullopt.
7419///
7420/// Used by selectArithImmed and selectNegArithImmed.
7421InstructionSelector::ComplexRendererFns
7422AArch64InstructionSelector::select12BitValueWithLeftShift(
7423 uint64_t Immed) const {
7424 unsigned ShiftAmt;
7425 if (Immed >> 12 == 0) {
7426 ShiftAmt = 0;
7427 } else if ((Immed & 0xfff) == 0 && Immed >> 24 == 0) {
7428 ShiftAmt = 12;
7429 Immed = Immed >> 12;
7430 } else
7431 return std::nullopt;
7432
7433 unsigned ShVal = AArch64_AM::getShifterImm(ST: AArch64_AM::LSL, Imm: ShiftAmt);
7434 return {{
7435 [=](MachineInstrBuilder &MIB) { MIB.addImm(Val: Immed); },
7436 [=](MachineInstrBuilder &MIB) { MIB.addImm(Val: ShVal); },
7437 }};
7438}
7439
7440/// SelectArithImmed - Select an immediate value that can be represented as
7441/// a 12-bit value shifted left by either 0 or 12. If so, return true with
7442/// Val set to the 12-bit value and Shift set to the shifter operand.
7443InstructionSelector::ComplexRendererFns
7444AArch64InstructionSelector::selectArithImmed(MachineOperand &Root) const {
7445 // This function is called from the addsub_shifted_imm ComplexPattern,
7446 // which lists [imm] as the list of opcode it's interested in, however
7447 // we still need to check whether the operand is actually an immediate
7448 // here because the ComplexPattern opcode list is only used in
7449 // root-level opcode matching.
7450 auto MaybeImmed = getImmedFromMO(Root);
7451 if (MaybeImmed == std::nullopt)
7452 return std::nullopt;
7453 return select12BitValueWithLeftShift(Immed: *MaybeImmed);
7454}
7455
7456/// SelectNegArithImmed - As above, but negates the value before trying to
7457/// select it.
7458InstructionSelector::ComplexRendererFns
7459AArch64InstructionSelector::selectNegArithImmed(MachineOperand &Root) const {
7460 // We need a register here, because we need to know if we have a 64 or 32
7461 // bit immediate.
7462 if (!Root.isReg())
7463 return std::nullopt;
7464 auto MaybeImmed = getImmedFromMO(Root);
7465 if (MaybeImmed == std::nullopt)
7466 return std::nullopt;
7467 uint64_t Immed = *MaybeImmed;
7468
7469 // This negation is almost always valid, but "cmp wN, #0" and "cmn wN, #0"
7470 // have the opposite effect on the C flag, so this pattern mustn't match under
7471 // those circumstances.
7472 if (Immed == 0)
7473 return std::nullopt;
7474
7475 // Check if we're dealing with a 32-bit type on the root or a 64-bit type on
7476 // the root.
7477 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
7478 if (MRI.getType(Reg: Root.getReg()).getSizeInBits() == 32)
7479 Immed = ~((uint32_t)Immed) + 1;
7480 else
7481 Immed = ~Immed + 1ULL;
7482
7483 if (Immed & 0xFFFFFFFFFF000000ULL)
7484 return std::nullopt;
7485
7486 Immed &= 0xFFFFFFULL;
7487 return select12BitValueWithLeftShift(Immed);
7488}
7489
7490/// Checks if we are sure that folding MI into load/store addressing mode is
7491/// beneficial or not.
7492///
7493/// Returns:
7494/// - true if folding MI would be beneficial.
7495/// - false if folding MI would be bad.
7496/// - std::nullopt if it is not sure whether folding MI is beneficial.
7497///
7498/// \p MI can be the offset operand of G_PTR_ADD, e.g. G_SHL in the example:
7499///
7500/// %13:gpr(s64) = G_CONSTANT i64 1
7501/// %8:gpr(s64) = G_SHL %6, %13(s64)
7502/// %9:gpr(p0) = G_PTR_ADD %0, %8(s64)
7503/// %12:gpr(s32) = G_LOAD %9(p0) :: (load (s16))
7504std::optional<bool> AArch64InstructionSelector::isWorthFoldingIntoAddrMode(
7505 const MachineInstr &MI, const MachineRegisterInfo &MRI) const {
7506 if (MI.getOpcode() == AArch64::G_SHL) {
7507 // Address operands with shifts are free, except for running on subtargets
7508 // with AddrLSLSlow14.
7509 if (const auto ValAndVeg = getIConstantVRegValWithLookThrough(
7510 VReg: MI.getOperand(i: 2).getReg(), MRI)) {
7511 const APInt ShiftVal = ValAndVeg->Value;
7512
7513 // Don't fold if we know this will be slow.
7514 return !(STI.hasAddrLSLSlow14() && (ShiftVal == 1 || ShiftVal == 4));
7515 }
7516 }
7517 return std::nullopt;
7518}
7519
7520/// Return true if it is worth folding MI into an extended register. That is,
7521/// if it's safe to pull it into the addressing mode of a load or store as a
7522/// shift.
7523/// \p IsAddrOperand whether the def of MI is used as an address operand
7524/// (e.g. feeding into an LDR/STR).
7525bool AArch64InstructionSelector::isWorthFoldingIntoExtendedReg(
7526 const MachineInstr &MI, const MachineRegisterInfo &MRI,
7527 bool IsAddrOperand) const {
7528
7529 // Always fold if there is one use, or if we're optimizing for size.
7530 Register DefReg = MI.getOperand(i: 0).getReg();
7531 if (MRI.hasOneNonDBGUse(RegNo: DefReg) ||
7532 MI.getParent()->getParent()->getFunction().hasOptSize())
7533 return true;
7534
7535 if (IsAddrOperand) {
7536 // If we are already sure that folding MI is good or bad, return the result.
7537 if (const auto Worth = isWorthFoldingIntoAddrMode(MI, MRI))
7538 return *Worth;
7539
7540 // Fold G_PTR_ADD if its offset operand can be folded
7541 if (MI.getOpcode() == AArch64::G_PTR_ADD) {
7542 MachineInstr *OffsetInst =
7543 getDefIgnoringCopies(Reg: MI.getOperand(i: 2).getReg(), MRI);
7544
7545 // Note, we already know G_PTR_ADD is used by at least two instructions.
7546 // If we are also sure about whether folding is beneficial or not,
7547 // return the result.
7548 if (const auto Worth = isWorthFoldingIntoAddrMode(MI: *OffsetInst, MRI))
7549 return *Worth;
7550 }
7551 }
7552
7553 // FIXME: Consider checking HasALULSLFast as appropriate.
7554
7555 // We have a fastpath, so folding a shift in and potentially computing it
7556 // many times may be beneficial. Check if this is only used in memory ops.
7557 // If it is, then we should fold.
7558 return all_of(Range: MRI.use_nodbg_instructions(Reg: DefReg),
7559 P: [](MachineInstr &Use) { return Use.mayLoadOrStore(); });
7560}
7561
7562InstructionSelector::ComplexRendererFns
7563AArch64InstructionSelector::selectExtendedSHL(
7564 MachineOperand &Root, MachineOperand &Base, MachineOperand &Offset,
7565 unsigned SizeInBytes, bool WantsExt) const {
7566 assert(Base.isReg() && "Expected base to be a register operand");
7567 assert(Offset.isReg() && "Expected offset to be a register operand");
7568
7569 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
7570 MachineInstr *OffsetInst = MRI.getVRegDef(Reg: Offset.getReg());
7571
7572 unsigned OffsetOpc = OffsetInst->getOpcode();
7573 bool LookedThroughZExt = false;
7574 if (OffsetOpc != TargetOpcode::G_SHL && OffsetOpc != TargetOpcode::G_MUL) {
7575 // Try to look through a ZEXT.
7576 if (OffsetOpc != TargetOpcode::G_ZEXT || !WantsExt)
7577 return std::nullopt;
7578
7579 OffsetInst = MRI.getVRegDef(Reg: OffsetInst->getOperand(i: 1).getReg());
7580 OffsetOpc = OffsetInst->getOpcode();
7581 LookedThroughZExt = true;
7582
7583 if (OffsetOpc != TargetOpcode::G_SHL && OffsetOpc != TargetOpcode::G_MUL)
7584 return std::nullopt;
7585 }
7586 // Make sure that the memory op is a valid size.
7587 int64_t LegalShiftVal = Log2_32(Value: SizeInBytes);
7588 if (LegalShiftVal == 0)
7589 return std::nullopt;
7590 if (!isWorthFoldingIntoExtendedReg(MI: *OffsetInst, MRI, IsAddrOperand: true))
7591 return std::nullopt;
7592
7593 // Now, try to find the specific G_CONSTANT. Start by assuming that the
7594 // register we will offset is the LHS, and the register containing the
7595 // constant is the RHS.
7596 Register OffsetReg = OffsetInst->getOperand(i: 1).getReg();
7597 Register ConstantReg = OffsetInst->getOperand(i: 2).getReg();
7598 auto ValAndVReg = getIConstantVRegValWithLookThrough(VReg: ConstantReg, MRI);
7599 if (!ValAndVReg) {
7600 // We didn't get a constant on the RHS. If the opcode is a shift, then
7601 // we're done.
7602 if (OffsetOpc == TargetOpcode::G_SHL)
7603 return std::nullopt;
7604
7605 // If we have a G_MUL, we can use either register. Try looking at the RHS.
7606 std::swap(a&: OffsetReg, b&: ConstantReg);
7607 ValAndVReg = getIConstantVRegValWithLookThrough(VReg: ConstantReg, MRI);
7608 if (!ValAndVReg)
7609 return std::nullopt;
7610 }
7611
7612 // The value must fit into 3 bits, and must be positive. Make sure that is
7613 // true.
7614 int64_t ImmVal = ValAndVReg->Value.getSExtValue();
7615
7616 // Since we're going to pull this into a shift, the constant value must be
7617 // a power of 2. If we got a multiply, then we need to check this.
7618 if (OffsetOpc == TargetOpcode::G_MUL) {
7619 if (!llvm::has_single_bit<uint32_t>(Value: ImmVal))
7620 return std::nullopt;
7621
7622 // Got a power of 2. So, the amount we'll shift is the log base-2 of that.
7623 ImmVal = Log2_32(Value: ImmVal);
7624 }
7625
7626 if ((ImmVal & 0x7) != ImmVal)
7627 return std::nullopt;
7628
7629 // We are only allowed to shift by LegalShiftVal. This shift value is built
7630 // into the instruction, so we can't just use whatever we want.
7631 if (ImmVal != LegalShiftVal)
7632 return std::nullopt;
7633
7634 unsigned SignExtend = 0;
7635 if (WantsExt) {
7636 // Check if the offset is defined by an extend, unless we looked through a
7637 // G_ZEXT earlier.
7638 if (!LookedThroughZExt) {
7639 MachineInstr *ExtInst = getDefIgnoringCopies(Reg: OffsetReg, MRI);
7640 auto Ext = getExtendTypeForInst(MI&: *ExtInst, MRI, IsLoadStore: true);
7641 if (Ext == AArch64_AM::InvalidShiftExtend)
7642 return std::nullopt;
7643
7644 SignExtend = AArch64_AM::isSignExtendShiftType(Type: Ext) ? 1 : 0;
7645 // We only support SXTW for signed extension here.
7646 if (SignExtend && Ext != AArch64_AM::SXTW)
7647 return std::nullopt;
7648 OffsetReg = ExtInst->getOperand(i: 1).getReg();
7649 }
7650
7651 // Need a 32-bit wide register here.
7652 MachineIRBuilder MIB(*MRI.getVRegDef(Reg: Root.getReg()));
7653 OffsetReg = moveScalarRegClass(Reg: OffsetReg, RC: AArch64::GPR32RegClass, MIB);
7654 }
7655
7656 // We can use the LHS of the GEP as the base, and the LHS of the shift as an
7657 // offset. Signify that we are shifting by setting the shift flag to 1.
7658 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(RegNo: Base.getReg()); },
7659 [=](MachineInstrBuilder &MIB) { MIB.addUse(RegNo: OffsetReg); },
7660 [=](MachineInstrBuilder &MIB) {
7661 // Need to add both immediates here to make sure that they are both
7662 // added to the instruction.
7663 MIB.addImm(Val: SignExtend);
7664 MIB.addImm(Val: 1);
7665 }}};
7666}
7667
7668/// This is used for computing addresses like this:
7669///
7670/// ldr x1, [x2, x3, lsl #3]
7671///
7672/// Where x2 is the base register, and x3 is an offset register. The shift-left
7673/// is a constant value specific to this load instruction. That is, we'll never
7674/// see anything other than a 3 here (which corresponds to the size of the
7675/// element being loaded.)
7676InstructionSelector::ComplexRendererFns
7677AArch64InstructionSelector::selectAddrModeShiftedExtendXReg(
7678 MachineOperand &Root, unsigned SizeInBytes) const {
7679 if (!Root.isReg())
7680 return std::nullopt;
7681 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
7682
7683 // We want to find something like this:
7684 //
7685 // val = G_CONSTANT LegalShiftVal
7686 // shift = G_SHL off_reg val
7687 // ptr = G_PTR_ADD base_reg shift
7688 // x = G_LOAD ptr
7689 //
7690 // And fold it into this addressing mode:
7691 //
7692 // ldr x, [base_reg, off_reg, lsl #LegalShiftVal]
7693
7694 // Check if we can find the G_PTR_ADD.
7695 MachineInstr *PtrAdd =
7696 getOpcodeDef(Opcode: TargetOpcode::G_PTR_ADD, Reg: Root.getReg(), MRI);
7697 if (!PtrAdd || !isWorthFoldingIntoExtendedReg(MI: *PtrAdd, MRI, IsAddrOperand: true))
7698 return std::nullopt;
7699
7700 // Now, try to match an opcode which will match our specific offset.
7701 // We want a G_SHL or a G_MUL.
7702 MachineInstr *OffsetInst =
7703 getDefIgnoringCopies(Reg: PtrAdd->getOperand(i: 2).getReg(), MRI);
7704 return selectExtendedSHL(Root, Base&: PtrAdd->getOperand(i: 1),
7705 Offset&: OffsetInst->getOperand(i: 0), SizeInBytes,
7706 /*WantsExt=*/false);
7707}
7708
7709/// This is used for computing addresses like this:
7710///
7711/// ldr x1, [x2, x3]
7712///
7713/// Where x2 is the base register, and x3 is an offset register.
7714///
7715/// When possible (or profitable) to fold a G_PTR_ADD into the address
7716/// calculation, this will do so. Otherwise, it will return std::nullopt.
7717InstructionSelector::ComplexRendererFns
7718AArch64InstructionSelector::selectAddrModeRegisterOffset(
7719 MachineOperand &Root) const {
7720 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
7721
7722 // We need a GEP.
7723 Register Base, Offset;
7724 if (!mi_match(R: Root.getReg(), MRI, P: m_GPtrAdd(L: m_Reg(R&: Base), R: m_Reg(R&: Offset))))
7725 return std::nullopt;
7726
7727 // If this is used more than once, let's not bother folding.
7728 // TODO: Check if they are memory ops. If they are, then we can still fold
7729 // without having to recompute anything.
7730 if (!MRI.hasOneNonDBGUse(RegNo: Root.getReg()))
7731 return std::nullopt;
7732
7733 // Base is the GEP's LHS, offset is its RHS.
7734 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(RegNo: Base); },
7735 [=](MachineInstrBuilder &MIB) { MIB.addUse(RegNo: Offset); },
7736 [=](MachineInstrBuilder &MIB) {
7737 // Need to add both immediates here to make sure that they are both
7738 // added to the instruction.
7739 MIB.addImm(Val: 0);
7740 MIB.addImm(Val: 0);
7741 }}};
7742}
7743
7744/// This is intended to be equivalent to selectAddrModeXRO in
7745/// AArch64ISelDAGtoDAG. It's used for selecting X register offset loads.
7746InstructionSelector::ComplexRendererFns
7747AArch64InstructionSelector::selectAddrModeXRO(MachineOperand &Root,
7748 unsigned SizeInBytes) const {
7749 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
7750 if (!Root.isReg())
7751 return std::nullopt;
7752 MachineInstr *PtrAdd =
7753 getOpcodeDef(Opcode: TargetOpcode::G_PTR_ADD, Reg: Root.getReg(), MRI);
7754 if (!PtrAdd)
7755 return std::nullopt;
7756
7757 // Check for an immediates which cannot be encoded in the [base + imm]
7758 // addressing mode, and can't be encoded in an add/sub. If this happens, we'll
7759 // end up with code like:
7760 //
7761 // mov x0, wide
7762 // add x1 base, x0
7763 // ldr x2, [x1, x0]
7764 //
7765 // In this situation, we can use the [base, xreg] addressing mode to save an
7766 // add/sub:
7767 //
7768 // mov x0, wide
7769 // ldr x2, [base, x0]
7770 auto ValAndVReg =
7771 getIConstantVRegValWithLookThrough(VReg: PtrAdd->getOperand(i: 2).getReg(), MRI);
7772 if (ValAndVReg) {
7773 unsigned Scale = Log2_32(Value: SizeInBytes);
7774 int64_t ImmOff = ValAndVReg->Value.getSExtValue();
7775
7776 // Skip immediates that can be selected in the load/store addressing
7777 // mode.
7778 if (ImmOff % SizeInBytes == 0 && ImmOff >= 0 &&
7779 ImmOff < (0x1000 << Scale))
7780 return std::nullopt;
7781
7782 // Helper lambda to decide whether or not it is preferable to emit an add.
7783 auto isPreferredADD = [](int64_t ImmOff) {
7784 // Constants in [0x0, 0xfff] can be encoded in an add.
7785 if ((ImmOff & 0xfffffffffffff000LL) == 0x0LL)
7786 return true;
7787
7788 // Can it be encoded in an add lsl #12?
7789 if ((ImmOff & 0xffffffffff000fffLL) != 0x0LL)
7790 return false;
7791
7792 // It can be encoded in an add lsl #12, but we may not want to. If it is
7793 // possible to select this as a single movz, then prefer that. A single
7794 // movz is faster than an add with a shift.
7795 return (ImmOff & 0xffffffffff00ffffLL) != 0x0LL &&
7796 (ImmOff & 0xffffffffffff0fffLL) != 0x0LL;
7797 };
7798
7799 // If the immediate can be encoded in a single add/sub, then bail out.
7800 if (isPreferredADD(ImmOff) || isPreferredADD(-ImmOff))
7801 return std::nullopt;
7802 }
7803
7804 // Try to fold shifts into the addressing mode.
7805 auto AddrModeFns = selectAddrModeShiftedExtendXReg(Root, SizeInBytes);
7806 if (AddrModeFns)
7807 return AddrModeFns;
7808
7809 // If that doesn't work, see if it's possible to fold in registers from
7810 // a GEP.
7811 return selectAddrModeRegisterOffset(Root);
7812}
7813
7814/// This is used for computing addresses like this:
7815///
7816/// ldr x0, [xBase, wOffset, sxtw #LegalShiftVal]
7817///
7818/// Where we have a 64-bit base register, a 32-bit offset register, and an
7819/// extend (which may or may not be signed).
7820InstructionSelector::ComplexRendererFns
7821AArch64InstructionSelector::selectAddrModeWRO(MachineOperand &Root,
7822 unsigned SizeInBytes) const {
7823 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
7824
7825 MachineInstr *PtrAdd =
7826 getOpcodeDef(Opcode: TargetOpcode::G_PTR_ADD, Reg: Root.getReg(), MRI);
7827 if (!PtrAdd || !isWorthFoldingIntoExtendedReg(MI: *PtrAdd, MRI, IsAddrOperand: true))
7828 return std::nullopt;
7829
7830 MachineOperand &LHS = PtrAdd->getOperand(i: 1);
7831 MachineOperand &RHS = PtrAdd->getOperand(i: 2);
7832 MachineInstr *OffsetInst = getDefIgnoringCopies(Reg: RHS.getReg(), MRI);
7833
7834 // The first case is the same as selectAddrModeXRO, except we need an extend.
7835 // In this case, we try to find a shift and extend, and fold them into the
7836 // addressing mode.
7837 //
7838 // E.g.
7839 //
7840 // off_reg = G_Z/S/ANYEXT ext_reg
7841 // val = G_CONSTANT LegalShiftVal
7842 // shift = G_SHL off_reg val
7843 // ptr = G_PTR_ADD base_reg shift
7844 // x = G_LOAD ptr
7845 //
7846 // In this case we can get a load like this:
7847 //
7848 // ldr x0, [base_reg, ext_reg, sxtw #LegalShiftVal]
7849 auto ExtendedShl = selectExtendedSHL(Root, Base&: LHS, Offset&: OffsetInst->getOperand(i: 0),
7850 SizeInBytes, /*WantsExt=*/true);
7851 if (ExtendedShl)
7852 return ExtendedShl;
7853
7854 // There was no shift. We can try and fold a G_Z/S/ANYEXT in alone though.
7855 //
7856 // e.g.
7857 // ldr something, [base_reg, ext_reg, sxtw]
7858 if (!isWorthFoldingIntoExtendedReg(MI: *OffsetInst, MRI, IsAddrOperand: true))
7859 return std::nullopt;
7860
7861 // Check if this is an extend. We'll get an extend type if it is.
7862 AArch64_AM::ShiftExtendType Ext =
7863 getExtendTypeForInst(MI&: *OffsetInst, MRI, /*IsLoadStore=*/true);
7864 if (Ext == AArch64_AM::InvalidShiftExtend)
7865 return std::nullopt;
7866
7867 // Need a 32-bit wide register.
7868 MachineIRBuilder MIB(*PtrAdd);
7869 Register ExtReg = moveScalarRegClass(Reg: OffsetInst->getOperand(i: 1).getReg(),
7870 RC: AArch64::GPR32RegClass, MIB);
7871 unsigned SignExtend = Ext == AArch64_AM::SXTW;
7872
7873 // Base is LHS, offset is ExtReg.
7874 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(RegNo: LHS.getReg()); },
7875 [=](MachineInstrBuilder &MIB) { MIB.addUse(RegNo: ExtReg); },
7876 [=](MachineInstrBuilder &MIB) {
7877 MIB.addImm(Val: SignExtend);
7878 MIB.addImm(Val: 0);
7879 }}};
7880}
7881
7882/// Select a "register plus unscaled signed 9-bit immediate" address. This
7883/// should only match when there is an offset that is not valid for a scaled
7884/// immediate addressing mode. The "Size" argument is the size in bytes of the
7885/// memory reference, which is needed here to know what is valid for a scaled
7886/// immediate.
7887InstructionSelector::ComplexRendererFns
7888AArch64InstructionSelector::selectAddrModeUnscaled(MachineOperand &Root,
7889 unsigned Size) const {
7890 MachineRegisterInfo &MRI =
7891 Root.getParent()->getParent()->getParent()->getRegInfo();
7892
7893 if (!Root.isReg())
7894 return std::nullopt;
7895
7896 if (!isBaseWithConstantOffset(Root, MRI))
7897 return std::nullopt;
7898
7899 MachineInstr *RootDef = MRI.getVRegDef(Reg: Root.getReg());
7900
7901 MachineOperand &OffImm = RootDef->getOperand(i: 2);
7902 if (!OffImm.isReg())
7903 return std::nullopt;
7904 MachineInstr *RHS = MRI.getVRegDef(Reg: OffImm.getReg());
7905 if (RHS->getOpcode() != TargetOpcode::G_CONSTANT)
7906 return std::nullopt;
7907 int64_t RHSC;
7908 MachineOperand &RHSOp1 = RHS->getOperand(i: 1);
7909 if (!RHSOp1.isCImm() || RHSOp1.getCImm()->getBitWidth() > 64)
7910 return std::nullopt;
7911 RHSC = RHSOp1.getCImm()->getSExtValue();
7912
7913 if (RHSC >= -256 && RHSC < 256) {
7914 MachineOperand &Base = RootDef->getOperand(i: 1);
7915 return {{
7916 [=](MachineInstrBuilder &MIB) { MIB.add(MO: Base); },
7917 [=](MachineInstrBuilder &MIB) { MIB.addImm(Val: RHSC); },
7918 }};
7919 }
7920 return std::nullopt;
7921}
7922
7923InstructionSelector::ComplexRendererFns
7924AArch64InstructionSelector::tryFoldAddLowIntoImm(MachineInstr &RootDef,
7925 unsigned Size,
7926 MachineRegisterInfo &MRI) const {
7927 if (RootDef.getOpcode() != AArch64::G_ADD_LOW)
7928 return std::nullopt;
7929 MachineInstr &Adrp = *MRI.getVRegDef(Reg: RootDef.getOperand(i: 1).getReg());
7930 if (Adrp.getOpcode() != AArch64::ADRP)
7931 return std::nullopt;
7932
7933 // TODO: add heuristics like isWorthFoldingADDlow() from SelectionDAG.
7934 auto Offset = Adrp.getOperand(i: 1).getOffset();
7935 if (Offset % Size != 0)
7936 return std::nullopt;
7937
7938 auto GV = Adrp.getOperand(i: 1).getGlobal();
7939 if (GV->isThreadLocal())
7940 return std::nullopt;
7941
7942 auto &MF = *RootDef.getParent()->getParent();
7943 if (GV->getPointerAlignment(DL: MF.getDataLayout()) < Size)
7944 return std::nullopt;
7945
7946 unsigned OpFlags = STI.ClassifyGlobalReference(GV, TM: MF.getTarget());
7947 MachineIRBuilder MIRBuilder(RootDef);
7948 Register AdrpReg = Adrp.getOperand(i: 0).getReg();
7949 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(RegNo: AdrpReg); },
7950 [=](MachineInstrBuilder &MIB) {
7951 MIB.addGlobalAddress(GV, Offset,
7952 TargetFlags: OpFlags | AArch64II::MO_PAGEOFF |
7953 AArch64II::MO_NC);
7954 }}};
7955}
7956
7957/// Select a "register plus scaled unsigned 12-bit immediate" address. The
7958/// "Size" argument is the size in bytes of the memory reference, which
7959/// determines the scale.
7960InstructionSelector::ComplexRendererFns
7961AArch64InstructionSelector::selectAddrModeIndexed(MachineOperand &Root,
7962 unsigned Size) const {
7963 MachineFunction &MF = *Root.getParent()->getParent()->getParent();
7964 MachineRegisterInfo &MRI = MF.getRegInfo();
7965
7966 if (!Root.isReg())
7967 return std::nullopt;
7968
7969 MachineInstr *RootDef = MRI.getVRegDef(Reg: Root.getReg());
7970 if (RootDef->getOpcode() == TargetOpcode::G_FRAME_INDEX) {
7971 return {{
7972 [=](MachineInstrBuilder &MIB) { MIB.add(MO: RootDef->getOperand(i: 1)); },
7973 [=](MachineInstrBuilder &MIB) { MIB.addImm(Val: 0); },
7974 }};
7975 }
7976
7977 CodeModel::Model CM = MF.getTarget().getCodeModel();
7978 // Check if we can fold in the ADD of small code model ADRP + ADD address.
7979 // HACK: ld64 on Darwin doesn't support relocations on PRFM, so we can't fold
7980 // globals into the offset.
7981 MachineInstr *RootParent = Root.getParent();
7982 if (CM == CodeModel::Small &&
7983 !(RootParent->getOpcode() == AArch64::G_AARCH64_PREFETCH &&
7984 STI.isTargetDarwin())) {
7985 auto OpFns = tryFoldAddLowIntoImm(RootDef&: *RootDef, Size, MRI);
7986 if (OpFns)
7987 return OpFns;
7988 }
7989
7990 if (isBaseWithConstantOffset(Root, MRI)) {
7991 MachineOperand &LHS = RootDef->getOperand(i: 1);
7992 MachineOperand &RHS = RootDef->getOperand(i: 2);
7993 MachineInstr *LHSDef = MRI.getVRegDef(Reg: LHS.getReg());
7994 MachineInstr *RHSDef = MRI.getVRegDef(Reg: RHS.getReg());
7995
7996 int64_t RHSC = (int64_t)RHSDef->getOperand(i: 1).getCImm()->getZExtValue();
7997 unsigned Scale = Log2_32(Value: Size);
7998 if ((RHSC & (Size - 1)) == 0 && RHSC >= 0 && RHSC < (0x1000 << Scale)) {
7999 if (LHSDef->getOpcode() == TargetOpcode::G_FRAME_INDEX)
8000 return {{
8001 [=](MachineInstrBuilder &MIB) { MIB.add(MO: LHSDef->getOperand(i: 1)); },
8002 [=](MachineInstrBuilder &MIB) { MIB.addImm(Val: RHSC >> Scale); },
8003 }};
8004
8005 return {{
8006 [=](MachineInstrBuilder &MIB) { MIB.add(MO: LHS); },
8007 [=](MachineInstrBuilder &MIB) { MIB.addImm(Val: RHSC >> Scale); },
8008 }};
8009 }
8010 }
8011
8012 // Before falling back to our general case, check if the unscaled
8013 // instructions can handle this. If so, that's preferable.
8014 if (selectAddrModeUnscaled(Root, Size))
8015 return std::nullopt;
8016
8017 return {{
8018 [=](MachineInstrBuilder &MIB) { MIB.add(MO: Root); },
8019 [=](MachineInstrBuilder &MIB) { MIB.addImm(Val: 0); },
8020 }};
8021}
8022
8023/// Given a shift instruction, return the correct shift type for that
8024/// instruction.
8025static AArch64_AM::ShiftExtendType getShiftTypeForInst(MachineInstr &MI) {
8026 switch (MI.getOpcode()) {
8027 default:
8028 return AArch64_AM::InvalidShiftExtend;
8029 case TargetOpcode::G_SHL:
8030 return AArch64_AM::LSL;
8031 case TargetOpcode::G_LSHR:
8032 return AArch64_AM::LSR;
8033 case TargetOpcode::G_ASHR:
8034 return AArch64_AM::ASR;
8035 case TargetOpcode::G_ROTR:
8036 return AArch64_AM::ROR;
8037 }
8038}
8039
8040/// Select a "shifted register" operand. If the value is not shifted, set the
8041/// shift operand to a default value of "lsl 0".
8042InstructionSelector::ComplexRendererFns
8043AArch64InstructionSelector::selectShiftedRegister(MachineOperand &Root,
8044 bool AllowROR) const {
8045 if (!Root.isReg())
8046 return std::nullopt;
8047 MachineRegisterInfo &MRI =
8048 Root.getParent()->getParent()->getParent()->getRegInfo();
8049
8050 // Check if the operand is defined by an instruction which corresponds to
8051 // a ShiftExtendType. E.g. a G_SHL, G_LSHR, etc.
8052 MachineInstr *ShiftInst = MRI.getVRegDef(Reg: Root.getReg());
8053 AArch64_AM::ShiftExtendType ShType = getShiftTypeForInst(MI&: *ShiftInst);
8054 if (ShType == AArch64_AM::InvalidShiftExtend)
8055 return std::nullopt;
8056 if (ShType == AArch64_AM::ROR && !AllowROR)
8057 return std::nullopt;
8058 if (!isWorthFoldingIntoExtendedReg(MI: *ShiftInst, MRI, IsAddrOperand: false))
8059 return std::nullopt;
8060
8061 // Need an immediate on the RHS.
8062 MachineOperand &ShiftRHS = ShiftInst->getOperand(i: 2);
8063 auto Immed = getImmedFromMO(Root: ShiftRHS);
8064 if (!Immed)
8065 return std::nullopt;
8066
8067 // We have something that we can fold. Fold in the shift's LHS and RHS into
8068 // the instruction.
8069 MachineOperand &ShiftLHS = ShiftInst->getOperand(i: 1);
8070 Register ShiftReg = ShiftLHS.getReg();
8071
8072 unsigned NumBits = MRI.getType(Reg: ShiftReg).getSizeInBits();
8073 unsigned Val = *Immed & (NumBits - 1);
8074 unsigned ShiftVal = AArch64_AM::getShifterImm(ST: ShType, Imm: Val);
8075
8076 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(RegNo: ShiftReg); },
8077 [=](MachineInstrBuilder &MIB) { MIB.addImm(Val: ShiftVal); }}};
8078}
8079
8080AArch64_AM::ShiftExtendType AArch64InstructionSelector::getExtendTypeForInst(
8081 MachineInstr &MI, MachineRegisterInfo &MRI, bool IsLoadStore) const {
8082 unsigned Opc = MI.getOpcode();
8083
8084 // Handle explicit extend instructions first.
8085 if (Opc == TargetOpcode::G_SEXT || Opc == TargetOpcode::G_SEXT_INREG) {
8086 unsigned Size;
8087 if (Opc == TargetOpcode::G_SEXT)
8088 Size = MRI.getType(Reg: MI.getOperand(i: 1).getReg()).getSizeInBits();
8089 else
8090 Size = MI.getOperand(i: 2).getImm();
8091 assert(Size != 64 && "Extend from 64 bits?");
8092 switch (Size) {
8093 case 8:
8094 return IsLoadStore ? AArch64_AM::InvalidShiftExtend : AArch64_AM::SXTB;
8095 case 16:
8096 return IsLoadStore ? AArch64_AM::InvalidShiftExtend : AArch64_AM::SXTH;
8097 case 32:
8098 return AArch64_AM::SXTW;
8099 default:
8100 return AArch64_AM::InvalidShiftExtend;
8101 }
8102 }
8103
8104 if (Opc == TargetOpcode::G_ZEXT || Opc == TargetOpcode::G_ANYEXT) {
8105 unsigned Size = MRI.getType(Reg: MI.getOperand(i: 1).getReg()).getSizeInBits();
8106 assert(Size != 64 && "Extend from 64 bits?");
8107 switch (Size) {
8108 case 8:
8109 return IsLoadStore ? AArch64_AM::InvalidShiftExtend : AArch64_AM::UXTB;
8110 case 16:
8111 return IsLoadStore ? AArch64_AM::InvalidShiftExtend : AArch64_AM::UXTH;
8112 case 32:
8113 return AArch64_AM::UXTW;
8114 default:
8115 return AArch64_AM::InvalidShiftExtend;
8116 }
8117 }
8118
8119 // Don't have an explicit extend. Try to handle a G_AND with a constant mask
8120 // on the RHS.
8121 if (Opc != TargetOpcode::G_AND)
8122 return AArch64_AM::InvalidShiftExtend;
8123
8124 std::optional<uint64_t> MaybeAndMask = getImmedFromMO(Root: MI.getOperand(i: 2));
8125 if (!MaybeAndMask)
8126 return AArch64_AM::InvalidShiftExtend;
8127 uint64_t AndMask = *MaybeAndMask;
8128 switch (AndMask) {
8129 default:
8130 return AArch64_AM::InvalidShiftExtend;
8131 case 0xFF:
8132 return !IsLoadStore ? AArch64_AM::UXTB : AArch64_AM::InvalidShiftExtend;
8133 case 0xFFFF:
8134 return !IsLoadStore ? AArch64_AM::UXTH : AArch64_AM::InvalidShiftExtend;
8135 case 0xFFFFFFFF:
8136 return AArch64_AM::UXTW;
8137 }
8138}
8139
8140Register AArch64InstructionSelector::moveScalarRegClass(
8141 Register Reg, const TargetRegisterClass &RC, MachineIRBuilder &MIB) const {
8142 MachineRegisterInfo &MRI = *MIB.getMRI();
8143 auto Ty = MRI.getType(Reg);
8144 assert(!Ty.isVector() && "Expected scalars only!");
8145 if (Ty.getSizeInBits() == TRI.getRegSizeInBits(RC))
8146 return Reg;
8147
8148 // Create a copy and immediately select it.
8149 // FIXME: We should have an emitCopy function?
8150 auto Copy = MIB.buildCopy(Res: {&RC}, Op: {Reg});
8151 selectCopy(I&: *Copy, TII, MRI, TRI, RBI);
8152 return Copy.getReg(Idx: 0);
8153}
8154
8155/// Select an "extended register" operand. This operand folds in an extend
8156/// followed by an optional left shift.
8157InstructionSelector::ComplexRendererFns
8158AArch64InstructionSelector::selectArithExtendedRegister(
8159 MachineOperand &Root) const {
8160 if (!Root.isReg())
8161 return std::nullopt;
8162 MachineRegisterInfo &MRI =
8163 Root.getParent()->getParent()->getParent()->getRegInfo();
8164
8165 uint64_t ShiftVal = 0;
8166 Register ExtReg;
8167 AArch64_AM::ShiftExtendType Ext;
8168 MachineInstr *RootDef = getDefIgnoringCopies(Reg: Root.getReg(), MRI);
8169 if (!RootDef)
8170 return std::nullopt;
8171
8172 if (!isWorthFoldingIntoExtendedReg(MI: *RootDef, MRI, IsAddrOperand: false))
8173 return std::nullopt;
8174
8175 // Check if we can fold a shift and an extend.
8176 if (RootDef->getOpcode() == TargetOpcode::G_SHL) {
8177 // Look for a constant on the RHS of the shift.
8178 MachineOperand &RHS = RootDef->getOperand(i: 2);
8179 std::optional<uint64_t> MaybeShiftVal = getImmedFromMO(Root: RHS);
8180 if (!MaybeShiftVal)
8181 return std::nullopt;
8182 ShiftVal = *MaybeShiftVal;
8183 if (ShiftVal > 4)
8184 return std::nullopt;
8185 // Look for a valid extend instruction on the LHS of the shift.
8186 MachineOperand &LHS = RootDef->getOperand(i: 1);
8187 MachineInstr *ExtDef = getDefIgnoringCopies(Reg: LHS.getReg(), MRI);
8188 if (!ExtDef)
8189 return std::nullopt;
8190 Ext = getExtendTypeForInst(MI&: *ExtDef, MRI);
8191 if (Ext == AArch64_AM::InvalidShiftExtend)
8192 return std::nullopt;
8193 ExtReg = ExtDef->getOperand(i: 1).getReg();
8194 } else {
8195 // Didn't get a shift. Try just folding an extend.
8196 Ext = getExtendTypeForInst(MI&: *RootDef, MRI);
8197 if (Ext == AArch64_AM::InvalidShiftExtend)
8198 return std::nullopt;
8199 ExtReg = RootDef->getOperand(i: 1).getReg();
8200
8201 // If we have a 32 bit instruction which zeroes out the high half of a
8202 // register, we get an implicit zero extend for free. Check if we have one.
8203 // FIXME: We actually emit the extend right now even though we don't have
8204 // to.
8205 if (Ext == AArch64_AM::UXTW && MRI.getType(Reg: ExtReg).getSizeInBits() == 32) {
8206 MachineInstr *ExtInst = MRI.getVRegDef(Reg: ExtReg);
8207 if (isDef32(MI: *ExtInst))
8208 return std::nullopt;
8209 }
8210 }
8211
8212 // We require a GPR32 here. Narrow the ExtReg if needed using a subregister
8213 // copy.
8214 MachineIRBuilder MIB(*RootDef);
8215 ExtReg = moveScalarRegClass(Reg: ExtReg, RC: AArch64::GPR32RegClass, MIB);
8216
8217 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(RegNo: ExtReg); },
8218 [=](MachineInstrBuilder &MIB) {
8219 MIB.addImm(Val: getArithExtendImm(ET: Ext, Imm: ShiftVal));
8220 }}};
8221}
8222
8223InstructionSelector::ComplexRendererFns
8224AArch64InstructionSelector::selectExtractHigh(MachineOperand &Root) const {
8225 if (!Root.isReg())
8226 return std::nullopt;
8227 MachineRegisterInfo &MRI =
8228 Root.getParent()->getParent()->getParent()->getRegInfo();
8229
8230 auto Extract = getDefSrcRegIgnoringCopies(Reg: Root.getReg(), MRI);
8231 while (Extract && Extract->MI->getOpcode() == TargetOpcode::G_BITCAST &&
8232 STI.isLittleEndian())
8233 Extract =
8234 getDefSrcRegIgnoringCopies(Reg: Extract->MI->getOperand(i: 1).getReg(), MRI);
8235 if (!Extract)
8236 return std::nullopt;
8237
8238 if (auto *Unmerge = dyn_cast<GUnmerge>(Val: Extract->MI)) {
8239 if (Unmerge->getNumDefs() == 2 &&
8240 Extract->Reg == Unmerge->getOperand(i: 1).getReg()) {
8241 Register ExtReg = Unmerge->getSourceReg();
8242 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(RegNo: ExtReg); }}};
8243 }
8244 }
8245 if (auto *ExtElt = dyn_cast<GExtractVectorElement>(Val: Extract->MI)) {
8246 LLT SrcTy = MRI.getType(Reg: ExtElt->getVectorReg());
8247 auto LaneIdx =
8248 getIConstantVRegValWithLookThrough(VReg: ExtElt->getIndexReg(), MRI);
8249 if (LaneIdx && SrcTy == LLT::fixed_vector(NumElements: 2, ScalarSizeInBits: 64) &&
8250 LaneIdx->Value.getSExtValue() == 1) {
8251 Register ExtReg = ExtElt->getVectorReg();
8252 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(RegNo: ExtReg); }}};
8253 }
8254 }
8255 if (auto *Subvec = dyn_cast<GExtractSubvector>(Val: Extract->MI)) {
8256 LLT SrcTy = MRI.getType(Reg: Subvec->getSrcVec());
8257 auto LaneIdx = Subvec->getIndexImm();
8258 if (LaneIdx == SrcTy.getNumElements() / 2) {
8259 Register ExtReg = Subvec->getSrcVec();
8260 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(RegNo: ExtReg); }}};
8261 }
8262 }
8263
8264 return std::nullopt;
8265}
8266
8267InstructionSelector::ComplexRendererFns
8268AArch64InstructionSelector::selectCVTFixedPointBase(const MachineOperand &Root,
8269 unsigned DstElemWidth,
8270 bool isReciprocal) const {
8271 if (!Root.isReg())
8272 return std::nullopt;
8273 const MachineRegisterInfo &MRI =
8274 Root.getParent()->getParent()->getParent()->getRegInfo();
8275
8276 Register Reg = Root.getReg();
8277 MachineInstr *Dup = getDefIgnoringCopies(Reg, MRI);
8278
8279 if (Dup && Dup->getOpcode() == AArch64::G_DUP)
8280 Reg = Dup->getOperand(i: 1).getReg();
8281
8282 std::optional<ValueAndVReg> CstVal =
8283 getAnyConstantVRegValWithLookThrough(VReg: Reg, MRI);
8284
8285 if (!CstVal)
8286 return std::nullopt;
8287
8288 unsigned CstElemWidth = MRI.getType(Reg).getScalarSizeInBits();
8289 APFloat FVal(0.0);
8290 switch (CstElemWidth) {
8291 case 16:
8292 FVal = APFloat(APFloat::IEEEhalf(), CstVal->Value);
8293 break;
8294 case 32:
8295 FVal = APFloat(APFloat::IEEEsingle(), CstVal->Value);
8296 break;
8297 case 64:
8298 FVal = APFloat(APFloat::IEEEdouble(), CstVal->Value);
8299 break;
8300 default:
8301 return std::nullopt;
8302 };
8303 if (unsigned FBits =
8304 CheckFixedPointOperandConstant(FVal, RegWidth: DstElemWidth, isReciprocal))
8305 return {{[=](MachineInstrBuilder &MIB) { MIB.addImm(Val: FBits); }}};
8306
8307 return std::nullopt;
8308}
8309
8310unsigned AArch64InstructionSelector::getFixedPointWidthFromOperand(
8311 const MachineOperand &Root) const {
8312 return Root.getParent()
8313 ->getMF()
8314 ->getRegInfo()
8315 .getType(Reg: Root.getReg())
8316 .getScalarSizeInBits();
8317}
8318
8319template <unsigned Width>
8320InstructionSelector::ComplexRendererFns
8321AArch64InstructionSelector::selectCVTFixedPoint(MachineOperand &Root) const {
8322 return selectCVTFixedPointBase(Root, DstElemWidth: Width, /*isReciprocal*/ false);
8323}
8324
8325template <unsigned Width>
8326InstructionSelector::ComplexRendererFns
8327AArch64InstructionSelector::selectCVTFixedPosRecipOperand(
8328 MachineOperand &Root) const {
8329 return selectCVTFixedPointBase(Root, DstElemWidth: Width, /*isReciprocal*/ true);
8330}
8331
8332InstructionSelector::ComplexRendererFns
8333AArch64InstructionSelector::selectCVTFixedPointVec(MachineOperand &Root) const {
8334 return selectCVTFixedPointBase(Root, DstElemWidth: getFixedPointWidthFromOperand(Root),
8335 /*isReciprocal*/ false);
8336}
8337
8338InstructionSelector::ComplexRendererFns
8339AArch64InstructionSelector::selectCVTFixedPosRecipOperandVec(
8340 MachineOperand &Root) const {
8341 return selectCVTFixedPointBase(Root, DstElemWidth: getFixedPointWidthFromOperand(Root),
8342 /*isReciprocal*/ true);
8343}
8344
8345void AArch64InstructionSelector::renderFixedPointScalarXForm(
8346 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
8347 assert(OpIdx == 3 && MI.getOperand(OpIdx).isImm() &&
8348 "Expected vecshift immediate operand");
8349 MIB.addImm(Val: MI.getOperand(i: OpIdx).getImm());
8350}
8351
8352void AArch64InstructionSelector::renderFixedPointImm(MachineInstrBuilder &MIB,
8353 const MachineOperand &Root,
8354 unsigned Width,
8355 bool isReciprocal) const {
8356 // FIXME: This is only needed to satisfy the type checking in tablegen, and
8357 // should be able to reuse the Renderers already calculated by
8358 // selectCVTFixedPointBase.
8359 InstructionSelector::ComplexRendererFns Renderer =
8360 selectCVTFixedPointBase(Root, DstElemWidth: Width, isReciprocal);
8361 assert((Renderer && Renderer->size() == 1) &&
8362 "Expected selectCVTFixedPointBase to provide a function\n");
8363 (Renderer->front())(MIB);
8364}
8365
8366void AArch64InstructionSelector::renderFixedPointXForm(MachineInstrBuilder &MIB,
8367 const MachineInstr &MI,
8368 int OpIdx) const {
8369 const MachineOperand &Root = MI.getOperand(i: OpIdx);
8370 renderFixedPointImm(MIB, Root, Width: getFixedPointWidthFromOperand(Root),
8371 /*isReciprocal*/ false);
8372}
8373
8374void AArch64InstructionSelector::renderFixedPointRecipXForm(
8375 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
8376 const MachineOperand &Root = MI.getOperand(i: OpIdx);
8377 renderFixedPointImm(MIB, Root, Width: getFixedPointWidthFromOperand(Root),
8378 /*isReciprocal*/ true);
8379}
8380
8381void AArch64InstructionSelector::renderTruncImm(MachineInstrBuilder &MIB,
8382 const MachineInstr &MI,
8383 int OpIdx) const {
8384 const MachineRegisterInfo &MRI = MI.getParent()->getParent()->getRegInfo();
8385 assert(MI.getOpcode() == TargetOpcode::G_CONSTANT && OpIdx == -1 &&
8386 "Expected G_CONSTANT");
8387 std::optional<int64_t> CstVal =
8388 getIConstantVRegSExtVal(VReg: MI.getOperand(i: 0).getReg(), MRI);
8389 assert(CstVal && "Expected constant value");
8390 MIB.addImm(Val: *CstVal);
8391}
8392
8393void AArch64InstructionSelector::renderLogicalImm32(
8394 MachineInstrBuilder &MIB, const MachineInstr &I, int OpIdx) const {
8395 assert(I.getOpcode() == TargetOpcode::G_CONSTANT && OpIdx == -1 &&
8396 "Expected G_CONSTANT");
8397 uint64_t CstVal = I.getOperand(i: 1).getCImm()->getZExtValue();
8398 uint64_t Enc = AArch64_AM::encodeLogicalImmediate(imm: CstVal, regSize: 32);
8399 MIB.addImm(Val: Enc);
8400}
8401
8402void AArch64InstructionSelector::renderLogicalImm64(
8403 MachineInstrBuilder &MIB, const MachineInstr &I, int OpIdx) const {
8404 assert(I.getOpcode() == TargetOpcode::G_CONSTANT && OpIdx == -1 &&
8405 "Expected G_CONSTANT");
8406 uint64_t CstVal = I.getOperand(i: 1).getCImm()->getZExtValue();
8407 uint64_t Enc = AArch64_AM::encodeLogicalImmediate(imm: CstVal, regSize: 64);
8408 MIB.addImm(Val: Enc);
8409}
8410
8411void AArch64InstructionSelector::renderUbsanTrap(MachineInstrBuilder &MIB,
8412 const MachineInstr &MI,
8413 int OpIdx) const {
8414 assert(MI.getOpcode() == TargetOpcode::G_UBSANTRAP && OpIdx == 0 &&
8415 "Expected G_UBSANTRAP");
8416 MIB.addImm(Val: MI.getOperand(i: 0).getImm() | ('U' << 8));
8417}
8418
8419void AArch64InstructionSelector::renderFPImm16(MachineInstrBuilder &MIB,
8420 const MachineInstr &MI,
8421 int OpIdx) const {
8422 assert(MI.getOpcode() == TargetOpcode::G_FCONSTANT && OpIdx == -1 &&
8423 "Expected G_FCONSTANT");
8424 MIB.addImm(
8425 Val: AArch64_AM::getFP16Imm(FPImm: MI.getOperand(i: 1).getFPImm()->getValueAPF()));
8426}
8427
8428void AArch64InstructionSelector::renderFPImm32(MachineInstrBuilder &MIB,
8429 const MachineInstr &MI,
8430 int OpIdx) const {
8431 assert(MI.getOpcode() == TargetOpcode::G_FCONSTANT && OpIdx == -1 &&
8432 "Expected G_FCONSTANT");
8433 MIB.addImm(
8434 Val: AArch64_AM::getFP32Imm(FPImm: MI.getOperand(i: 1).getFPImm()->getValueAPF()));
8435}
8436
8437void AArch64InstructionSelector::renderFPImm64(MachineInstrBuilder &MIB,
8438 const MachineInstr &MI,
8439 int OpIdx) const {
8440 assert(MI.getOpcode() == TargetOpcode::G_FCONSTANT && OpIdx == -1 &&
8441 "Expected G_FCONSTANT");
8442 MIB.addImm(
8443 Val: AArch64_AM::getFP64Imm(FPImm: MI.getOperand(i: 1).getFPImm()->getValueAPF()));
8444}
8445
8446void AArch64InstructionSelector::renderFPImm32SIMDModImmType4(
8447 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
8448 assert(MI.getOpcode() == TargetOpcode::G_FCONSTANT && OpIdx == -1 &&
8449 "Expected G_FCONSTANT");
8450 MIB.addImm(Val: AArch64_AM::encodeAdvSIMDModImmType4(Imm: MI.getOperand(i: 1)
8451 .getFPImm()
8452 ->getValueAPF()
8453 .bitcastToAPInt()
8454 .getZExtValue()));
8455}
8456
8457bool AArch64InstructionSelector::isLoadStoreOfNumBytes(
8458 const MachineInstr &MI, unsigned NumBytes) const {
8459 if (!MI.mayLoadOrStore())
8460 return false;
8461 assert(MI.hasOneMemOperand() &&
8462 "Expected load/store to have only one mem op!");
8463 return (*MI.memoperands_begin())->getSize() == NumBytes;
8464}
8465
8466bool AArch64InstructionSelector::isDef32(const MachineInstr &MI) const {
8467 const MachineRegisterInfo &MRI = MI.getParent()->getParent()->getRegInfo();
8468 if (MRI.getType(Reg: MI.getOperand(i: 0).getReg()).getSizeInBits() != 32)
8469 return false;
8470
8471 // Only return true if we know the operation will zero-out the high half of
8472 // the 64-bit register. Truncates can be subregister copies, which don't
8473 // zero out the high bits. Copies and other copy-like instructions can be
8474 // fed by truncates, or could be lowered as subregister copies.
8475 switch (MI.getOpcode()) {
8476 default:
8477 return true;
8478 case TargetOpcode::COPY:
8479 case TargetOpcode::G_BITCAST:
8480 case TargetOpcode::G_TRUNC:
8481 case TargetOpcode::G_PHI:
8482 return false;
8483 }
8484}
8485
8486
8487// Perform fixups on the given PHI instruction's operands to force them all
8488// to be the same as the destination regbank.
8489static void fixupPHIOpBanks(MachineInstr &MI, MachineRegisterInfo &MRI,
8490 const AArch64RegisterBankInfo &RBI) {
8491 assert(MI.getOpcode() == TargetOpcode::G_PHI && "Expected a G_PHI");
8492 Register DstReg = MI.getOperand(i: 0).getReg();
8493 const RegisterBank *DstRB = MRI.getRegBankOrNull(Reg: DstReg);
8494 assert(DstRB && "Expected PHI dst to have regbank assigned");
8495 MachineIRBuilder MIB(MI);
8496
8497 // Go through each operand and ensure it has the same regbank.
8498 for (MachineOperand &MO : llvm::drop_begin(RangeOrContainer: MI.operands())) {
8499 if (!MO.isReg())
8500 continue;
8501 Register OpReg = MO.getReg();
8502 const RegisterBank *RB = MRI.getRegBankOrNull(Reg: OpReg);
8503 if (RB != DstRB) {
8504 // Insert a cross-bank copy.
8505 auto *OpDef = MRI.getVRegDef(Reg: OpReg);
8506 const LLT &Ty = MRI.getType(Reg: OpReg);
8507 MachineBasicBlock &OpDefBB = *OpDef->getParent();
8508
8509 // Any instruction we insert must appear after all PHIs in the block
8510 // for the block to be valid MIR.
8511 MachineBasicBlock::iterator InsertPt = std::next(x: OpDef->getIterator());
8512 if (InsertPt != OpDefBB.end() && InsertPt->isPHI())
8513 InsertPt = OpDefBB.getFirstNonPHI();
8514 MIB.setInsertPt(MBB&: *OpDef->getParent(), II: InsertPt);
8515 auto Copy = MIB.buildCopy(Res: Ty, Op: OpReg);
8516 MRI.setRegBank(Reg: Copy.getReg(Idx: 0), RegBank: *DstRB);
8517 MO.setReg(Copy.getReg(Idx: 0));
8518 }
8519 }
8520}
8521
8522void AArch64InstructionSelector::processPHIs(MachineFunction &MF) {
8523 // We're looking for PHIs, build a list so we don't invalidate iterators.
8524 MachineRegisterInfo &MRI = MF.getRegInfo();
8525 SmallVector<MachineInstr *, 32> Phis;
8526 for (auto &BB : MF) {
8527 for (auto &MI : BB) {
8528 if (MI.getOpcode() == TargetOpcode::G_PHI)
8529 Phis.emplace_back(Args: &MI);
8530 }
8531 }
8532
8533 for (auto *MI : Phis) {
8534 // We need to do some work here if the operand types are < 16 bit and they
8535 // are split across fpr/gpr banks. Since all types <32b on gpr
8536 // end up being assigned gpr32 regclasses, we can end up with PHIs here
8537 // which try to select between a gpr32 and an fpr16. Ideally RBS shouldn't
8538 // be selecting heterogenous regbanks for operands if possible, but we
8539 // still need to be able to deal with it here.
8540 //
8541 // To fix this, if we have a gpr-bank operand < 32b in size and at least
8542 // one other operand is on the fpr bank, then we add cross-bank copies
8543 // to homogenize the operand banks. For simplicity the bank that we choose
8544 // to settle on is whatever bank the def operand has. For example:
8545 //
8546 // %endbb:
8547 // %dst:gpr(s16) = G_PHI %in1:gpr(s16), %bb1, %in2:fpr(s16), %bb2
8548 // =>
8549 // %bb2:
8550 // ...
8551 // %in2_copy:gpr(s16) = COPY %in2:fpr(s16)
8552 // ...
8553 // %endbb:
8554 // %dst:gpr(s16) = G_PHI %in1:gpr(s16), %bb1, %in2_copy:gpr(s16), %bb2
8555 bool HasGPROp = false, HasFPROp = false;
8556 for (const MachineOperand &MO : llvm::drop_begin(RangeOrContainer: MI->operands())) {
8557 if (!MO.isReg())
8558 continue;
8559 const LLT &Ty = MRI.getType(Reg: MO.getReg());
8560 if (!Ty.isValid() || !Ty.isScalar())
8561 break;
8562 if (Ty.getSizeInBits() >= 32)
8563 break;
8564 const RegisterBank *RB = MRI.getRegBankOrNull(Reg: MO.getReg());
8565 // If for some reason we don't have a regbank yet. Don't try anything.
8566 if (!RB)
8567 break;
8568
8569 if (RB->getID() == AArch64::GPRRegBankID)
8570 HasGPROp = true;
8571 else
8572 HasFPROp = true;
8573 }
8574 // We have heterogenous regbanks, need to fixup.
8575 if (HasGPROp && HasFPROp)
8576 fixupPHIOpBanks(MI&: *MI, MRI, RBI);
8577 }
8578}
8579
8580namespace llvm {
8581InstructionSelector *
8582createAArch64InstructionSelector(const AArch64TargetMachine &TM,
8583 const AArch64Subtarget &Subtarget,
8584 const AArch64RegisterBankInfo &RBI) {
8585 return new AArch64InstructionSelector(TM, Subtarget, RBI);
8586}
8587}
8588