1//===- MipsSEISelLowering.cpp - MipsSE DAG Lowering Interface -------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// Subclass of MipsTargetLowering specialized for mips32/64.
10//
11//===----------------------------------------------------------------------===//
12
13#include "MipsSEISelLowering.h"
14#include "MipsMachineFunction.h"
15#include "MipsRegisterInfo.h"
16#include "MipsSubtarget.h"
17#include "llvm/ADT/APInt.h"
18#include "llvm/ADT/SmallVector.h"
19#include "llvm/CodeGen/CallingConvLower.h"
20#include "llvm/CodeGen/ISDOpcodes.h"
21#include "llvm/CodeGen/MachineBasicBlock.h"
22#include "llvm/CodeGen/MachineFunction.h"
23#include "llvm/CodeGen/MachineInstr.h"
24#include "llvm/CodeGen/MachineInstrBuilder.h"
25#include "llvm/CodeGen/MachineMemOperand.h"
26#include "llvm/CodeGen/MachineRegisterInfo.h"
27#include "llvm/CodeGen/SelectionDAG.h"
28#include "llvm/CodeGen/SelectionDAGNodes.h"
29#include "llvm/CodeGen/TargetInstrInfo.h"
30#include "llvm/CodeGen/TargetLowering.h"
31#include "llvm/CodeGen/TargetSubtargetInfo.h"
32#include "llvm/CodeGen/ValueTypes.h"
33#include "llvm/CodeGenTypes/MachineValueType.h"
34#include "llvm/IR/DebugLoc.h"
35#include "llvm/IR/Intrinsics.h"
36#include "llvm/IR/IntrinsicsMips.h"
37#include "llvm/Support/Casting.h"
38#include "llvm/Support/CommandLine.h"
39#include "llvm/Support/Debug.h"
40#include "llvm/Support/ErrorHandling.h"
41#include "llvm/Support/raw_ostream.h"
42#include "llvm/TargetParser/Triple.h"
43#include <algorithm>
44#include <cassert>
45#include <cstddef>
46#include <cstdint>
47#include <iterator>
48#include <utility>
49
50using namespace llvm;
51
52#define DEBUG_TYPE "mips-isel"
53
54static cl::opt<bool> NoDPLoadStore("mno-ldc1-sdc1", cl::init(Val: false),
55 cl::desc("Expand double precision loads and "
56 "stores to their single precision "
57 "counterparts"));
58
59// Widen the v2 vectors to the register width, i.e. v2i16 -> v8i16,
60// v2i32 -> v4i32, etc, to ensure the correct rail size is used, i.e.
61// INST.h for v16, INST.w for v32, INST.d for v64.
62TargetLoweringBase::LegalizeTypeAction
63MipsSETargetLowering::getPreferredVectorAction(MVT VT) const {
64 if (this->Subtarget.hasMSA()) {
65 switch (VT.SimpleTy) {
66 // Leave v2i1 vectors to be promoted to larger ones.
67 // Other i1 types will be promoted by default.
68 case MVT::v2i1:
69 return TypePromoteInteger;
70 break;
71 // 16-bit vector types (v2 and longer)
72 case MVT::v2i8:
73 // 32-bit vector types (v2 and longer)
74 case MVT::v2i16:
75 case MVT::v4i8:
76 // 64-bit vector types (v2 and longer)
77 case MVT::v2i32:
78 case MVT::v4i16:
79 case MVT::v8i8:
80 return TypeWidenVector;
81 break;
82 // Only word (.w) and doubleword (.d) are available for floating point
83 // vectors. That means floating point vectors should be either v2f64
84 // or v4f32.
85 // Here we only explicitly widen the f32 types - f16 will be promoted
86 // by default.
87 case MVT::v2f32:
88 case MVT::v3f32:
89 return TypeWidenVector;
90 // v2i64 is already 128-bit wide.
91 default:
92 break;
93 }
94 }
95 return TargetLoweringBase::getPreferredVectorAction(VT);
96}
97
98MipsSETargetLowering::MipsSETargetLowering(const MipsTargetMachine &TM,
99 const MipsSubtarget &STI)
100 : MipsTargetLowering(TM, STI) {
101 // Set up the register classes
102 addRegisterClass(VT: MVT::i32, RC: &Mips::GPR32RegClass);
103
104 if (Subtarget.isGP64bit())
105 addRegisterClass(VT: MVT::i64, RC: &Mips::GPR64RegClass);
106
107 if (Subtarget.hasDSP() || Subtarget.hasMSA()) {
108 // Expand all truncating stores and extending loads.
109 for (MVT VT0 : MVT::fixedlen_vector_valuetypes()) {
110 for (MVT VT1 : MVT::fixedlen_vector_valuetypes()) {
111 setTruncStoreAction(ValVT: VT0, MemVT: VT1, Action: Expand);
112 setLoadExtAction(ExtType: ISD::SEXTLOAD, ValVT: VT0, MemVT: VT1, Action: Expand);
113 setLoadExtAction(ExtType: ISD::ZEXTLOAD, ValVT: VT0, MemVT: VT1, Action: Expand);
114 setLoadExtAction(ExtType: ISD::EXTLOAD, ValVT: VT0, MemVT: VT1, Action: Expand);
115 }
116 }
117 }
118
119 if (Subtarget.hasDSP()) {
120 MVT::SimpleValueType VecTys[2] = {MVT::v2i16, MVT::v4i8};
121
122 for (const auto &VecTy : VecTys) {
123 addRegisterClass(VT: VecTy, RC: &Mips::DSPRRegClass);
124
125 // Expand all builtin opcodes.
126 for (unsigned Opc = 0; Opc < ISD::BUILTIN_OP_END; ++Opc)
127 setOperationAction(Op: Opc, VT: VecTy, Action: Expand);
128
129 setOperationAction(Op: ISD::ADD, VT: VecTy, Action: Legal);
130 setOperationAction(Op: ISD::SUB, VT: VecTy, Action: Legal);
131 setOperationAction(Op: ISD::LOAD, VT: VecTy, Action: Legal);
132 setOperationAction(Op: ISD::STORE, VT: VecTy, Action: Legal);
133 setOperationAction(Op: ISD::BITCAST, VT: VecTy, Action: Legal);
134 }
135
136 setTargetDAGCombine(
137 {ISD::SHL, ISD::SRA, ISD::SRL, ISD::SETCC, ISD::VSELECT});
138
139 if (Subtarget.hasMips32r2()) {
140 setOperationAction(Op: ISD::ADDC, VT: MVT::i32, Action: Legal);
141 setOperationAction(Op: ISD::ADDE, VT: MVT::i32, Action: Legal);
142 }
143 }
144
145 if (Subtarget.hasDSPR2())
146 setOperationAction(Op: ISD::MUL, VT: MVT::v2i16, Action: Legal);
147
148 if (Subtarget.hasMSA()) {
149 addMSAIntType(Ty: MVT::v16i8, RC: &Mips::MSA128BRegClass);
150 addMSAIntType(Ty: MVT::v8i16, RC: &Mips::MSA128HRegClass);
151 addMSAIntType(Ty: MVT::v4i32, RC: &Mips::MSA128WRegClass);
152 addMSAIntType(Ty: MVT::v2i64, RC: &Mips::MSA128DRegClass);
153 addMSAFloatType(Ty: MVT::v8f16, RC: &Mips::MSA128HRegClass);
154 addMSAFloatType(Ty: MVT::v4f32, RC: &Mips::MSA128WRegClass);
155 addMSAFloatType(Ty: MVT::v2f64, RC: &Mips::MSA128DRegClass);
156
157 // Shuffle half vectors as integers to avoid expanding them through
158 // EXTRACT_VECTOR_ELT and BUILD_VECTOR with an illegal scalar f16 type.
159 setOperationPromotedToType(Opc: ISD::VECTOR_SHUFFLE, OrigVT: MVT::v8f16, DestVT: MVT::v8i16);
160
161 // We're using soft promotion for f16, but msa has some instructions for
162 // conversion to/from f16. Mark those conversions as custom so we can take
163 // advantage of these instructions.
164 for (MVT VT : {MVT::f32, MVT::f64}) {
165 setOperationAction(Op: ISD::FP16_TO_FP, VT, Action: Custom);
166 setOperationAction(Op: ISD::FP_TO_FP16, VT, Action: Custom);
167 }
168
169 setTargetDAGCombine(
170 {ISD::AND, ISD::OR, ISD::SRA, ISD::VSELECT, ISD::XOR, ISD::FP_TO_UINT});
171 }
172
173 if (!Subtarget.useSoftFloat()) {
174 addRegisterClass(VT: MVT::f32, RC: &Mips::FGR32RegClass);
175
176 // When dealing with single precision only, use libcalls
177 if (!Subtarget.isSingleFloat()) {
178 if (Subtarget.isFP64bit())
179 addRegisterClass(VT: MVT::f64, RC: &Mips::FGR64RegClass);
180 else
181 addRegisterClass(VT: MVT::f64, RC: &Mips::AFGR64RegClass);
182 }
183
184 for (auto Op : {ISD::STRICT_FADD, ISD::STRICT_FSUB, ISD::STRICT_FMUL,
185 ISD::STRICT_FDIV, ISD::STRICT_FSQRT}) {
186 setOperationAction(Op, VT: MVT::f32, Action: Legal);
187 setOperationAction(Op, VT: MVT::f64, Action: Legal);
188 }
189 }
190
191 // Targets with 64bits integer registers, but no 64bit floating point register
192 // do not support conversion between them
193 if (Subtarget.isGP64bit() && Subtarget.isSingleFloat() &&
194 !Subtarget.useSoftFloat()) {
195 setOperationAction(Op: ISD::FP_TO_SINT, VT: MVT::i64, Action: Expand);
196 setOperationAction(Op: ISD::FP_TO_UINT, VT: MVT::i64, Action: Expand);
197 setOperationAction(Op: ISD::SINT_TO_FP, VT: MVT::i64, Action: Expand);
198 setOperationAction(Op: ISD::UINT_TO_FP, VT: MVT::i64, Action: Expand);
199 }
200
201 setOperationAction(Op: ISD::SMUL_LOHI, VT: MVT::i32, Action: Custom);
202 setOperationAction(Op: ISD::UMUL_LOHI, VT: MVT::i32, Action: Custom);
203 setOperationAction(Op: ISD::MULHS, VT: MVT::i32, Action: Custom);
204 setOperationAction(Op: ISD::MULHU, VT: MVT::i32, Action: Custom);
205
206 if (Subtarget.hasCnMips())
207 setOperationAction(Op: ISD::MUL, VT: MVT::i64, Action: Legal);
208 else if (Subtarget.isR5900()) {
209 // R5900 doesn't have DMULT/DMULTU/DDIV/DDIVU - expand to 32-bit ops
210 setOperationAction(Op: ISD::MUL, VT: MVT::i64, Action: Expand);
211 setOperationAction(Op: ISD::SMUL_LOHI, VT: MVT::i64, Action: Expand);
212 setOperationAction(Op: ISD::UMUL_LOHI, VT: MVT::i64, Action: Expand);
213 setOperationAction(Op: ISD::MULHS, VT: MVT::i64, Action: Expand);
214 setOperationAction(Op: ISD::MULHU, VT: MVT::i64, Action: Expand);
215 setOperationAction(Op: ISD::SDIVREM, VT: MVT::i64, Action: Expand);
216 setOperationAction(Op: ISD::UDIVREM, VT: MVT::i64, Action: Expand);
217 } else if (Subtarget.isGP64bit())
218 setOperationAction(Op: ISD::MUL, VT: MVT::i64, Action: Custom);
219
220 if (Subtarget.isGP64bit() && !Subtarget.isR5900()) {
221 setOperationAction(Op: ISD::SMUL_LOHI, VT: MVT::i64, Action: Custom);
222 setOperationAction(Op: ISD::UMUL_LOHI, VT: MVT::i64, Action: Custom);
223 setOperationAction(Op: ISD::MULHS, VT: MVT::i64, Action: Custom);
224 setOperationAction(Op: ISD::MULHU, VT: MVT::i64, Action: Custom);
225 setOperationAction(Op: ISD::SDIVREM, VT: MVT::i64, Action: Custom);
226 setOperationAction(Op: ISD::UDIVREM, VT: MVT::i64, Action: Custom);
227 }
228
229 setOperationAction(Op: ISD::INTRINSIC_WO_CHAIN, VT: MVT::i64, Action: Custom);
230 setOperationAction(Op: ISD::INTRINSIC_W_CHAIN, VT: MVT::i64, Action: Custom);
231
232 setOperationAction(Op: ISD::SDIVREM, VT: MVT::i32, Action: Custom);
233 setOperationAction(Op: ISD::UDIVREM, VT: MVT::i32, Action: Custom);
234 setOperationAction(Op: ISD::ATOMIC_FENCE, VT: MVT::Other, Action: Custom);
235 if (Subtarget.hasMips32r6()) {
236 setOperationAction(Op: ISD::LOAD, VT: MVT::i32, Action: Legal);
237 setOperationAction(Op: ISD::STORE, VT: MVT::i32, Action: Legal);
238 } else {
239 setOperationAction(Op: ISD::LOAD, VT: MVT::i32, Action: Custom);
240 setOperationAction(Op: ISD::STORE, VT: MVT::i32, Action: Custom);
241 }
242
243 setTargetDAGCombine(ISD::MUL);
244
245 setOperationAction(Op: ISD::INTRINSIC_WO_CHAIN, VT: MVT::Other, Action: Custom);
246 setOperationAction(Op: ISD::INTRINSIC_W_CHAIN, VT: MVT::Other, Action: Custom);
247 setOperationAction(Op: ISD::INTRINSIC_VOID, VT: MVT::Other, Action: Custom);
248
249 if (Subtarget.hasMips32r2() && !Subtarget.useSoftFloat() &&
250 !Subtarget.hasMips64()) {
251 setOperationAction(Op: ISD::BITCAST, VT: MVT::i64, Action: Custom);
252 }
253
254 if (NoDPLoadStore || (Subtarget.hasMips1() && !Subtarget.hasMips2())) {
255 setOperationAction(Op: ISD::LOAD, VT: MVT::f64, Action: Custom);
256 setOperationAction(Op: ISD::STORE, VT: MVT::f64, Action: Custom);
257 }
258
259 if (Subtarget.hasMips32r6()) {
260 // MIPS32r6 replaces the accumulator-based multiplies with a three register
261 // instruction
262 setOperationAction(Op: ISD::SMUL_LOHI, VT: MVT::i32, Action: Expand);
263 setOperationAction(Op: ISD::UMUL_LOHI, VT: MVT::i32, Action: Expand);
264 setOperationAction(Op: ISD::MUL, VT: MVT::i32, Action: Legal);
265 setOperationAction(Op: ISD::MULHS, VT: MVT::i32, Action: Legal);
266 setOperationAction(Op: ISD::MULHU, VT: MVT::i32, Action: Legal);
267
268 // MIPS32r6 replaces the accumulator-based division/remainder with separate
269 // three register division and remainder instructions.
270 setOperationAction(Op: ISD::SDIVREM, VT: MVT::i32, Action: Expand);
271 setOperationAction(Op: ISD::UDIVREM, VT: MVT::i32, Action: Expand);
272 setOperationAction(Op: ISD::SDIV, VT: MVT::i32, Action: Legal);
273 setOperationAction(Op: ISD::UDIV, VT: MVT::i32, Action: Legal);
274 setOperationAction(Op: ISD::SREM, VT: MVT::i32, Action: Legal);
275 setOperationAction(Op: ISD::UREM, VT: MVT::i32, Action: Legal);
276
277 // MIPS32r6 replaces conditional moves with an equivalent that removes the
278 // need for three GPR read ports.
279 setOperationAction(Op: ISD::SETCC, VT: MVT::i32, Action: Legal);
280 setOperationAction(Op: ISD::SELECT, VT: MVT::i32, Action: Legal);
281 setOperationAction(Op: ISD::SELECT_CC, VT: MVT::i32, Action: Expand);
282
283 setOperationAction(Op: ISD::SETCC, VT: MVT::f32, Action: Legal);
284 setOperationAction(Op: ISD::SELECT, VT: MVT::f32, Action: Legal);
285 setOperationAction(Op: ISD::SELECT_CC, VT: MVT::f32, Action: Expand);
286
287 assert(Subtarget.isFP64bit() && "FR=1 is required for MIPS32r6");
288 setOperationAction(Op: ISD::SETCC, VT: MVT::f64, Action: Legal);
289 setOperationAction(Op: ISD::SELECT, VT: MVT::f64, Action: Legal);
290 setOperationAction(Op: ISD::SELECT_CC, VT: MVT::f64, Action: Expand);
291
292 setOperationAction(Op: ISD::BRCOND, VT: MVT::Other, Action: Legal);
293
294 // Floating point > and >= are supported via < and <=
295 setCondCodeAction(CCs: ISD::SETOGE, VT: MVT::f32, Action: Expand);
296 setCondCodeAction(CCs: ISD::SETOGT, VT: MVT::f32, Action: Expand);
297 setCondCodeAction(CCs: ISD::SETUGE, VT: MVT::f32, Action: Expand);
298 setCondCodeAction(CCs: ISD::SETUGT, VT: MVT::f32, Action: Expand);
299 setCondCodeAction(CCs: ISD::SETONE, VT: MVT::f32, Action: Expand);
300 setCondCodeAction(CCs: ISD::SETO, VT: MVT::f32, Action: Expand);
301 setCondCodeAction(CCs: ISD::SETUNE, VT: MVT::f32, Action: Expand);
302 setCondCodeAction(CCs: ISD::SETNE, VT: MVT::f32, Action: Expand);
303
304 setCondCodeAction(CCs: ISD::SETOGE, VT: MVT::f64, Action: Expand);
305 setCondCodeAction(CCs: ISD::SETOGT, VT: MVT::f64, Action: Expand);
306 setCondCodeAction(CCs: ISD::SETUGE, VT: MVT::f64, Action: Expand);
307 setCondCodeAction(CCs: ISD::SETUGT, VT: MVT::f64, Action: Expand);
308 setCondCodeAction(CCs: ISD::SETONE, VT: MVT::f64, Action: Expand);
309 setCondCodeAction(CCs: ISD::SETO, VT: MVT::f64, Action: Expand);
310 setCondCodeAction(CCs: ISD::SETUNE, VT: MVT::f64, Action: Expand);
311 setCondCodeAction(CCs: ISD::SETNE, VT: MVT::f64, Action: Expand);
312 }
313
314 if (Subtarget.hasMips64r6()) {
315 // MIPS64r6 replaces the accumulator-based multiplies with a three register
316 // instruction
317 setOperationAction(Op: ISD::SMUL_LOHI, VT: MVT::i64, Action: Expand);
318 setOperationAction(Op: ISD::UMUL_LOHI, VT: MVT::i64, Action: Expand);
319 setOperationAction(Op: ISD::MUL, VT: MVT::i64, Action: Legal);
320 setOperationAction(Op: ISD::MULHS, VT: MVT::i64, Action: Legal);
321 setOperationAction(Op: ISD::MULHU, VT: MVT::i64, Action: Legal);
322
323 // MIPS32r6 replaces the accumulator-based division/remainder with separate
324 // three register division and remainder instructions.
325 setOperationAction(Op: ISD::SDIVREM, VT: MVT::i64, Action: Expand);
326 setOperationAction(Op: ISD::UDIVREM, VT: MVT::i64, Action: Expand);
327 setOperationAction(Op: ISD::SDIV, VT: MVT::i64, Action: Legal);
328 setOperationAction(Op: ISD::UDIV, VT: MVT::i64, Action: Legal);
329 setOperationAction(Op: ISD::SREM, VT: MVT::i64, Action: Legal);
330 setOperationAction(Op: ISD::UREM, VT: MVT::i64, Action: Legal);
331
332 // MIPS64r6 replaces conditional moves with an equivalent that removes the
333 // need for three GPR read ports.
334 setOperationAction(Op: ISD::SETCC, VT: MVT::i64, Action: Legal);
335 setOperationAction(Op: ISD::SELECT, VT: MVT::i64, Action: Legal);
336 setOperationAction(Op: ISD::SELECT_CC, VT: MVT::i64, Action: Expand);
337 }
338
339 if (Subtarget.isR5900()) {
340 // R5900 FPU only supports 4 compare conditions: C.F, C.EQ, C.OLT, C.OLE
341 // (and their inversions via bc1t/bc1f). Expand all conditions that would
342 // require C.UN, C.UEQ, C.ULT, or C.ULE instructions (not available on
343 // R5900). The legalizer resolves these via operand swapping, condition
344 // inversion, and decomposition into supported conditions.
345 setCondCodeAction(CCs: ISD::SETOGT, VT: MVT::f32, Action: Expand);
346 setCondCodeAction(CCs: ISD::SETOGE, VT: MVT::f32, Action: Expand);
347 setCondCodeAction(CCs: ISD::SETGT, VT: MVT::f32, Action: Expand);
348 setCondCodeAction(CCs: ISD::SETGE, VT: MVT::f32, Action: Expand);
349 setCondCodeAction(CCs: ISD::SETULT, VT: MVT::f32, Action: Expand);
350 setCondCodeAction(CCs: ISD::SETULE, VT: MVT::f32, Action: Expand);
351 setCondCodeAction(CCs: ISD::SETUO, VT: MVT::f32, Action: Expand);
352 setCondCodeAction(CCs: ISD::SETO, VT: MVT::f32, Action: Expand);
353 setCondCodeAction(CCs: ISD::SETONE, VT: MVT::f32, Action: Expand);
354 setCondCodeAction(CCs: ISD::SETUEQ, VT: MVT::f32, Action: Expand);
355 setCondCodeAction(CCs: ISD::SETNE, VT: MVT::f32, Action: Expand);
356
357 // R5900 FPU does not support IEEE 754 special values (NaN, infinity). Use
358 // custom lowering to decide per-instruction: hardware when nnan+ninf flags
359 // guarantee no NaN or infinity, software libcall otherwise.
360 setOperationAction(Op: ISD::FADD, VT: MVT::f32, Action: Custom);
361 setOperationAction(Op: ISD::FSUB, VT: MVT::f32, Action: Custom);
362 setOperationAction(Op: ISD::FMUL, VT: MVT::f32, Action: Custom);
363 setOperationAction(Op: ISD::FDIV, VT: MVT::f32, Action: Custom);
364 setOperationAction(Op: ISD::FSQRT, VT: MVT::f32, Action: Custom);
365 }
366
367 computeRegisterProperties(TRI: Subtarget.getRegisterInfo());
368}
369
370const MipsTargetLowering *
371llvm::createMipsSETargetLowering(const MipsTargetMachine &TM,
372 const MipsSubtarget &STI) {
373 return new MipsSETargetLowering(TM, STI);
374}
375
376const TargetRegisterClass *
377MipsSETargetLowering::getRepRegClassFor(MVT VT) const {
378 if (VT == MVT::Untyped)
379 return Subtarget.hasDSP() ? &Mips::ACC64DSPRegClass : &Mips::ACC64RegClass;
380
381 return TargetLowering::getRepRegClassFor(VT);
382}
383
384// Enable MSA support for the given integer type and Register class.
385void MipsSETargetLowering::
386addMSAIntType(MVT::SimpleValueType Ty, const TargetRegisterClass *RC) {
387 addRegisterClass(VT: Ty, RC);
388
389 // Expand all builtin opcodes.
390 for (unsigned Opc = 0; Opc < ISD::BUILTIN_OP_END; ++Opc)
391 setOperationAction(Op: Opc, VT: Ty, Action: Expand);
392
393 setOperationAction(Op: ISD::BITCAST, VT: Ty, Action: Legal);
394 setOperationAction(Op: ISD::LOAD, VT: Ty, Action: Legal);
395 setOperationAction(Op: ISD::STORE, VT: Ty, Action: Legal);
396 setOperationAction(Op: ISD::EXTRACT_VECTOR_ELT, VT: Ty, Action: Custom);
397 setOperationAction(Op: ISD::INSERT_VECTOR_ELT, VT: Ty, Action: Legal);
398 setOperationAction(Op: ISD::BUILD_VECTOR, VT: Ty, Action: Custom);
399 setOperationAction(Ops: {ISD::UNDEF, ISD::POISON}, VT: Ty, Action: Legal);
400
401 setOperationAction(Op: ISD::ADD, VT: Ty, Action: Legal);
402 setOperationAction(Op: ISD::AND, VT: Ty, Action: Legal);
403 setOperationAction(Op: ISD::CTLZ, VT: Ty, Action: Legal);
404 setOperationAction(Op: ISD::CTPOP, VT: Ty, Action: Legal);
405 setOperationAction(Op: ISD::MUL, VT: Ty, Action: Legal);
406 setOperationAction(Op: ISD::OR, VT: Ty, Action: Legal);
407 setOperationAction(Op: ISD::SDIV, VT: Ty, Action: Legal);
408 setOperationAction(Op: ISD::SREM, VT: Ty, Action: Legal);
409 setOperationAction(Op: ISD::SHL, VT: Ty, Action: Legal);
410 setOperationAction(Op: ISD::SRA, VT: Ty, Action: Legal);
411 setOperationAction(Op: ISD::SRL, VT: Ty, Action: Legal);
412 setOperationAction(Op: ISD::SUB, VT: Ty, Action: Legal);
413 setOperationAction(Op: ISD::SMAX, VT: Ty, Action: Legal);
414 setOperationAction(Op: ISD::SMIN, VT: Ty, Action: Legal);
415 setOperationAction(Op: ISD::UDIV, VT: Ty, Action: Legal);
416 setOperationAction(Op: ISD::UREM, VT: Ty, Action: Legal);
417 setOperationAction(Op: ISD::UMAX, VT: Ty, Action: Legal);
418 setOperationAction(Op: ISD::UMIN, VT: Ty, Action: Legal);
419 setOperationAction(Op: ISD::VECTOR_SHUFFLE, VT: Ty, Action: Custom);
420 setOperationAction(Op: ISD::VSELECT, VT: Ty, Action: Legal);
421 setOperationAction(Op: ISD::XOR, VT: Ty, Action: Legal);
422
423 if (Ty == MVT::v4i32 || Ty == MVT::v2i64) {
424 setOperationAction(Op: ISD::FP_TO_SINT, VT: Ty, Action: Legal);
425 setOperationAction(Op: ISD::FP_TO_UINT, VT: Ty, Action: Legal);
426 setOperationAction(Op: ISD::SINT_TO_FP, VT: Ty, Action: Legal);
427 setOperationAction(Op: ISD::UINT_TO_FP, VT: Ty, Action: Legal);
428 }
429
430 setOperationAction(Op: ISD::SETCC, VT: Ty, Action: Legal);
431 setCondCodeAction(CCs: ISD::SETNE, VT: Ty, Action: Expand);
432 setCondCodeAction(CCs: ISD::SETGE, VT: Ty, Action: Expand);
433 setCondCodeAction(CCs: ISD::SETGT, VT: Ty, Action: Expand);
434 setCondCodeAction(CCs: ISD::SETUGE, VT: Ty, Action: Expand);
435 setCondCodeAction(CCs: ISD::SETUGT, VT: Ty, Action: Expand);
436}
437
438// Enable MSA support for the given floating-point type and Register class.
439void MipsSETargetLowering::
440addMSAFloatType(MVT::SimpleValueType Ty, const TargetRegisterClass *RC) {
441 addRegisterClass(VT: Ty, RC);
442
443 // Expand all builtin opcodes.
444 for (unsigned Opc = 0; Opc < ISD::BUILTIN_OP_END; ++Opc)
445 setOperationAction(Op: Opc, VT: Ty, Action: Expand);
446
447 setOperationAction(Op: ISD::LOAD, VT: Ty, Action: Legal);
448 setOperationAction(Op: ISD::STORE, VT: Ty, Action: Legal);
449 setOperationAction(Op: ISD::BITCAST, VT: Ty, Action: Legal);
450 setOperationAction(Op: ISD::EXTRACT_VECTOR_ELT, VT: Ty, Action: Legal);
451 setOperationAction(Op: ISD::INSERT_VECTOR_ELT, VT: Ty, Action: Legal);
452 setOperationAction(Op: ISD::BUILD_VECTOR, VT: Ty, Action: Custom);
453 setOperationAction(Op: ISD::UNDEF, VT: Ty, Action: Legal);
454
455 if (Ty != MVT::v8f16) {
456 setOperationAction(Op: ISD::FABS, VT: Ty, Action: Legal);
457 setOperationAction(Op: ISD::FADD, VT: Ty, Action: Legal);
458 setOperationAction(Op: ISD::FDIV, VT: Ty, Action: Legal);
459 setOperationAction(Op: ISD::FEXP2, VT: Ty, Action: Legal);
460 setOperationAction(Op: ISD::FLOG2, VT: Ty, Action: Legal);
461 setOperationAction(Op: ISD::FMA, VT: Ty, Action: Legal);
462 setOperationAction(Op: ISD::FMUL, VT: Ty, Action: Legal);
463 setOperationAction(Op: ISD::FRINT, VT: Ty, Action: Legal);
464 setOperationAction(Op: ISD::FSQRT, VT: Ty, Action: Legal);
465 setOperationAction(Op: ISD::FSUB, VT: Ty, Action: Legal);
466 setOperationAction(Op: ISD::VSELECT, VT: Ty, Action: Legal);
467
468 setOperationAction(Op: ISD::SETCC, VT: Ty, Action: Legal);
469 setCondCodeAction(CCs: ISD::SETOGE, VT: Ty, Action: Expand);
470 setCondCodeAction(CCs: ISD::SETOGT, VT: Ty, Action: Expand);
471 setCondCodeAction(CCs: ISD::SETUGE, VT: Ty, Action: Expand);
472 setCondCodeAction(CCs: ISD::SETUGT, VT: Ty, Action: Expand);
473 setCondCodeAction(CCs: ISD::SETGE, VT: Ty, Action: Expand);
474 setCondCodeAction(CCs: ISD::SETGT, VT: Ty, Action: Expand);
475 }
476}
477
478SDValue MipsSETargetLowering::lowerSELECT(SDValue Op, SelectionDAG &DAG) const {
479 if(!Subtarget.hasMips32r6())
480 return MipsTargetLowering::LowerOperation(Op, DAG);
481
482 EVT ResTy = Op->getValueType(ResNo: 0);
483 SDLoc DL(Op);
484
485 // Although MTC1_D64 takes an i32 and writes an f64, the upper 32 bits of the
486 // floating point register are undefined. Not really an issue as sel.d, which
487 // is produced from an FSELECT node, only looks at bit 0.
488 SDValue Tmp = DAG.getNode(Opcode: MipsISD::MTC1_D64, DL, VT: MVT::f64, Operand: Op->getOperand(Num: 0));
489 return DAG.getNode(Opcode: MipsISD::FSELECT, DL, VT: ResTy, N1: Tmp, N2: Op->getOperand(Num: 1),
490 N3: Op->getOperand(Num: 2));
491}
492
493// Lower FP16_TO_FP (the soft-promote-half representation of an f16 -> f32/f64
494// conversion).
495SDValue MipsSETargetLowering::lowerFP16_TO_FP(SDValue Op,
496 SelectionDAG &DAG) const {
497 SDLoc DL(Op);
498 EVT ResTy = Op.getValueType();
499 assert((ResTy == MVT::f32 || ResTy == MVT::f64) && "Unexpected FP16_TO_FP");
500
501 // The operand type is i32 because i16 isn't actually legal on MIPS.
502 SDValue In = Op.getOperand(i: 0);
503 assert(In.getValueType() == MVT::i32 && "Unexpected FP16_TO_FP operand type");
504
505 // Splat into a v8i16 (the 32-bit In value is truncated to the lower 16 bits).
506 SDValue Splatted = DAG.getSplatBuildVector(VT: MVT::v8i16, DL, Op: In);
507
508 // Bitcast from v8i16 to v8f16.
509 SDValue HVec = DAG.getNode(Opcode: ISD::BITCAST, DL, VT: MVT::v8f16, Operand: Splatted);
510
511 // Convert from v8f16 to v4f32.
512 SDValue F32Vec = DAG.getNode(
513 Opcode: ISD::INTRINSIC_WO_CHAIN, DL, VT: MVT::v4f32,
514 N1: DAG.getConstant(Val: Intrinsic::mips_fexupr_w, DL, VT: MVT::i32), N2: HVec);
515 SDValue Res;
516 if (ResTy == MVT::f32) {
517 // Every lane has the converted value, just read it from lane 0.
518 Res = DAG.getNode(Opcode: ISD::EXTRACT_VECTOR_ELT, DL, VT: MVT::f32, N1: F32Vec,
519 N2: DAG.getVectorIdxConstant(Val: 0, DL));
520 } else {
521 // Convert from v4f32 to v2f64.
522 SDValue F64Vec = DAG.getNode(
523 Opcode: ISD::INTRINSIC_WO_CHAIN, DL, VT: MVT::v2f64,
524 N1: DAG.getConstant(Val: Intrinsic::mips_fexupr_d, DL, VT: MVT::i32), N2: F32Vec);
525 // Every lane has the converted value, just read it from lane 0.
526 Res = DAG.getNode(Opcode: ISD::EXTRACT_VECTOR_ELT, DL, VT: MVT::f64, N1: F64Vec,
527 N2: DAG.getVectorIdxConstant(Val: 0, DL));
528 }
529
530 return Res;
531}
532
533// Lower FP_TO_FP16 (the soft-promote-half representation of an f32/f64 -> f16
534// conversion)
535SDValue MipsSETargetLowering::lowerFP_TO_FP16(SDValue Op,
536 SelectionDAG &DAG) const {
537 SDLoc DL(Op);
538 EVT ResTy = Op.getValueType();
539 SDValue In = Op.getOperand(i: 0);
540 assert((In.getValueType() == MVT::f32 || In.getValueType() == MVT::f64) &&
541 "Unexpected FP_TO_FP16");
542
543 SDValue F32Vec;
544 if (In.getValueType() == MVT::f64) {
545 // Splat f64 to v2f64, then convert to v4f32.
546 SDValue F64Vec = DAG.getSplatBuildVector(VT: MVT::v2f64, DL, Op: In);
547 F32Vec = DAG.getNode(Opcode: ISD::INTRINSIC_WO_CHAIN, DL, VT: MVT::v4f32,
548 N1: DAG.getConstant(Val: Intrinsic::mips_fexdo_w, DL, VT: MVT::i32),
549 N2: F64Vec, N3: F64Vec);
550 } else {
551 // Splat f32 to v4f32.
552 F32Vec = DAG.getSplatBuildVector(VT: MVT::v4f32, DL, Op: In);
553 }
554
555 // Then convert from v4f32 to v8f16.
556 SDValue HVec = DAG.getNode(
557 Opcode: ISD::INTRINSIC_WO_CHAIN, DL, VT: MVT::v8f16,
558 N1: DAG.getConstant(Val: Intrinsic::mips_fexdo_h, DL, VT: MVT::i32), N2: F32Vec, N3: F32Vec);
559
560 // Finally cast to v8i16 (f16 is soft-promoted).
561 SDValue IVec = DAG.getNode(Opcode: ISD::BITCAST, DL, VT: MVT::v8i16, Operand: HVec);
562 SDValue Res = DAG.getNode(Opcode: ISD::EXTRACT_VECTOR_ELT, DL, VT: ResTy, N1: IVec,
563 N2: DAG.getVectorIdxConstant(Val: 0, DL));
564
565 return Res;
566}
567
568bool MipsSETargetLowering::allowsMisalignedMemoryAccesses(
569 EVT VT, unsigned, Align, MachineMemOperand::Flags, unsigned *Fast) const {
570 MVT::SimpleValueType SVT = VT.getSimpleVT().SimpleTy;
571
572 if (Subtarget.systemSupportsUnalignedAccess()) {
573 // MIPS32r6/MIPS64r6 is required to support unaligned access. It's
574 // implementation defined whether this is handled by hardware, software, or
575 // a hybrid of the two but it's expected that most implementations will
576 // handle the majority of cases in hardware.
577 if (Fast)
578 *Fast = 1;
579 return true;
580 } else if (Subtarget.hasMips32r6()) {
581 return false;
582 }
583
584 switch (SVT) {
585 case MVT::i64:
586 case MVT::i32:
587 if (Fast)
588 *Fast = 1;
589 return true;
590 default:
591 return false;
592 }
593}
594
595SDValue MipsSETargetLowering::LowerOperation(SDValue Op,
596 SelectionDAG &DAG) const {
597 switch(Op.getOpcode()) {
598 case ISD::LOAD: return lowerLOAD(Op, DAG);
599 case ISD::STORE: return lowerSTORE(Op, DAG);
600 case ISD::SMUL_LOHI: return lowerMulDiv(Op, NewOpc: MipsISD::Mult, HasLo: true, HasHi: true, DAG);
601 case ISD::UMUL_LOHI: return lowerMulDiv(Op, NewOpc: MipsISD::Multu, HasLo: true, HasHi: true, DAG);
602 case ISD::MULHS: return lowerMulDiv(Op, NewOpc: MipsISD::Mult, HasLo: false, HasHi: true, DAG);
603 case ISD::MULHU: return lowerMulDiv(Op, NewOpc: MipsISD::Multu, HasLo: false, HasHi: true, DAG);
604 case ISD::MUL: return lowerMulDiv(Op, NewOpc: MipsISD::Mult, HasLo: true, HasHi: false, DAG);
605 case ISD::SDIVREM: return lowerMulDiv(Op, NewOpc: MipsISD::DivRem, HasLo: true, HasHi: true, DAG);
606 case ISD::UDIVREM: return lowerMulDiv(Op, NewOpc: MipsISD::DivRemU, HasLo: true, HasHi: true,
607 DAG);
608 case ISD::INTRINSIC_WO_CHAIN: return lowerINTRINSIC_WO_CHAIN(Op, DAG);
609 case ISD::INTRINSIC_W_CHAIN: return lowerINTRINSIC_W_CHAIN(Op, DAG);
610 case ISD::INTRINSIC_VOID: return lowerINTRINSIC_VOID(Op, DAG);
611 case ISD::EXTRACT_VECTOR_ELT: return lowerEXTRACT_VECTOR_ELT(Op, DAG);
612 case ISD::BUILD_VECTOR: return lowerBUILD_VECTOR(Op, DAG);
613 case ISD::VECTOR_SHUFFLE: return lowerVECTOR_SHUFFLE(Op, DAG);
614 case ISD::SELECT:
615 return lowerSELECT(Op, DAG);
616 case ISD::FP16_TO_FP:
617 case ISD::STRICT_FP16_TO_FP:
618 return lowerFP16_TO_FP(Op, DAG);
619 case ISD::FP_TO_FP16:
620 case ISD::STRICT_FP_TO_FP16:
621 return lowerFP_TO_FP16(Op, DAG);
622 case ISD::BITCAST: return lowerBITCAST(Op, DAG);
623 case ISD::FADD:
624 return lowerR5900FPOp(Op, DAG, LC: RTLIB::ADD_F32);
625 case ISD::FSUB:
626 return lowerR5900FPOp(Op, DAG, LC: RTLIB::SUB_F32);
627 case ISD::FMUL:
628 return lowerR5900FPOp(Op, DAG, LC: RTLIB::MUL_F32);
629 case ISD::FDIV:
630 return lowerR5900FPOp(Op, DAG, LC: RTLIB::DIV_F32);
631 case ISD::FSQRT:
632 return lowerR5900FPOp(Op, DAG, LC: RTLIB::SQRT_F32);
633 }
634
635 return MipsTargetLowering::LowerOperation(Op, DAG);
636}
637
638SDValue MipsSETargetLowering::lowerR5900FPOp(SDValue Op, SelectionDAG &DAG,
639 RTLIB::Libcall LC) const {
640 assert(Subtarget.isR5900());
641 SDNodeFlags Flags = Op->getFlags();
642
643 if (Flags.hasNoNaNs() && Flags.hasNoInfs()) {
644 // Use the hardware FPU instruction if the operation is guaranteed to have
645 // no NaN or infinity inputs/outputs (nnan+ninf flags).
646 return Op;
647 }
648
649 // Fall back to a software libcall for IEEE correctness.
650 SDLoc DL(Op);
651 MVT VT = Op.getSimpleValueType();
652 SmallVector<SDValue, 2> Ops(Op->op_begin(), Op->op_end());
653 TargetLowering::MakeLibCallOptions CallOptions;
654 auto [Result, Chain] = makeLibCall(DAG, LC, RetVT: VT, Ops, CallOptions, dl: DL);
655 return Result;
656}
657
658// Fold zero extensions into MipsISD::VEXTRACT_[SZ]EXT_ELT
659//
660// Performs the following transformations:
661// - Changes MipsISD::VEXTRACT_[SZ]EXT_ELT to zero extension if its
662// sign/zero-extension is completely overwritten by the new one performed by
663// the ISD::AND.
664// - Removes redundant zero extensions performed by an ISD::AND.
665static SDValue performANDCombine(SDNode *N, SelectionDAG &DAG,
666 TargetLowering::DAGCombinerInfo &DCI,
667 const MipsSubtarget &Subtarget) {
668 if (!Subtarget.hasMSA())
669 return SDValue();
670
671 SDValue Op0 = N->getOperand(Num: 0);
672 SDValue Op1 = N->getOperand(Num: 1);
673 unsigned Op0Opcode = Op0->getOpcode();
674
675 // (and (MipsVExtract[SZ]Ext $a, $b, $c), imm:$d)
676 // where $d + 1 == 2^n and n == 32
677 // or $d + 1 == 2^n and n <= 32 and ZExt
678 // -> (MipsVExtractZExt $a, $b, $c)
679 if (Op0Opcode == MipsISD::VEXTRACT_SEXT_ELT ||
680 Op0Opcode == MipsISD::VEXTRACT_ZEXT_ELT) {
681 ConstantSDNode *Mask = dyn_cast<ConstantSDNode>(Val&: Op1);
682
683 if (!Mask)
684 return SDValue();
685
686 int32_t Log2IfPositive = (Mask->getAPIntValue() + 1).exactLogBase2();
687
688 if (Log2IfPositive <= 0)
689 return SDValue(); // Mask+1 is not a power of 2
690
691 SDValue Op0Op2 = Op0->getOperand(Num: 2);
692 EVT ExtendTy = cast<VTSDNode>(Val&: Op0Op2)->getVT();
693 unsigned ExtendTySize = ExtendTy.getSizeInBits();
694 unsigned Log2 = Log2IfPositive;
695
696 if ((Op0Opcode == MipsISD::VEXTRACT_ZEXT_ELT && Log2 >= ExtendTySize) ||
697 Log2 == ExtendTySize) {
698 SDValue Ops[] = { Op0->getOperand(Num: 0), Op0->getOperand(Num: 1), Op0Op2 };
699 return DAG.getNode(Opcode: MipsISD::VEXTRACT_ZEXT_ELT, DL: SDLoc(Op0),
700 VTList: Op0->getVTList(),
701 Ops: ArrayRef(Ops, Op0->getNumOperands()));
702 }
703 }
704
705 return SDValue();
706}
707
708// Determine if the specified node is a constant vector splat.
709//
710// Returns true and sets Imm if:
711// * N is a ISD::BUILD_VECTOR representing a constant splat
712//
713// This function is quite similar to MipsSEDAGToDAGISel::selectVSplat. The
714// differences are that it assumes the MSA has already been checked and the
715// arbitrary requirement for a maximum of 32-bit integers isn't applied (and
716// must not be in order for binsri.d to be selectable).
717static bool isVSplat(SDValue N, APInt &Imm, bool IsLittleEndian) {
718 BuildVectorSDNode *Node = dyn_cast<BuildVectorSDNode>(Val: N.getNode());
719
720 if (!Node)
721 return false;
722
723 APInt SplatValue, SplatUndef;
724 unsigned SplatBitSize;
725 bool HasAnyUndefs;
726
727 if (!Node->isConstantSplat(SplatValue, SplatUndef, SplatBitSize, HasAnyUndefs,
728 MinSplatBits: 8, isBigEndian: !IsLittleEndian))
729 return false;
730
731 Imm = SplatValue;
732
733 return true;
734}
735
736// Test whether the given node is an all-ones build_vector.
737static bool isVectorAllOnes(SDValue N) {
738 // Look through bitcasts. Endianness doesn't matter because we are looking
739 // for an all-ones value.
740 if (N->getOpcode() == ISD::BITCAST)
741 N = N->getOperand(Num: 0);
742
743 BuildVectorSDNode *BVN = dyn_cast<BuildVectorSDNode>(Val&: N);
744
745 if (!BVN)
746 return false;
747
748 APInt SplatValue, SplatUndef;
749 unsigned SplatBitSize;
750 bool HasAnyUndefs;
751
752 // Endianness doesn't matter in this context because we are looking for
753 // an all-ones value.
754 if (BVN->isConstantSplat(SplatValue, SplatUndef, SplatBitSize, HasAnyUndefs))
755 return SplatValue.isAllOnes();
756
757 return false;
758}
759
760// Test whether N is the bitwise inverse of OfNode.
761static bool isBitwiseInverse(SDValue N, SDValue OfNode) {
762 if (N->getOpcode() != ISD::XOR)
763 return false;
764
765 if (isVectorAllOnes(N: N->getOperand(Num: 0)))
766 return N->getOperand(Num: 1) == OfNode;
767
768 if (isVectorAllOnes(N: N->getOperand(Num: 1)))
769 return N->getOperand(Num: 0) == OfNode;
770
771 return false;
772}
773
774// Perform combines where ISD::OR is the root node.
775//
776// Performs the following transformations:
777// - (or (and $a, $mask), (and $b, $inv_mask)) => (vselect $mask, $a, $b)
778// where $inv_mask is the bitwise inverse of $mask and the 'or' has a 128-bit
779// vector type.
780static SDValue performORCombine(SDNode *N, SelectionDAG &DAG,
781 TargetLowering::DAGCombinerInfo &DCI,
782 const MipsSubtarget &Subtarget) {
783 if (!Subtarget.hasMSA())
784 return SDValue();
785
786 EVT Ty = N->getValueType(ResNo: 0);
787
788 if (!Ty.is128BitVector())
789 return SDValue();
790
791 SDValue Op0 = N->getOperand(Num: 0);
792 SDValue Op1 = N->getOperand(Num: 1);
793
794 if (Op0->getOpcode() == ISD::AND && Op1->getOpcode() == ISD::AND) {
795 SDValue Op0Op0 = Op0->getOperand(Num: 0);
796 SDValue Op0Op1 = Op0->getOperand(Num: 1);
797 SDValue Op1Op0 = Op1->getOperand(Num: 0);
798 SDValue Op1Op1 = Op1->getOperand(Num: 1);
799 bool IsLittleEndian = !Subtarget.isLittle();
800
801 SDValue IfSet, IfClr, Cond;
802 bool IsConstantMask = false;
803 APInt Mask, InvMask;
804
805 // If Op0Op0 is an appropriate mask, try to find it's inverse in either
806 // Op1Op0, or Op1Op1. Keep track of the Cond, IfSet, and IfClr nodes, while
807 // looking.
808 // IfClr will be set if we find a valid match.
809 if (isVSplat(N: Op0Op0, Imm&: Mask, IsLittleEndian)) {
810 Cond = Op0Op0;
811 IfSet = Op0Op1;
812
813 if (isVSplat(N: Op1Op0, Imm&: InvMask, IsLittleEndian) &&
814 Mask.getBitWidth() == InvMask.getBitWidth() && Mask == ~InvMask)
815 IfClr = Op1Op1;
816 else if (isVSplat(N: Op1Op1, Imm&: InvMask, IsLittleEndian) &&
817 Mask.getBitWidth() == InvMask.getBitWidth() && Mask == ~InvMask)
818 IfClr = Op1Op0;
819
820 IsConstantMask = true;
821 }
822
823 // If IfClr is not yet set, and Op0Op1 is an appropriate mask, try the same
824 // thing again using this mask.
825 // IfClr will be set if we find a valid match.
826 if (!IfClr.getNode() && isVSplat(N: Op0Op1, Imm&: Mask, IsLittleEndian)) {
827 Cond = Op0Op1;
828 IfSet = Op0Op0;
829
830 if (isVSplat(N: Op1Op0, Imm&: InvMask, IsLittleEndian) &&
831 Mask.getBitWidth() == InvMask.getBitWidth() && Mask == ~InvMask)
832 IfClr = Op1Op1;
833 else if (isVSplat(N: Op1Op1, Imm&: InvMask, IsLittleEndian) &&
834 Mask.getBitWidth() == InvMask.getBitWidth() && Mask == ~InvMask)
835 IfClr = Op1Op0;
836
837 IsConstantMask = true;
838 }
839
840 // If IfClr is not yet set, try looking for a non-constant match.
841 // IfClr will be set if we find a valid match amongst the eight
842 // possibilities.
843 if (!IfClr.getNode()) {
844 if (isBitwiseInverse(N: Op0Op0, OfNode: Op1Op0)) {
845 Cond = Op1Op0;
846 IfSet = Op1Op1;
847 IfClr = Op0Op1;
848 } else if (isBitwiseInverse(N: Op0Op1, OfNode: Op1Op0)) {
849 Cond = Op1Op0;
850 IfSet = Op1Op1;
851 IfClr = Op0Op0;
852 } else if (isBitwiseInverse(N: Op0Op0, OfNode: Op1Op1)) {
853 Cond = Op1Op1;
854 IfSet = Op1Op0;
855 IfClr = Op0Op1;
856 } else if (isBitwiseInverse(N: Op0Op1, OfNode: Op1Op1)) {
857 Cond = Op1Op1;
858 IfSet = Op1Op0;
859 IfClr = Op0Op0;
860 } else if (isBitwiseInverse(N: Op1Op0, OfNode: Op0Op0)) {
861 Cond = Op0Op0;
862 IfSet = Op0Op1;
863 IfClr = Op1Op1;
864 } else if (isBitwiseInverse(N: Op1Op1, OfNode: Op0Op0)) {
865 Cond = Op0Op0;
866 IfSet = Op0Op1;
867 IfClr = Op1Op0;
868 } else if (isBitwiseInverse(N: Op1Op0, OfNode: Op0Op1)) {
869 Cond = Op0Op1;
870 IfSet = Op0Op0;
871 IfClr = Op1Op1;
872 } else if (isBitwiseInverse(N: Op1Op1, OfNode: Op0Op1)) {
873 Cond = Op0Op1;
874 IfSet = Op0Op0;
875 IfClr = Op1Op0;
876 }
877 }
878
879 // At this point, IfClr will be set if we have a valid match.
880 if (!IfClr.getNode())
881 return SDValue();
882
883 assert(Cond.getNode() && IfSet.getNode());
884
885 // Fold degenerate cases.
886 if (IsConstantMask) {
887 if (Mask.isAllOnes())
888 return IfSet;
889 else if (Mask == 0)
890 return IfClr;
891 }
892
893 // Transform the DAG into an equivalent VSELECT.
894 return DAG.getNode(Opcode: ISD::VSELECT, DL: SDLoc(N), VT: Ty, N1: Cond, N2: IfSet, N3: IfClr);
895 }
896
897 return SDValue();
898}
899
900static bool shouldTransformMulToShiftsAddsSubs(APInt C, EVT VT,
901 SelectionDAG &DAG,
902 const MipsSubtarget &Subtarget) {
903 // Estimate the number of operations the below transform will turn a
904 // constant multiply into. The number is approximately equal to the minimal
905 // number of powers of two that constant can be broken down to by adding
906 // or subtracting them.
907 //
908 // If we have taken more than 12[1] / 8[2] steps to attempt the
909 // optimization for a native sized value, it is more than likely that this
910 // optimization will make things worse.
911 //
912 // [1] MIPS64 requires 6 instructions at most to materialize any constant,
913 // multiplication requires at least 4 cycles, but another cycle (or two)
914 // to retrieve the result from the HI/LO registers.
915 //
916 // [2] For MIPS32, more than 8 steps is expensive as the constant could be
917 // materialized in 2 instructions, multiplication requires at least 4
918 // cycles, but another cycle (or two) to retrieve the result from the
919 // HI/LO registers.
920 //
921 // TODO:
922 // - MaxSteps needs to consider the `VT` of the constant for the current
923 // target.
924 // - Consider to perform this optimization after type legalization.
925 // That allows to remove a workaround for types not supported natively.
926 // - Take in account `-Os, -Oz` flags because this optimization
927 // increases code size.
928 unsigned MaxSteps = Subtarget.isABI_O32() ? 8 : 12;
929
930 SmallVector<APInt, 16> WorkStack(1, C);
931 unsigned Steps = 0;
932 unsigned BitWidth = C.getBitWidth();
933
934 while (!WorkStack.empty()) {
935 APInt Val = WorkStack.pop_back_val();
936
937 if (Val == 0 || Val == 1)
938 continue;
939
940 if (Steps >= MaxSteps)
941 return false;
942
943 if (Val.isPowerOf2()) {
944 ++Steps;
945 continue;
946 }
947
948 APInt Floor = APInt(BitWidth, 1) << Val.logBase2();
949 APInt Ceil = Val.isNegative() ? APInt(BitWidth, 0)
950 : APInt(BitWidth, 1) << C.ceilLogBase2();
951 if ((Val - Floor).ule(RHS: Ceil - Val)) {
952 WorkStack.push_back(Elt: Floor);
953 WorkStack.push_back(Elt: Val - Floor);
954 } else {
955 WorkStack.push_back(Elt: Ceil);
956 WorkStack.push_back(Elt: Ceil - Val);
957 }
958
959 ++Steps;
960 }
961
962 // If the value being multiplied is not supported natively, we have to pay
963 // an additional legalization cost, conservatively assume an increase in the
964 // cost of 3 instructions per step. This values for this heuristic were
965 // determined experimentally.
966 unsigned RegisterSize = DAG.getTargetLoweringInfo()
967 .getRegisterType(Context&: *DAG.getContext(), VT)
968 .getSizeInBits();
969 Steps *= (VT.getSizeInBits() != RegisterSize) * 3;
970 if (Steps > 27)
971 return false;
972
973 return true;
974}
975
976static SDValue genConstMult(SDValue X, APInt C, const SDLoc &DL, EVT VT,
977 EVT ShiftTy, SelectionDAG &DAG) {
978 // Return 0.
979 if (C == 0)
980 return DAG.getConstant(Val: 0, DL, VT);
981
982 // Return x.
983 if (C == 1)
984 return X;
985
986 // If c is power of 2, return (shl x, log2(c)).
987 if (C.isPowerOf2())
988 return DAG.getNode(Opcode: ISD::SHL, DL, VT, N1: X,
989 N2: DAG.getConstant(Val: C.logBase2(), DL, VT: ShiftTy));
990
991 unsigned BitWidth = C.getBitWidth();
992 APInt Floor = APInt(BitWidth, 1) << C.logBase2();
993 APInt Ceil = C.isNegative() ? APInt(BitWidth, 0) :
994 APInt(BitWidth, 1) << C.ceilLogBase2();
995
996 // If |c - floor_c| <= |c - ceil_c|,
997 // where floor_c = pow(2, floor(log2(c))) and ceil_c = pow(2, ceil(log2(c))),
998 // return (add constMult(x, floor_c), constMult(x, c - floor_c)).
999 if ((C - Floor).ule(RHS: Ceil - C)) {
1000 SDValue Op0 = genConstMult(X, C: Floor, DL, VT, ShiftTy, DAG);
1001 SDValue Op1 = genConstMult(X, C: C - Floor, DL, VT, ShiftTy, DAG);
1002 return DAG.getNode(Opcode: ISD::ADD, DL, VT, N1: Op0, N2: Op1);
1003 }
1004
1005 // If |c - floor_c| > |c - ceil_c|,
1006 // return (sub constMult(x, ceil_c), constMult(x, ceil_c - c)).
1007 SDValue Op0 = genConstMult(X, C: Ceil, DL, VT, ShiftTy, DAG);
1008 SDValue Op1 = genConstMult(X, C: Ceil - C, DL, VT, ShiftTy, DAG);
1009 return DAG.getNode(Opcode: ISD::SUB, DL, VT, N1: Op0, N2: Op1);
1010}
1011
1012static SDValue performMULCombine(SDNode *N, SelectionDAG &DAG,
1013 const TargetLowering::DAGCombinerInfo &DCI,
1014 const MipsSETargetLowering *TL,
1015 const MipsSubtarget &Subtarget) {
1016 EVT VT = N->getValueType(ResNo: 0);
1017
1018 if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(Val: N->getOperand(Num: 1)))
1019 if (!VT.isVector() && shouldTransformMulToShiftsAddsSubs(
1020 C: C->getAPIntValue(), VT, DAG, Subtarget))
1021 return genConstMult(X: N->getOperand(Num: 0), C: C->getAPIntValue(), DL: SDLoc(N), VT,
1022 ShiftTy: TL->getScalarShiftAmountTy(DAG.getDataLayout(), VT),
1023 DAG);
1024
1025 return SDValue(N, 0);
1026}
1027
1028static SDValue performDSPShiftCombine(unsigned Opc, SDNode *N, EVT Ty,
1029 SelectionDAG &DAG,
1030 const MipsSubtarget &Subtarget) {
1031 // See if this is a vector splat immediate node.
1032 APInt SplatValue, SplatUndef;
1033 unsigned SplatBitSize;
1034 bool HasAnyUndefs;
1035 unsigned EltSize = Ty.getScalarSizeInBits();
1036 BuildVectorSDNode *BV = dyn_cast<BuildVectorSDNode>(Val: N->getOperand(Num: 1));
1037
1038 if (!Subtarget.hasDSP())
1039 return SDValue();
1040
1041 if (!BV ||
1042 !BV->isConstantSplat(SplatValue, SplatUndef, SplatBitSize, HasAnyUndefs,
1043 MinSplatBits: EltSize, isBigEndian: !Subtarget.isLittle()) ||
1044 (SplatBitSize != EltSize) ||
1045 (SplatValue.getZExtValue() >= EltSize))
1046 return SDValue();
1047
1048 SDLoc DL(N);
1049 return DAG.getNode(Opcode: Opc, DL, VT: Ty, N1: N->getOperand(Num: 0),
1050 N2: DAG.getConstant(Val: SplatValue.getZExtValue(), DL, VT: MVT::i32));
1051}
1052
1053static SDValue performSHLCombine(SDNode *N, SelectionDAG &DAG,
1054 TargetLowering::DAGCombinerInfo &DCI,
1055 const MipsSubtarget &Subtarget) {
1056 EVT Ty = N->getValueType(ResNo: 0);
1057
1058 if ((Ty != MVT::v2i16) && (Ty != MVT::v4i8))
1059 return SDValue();
1060
1061 return performDSPShiftCombine(Opc: MipsISD::SHLL_DSP, N, Ty, DAG, Subtarget);
1062}
1063
1064// Fold sign-extensions into MipsISD::VEXTRACT_[SZ]EXT_ELT for MSA and fold
1065// constant splats into MipsISD::SHRA_DSP for DSPr2.
1066//
1067// Performs the following transformations:
1068// - Changes MipsISD::VEXTRACT_[SZ]EXT_ELT to sign extension if its
1069// sign/zero-extension is completely overwritten by the new one performed by
1070// the ISD::SRA and ISD::SHL nodes.
1071// - Removes redundant sign extensions performed by an ISD::SRA and ISD::SHL
1072// sequence.
1073//
1074// See performDSPShiftCombine for more information about the transformation
1075// used for DSPr2.
1076static SDValue performSRACombine(SDNode *N, SelectionDAG &DAG,
1077 TargetLowering::DAGCombinerInfo &DCI,
1078 const MipsSubtarget &Subtarget) {
1079 EVT Ty = N->getValueType(ResNo: 0);
1080
1081 if (Subtarget.hasMSA()) {
1082 SDValue Op0 = N->getOperand(Num: 0);
1083 SDValue Op1 = N->getOperand(Num: 1);
1084
1085 // (sra (shl (MipsVExtract[SZ]Ext $a, $b, $c), imm:$d), imm:$d)
1086 // where $d + sizeof($c) == 32
1087 // or $d + sizeof($c) <= 32 and SExt
1088 // -> (MipsVExtractSExt $a, $b, $c)
1089 if (Op0->getOpcode() == ISD::SHL && Op1 == Op0->getOperand(Num: 1)) {
1090 SDValue Op0Op0 = Op0->getOperand(Num: 0);
1091 ConstantSDNode *ShAmount = dyn_cast<ConstantSDNode>(Val&: Op1);
1092
1093 if (!ShAmount)
1094 return SDValue();
1095
1096 if (Op0Op0->getOpcode() != MipsISD::VEXTRACT_SEXT_ELT &&
1097 Op0Op0->getOpcode() != MipsISD::VEXTRACT_ZEXT_ELT)
1098 return SDValue();
1099
1100 EVT ExtendTy = cast<VTSDNode>(Val: Op0Op0->getOperand(Num: 2))->getVT();
1101 unsigned TotalBits = ShAmount->getZExtValue() + ExtendTy.getSizeInBits();
1102
1103 if (TotalBits == 32 ||
1104 (Op0Op0->getOpcode() == MipsISD::VEXTRACT_SEXT_ELT &&
1105 TotalBits <= 32)) {
1106 SDValue Ops[] = { Op0Op0->getOperand(Num: 0), Op0Op0->getOperand(Num: 1),
1107 Op0Op0->getOperand(Num: 2) };
1108 return DAG.getNode(Opcode: MipsISD::VEXTRACT_SEXT_ELT, DL: SDLoc(Op0Op0),
1109 VTList: Op0Op0->getVTList(),
1110 Ops: ArrayRef(Ops, Op0Op0->getNumOperands()));
1111 }
1112 }
1113 }
1114
1115 if ((Ty != MVT::v2i16) && ((Ty != MVT::v4i8) || !Subtarget.hasDSPR2()))
1116 return SDValue();
1117
1118 return performDSPShiftCombine(Opc: MipsISD::SHRA_DSP, N, Ty, DAG, Subtarget);
1119}
1120
1121
1122static SDValue performSRLCombine(SDNode *N, SelectionDAG &DAG,
1123 TargetLowering::DAGCombinerInfo &DCI,
1124 const MipsSubtarget &Subtarget) {
1125 EVT Ty = N->getValueType(ResNo: 0);
1126
1127 if (((Ty != MVT::v2i16) || !Subtarget.hasDSPR2()) && (Ty != MVT::v4i8))
1128 return SDValue();
1129
1130 return performDSPShiftCombine(Opc: MipsISD::SHRL_DSP, N, Ty, DAG, Subtarget);
1131}
1132
1133static bool isLegalDSPCondCode(EVT Ty, ISD::CondCode CC) {
1134 bool IsV216 = (Ty == MVT::v2i16);
1135
1136 switch (CC) {
1137 case ISD::SETEQ:
1138 case ISD::SETNE: return true;
1139 case ISD::SETLT:
1140 case ISD::SETLE:
1141 case ISD::SETGT:
1142 case ISD::SETGE: return IsV216;
1143 case ISD::SETULT:
1144 case ISD::SETULE:
1145 case ISD::SETUGT:
1146 case ISD::SETUGE: return !IsV216;
1147 default: return false;
1148 }
1149}
1150
1151static SDValue performSETCCCombine(SDNode *N, SelectionDAG &DAG) {
1152 EVT Ty = N->getValueType(ResNo: 0);
1153
1154 if ((Ty != MVT::v2i16) && (Ty != MVT::v4i8))
1155 return SDValue();
1156
1157 if (!isLegalDSPCondCode(Ty, CC: cast<CondCodeSDNode>(Val: N->getOperand(Num: 2))->get()))
1158 return SDValue();
1159
1160 return DAG.getNode(Opcode: MipsISD::SETCC_DSP, DL: SDLoc(N), VT: Ty, N1: N->getOperand(Num: 0),
1161 N2: N->getOperand(Num: 1), N3: N->getOperand(Num: 2));
1162}
1163
1164static SDValue performVSELECTCombine(SDNode *N, SelectionDAG &DAG) {
1165 EVT Ty = N->getValueType(ResNo: 0);
1166
1167 if (Ty == MVT::v2i16 || Ty == MVT::v4i8) {
1168 SDValue SetCC = N->getOperand(Num: 0);
1169
1170 if (SetCC.getOpcode() != MipsISD::SETCC_DSP)
1171 return SDValue();
1172
1173 return DAG.getNode(Opcode: MipsISD::SELECT_CC_DSP, DL: SDLoc(N), VT: Ty,
1174 N1: SetCC.getOperand(i: 0), N2: SetCC.getOperand(i: 1),
1175 N3: N->getOperand(Num: 1), N4: N->getOperand(Num: 2), N5: SetCC.getOperand(i: 2));
1176 }
1177
1178 return SDValue();
1179}
1180
1181static SDValue performXORCombine(SDNode *N, SelectionDAG &DAG,
1182 const MipsSubtarget &Subtarget) {
1183 EVT Ty = N->getValueType(ResNo: 0);
1184
1185 if (Subtarget.hasMSA() && Ty.is128BitVector() && Ty.isInteger()) {
1186 // Try the following combines:
1187 // (xor (or $a, $b), (build_vector allones))
1188 // (xor (or $a, $b), (bitcast (build_vector allones)))
1189 SDValue Op0 = N->getOperand(Num: 0);
1190 SDValue Op1 = N->getOperand(Num: 1);
1191 SDValue NotOp;
1192
1193 if (ISD::isBuildVectorAllOnes(N: Op0.getNode()))
1194 NotOp = Op1;
1195 else if (ISD::isBuildVectorAllOnes(N: Op1.getNode()))
1196 NotOp = Op0;
1197 else
1198 return SDValue();
1199
1200 if (NotOp->getOpcode() == ISD::OR)
1201 return DAG.getNode(Opcode: MipsISD::VNOR, DL: SDLoc(N), VT: Ty, N1: NotOp->getOperand(Num: 0),
1202 N2: NotOp->getOperand(Num: 1));
1203 }
1204
1205 return SDValue();
1206}
1207
1208// Convert (fp_to_uint (fp16_to_fp x)) into (fp_to_sint (fp16_to_fp x)).
1209static SDValue performFP_TO_UINTCombine(SDNode *N, SelectionDAG &DAG) {
1210 SDValue Src = N->getOperand(Num: 0);
1211 EVT VT = N->getValueType(ResNo: 0);
1212
1213 // Use a trick from TargetLowering::expandFP_TO_UINT: we know that every
1214 // integer value that can be represented by f16 is <= 65504, i.e. a signed
1215 // integer of 17 bits or more can represent all values and fptoui and fptosi
1216 // are equivalent.
1217 //
1218 // NOTE: the result of fptoui is poison when the value does not fit in the
1219 // destination type (e.g. because it is negative).
1220 if (Src.getOpcode() != ISD::FP16_TO_FP || VT.getScalarSizeInBits() < 17)
1221 return SDValue();
1222 return DAG.getNode(Opcode: ISD::FP_TO_SINT, DL: SDLoc(N), VT, Operand: Src);
1223}
1224
1225SDValue
1226MipsSETargetLowering::PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const {
1227 SelectionDAG &DAG = DCI.DAG;
1228 SDValue Val;
1229
1230 switch (N->getOpcode()) {
1231 case ISD::AND:
1232 Val = performANDCombine(N, DAG, DCI, Subtarget);
1233 break;
1234 case ISD::OR:
1235 Val = performORCombine(N, DAG, DCI, Subtarget);
1236 break;
1237 case ISD::MUL:
1238 return performMULCombine(N, DAG, DCI, TL: this, Subtarget);
1239 case ISD::SHL:
1240 Val = performSHLCombine(N, DAG, DCI, Subtarget);
1241 break;
1242 case ISD::SRA:
1243 return performSRACombine(N, DAG, DCI, Subtarget);
1244 case ISD::SRL:
1245 return performSRLCombine(N, DAG, DCI, Subtarget);
1246 case ISD::VSELECT:
1247 return performVSELECTCombine(N, DAG);
1248 case ISD::XOR:
1249 Val = performXORCombine(N, DAG, Subtarget);
1250 break;
1251 case ISD::SETCC:
1252 Val = performSETCCCombine(N, DAG);
1253 break;
1254 case ISD::FP_TO_UINT:
1255 Val = performFP_TO_UINTCombine(N, DAG);
1256 break;
1257 }
1258
1259 if (Val.getNode()) {
1260 LLVM_DEBUG(dbgs() << "\nMipsSE DAG Combine:\n";
1261 N->printrWithDepth(dbgs(), &DAG); dbgs() << "\n=> \n";
1262 Val.getNode()->printrWithDepth(dbgs(), &DAG); dbgs() << "\n");
1263 return Val;
1264 }
1265
1266 return MipsTargetLowering::PerformDAGCombine(N, DCI);
1267}
1268
1269MachineBasicBlock *
1270MipsSETargetLowering::EmitInstrWithCustomInserter(MachineInstr &MI,
1271 MachineBasicBlock *BB) const {
1272 switch (MI.getOpcode()) {
1273 default:
1274 return MipsTargetLowering::EmitInstrWithCustomInserter(MI, MBB: BB);
1275 case Mips::BPOSGE32_PSEUDO:
1276 return emitBPOSGE32(MI, BB);
1277 case Mips::SNZ_B_PSEUDO:
1278 return emitMSACBranchPseudo(MI, BB, BranchOp: Mips::BNZ_B);
1279 case Mips::SNZ_H_PSEUDO:
1280 return emitMSACBranchPseudo(MI, BB, BranchOp: Mips::BNZ_H);
1281 case Mips::SNZ_W_PSEUDO:
1282 return emitMSACBranchPseudo(MI, BB, BranchOp: Mips::BNZ_W);
1283 case Mips::SNZ_D_PSEUDO:
1284 return emitMSACBranchPseudo(MI, BB, BranchOp: Mips::BNZ_D);
1285 case Mips::SNZ_V_PSEUDO:
1286 return emitMSACBranchPseudo(MI, BB, BranchOp: Mips::BNZ_V);
1287 case Mips::SZ_B_PSEUDO:
1288 return emitMSACBranchPseudo(MI, BB, BranchOp: Mips::BZ_B);
1289 case Mips::SZ_H_PSEUDO:
1290 return emitMSACBranchPseudo(MI, BB, BranchOp: Mips::BZ_H);
1291 case Mips::SZ_W_PSEUDO:
1292 return emitMSACBranchPseudo(MI, BB, BranchOp: Mips::BZ_W);
1293 case Mips::SZ_D_PSEUDO:
1294 return emitMSACBranchPseudo(MI, BB, BranchOp: Mips::BZ_D);
1295 case Mips::SZ_V_PSEUDO:
1296 return emitMSACBranchPseudo(MI, BB, BranchOp: Mips::BZ_V);
1297 case Mips::COPY_FW_PSEUDO:
1298 return emitCOPY_FW(MI, BB);
1299 case Mips::COPY_FD_PSEUDO:
1300 return emitCOPY_FD(MI, BB);
1301 case Mips::INSERT_FW_PSEUDO:
1302 return emitINSERT_FW(MI, BB);
1303 case Mips::INSERT_FD_PSEUDO:
1304 return emitINSERT_FD(MI, BB);
1305 case Mips::INSERT_B_VIDX_PSEUDO:
1306 case Mips::INSERT_B_VIDX64_PSEUDO:
1307 return emitINSERT_DF_VIDX(MI, BB, EltSizeInBytes: 1, IsFP: false);
1308 case Mips::INSERT_H_VIDX_PSEUDO:
1309 case Mips::INSERT_H_VIDX64_PSEUDO:
1310 return emitINSERT_DF_VIDX(MI, BB, EltSizeInBytes: 2, IsFP: false);
1311 case Mips::INSERT_W_VIDX_PSEUDO:
1312 case Mips::INSERT_W_VIDX64_PSEUDO:
1313 return emitINSERT_DF_VIDX(MI, BB, EltSizeInBytes: 4, IsFP: false);
1314 case Mips::INSERT_D_VIDX_PSEUDO:
1315 case Mips::INSERT_D_VIDX64_PSEUDO:
1316 return emitINSERT_DF_VIDX(MI, BB, EltSizeInBytes: 8, IsFP: false);
1317 case Mips::INSERT_FW_VIDX_PSEUDO:
1318 case Mips::INSERT_FW_VIDX64_PSEUDO:
1319 return emitINSERT_DF_VIDX(MI, BB, EltSizeInBytes: 4, IsFP: true);
1320 case Mips::INSERT_FD_VIDX_PSEUDO:
1321 case Mips::INSERT_FD_VIDX64_PSEUDO:
1322 return emitINSERT_DF_VIDX(MI, BB, EltSizeInBytes: 8, IsFP: true);
1323 case Mips::FILL_FW_PSEUDO:
1324 return emitFILL_FW(MI, BB);
1325 case Mips::FILL_FD_PSEUDO:
1326 return emitFILL_FD(MI, BB);
1327 case Mips::FEXP2_W_1_PSEUDO:
1328 return emitFEXP2_W_1(MI, BB);
1329 case Mips::FEXP2_D_1_PSEUDO:
1330 return emitFEXP2_D_1(MI, BB);
1331 }
1332}
1333
1334bool MipsSETargetLowering::isEligibleForTailCallOptimization(
1335 const CCState &CCInfo, unsigned NextStackOffset,
1336 const MipsFunctionInfo &FI) const {
1337 // Exception has to be cleared with eret.
1338 if (FI.isISR())
1339 return false;
1340
1341 // Return false if either the callee or caller has a byval argument.
1342 if (CCInfo.getInRegsParamsCount() > 0 || FI.hasByvalArg())
1343 return false;
1344
1345 // Return true if the callee's argument area is no larger than the caller's.
1346 return NextStackOffset <= FI.getIncomingArgSize();
1347}
1348
1349void MipsSETargetLowering::getOpndList(
1350 SmallVectorImpl<SDValue> &Ops,
1351 std::deque<std::pair<unsigned, SDValue>> &RegsToPass, bool IsPICCall,
1352 bool GlobalOrExternal, bool LocalLinkage, bool IsCallReloc,
1353 CallLoweringInfo &CLI, SDValue Callee, SDValue Chain) const {
1354 Ops.push_back(Elt: Callee);
1355 MipsTargetLowering::getOpndList(Ops, RegsToPass, IsPICCall, GlobalOrExternal,
1356 LocalLinkage, IsCallReloc, CLI, Callee,
1357 Chain);
1358}
1359
1360SDValue MipsSETargetLowering::lowerLOAD(SDValue Op, SelectionDAG &DAG) const {
1361 LoadSDNode &Nd = *cast<LoadSDNode>(Val&: Op);
1362
1363 if (Nd.getMemoryVT() != MVT::f64 || (!NoDPLoadStore && Subtarget.hasMips2()))
1364 return MipsTargetLowering::lowerLOAD(Op, DAG);
1365
1366 // Replace a double precision load with two i32 loads and a buildpair64.
1367 SDLoc DL(Op);
1368 SDValue Ptr = Nd.getBasePtr(), Chain = Nd.getChain();
1369 EVT PtrVT = Ptr.getValueType();
1370 EVT VT = Subtarget.hasMips2() ? MVT::i32 : MVT::f32;
1371
1372 // i32 load from lower address.
1373 SDValue Lo = DAG.getLoad(VT, dl: DL, Chain, Ptr, PtrInfo: MachinePointerInfo(),
1374 Alignment: Nd.getAlign(), MMOFlags: Nd.getMemOperand()->getFlags());
1375
1376 // i32 load from higher address.
1377 Ptr = DAG.getNode(Opcode: ISD::ADD, DL, VT: PtrVT, N1: Ptr, N2: DAG.getConstant(Val: 4, DL, VT: PtrVT));
1378 SDValue Hi = DAG.getLoad(VT, dl: DL, Chain: Lo.getValue(R: 1), Ptr, PtrInfo: MachinePointerInfo(),
1379 Alignment: commonAlignment(A: Nd.getAlign(), Offset: 4),
1380 MMOFlags: Nd.getMemOperand()->getFlags());
1381
1382 if (!Subtarget.isLittle())
1383 std::swap(a&: Lo, b&: Hi);
1384
1385 SDValue BP;
1386 if (Subtarget.hasMips2())
1387 BP = DAG.getNode(Opcode: MipsISD::BuildPairF64, DL, VT: MVT::f64, N1: Lo, N2: Hi);
1388 else
1389 BP = DAG.getNode(Opcode: MipsISD::BuildPairF64_FPR, DL, VT: MVT::f64, N1: Hi, N2: Lo);
1390
1391 SDValue Ops[2] = {BP, Hi.getValue(R: 1)};
1392 return DAG.getMergeValues(Ops, dl: DL);
1393}
1394
1395SDValue MipsSETargetLowering::lowerSTORE(SDValue Op, SelectionDAG &DAG) const {
1396 StoreSDNode &Nd = *cast<StoreSDNode>(Val&: Op);
1397
1398 if (Nd.getMemoryVT() != MVT::f64 || (!NoDPLoadStore && Subtarget.hasMips2()))
1399 return MipsTargetLowering::lowerSTORE(Op, DAG);
1400
1401 // Replace a double precision store with two extractelement64s and i32 stores.
1402 SDLoc DL(Op);
1403 SDValue Val = Nd.getValue(), Ptr = Nd.getBasePtr(), Chain = Nd.getChain();
1404 EVT PtrVT = Ptr.getValueType();
1405 EVT VT = Subtarget.hasMips2() ? MVT::i32 : MVT::f32;
1406
1407 unsigned ExtractOp = Subtarget.hasMips2() ? MipsISD::ExtractElementF64
1408 : MipsISD::ExtractElementF64_FPR;
1409 SDValue Lo =
1410 DAG.getNode(Opcode: ExtractOp, DL, VT, N1: Val, N2: DAG.getConstant(Val: 0, DL, VT: MVT::i32));
1411 SDValue Hi =
1412 DAG.getNode(Opcode: ExtractOp, DL, VT, N1: Val, N2: DAG.getConstant(Val: 1, DL, VT: MVT::i32));
1413
1414 if (!Subtarget.isLittle())
1415 std::swap(a&: Lo, b&: Hi);
1416
1417 // i32 store to lower address.
1418 Chain = DAG.getStore(Chain, dl: DL, Val: Lo, Ptr, PtrInfo: MachinePointerInfo(), Alignment: Nd.getAlign(),
1419 MMOFlags: Nd.getMemOperand()->getFlags(), Metadata: Nd.getAAInfo());
1420
1421 // i32 store to higher address.
1422 Ptr = DAG.getNode(Opcode: ISD::ADD, DL, VT: PtrVT, N1: Ptr, N2: DAG.getConstant(Val: 4, DL, VT: PtrVT));
1423 return DAG.getStore(Chain, dl: DL, Val: Hi, Ptr, PtrInfo: MachinePointerInfo(),
1424 Alignment: commonAlignment(A: Nd.getAlign(), Offset: 4),
1425 MMOFlags: Nd.getMemOperand()->getFlags(), Metadata: Nd.getAAInfo());
1426}
1427
1428SDValue MipsSETargetLowering::lowerBITCAST(SDValue Op,
1429 SelectionDAG &DAG) const {
1430 SDLoc DL(Op);
1431 MVT Src = Op.getOperand(i: 0).getValueType().getSimpleVT();
1432 MVT Dest = Op.getValueType().getSimpleVT();
1433
1434 // Bitcast i64 to double.
1435 if (Src == MVT::i64 && Dest == MVT::f64) {
1436 SDValue Lo, Hi;
1437 std::tie(args&: Lo, args&: Hi) =
1438 DAG.SplitScalar(N: Op.getOperand(i: 0), DL, LoVT: MVT::i32, HiVT: MVT::i32);
1439 return DAG.getNode(Opcode: MipsISD::BuildPairF64, DL, VT: MVT::f64, N1: Lo, N2: Hi);
1440 }
1441
1442 // Bitcast double to i64.
1443 if (Src == MVT::f64 && Dest == MVT::i64) {
1444 // Skip lower bitcast when operand0 has converted float results to integer
1445 // which was done by function SoftenFloatResult.
1446 if (getTypeAction(Context&: *DAG.getContext(), VT: Op.getOperand(i: 0).getValueType()) ==
1447 TargetLowering::TypeSoftenFloat)
1448 return SDValue();
1449 SDValue Lo =
1450 DAG.getNode(Opcode: MipsISD::ExtractElementF64, DL, VT: MVT::i32, N1: Op.getOperand(i: 0),
1451 N2: DAG.getConstant(Val: 0, DL, VT: MVT::i32));
1452 SDValue Hi =
1453 DAG.getNode(Opcode: MipsISD::ExtractElementF64, DL, VT: MVT::i32, N1: Op.getOperand(i: 0),
1454 N2: DAG.getConstant(Val: 1, DL, VT: MVT::i32));
1455 return DAG.getNode(Opcode: ISD::BUILD_PAIR, DL, VT: MVT::i64, N1: Lo, N2: Hi);
1456 }
1457
1458 // Skip other cases of bitcast and use default lowering.
1459 return SDValue();
1460}
1461
1462SDValue MipsSETargetLowering::lowerMulDiv(SDValue Op, unsigned NewOpc,
1463 bool HasLo, bool HasHi,
1464 SelectionDAG &DAG) const {
1465 // MIPS32r6/MIPS64r6 removed accumulator based multiplies.
1466 assert(!Subtarget.hasMips32r6());
1467
1468 EVT Ty = Op.getOperand(i: 0).getValueType();
1469 SDLoc DL(Op);
1470 SDValue Mult = DAG.getNode(Opcode: NewOpc, DL, VT: MVT::Untyped,
1471 N1: Op.getOperand(i: 0), N2: Op.getOperand(i: 1));
1472 SDValue Lo, Hi;
1473
1474 if (HasLo)
1475 Lo = DAG.getNode(Opcode: MipsISD::MFLO, DL, VT: Ty, Operand: Mult);
1476 if (HasHi)
1477 Hi = DAG.getNode(Opcode: MipsISD::MFHI, DL, VT: Ty, Operand: Mult);
1478
1479 if (!HasLo || !HasHi)
1480 return HasLo ? Lo : Hi;
1481
1482 SDValue Vals[] = { Lo, Hi };
1483 return DAG.getMergeValues(Ops: Vals, dl: DL);
1484}
1485
1486static SDValue initAccumulator(SDValue In, const SDLoc &DL, SelectionDAG &DAG) {
1487 SDValue InLo, InHi;
1488 std::tie(args&: InLo, args&: InHi) = DAG.SplitScalar(N: In, DL, LoVT: MVT::i32, HiVT: MVT::i32);
1489 return DAG.getNode(Opcode: MipsISD::MTLOHI, DL, VT: MVT::Untyped, N1: InLo, N2: InHi);
1490}
1491
1492static SDValue extractLOHI(SDValue Op, const SDLoc &DL, SelectionDAG &DAG) {
1493 SDValue Lo = DAG.getNode(Opcode: MipsISD::MFLO, DL, VT: MVT::i32, Operand: Op);
1494 SDValue Hi = DAG.getNode(Opcode: MipsISD::MFHI, DL, VT: MVT::i32, Operand: Op);
1495 return DAG.getNode(Opcode: ISD::BUILD_PAIR, DL, VT: MVT::i64, N1: Lo, N2: Hi);
1496}
1497
1498// This function expands mips intrinsic nodes which have 64-bit input operands
1499// or output values.
1500//
1501// out64 = intrinsic-node in64
1502// =>
1503// lo = copy (extract-element (in64, 0))
1504// hi = copy (extract-element (in64, 1))
1505// mips-specific-node
1506// v0 = copy lo
1507// v1 = copy hi
1508// out64 = merge-values (v0, v1)
1509//
1510static SDValue lowerDSPIntr(SDValue Op, SelectionDAG &DAG, unsigned Opc) {
1511 SDLoc DL(Op);
1512 bool HasChainIn = Op->getOperand(Num: 0).getValueType() == MVT::Other;
1513 SmallVector<SDValue, 3> Ops;
1514 unsigned OpNo = 0;
1515
1516 // See if Op has a chain input.
1517 if (HasChainIn)
1518 Ops.push_back(Elt: Op->getOperand(Num: OpNo++));
1519
1520 // The next operand is the intrinsic opcode.
1521 assert(Op->getOperand(OpNo).getOpcode() == ISD::TargetConstant);
1522
1523 // See if the next operand has type i64.
1524 SDValue Opnd = Op->getOperand(Num: ++OpNo), In64;
1525
1526 if (Opnd.getValueType() == MVT::i64)
1527 In64 = initAccumulator(In: Opnd, DL, DAG);
1528 else
1529 Ops.push_back(Elt: Opnd);
1530
1531 // Push the remaining operands.
1532 for (++OpNo ; OpNo < Op->getNumOperands(); ++OpNo)
1533 Ops.push_back(Elt: Op->getOperand(Num: OpNo));
1534
1535 // Add In64 to the end of the list.
1536 if (In64.getNode())
1537 Ops.push_back(Elt: In64);
1538
1539 // Scan output.
1540 SmallVector<EVT, 2> ResTys;
1541
1542 for (EVT Ty : Op->values())
1543 ResTys.push_back(Elt: (Ty == MVT::i64) ? MVT::Untyped : Ty);
1544
1545 // Create node.
1546 SDValue Val = DAG.getNode(Opcode: Opc, DL, ResultTys: ResTys, Ops);
1547 SDValue Out = (ResTys[0] == MVT::Untyped) ? extractLOHI(Op: Val, DL, DAG) : Val;
1548
1549 if (!HasChainIn)
1550 return Out;
1551
1552 assert(Val->getValueType(1) == MVT::Other);
1553 SDValue Vals[] = { Out, SDValue(Val.getNode(), 1) };
1554 return DAG.getMergeValues(Ops: Vals, dl: DL);
1555}
1556
1557// Lower an MSA copy intrinsic into the specified SelectionDAG node
1558static SDValue lowerMSACopyIntr(SDValue Op, SelectionDAG &DAG, unsigned Opc) {
1559 SDLoc DL(Op);
1560 SDValue Vec = Op->getOperand(Num: 1);
1561 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
1562 SDValue Idx = DAG.getZExtOrTrunc(Op: Op->getOperand(Num: 2), DL,
1563 VT: TLI.getVectorIdxTy(DL: DAG.getDataLayout()));
1564 EVT ResTy = Op->getValueType(ResNo: 0);
1565 EVT EltTy = Vec->getValueType(ResNo: 0).getVectorElementType();
1566
1567 SDValue Result = DAG.getNode(Opcode: Opc, DL, VT: ResTy, N1: Vec, N2: Idx,
1568 N3: DAG.getValueType(EltTy));
1569
1570 return Result;
1571}
1572
1573static SDValue lowerMSASplatZExt(SDValue Op, unsigned OpNr, SelectionDAG &DAG) {
1574 EVT ResVecTy = Op->getValueType(ResNo: 0);
1575 EVT ViaVecTy = ResVecTy;
1576 bool BigEndian = !DAG.getSubtarget().getTargetTriple().isLittleEndian();
1577 SDLoc DL(Op);
1578
1579 // When ResVecTy == MVT::v2i64, LaneA is the upper 32 bits of the lane and
1580 // LaneB is the lower 32-bits. Otherwise LaneA and LaneB are alternating
1581 // lanes.
1582 SDValue LaneA = Op->getOperand(Num: OpNr);
1583 SDValue LaneB;
1584
1585 if (ResVecTy == MVT::v2i64) {
1586 // In case of the index being passed as an immediate value, set the upper
1587 // lane to 0 so that the splati.d instruction can be matched.
1588 if (isa<ConstantSDNode>(Val: LaneA))
1589 LaneB = DAG.getConstant(Val: 0, DL, VT: MVT::i32);
1590 // Having the index passed in a register, set the upper lane to the same
1591 // value as the lower - this results in the BUILD_VECTOR node not being
1592 // expanded through stack. This way we are able to pattern match the set of
1593 // nodes created here to splat.d.
1594 else
1595 LaneB = LaneA;
1596 ViaVecTy = MVT::v4i32;
1597 if(BigEndian)
1598 std::swap(a&: LaneA, b&: LaneB);
1599 } else
1600 LaneB = LaneA;
1601
1602 SDValue Ops[16] = { LaneA, LaneB, LaneA, LaneB, LaneA, LaneB, LaneA, LaneB,
1603 LaneA, LaneB, LaneA, LaneB, LaneA, LaneB, LaneA, LaneB };
1604
1605 SDValue Result = DAG.getBuildVector(
1606 VT: ViaVecTy, DL, Ops: ArrayRef(Ops, ViaVecTy.getVectorNumElements()));
1607
1608 if (ViaVecTy != ResVecTy) {
1609 SDValue One = DAG.getConstant(Val: 1, DL, VT: ViaVecTy);
1610 Result = DAG.getNode(Opcode: ISD::BITCAST, DL, VT: ResVecTy,
1611 Operand: DAG.getNode(Opcode: ISD::AND, DL, VT: ViaVecTy, N1: Result, N2: One));
1612 }
1613
1614 return Result;
1615}
1616
1617static SDValue lowerMSASplatImm(SDValue Op, unsigned ImmOp, SelectionDAG &DAG,
1618 bool IsSigned = false) {
1619 auto *CImm = cast<ConstantSDNode>(Val: Op->getOperand(Num: ImmOp));
1620 return DAG.getConstant(
1621 Val: APInt(Op->getValueType(ResNo: 0).getScalarType().getSizeInBits(),
1622 IsSigned ? CImm->getSExtValue() : CImm->getZExtValue(), IsSigned),
1623 DL: SDLoc(Op), VT: Op->getValueType(ResNo: 0));
1624}
1625
1626static SDValue getBuildVectorSplat(EVT VecTy, SDValue SplatValue,
1627 bool BigEndian, SelectionDAG &DAG) {
1628 EVT ViaVecTy = VecTy;
1629 SDValue SplatValueA = SplatValue;
1630 SDValue SplatValueB = SplatValue;
1631 SDLoc DL(SplatValue);
1632
1633 if (VecTy == MVT::v2i64) {
1634 // v2i64 BUILD_VECTOR must be performed via v4i32 so split into i32's.
1635 ViaVecTy = MVT::v4i32;
1636
1637 SplatValueA = DAG.getNode(Opcode: ISD::TRUNCATE, DL, VT: MVT::i32, Operand: SplatValue);
1638 SplatValueB = DAG.getNode(Opcode: ISD::SRL, DL, VT: MVT::i64, N1: SplatValue,
1639 N2: DAG.getConstant(Val: 32, DL, VT: MVT::i32));
1640 SplatValueB = DAG.getNode(Opcode: ISD::TRUNCATE, DL, VT: MVT::i32, Operand: SplatValueB);
1641 }
1642
1643 // We currently hold the parts in little endian order. Swap them if
1644 // necessary.
1645 if (BigEndian)
1646 std::swap(a&: SplatValueA, b&: SplatValueB);
1647
1648 SDValue Ops[16] = { SplatValueA, SplatValueB, SplatValueA, SplatValueB,
1649 SplatValueA, SplatValueB, SplatValueA, SplatValueB,
1650 SplatValueA, SplatValueB, SplatValueA, SplatValueB,
1651 SplatValueA, SplatValueB, SplatValueA, SplatValueB };
1652
1653 SDValue Result = DAG.getBuildVector(
1654 VT: ViaVecTy, DL, Ops: ArrayRef(Ops, ViaVecTy.getVectorNumElements()));
1655
1656 if (VecTy != ViaVecTy)
1657 Result = DAG.getNode(Opcode: ISD::BITCAST, DL, VT: VecTy, Operand: Result);
1658
1659 return Result;
1660}
1661
1662static SDValue lowerMSABinaryBitImmIntr(SDValue Op, SelectionDAG &DAG,
1663 unsigned Opc, SDValue Imm,
1664 bool BigEndian) {
1665 EVT VecTy = Op->getValueType(ResNo: 0);
1666 SDValue Exp2Imm;
1667 SDLoc DL(Op);
1668
1669 // The DAG Combiner can't constant fold bitcasted vectors yet so we must do it
1670 // here for now.
1671 if (VecTy == MVT::v2i64) {
1672 if (ConstantSDNode *CImm = dyn_cast<ConstantSDNode>(Val&: Imm)) {
1673 APInt BitImm = APInt(64, 1) << CImm->getAPIntValue();
1674
1675 SDValue BitImmHiOp = DAG.getConstant(Val: BitImm.lshr(shiftAmt: 32).trunc(width: 32), DL,
1676 VT: MVT::i32);
1677 SDValue BitImmLoOp = DAG.getConstant(Val: BitImm.trunc(width: 32), DL, VT: MVT::i32);
1678
1679 if (BigEndian)
1680 std::swap(a&: BitImmLoOp, b&: BitImmHiOp);
1681
1682 Exp2Imm = DAG.getNode(
1683 Opcode: ISD::BITCAST, DL, VT: MVT::v2i64,
1684 Operand: DAG.getBuildVector(VT: MVT::v4i32, DL,
1685 Ops: {BitImmLoOp, BitImmHiOp, BitImmLoOp, BitImmHiOp}));
1686 }
1687 }
1688
1689 if (!Exp2Imm.getNode()) {
1690 // We couldnt constant fold, do a vector shift instead
1691
1692 // Extend i32 to i64 if necessary. Sign or zero extend doesn't matter since
1693 // only values 0-63 are valid.
1694 if (VecTy == MVT::v2i64)
1695 Imm = DAG.getNode(Opcode: ISD::ZERO_EXTEND, DL, VT: MVT::i64, Operand: Imm);
1696
1697 Exp2Imm = getBuildVectorSplat(VecTy, SplatValue: Imm, BigEndian, DAG);
1698
1699 Exp2Imm = DAG.getNode(Opcode: ISD::SHL, DL, VT: VecTy, N1: DAG.getConstant(Val: 1, DL, VT: VecTy),
1700 N2: Exp2Imm);
1701 }
1702
1703 return DAG.getNode(Opcode: Opc, DL, VT: VecTy, N1: Op->getOperand(Num: 1), N2: Exp2Imm);
1704}
1705
1706static SDValue truncateVecElts(SDValue Op, SelectionDAG &DAG) {
1707 SDLoc DL(Op);
1708 EVT ResTy = Op->getValueType(ResNo: 0);
1709 SDValue Vec = Op->getOperand(Num: 2);
1710 bool BigEndian = !DAG.getSubtarget().getTargetTriple().isLittleEndian();
1711 MVT ResEltTy = ResTy == MVT::v2i64 ? MVT::i64 : MVT::i32;
1712 SDValue ConstValue = DAG.getConstant(Val: Vec.getScalarValueSizeInBits() - 1,
1713 DL, VT: ResEltTy);
1714 SDValue SplatVec = getBuildVectorSplat(VecTy: ResTy, SplatValue: ConstValue, BigEndian, DAG);
1715
1716 return DAG.getNode(Opcode: ISD::AND, DL, VT: ResTy, N1: Vec, N2: SplatVec);
1717}
1718
1719static SDValue lowerMSABitClear(SDValue Op, SelectionDAG &DAG) {
1720 EVT ResTy = Op->getValueType(ResNo: 0);
1721 SDLoc DL(Op);
1722 SDValue One = DAG.getConstant(Val: 1, DL, VT: ResTy);
1723 SDValue Bit = DAG.getNode(Opcode: ISD::SHL, DL, VT: ResTy, N1: One, N2: truncateVecElts(Op, DAG));
1724
1725 return DAG.getNode(Opcode: ISD::AND, DL, VT: ResTy, N1: Op->getOperand(Num: 1),
1726 N2: DAG.getNOT(DL, Val: Bit, VT: ResTy));
1727}
1728
1729static SDValue lowerMSABitClearImm(SDValue Op, SelectionDAG &DAG) {
1730 SDLoc DL(Op);
1731 EVT ResTy = Op->getValueType(ResNo: 0);
1732 APInt BitImm = APInt(ResTy.getScalarSizeInBits(), 1)
1733 << Op->getConstantOperandAPInt(Num: 2);
1734 SDValue BitMask = DAG.getConstant(Val: ~BitImm, DL, VT: ResTy);
1735
1736 return DAG.getNode(Opcode: ISD::AND, DL, VT: ResTy, N1: Op->getOperand(Num: 1), N2: BitMask);
1737}
1738
1739SDValue MipsSETargetLowering::lowerINTRINSIC_WO_CHAIN(SDValue Op,
1740 SelectionDAG &DAG) const {
1741 SDLoc DL(Op);
1742 unsigned Intrinsic = Op->getConstantOperandVal(Num: 0);
1743 switch (Intrinsic) {
1744 default:
1745 return SDValue();
1746 case Intrinsic::mips_shilo:
1747 return lowerDSPIntr(Op, DAG, Opc: MipsISD::SHILO);
1748 case Intrinsic::mips_dpau_h_qbl:
1749 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPAU_H_QBL);
1750 case Intrinsic::mips_dpau_h_qbr:
1751 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPAU_H_QBR);
1752 case Intrinsic::mips_dpsu_h_qbl:
1753 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPSU_H_QBL);
1754 case Intrinsic::mips_dpsu_h_qbr:
1755 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPSU_H_QBR);
1756 case Intrinsic::mips_dpa_w_ph:
1757 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPA_W_PH);
1758 case Intrinsic::mips_dps_w_ph:
1759 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPS_W_PH);
1760 case Intrinsic::mips_dpax_w_ph:
1761 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPAX_W_PH);
1762 case Intrinsic::mips_dpsx_w_ph:
1763 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPSX_W_PH);
1764 case Intrinsic::mips_mulsa_w_ph:
1765 return lowerDSPIntr(Op, DAG, Opc: MipsISD::MULSA_W_PH);
1766 case Intrinsic::mips_mult:
1767 return lowerDSPIntr(Op, DAG, Opc: MipsISD::Mult);
1768 case Intrinsic::mips_multu:
1769 return lowerDSPIntr(Op, DAG, Opc: MipsISD::Multu);
1770 case Intrinsic::mips_madd:
1771 return lowerDSPIntr(Op, DAG, Opc: MipsISD::MAdd);
1772 case Intrinsic::mips_maddu:
1773 return lowerDSPIntr(Op, DAG, Opc: MipsISD::MAddu);
1774 case Intrinsic::mips_msub:
1775 return lowerDSPIntr(Op, DAG, Opc: MipsISD::MSub);
1776 case Intrinsic::mips_msubu:
1777 return lowerDSPIntr(Op, DAG, Opc: MipsISD::MSubu);
1778 case Intrinsic::mips_addv_b:
1779 case Intrinsic::mips_addv_h:
1780 case Intrinsic::mips_addv_w:
1781 case Intrinsic::mips_addv_d:
1782 return DAG.getNode(Opcode: ISD::ADD, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
1783 N2: Op->getOperand(Num: 2));
1784 case Intrinsic::mips_addvi_b:
1785 case Intrinsic::mips_addvi_h:
1786 case Intrinsic::mips_addvi_w:
1787 case Intrinsic::mips_addvi_d:
1788 return DAG.getNode(Opcode: ISD::ADD, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
1789 N2: lowerMSASplatImm(Op, ImmOp: 2, DAG));
1790 case Intrinsic::mips_and_v:
1791 return DAG.getNode(Opcode: ISD::AND, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
1792 N2: Op->getOperand(Num: 2));
1793 case Intrinsic::mips_andi_b:
1794 return DAG.getNode(Opcode: ISD::AND, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
1795 N2: lowerMSASplatImm(Op, ImmOp: 2, DAG));
1796 case Intrinsic::mips_bclr_b:
1797 case Intrinsic::mips_bclr_h:
1798 case Intrinsic::mips_bclr_w:
1799 case Intrinsic::mips_bclr_d:
1800 return lowerMSABitClear(Op, DAG);
1801 case Intrinsic::mips_bclri_b:
1802 case Intrinsic::mips_bclri_h:
1803 case Intrinsic::mips_bclri_w:
1804 case Intrinsic::mips_bclri_d:
1805 return lowerMSABitClearImm(Op, DAG);
1806 case Intrinsic::mips_binsli_b:
1807 case Intrinsic::mips_binsli_h:
1808 case Intrinsic::mips_binsli_w:
1809 case Intrinsic::mips_binsli_d: {
1810 // binsli_x(IfClear, IfSet, nbits) -> (vselect LBitsMask, IfSet, IfClear)
1811 EVT VecTy = Op->getValueType(ResNo: 0);
1812 EVT EltTy = VecTy.getVectorElementType();
1813 if (Op->getConstantOperandVal(Num: 3) >= EltTy.getSizeInBits())
1814 report_fatal_error(reason: "Immediate out of range");
1815 APInt Mask = APInt::getHighBitsSet(numBits: EltTy.getSizeInBits(),
1816 hiBitsSet: Op->getConstantOperandVal(Num: 3) + 1);
1817 return DAG.getNode(Opcode: ISD::VSELECT, DL, VT: VecTy,
1818 N1: DAG.getConstant(Val: Mask, DL, VT: VecTy, isTarget: true),
1819 N2: Op->getOperand(Num: 2), N3: Op->getOperand(Num: 1));
1820 }
1821 case Intrinsic::mips_binsri_b:
1822 case Intrinsic::mips_binsri_h:
1823 case Intrinsic::mips_binsri_w:
1824 case Intrinsic::mips_binsri_d: {
1825 // binsri_x(IfClear, IfSet, nbits) -> (vselect RBitsMask, IfSet, IfClear)
1826 EVT VecTy = Op->getValueType(ResNo: 0);
1827 EVT EltTy = VecTy.getVectorElementType();
1828 if (Op->getConstantOperandVal(Num: 3) >= EltTy.getSizeInBits())
1829 report_fatal_error(reason: "Immediate out of range");
1830 APInt Mask = APInt::getLowBitsSet(numBits: EltTy.getSizeInBits(),
1831 loBitsSet: Op->getConstantOperandVal(Num: 3) + 1);
1832 return DAG.getNode(Opcode: ISD::VSELECT, DL, VT: VecTy,
1833 N1: DAG.getConstant(Val: Mask, DL, VT: VecTy, isTarget: true),
1834 N2: Op->getOperand(Num: 2), N3: Op->getOperand(Num: 1));
1835 }
1836 case Intrinsic::mips_bmnz_v:
1837 return DAG.getNode(Opcode: ISD::VSELECT, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 3),
1838 N2: Op->getOperand(Num: 2), N3: Op->getOperand(Num: 1));
1839 case Intrinsic::mips_bmnzi_b:
1840 return DAG.getNode(Opcode: ISD::VSELECT, DL, VT: Op->getValueType(ResNo: 0),
1841 N1: lowerMSASplatImm(Op, ImmOp: 3, DAG), N2: Op->getOperand(Num: 2),
1842 N3: Op->getOperand(Num: 1));
1843 case Intrinsic::mips_bmz_v:
1844 return DAG.getNode(Opcode: ISD::VSELECT, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 3),
1845 N2: Op->getOperand(Num: 1), N3: Op->getOperand(Num: 2));
1846 case Intrinsic::mips_bmzi_b:
1847 return DAG.getNode(Opcode: ISD::VSELECT, DL, VT: Op->getValueType(ResNo: 0),
1848 N1: lowerMSASplatImm(Op, ImmOp: 3, DAG), N2: Op->getOperand(Num: 1),
1849 N3: Op->getOperand(Num: 2));
1850 case Intrinsic::mips_bneg_b:
1851 case Intrinsic::mips_bneg_h:
1852 case Intrinsic::mips_bneg_w:
1853 case Intrinsic::mips_bneg_d: {
1854 EVT VecTy = Op->getValueType(ResNo: 0);
1855 SDValue One = DAG.getConstant(Val: 1, DL, VT: VecTy);
1856
1857 return DAG.getNode(Opcode: ISD::XOR, DL, VT: VecTy, N1: Op->getOperand(Num: 1),
1858 N2: DAG.getNode(Opcode: ISD::SHL, DL, VT: VecTy, N1: One,
1859 N2: truncateVecElts(Op, DAG)));
1860 }
1861 case Intrinsic::mips_bnegi_b:
1862 case Intrinsic::mips_bnegi_h:
1863 case Intrinsic::mips_bnegi_w:
1864 case Intrinsic::mips_bnegi_d:
1865 return lowerMSABinaryBitImmIntr(Op, DAG, Opc: ISD::XOR, Imm: Op->getOperand(Num: 2),
1866 BigEndian: !Subtarget.isLittle());
1867 case Intrinsic::mips_bnz_b:
1868 case Intrinsic::mips_bnz_h:
1869 case Intrinsic::mips_bnz_w:
1870 case Intrinsic::mips_bnz_d:
1871 return DAG.getNode(Opcode: MipsISD::VALL_NONZERO, DL, VT: Op->getValueType(ResNo: 0),
1872 Operand: Op->getOperand(Num: 1));
1873 case Intrinsic::mips_bnz_v:
1874 return DAG.getNode(Opcode: MipsISD::VANY_NONZERO, DL, VT: Op->getValueType(ResNo: 0),
1875 Operand: Op->getOperand(Num: 1));
1876 case Intrinsic::mips_bsel_v:
1877 // bsel_v(Mask, IfClear, IfSet) -> (vselect Mask, IfSet, IfClear)
1878 return DAG.getNode(Opcode: ISD::VSELECT, DL, VT: Op->getValueType(ResNo: 0),
1879 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 3),
1880 N3: Op->getOperand(Num: 2));
1881 case Intrinsic::mips_bseli_b:
1882 // bseli_v(Mask, IfClear, IfSet) -> (vselect Mask, IfSet, IfClear)
1883 return DAG.getNode(Opcode: ISD::VSELECT, DL, VT: Op->getValueType(ResNo: 0),
1884 N1: Op->getOperand(Num: 1), N2: lowerMSASplatImm(Op, ImmOp: 3, DAG),
1885 N3: Op->getOperand(Num: 2));
1886 case Intrinsic::mips_bset_b:
1887 case Intrinsic::mips_bset_h:
1888 case Intrinsic::mips_bset_w:
1889 case Intrinsic::mips_bset_d: {
1890 EVT VecTy = Op->getValueType(ResNo: 0);
1891 SDValue One = DAG.getConstant(Val: 1, DL, VT: VecTy);
1892
1893 return DAG.getNode(Opcode: ISD::OR, DL, VT: VecTy, N1: Op->getOperand(Num: 1),
1894 N2: DAG.getNode(Opcode: ISD::SHL, DL, VT: VecTy, N1: One,
1895 N2: truncateVecElts(Op, DAG)));
1896 }
1897 case Intrinsic::mips_bseti_b:
1898 case Intrinsic::mips_bseti_h:
1899 case Intrinsic::mips_bseti_w:
1900 case Intrinsic::mips_bseti_d:
1901 return lowerMSABinaryBitImmIntr(Op, DAG, Opc: ISD::OR, Imm: Op->getOperand(Num: 2),
1902 BigEndian: !Subtarget.isLittle());
1903 case Intrinsic::mips_bz_b:
1904 case Intrinsic::mips_bz_h:
1905 case Intrinsic::mips_bz_w:
1906 case Intrinsic::mips_bz_d:
1907 return DAG.getNode(Opcode: MipsISD::VALL_ZERO, DL, VT: Op->getValueType(ResNo: 0),
1908 Operand: Op->getOperand(Num: 1));
1909 case Intrinsic::mips_bz_v:
1910 return DAG.getNode(Opcode: MipsISD::VANY_ZERO, DL, VT: Op->getValueType(ResNo: 0),
1911 Operand: Op->getOperand(Num: 1));
1912 case Intrinsic::mips_ceq_b:
1913 case Intrinsic::mips_ceq_h:
1914 case Intrinsic::mips_ceq_w:
1915 case Intrinsic::mips_ceq_d:
1916 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
1917 RHS: Op->getOperand(Num: 2), Cond: ISD::SETEQ);
1918 case Intrinsic::mips_ceqi_b:
1919 case Intrinsic::mips_ceqi_h:
1920 case Intrinsic::mips_ceqi_w:
1921 case Intrinsic::mips_ceqi_d:
1922 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
1923 RHS: lowerMSASplatImm(Op, ImmOp: 2, DAG, IsSigned: true), Cond: ISD::SETEQ);
1924 case Intrinsic::mips_cle_s_b:
1925 case Intrinsic::mips_cle_s_h:
1926 case Intrinsic::mips_cle_s_w:
1927 case Intrinsic::mips_cle_s_d:
1928 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
1929 RHS: Op->getOperand(Num: 2), Cond: ISD::SETLE);
1930 case Intrinsic::mips_clei_s_b:
1931 case Intrinsic::mips_clei_s_h:
1932 case Intrinsic::mips_clei_s_w:
1933 case Intrinsic::mips_clei_s_d:
1934 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
1935 RHS: lowerMSASplatImm(Op, ImmOp: 2, DAG, IsSigned: true), Cond: ISD::SETLE);
1936 case Intrinsic::mips_cle_u_b:
1937 case Intrinsic::mips_cle_u_h:
1938 case Intrinsic::mips_cle_u_w:
1939 case Intrinsic::mips_cle_u_d:
1940 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
1941 RHS: Op->getOperand(Num: 2), Cond: ISD::SETULE);
1942 case Intrinsic::mips_clei_u_b:
1943 case Intrinsic::mips_clei_u_h:
1944 case Intrinsic::mips_clei_u_w:
1945 case Intrinsic::mips_clei_u_d:
1946 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
1947 RHS: lowerMSASplatImm(Op, ImmOp: 2, DAG), Cond: ISD::SETULE);
1948 case Intrinsic::mips_clt_s_b:
1949 case Intrinsic::mips_clt_s_h:
1950 case Intrinsic::mips_clt_s_w:
1951 case Intrinsic::mips_clt_s_d:
1952 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
1953 RHS: Op->getOperand(Num: 2), Cond: ISD::SETLT);
1954 case Intrinsic::mips_clti_s_b:
1955 case Intrinsic::mips_clti_s_h:
1956 case Intrinsic::mips_clti_s_w:
1957 case Intrinsic::mips_clti_s_d:
1958 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
1959 RHS: lowerMSASplatImm(Op, ImmOp: 2, DAG, IsSigned: true), Cond: ISD::SETLT);
1960 case Intrinsic::mips_clt_u_b:
1961 case Intrinsic::mips_clt_u_h:
1962 case Intrinsic::mips_clt_u_w:
1963 case Intrinsic::mips_clt_u_d:
1964 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
1965 RHS: Op->getOperand(Num: 2), Cond: ISD::SETULT);
1966 case Intrinsic::mips_clti_u_b:
1967 case Intrinsic::mips_clti_u_h:
1968 case Intrinsic::mips_clti_u_w:
1969 case Intrinsic::mips_clti_u_d:
1970 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
1971 RHS: lowerMSASplatImm(Op, ImmOp: 2, DAG), Cond: ISD::SETULT);
1972 case Intrinsic::mips_copy_s_b:
1973 case Intrinsic::mips_copy_s_h:
1974 case Intrinsic::mips_copy_s_w:
1975 return lowerMSACopyIntr(Op, DAG, Opc: MipsISD::VEXTRACT_SEXT_ELT);
1976 case Intrinsic::mips_copy_s_d:
1977 if (Subtarget.hasMips64())
1978 // Lower directly into VEXTRACT_SEXT_ELT since i64 is legal on Mips64.
1979 return lowerMSACopyIntr(Op, DAG, Opc: MipsISD::VEXTRACT_SEXT_ELT);
1980 else {
1981 // Lower into the generic EXTRACT_VECTOR_ELT node and let the type
1982 // legalizer and EXTRACT_VECTOR_ELT lowering sort it out.
1983 return DAG.getNode(Opcode: ISD::EXTRACT_VECTOR_ELT, DL: SDLoc(Op),
1984 VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
1985 N2: Op->getOperand(Num: 2));
1986 }
1987 case Intrinsic::mips_copy_u_b:
1988 case Intrinsic::mips_copy_u_h:
1989 case Intrinsic::mips_copy_u_w:
1990 return lowerMSACopyIntr(Op, DAG, Opc: MipsISD::VEXTRACT_ZEXT_ELT);
1991 case Intrinsic::mips_copy_u_d:
1992 if (Subtarget.hasMips64())
1993 // Lower directly into VEXTRACT_ZEXT_ELT since i64 is legal on Mips64.
1994 return lowerMSACopyIntr(Op, DAG, Opc: MipsISD::VEXTRACT_ZEXT_ELT);
1995 else {
1996 // Lower into the generic EXTRACT_VECTOR_ELT node and let the type
1997 // legalizer and EXTRACT_VECTOR_ELT lowering sort it out.
1998 // Note: When i64 is illegal, this results in copy_s.w instructions
1999 // instead of copy_u.w instructions. This makes no difference to the
2000 // behaviour since i64 is only illegal when the register file is 32-bit.
2001 return DAG.getNode(Opcode: ISD::EXTRACT_VECTOR_ELT, DL: SDLoc(Op),
2002 VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2003 N2: Op->getOperand(Num: 2));
2004 }
2005 case Intrinsic::mips_div_s_b:
2006 case Intrinsic::mips_div_s_h:
2007 case Intrinsic::mips_div_s_w:
2008 case Intrinsic::mips_div_s_d:
2009 return DAG.getNode(Opcode: ISD::SDIV, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2010 N2: Op->getOperand(Num: 2));
2011 case Intrinsic::mips_div_u_b:
2012 case Intrinsic::mips_div_u_h:
2013 case Intrinsic::mips_div_u_w:
2014 case Intrinsic::mips_div_u_d:
2015 return DAG.getNode(Opcode: ISD::UDIV, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2016 N2: Op->getOperand(Num: 2));
2017 case Intrinsic::mips_fadd_w:
2018 case Intrinsic::mips_fadd_d:
2019 return DAG.getNode(Opcode: ISD::FADD, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2020 N2: Op->getOperand(Num: 2), Flags: Op->getFlags());
2021 // Don't lower mips_fcaf_[wd] since LLVM folds SETFALSE condcodes away
2022 case Intrinsic::mips_fceq_w:
2023 case Intrinsic::mips_fceq_d:
2024 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
2025 RHS: Op->getOperand(Num: 2), Cond: ISD::SETOEQ);
2026 case Intrinsic::mips_fcle_w:
2027 case Intrinsic::mips_fcle_d:
2028 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
2029 RHS: Op->getOperand(Num: 2), Cond: ISD::SETOLE);
2030 case Intrinsic::mips_fclt_w:
2031 case Intrinsic::mips_fclt_d:
2032 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
2033 RHS: Op->getOperand(Num: 2), Cond: ISD::SETOLT);
2034 case Intrinsic::mips_fcne_w:
2035 case Intrinsic::mips_fcne_d:
2036 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
2037 RHS: Op->getOperand(Num: 2), Cond: ISD::SETONE);
2038 case Intrinsic::mips_fcor_w:
2039 case Intrinsic::mips_fcor_d:
2040 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
2041 RHS: Op->getOperand(Num: 2), Cond: ISD::SETO);
2042 case Intrinsic::mips_fcueq_w:
2043 case Intrinsic::mips_fcueq_d:
2044 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
2045 RHS: Op->getOperand(Num: 2), Cond: ISD::SETUEQ);
2046 case Intrinsic::mips_fcule_w:
2047 case Intrinsic::mips_fcule_d:
2048 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
2049 RHS: Op->getOperand(Num: 2), Cond: ISD::SETULE);
2050 case Intrinsic::mips_fcult_w:
2051 case Intrinsic::mips_fcult_d:
2052 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
2053 RHS: Op->getOperand(Num: 2), Cond: ISD::SETULT);
2054 case Intrinsic::mips_fcun_w:
2055 case Intrinsic::mips_fcun_d:
2056 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
2057 RHS: Op->getOperand(Num: 2), Cond: ISD::SETUO);
2058 case Intrinsic::mips_fcune_w:
2059 case Intrinsic::mips_fcune_d:
2060 return DAG.getSetCC(DL, VT: Op->getValueType(ResNo: 0), LHS: Op->getOperand(Num: 1),
2061 RHS: Op->getOperand(Num: 2), Cond: ISD::SETUNE);
2062 case Intrinsic::mips_fdiv_w:
2063 case Intrinsic::mips_fdiv_d:
2064 // TODO: If intrinsics have fast-math-flags, propagate them.
2065 return DAG.getNode(Opcode: ISD::FDIV, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2066 N2: Op->getOperand(Num: 2));
2067 case Intrinsic::mips_ffint_u_w:
2068 case Intrinsic::mips_ffint_u_d:
2069 return DAG.getNode(Opcode: ISD::UINT_TO_FP, DL, VT: Op->getValueType(ResNo: 0),
2070 Operand: Op->getOperand(Num: 1));
2071 case Intrinsic::mips_ffint_s_w:
2072 case Intrinsic::mips_ffint_s_d:
2073 return DAG.getNode(Opcode: ISD::SINT_TO_FP, DL, VT: Op->getValueType(ResNo: 0),
2074 Operand: Op->getOperand(Num: 1));
2075 case Intrinsic::mips_fill_b:
2076 case Intrinsic::mips_fill_h:
2077 case Intrinsic::mips_fill_w:
2078 case Intrinsic::mips_fill_d: {
2079 EVT ResTy = Op->getValueType(ResNo: 0);
2080 SmallVector<SDValue, 16> Ops(ResTy.getVectorNumElements(),
2081 Op->getOperand(Num: 1));
2082
2083 // If ResTy is v2i64 then the type legalizer will break this node down into
2084 // an equivalent v4i32.
2085 return DAG.getBuildVector(VT: ResTy, DL, Ops);
2086 }
2087 case Intrinsic::mips_fexp2_w:
2088 case Intrinsic::mips_fexp2_d: {
2089 // TODO: If intrinsics have fast-math-flags, propagate them.
2090 EVT ResTy = Op->getValueType(ResNo: 0);
2091 return DAG.getNode(
2092 Opcode: ISD::FMUL, DL: SDLoc(Op), VT: ResTy, N1: Op->getOperand(Num: 1),
2093 N2: DAG.getNode(Opcode: ISD::FEXP2, DL: SDLoc(Op), VT: ResTy, Operand: Op->getOperand(Num: 2)));
2094 }
2095 case Intrinsic::mips_flog2_w:
2096 case Intrinsic::mips_flog2_d:
2097 return DAG.getNode(Opcode: ISD::FLOG2, DL, VT: Op->getValueType(ResNo: 0), Operand: Op->getOperand(Num: 1));
2098 case Intrinsic::mips_fmadd_w:
2099 case Intrinsic::mips_fmadd_d:
2100 return DAG.getNode(Opcode: ISD::FMA, DL: SDLoc(Op), VT: Op->getValueType(ResNo: 0),
2101 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 2), N3: Op->getOperand(Num: 3));
2102 case Intrinsic::mips_fmul_w:
2103 case Intrinsic::mips_fmul_d:
2104 return DAG.getNode(Opcode: ISD::FMUL, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2105 N2: Op->getOperand(Num: 2), Flags: Op->getFlags());
2106 case Intrinsic::mips_fmsub_w:
2107 case Intrinsic::mips_fmsub_d: {
2108 // TODO: If intrinsics have fast-math-flags, propagate them.
2109 return DAG.getNode(Opcode: MipsISD::FMS, DL: SDLoc(Op), VT: Op->getValueType(ResNo: 0),
2110 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 2), N3: Op->getOperand(Num: 3));
2111 }
2112 case Intrinsic::mips_frint_w:
2113 case Intrinsic::mips_frint_d:
2114 return DAG.getNode(Opcode: ISD::FRINT, DL, VT: Op->getValueType(ResNo: 0), Operand: Op->getOperand(Num: 1));
2115 case Intrinsic::mips_fsqrt_w:
2116 case Intrinsic::mips_fsqrt_d:
2117 return DAG.getNode(Opcode: ISD::FSQRT, DL, VT: Op->getValueType(ResNo: 0), Operand: Op->getOperand(Num: 1));
2118 case Intrinsic::mips_fsub_w:
2119 case Intrinsic::mips_fsub_d:
2120 return DAG.getNode(Opcode: ISD::FSUB, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2121 N2: Op->getOperand(Num: 2), Flags: Op->getFlags());
2122 case Intrinsic::mips_ftrunc_u_w:
2123 case Intrinsic::mips_ftrunc_u_d:
2124 return DAG.getNode(Opcode: ISD::FP_TO_UINT, DL, VT: Op->getValueType(ResNo: 0),
2125 Operand: Op->getOperand(Num: 1));
2126 case Intrinsic::mips_ftrunc_s_w:
2127 case Intrinsic::mips_ftrunc_s_d:
2128 return DAG.getNode(Opcode: ISD::FP_TO_SINT, DL, VT: Op->getValueType(ResNo: 0),
2129 Operand: Op->getOperand(Num: 1));
2130 case Intrinsic::mips_ilvev_b:
2131 case Intrinsic::mips_ilvev_h:
2132 case Intrinsic::mips_ilvev_w:
2133 case Intrinsic::mips_ilvev_d:
2134 return DAG.getNode(Opcode: MipsISD::ILVEV, DL, VT: Op->getValueType(ResNo: 0),
2135 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 2));
2136 case Intrinsic::mips_ilvl_b:
2137 case Intrinsic::mips_ilvl_h:
2138 case Intrinsic::mips_ilvl_w:
2139 case Intrinsic::mips_ilvl_d:
2140 return DAG.getNode(Opcode: MipsISD::ILVL, DL, VT: Op->getValueType(ResNo: 0),
2141 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 2));
2142 case Intrinsic::mips_ilvod_b:
2143 case Intrinsic::mips_ilvod_h:
2144 case Intrinsic::mips_ilvod_w:
2145 case Intrinsic::mips_ilvod_d:
2146 return DAG.getNode(Opcode: MipsISD::ILVOD, DL, VT: Op->getValueType(ResNo: 0),
2147 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 2));
2148 case Intrinsic::mips_ilvr_b:
2149 case Intrinsic::mips_ilvr_h:
2150 case Intrinsic::mips_ilvr_w:
2151 case Intrinsic::mips_ilvr_d:
2152 return DAG.getNode(Opcode: MipsISD::ILVR, DL, VT: Op->getValueType(ResNo: 0),
2153 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 2));
2154 case Intrinsic::mips_insert_b:
2155 case Intrinsic::mips_insert_h:
2156 case Intrinsic::mips_insert_w:
2157 case Intrinsic::mips_insert_d:
2158 return DAG.getNode(Opcode: ISD::INSERT_VECTOR_ELT, DL: SDLoc(Op), VT: Op->getValueType(ResNo: 0),
2159 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 3), N3: Op->getOperand(Num: 2));
2160 case Intrinsic::mips_insve_b:
2161 case Intrinsic::mips_insve_h:
2162 case Intrinsic::mips_insve_w:
2163 case Intrinsic::mips_insve_d: {
2164 // Report an error for out of range values.
2165 int64_t Max;
2166 switch (Intrinsic) {
2167 case Intrinsic::mips_insve_b: Max = 15; break;
2168 case Intrinsic::mips_insve_h: Max = 7; break;
2169 case Intrinsic::mips_insve_w: Max = 3; break;
2170 case Intrinsic::mips_insve_d: Max = 1; break;
2171 default: llvm_unreachable("Unmatched intrinsic");
2172 }
2173 int64_t Value = cast<ConstantSDNode>(Val: Op->getOperand(Num: 2))->getSExtValue();
2174 if (Value < 0 || Value > Max)
2175 report_fatal_error(reason: "Immediate out of range");
2176 return DAG.getNode(Opcode: MipsISD::INSVE, DL, VT: Op->getValueType(ResNo: 0),
2177 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 2), N3: Op->getOperand(Num: 3),
2178 N4: DAG.getConstant(Val: 0, DL, VT: MVT::i32));
2179 }
2180 case Intrinsic::mips_ldi_b:
2181 case Intrinsic::mips_ldi_h:
2182 case Intrinsic::mips_ldi_w:
2183 case Intrinsic::mips_ldi_d:
2184 return lowerMSASplatImm(Op, ImmOp: 1, DAG, IsSigned: true);
2185 case Intrinsic::mips_lsa:
2186 case Intrinsic::mips_dlsa: {
2187 EVT ResTy = Op->getValueType(ResNo: 0);
2188 return DAG.getNode(Opcode: ISD::ADD, DL: SDLoc(Op), VT: ResTy, N1: Op->getOperand(Num: 1),
2189 N2: DAG.getNode(Opcode: ISD::SHL, DL: SDLoc(Op), VT: ResTy,
2190 N1: Op->getOperand(Num: 2), N2: Op->getOperand(Num: 3)));
2191 }
2192 case Intrinsic::mips_maddv_b:
2193 case Intrinsic::mips_maddv_h:
2194 case Intrinsic::mips_maddv_w:
2195 case Intrinsic::mips_maddv_d: {
2196 EVT ResTy = Op->getValueType(ResNo: 0);
2197 return DAG.getNode(Opcode: ISD::ADD, DL: SDLoc(Op), VT: ResTy, N1: Op->getOperand(Num: 1),
2198 N2: DAG.getNode(Opcode: ISD::MUL, DL: SDLoc(Op), VT: ResTy,
2199 N1: Op->getOperand(Num: 2), N2: Op->getOperand(Num: 3)));
2200 }
2201 case Intrinsic::mips_max_s_b:
2202 case Intrinsic::mips_max_s_h:
2203 case Intrinsic::mips_max_s_w:
2204 case Intrinsic::mips_max_s_d:
2205 return DAG.getNode(Opcode: ISD::SMAX, DL, VT: Op->getValueType(ResNo: 0),
2206 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 2));
2207 case Intrinsic::mips_max_u_b:
2208 case Intrinsic::mips_max_u_h:
2209 case Intrinsic::mips_max_u_w:
2210 case Intrinsic::mips_max_u_d:
2211 return DAG.getNode(Opcode: ISD::UMAX, DL, VT: Op->getValueType(ResNo: 0),
2212 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 2));
2213 case Intrinsic::mips_maxi_s_b:
2214 case Intrinsic::mips_maxi_s_h:
2215 case Intrinsic::mips_maxi_s_w:
2216 case Intrinsic::mips_maxi_s_d:
2217 return DAG.getNode(Opcode: ISD::SMAX, DL, VT: Op->getValueType(ResNo: 0),
2218 N1: Op->getOperand(Num: 1), N2: lowerMSASplatImm(Op, ImmOp: 2, DAG, IsSigned: true));
2219 case Intrinsic::mips_maxi_u_b:
2220 case Intrinsic::mips_maxi_u_h:
2221 case Intrinsic::mips_maxi_u_w:
2222 case Intrinsic::mips_maxi_u_d:
2223 return DAG.getNode(Opcode: ISD::UMAX, DL, VT: Op->getValueType(ResNo: 0),
2224 N1: Op->getOperand(Num: 1), N2: lowerMSASplatImm(Op, ImmOp: 2, DAG));
2225 case Intrinsic::mips_min_s_b:
2226 case Intrinsic::mips_min_s_h:
2227 case Intrinsic::mips_min_s_w:
2228 case Intrinsic::mips_min_s_d:
2229 return DAG.getNode(Opcode: ISD::SMIN, DL, VT: Op->getValueType(ResNo: 0),
2230 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 2));
2231 case Intrinsic::mips_min_u_b:
2232 case Intrinsic::mips_min_u_h:
2233 case Intrinsic::mips_min_u_w:
2234 case Intrinsic::mips_min_u_d:
2235 return DAG.getNode(Opcode: ISD::UMIN, DL, VT: Op->getValueType(ResNo: 0),
2236 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 2));
2237 case Intrinsic::mips_mini_s_b:
2238 case Intrinsic::mips_mini_s_h:
2239 case Intrinsic::mips_mini_s_w:
2240 case Intrinsic::mips_mini_s_d:
2241 return DAG.getNode(Opcode: ISD::SMIN, DL, VT: Op->getValueType(ResNo: 0),
2242 N1: Op->getOperand(Num: 1), N2: lowerMSASplatImm(Op, ImmOp: 2, DAG, IsSigned: true));
2243 case Intrinsic::mips_mini_u_b:
2244 case Intrinsic::mips_mini_u_h:
2245 case Intrinsic::mips_mini_u_w:
2246 case Intrinsic::mips_mini_u_d:
2247 return DAG.getNode(Opcode: ISD::UMIN, DL, VT: Op->getValueType(ResNo: 0),
2248 N1: Op->getOperand(Num: 1), N2: lowerMSASplatImm(Op, ImmOp: 2, DAG));
2249 case Intrinsic::mips_mod_s_b:
2250 case Intrinsic::mips_mod_s_h:
2251 case Intrinsic::mips_mod_s_w:
2252 case Intrinsic::mips_mod_s_d:
2253 return DAG.getNode(Opcode: ISD::SREM, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2254 N2: Op->getOperand(Num: 2));
2255 case Intrinsic::mips_mod_u_b:
2256 case Intrinsic::mips_mod_u_h:
2257 case Intrinsic::mips_mod_u_w:
2258 case Intrinsic::mips_mod_u_d:
2259 return DAG.getNode(Opcode: ISD::UREM, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2260 N2: Op->getOperand(Num: 2));
2261 case Intrinsic::mips_mulv_b:
2262 case Intrinsic::mips_mulv_h:
2263 case Intrinsic::mips_mulv_w:
2264 case Intrinsic::mips_mulv_d:
2265 return DAG.getNode(Opcode: ISD::MUL, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2266 N2: Op->getOperand(Num: 2));
2267 case Intrinsic::mips_msubv_b:
2268 case Intrinsic::mips_msubv_h:
2269 case Intrinsic::mips_msubv_w:
2270 case Intrinsic::mips_msubv_d: {
2271 EVT ResTy = Op->getValueType(ResNo: 0);
2272 return DAG.getNode(Opcode: ISD::SUB, DL: SDLoc(Op), VT: ResTy, N1: Op->getOperand(Num: 1),
2273 N2: DAG.getNode(Opcode: ISD::MUL, DL: SDLoc(Op), VT: ResTy,
2274 N1: Op->getOperand(Num: 2), N2: Op->getOperand(Num: 3)));
2275 }
2276 case Intrinsic::mips_nlzc_b:
2277 case Intrinsic::mips_nlzc_h:
2278 case Intrinsic::mips_nlzc_w:
2279 case Intrinsic::mips_nlzc_d:
2280 return DAG.getNode(Opcode: ISD::CTLZ, DL, VT: Op->getValueType(ResNo: 0), Operand: Op->getOperand(Num: 1));
2281 case Intrinsic::mips_nor_v: {
2282 SDValue Res = DAG.getNode(Opcode: ISD::OR, DL, VT: Op->getValueType(ResNo: 0),
2283 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 2));
2284 return DAG.getNOT(DL, Val: Res, VT: Res->getValueType(ResNo: 0));
2285 }
2286 case Intrinsic::mips_nori_b: {
2287 SDValue Res = DAG.getNode(Opcode: ISD::OR, DL, VT: Op->getValueType(ResNo: 0),
2288 N1: Op->getOperand(Num: 1),
2289 N2: lowerMSASplatImm(Op, ImmOp: 2, DAG));
2290 return DAG.getNOT(DL, Val: Res, VT: Res->getValueType(ResNo: 0));
2291 }
2292 case Intrinsic::mips_or_v:
2293 return DAG.getNode(Opcode: ISD::OR, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2294 N2: Op->getOperand(Num: 2));
2295 case Intrinsic::mips_ori_b:
2296 return DAG.getNode(Opcode: ISD::OR, DL, VT: Op->getValueType(ResNo: 0),
2297 N1: Op->getOperand(Num: 1), N2: lowerMSASplatImm(Op, ImmOp: 2, DAG));
2298 case Intrinsic::mips_pckev_b:
2299 case Intrinsic::mips_pckev_h:
2300 case Intrinsic::mips_pckev_w:
2301 case Intrinsic::mips_pckev_d:
2302 return DAG.getNode(Opcode: MipsISD::PCKEV, DL, VT: Op->getValueType(ResNo: 0),
2303 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 2));
2304 case Intrinsic::mips_pckod_b:
2305 case Intrinsic::mips_pckod_h:
2306 case Intrinsic::mips_pckod_w:
2307 case Intrinsic::mips_pckod_d:
2308 return DAG.getNode(Opcode: MipsISD::PCKOD, DL, VT: Op->getValueType(ResNo: 0),
2309 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 2));
2310 case Intrinsic::mips_pcnt_b:
2311 case Intrinsic::mips_pcnt_h:
2312 case Intrinsic::mips_pcnt_w:
2313 case Intrinsic::mips_pcnt_d:
2314 return DAG.getNode(Opcode: ISD::CTPOP, DL, VT: Op->getValueType(ResNo: 0), Operand: Op->getOperand(Num: 1));
2315 case Intrinsic::mips_sat_s_b:
2316 case Intrinsic::mips_sat_s_h:
2317 case Intrinsic::mips_sat_s_w:
2318 case Intrinsic::mips_sat_s_d:
2319 case Intrinsic::mips_sat_u_b:
2320 case Intrinsic::mips_sat_u_h:
2321 case Intrinsic::mips_sat_u_w:
2322 case Intrinsic::mips_sat_u_d: {
2323 // Report an error for out of range values.
2324 int64_t Max;
2325 switch (Intrinsic) {
2326 case Intrinsic::mips_sat_s_b:
2327 case Intrinsic::mips_sat_u_b: Max = 7; break;
2328 case Intrinsic::mips_sat_s_h:
2329 case Intrinsic::mips_sat_u_h: Max = 15; break;
2330 case Intrinsic::mips_sat_s_w:
2331 case Intrinsic::mips_sat_u_w: Max = 31; break;
2332 case Intrinsic::mips_sat_s_d:
2333 case Intrinsic::mips_sat_u_d: Max = 63; break;
2334 default: llvm_unreachable("Unmatched intrinsic");
2335 }
2336 int64_t Value = cast<ConstantSDNode>(Val: Op->getOperand(Num: 2))->getSExtValue();
2337 if (Value < 0 || Value > Max)
2338 report_fatal_error(reason: "Immediate out of range");
2339 return SDValue();
2340 }
2341 case Intrinsic::mips_shf_b:
2342 case Intrinsic::mips_shf_h:
2343 case Intrinsic::mips_shf_w: {
2344 int64_t Value = cast<ConstantSDNode>(Val: Op->getOperand(Num: 2))->getSExtValue();
2345 if (Value < 0 || Value > 255)
2346 report_fatal_error(reason: "Immediate out of range");
2347 return DAG.getNode(Opcode: MipsISD::SHF, DL, VT: Op->getValueType(ResNo: 0),
2348 N1: Op->getOperand(Num: 2), N2: Op->getOperand(Num: 1));
2349 }
2350 case Intrinsic::mips_sldi_b:
2351 case Intrinsic::mips_sldi_h:
2352 case Intrinsic::mips_sldi_w:
2353 case Intrinsic::mips_sldi_d: {
2354 // Report an error for out of range values.
2355 int64_t Max;
2356 switch (Intrinsic) {
2357 case Intrinsic::mips_sldi_b: Max = 15; break;
2358 case Intrinsic::mips_sldi_h: Max = 7; break;
2359 case Intrinsic::mips_sldi_w: Max = 3; break;
2360 case Intrinsic::mips_sldi_d: Max = 1; break;
2361 default: llvm_unreachable("Unmatched intrinsic");
2362 }
2363 int64_t Value = cast<ConstantSDNode>(Val: Op->getOperand(Num: 3))->getSExtValue();
2364 if (Value < 0 || Value > Max)
2365 report_fatal_error(reason: "Immediate out of range");
2366 return SDValue();
2367 }
2368 case Intrinsic::mips_sll_b:
2369 case Intrinsic::mips_sll_h:
2370 case Intrinsic::mips_sll_w:
2371 case Intrinsic::mips_sll_d:
2372 return DAG.getNode(Opcode: ISD::SHL, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2373 N2: truncateVecElts(Op, DAG));
2374 case Intrinsic::mips_slli_b:
2375 case Intrinsic::mips_slli_h:
2376 case Intrinsic::mips_slli_w:
2377 case Intrinsic::mips_slli_d:
2378 return DAG.getNode(Opcode: ISD::SHL, DL, VT: Op->getValueType(ResNo: 0),
2379 N1: Op->getOperand(Num: 1), N2: lowerMSASplatImm(Op, ImmOp: 2, DAG));
2380 case Intrinsic::mips_splat_b:
2381 case Intrinsic::mips_splat_h:
2382 case Intrinsic::mips_splat_w:
2383 case Intrinsic::mips_splat_d:
2384 // We can't lower via VECTOR_SHUFFLE because it requires constant shuffle
2385 // masks, nor can we lower via BUILD_VECTOR & EXTRACT_VECTOR_ELT because
2386 // EXTRACT_VECTOR_ELT can't extract i64's on MIPS32.
2387 // Instead we lower to MipsISD::VSHF and match from there.
2388 return DAG.getNode(Opcode: MipsISD::VSHF, DL, VT: Op->getValueType(ResNo: 0),
2389 N1: lowerMSASplatZExt(Op, OpNr: 2, DAG), N2: Op->getOperand(Num: 1),
2390 N3: Op->getOperand(Num: 1));
2391 case Intrinsic::mips_splati_b:
2392 case Intrinsic::mips_splati_h:
2393 case Intrinsic::mips_splati_w:
2394 case Intrinsic::mips_splati_d:
2395 return DAG.getNode(Opcode: MipsISD::VSHF, DL, VT: Op->getValueType(ResNo: 0),
2396 N1: lowerMSASplatImm(Op, ImmOp: 2, DAG), N2: Op->getOperand(Num: 1),
2397 N3: Op->getOperand(Num: 1));
2398 case Intrinsic::mips_sra_b:
2399 case Intrinsic::mips_sra_h:
2400 case Intrinsic::mips_sra_w:
2401 case Intrinsic::mips_sra_d:
2402 return DAG.getNode(Opcode: ISD::SRA, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2403 N2: truncateVecElts(Op, DAG));
2404 case Intrinsic::mips_srai_b:
2405 case Intrinsic::mips_srai_h:
2406 case Intrinsic::mips_srai_w:
2407 case Intrinsic::mips_srai_d:
2408 return DAG.getNode(Opcode: ISD::SRA, DL, VT: Op->getValueType(ResNo: 0),
2409 N1: Op->getOperand(Num: 1), N2: lowerMSASplatImm(Op, ImmOp: 2, DAG));
2410 case Intrinsic::mips_srari_b:
2411 case Intrinsic::mips_srari_h:
2412 case Intrinsic::mips_srari_w:
2413 case Intrinsic::mips_srari_d: {
2414 // Report an error for out of range values.
2415 int64_t Max;
2416 switch (Intrinsic) {
2417 case Intrinsic::mips_srari_b: Max = 7; break;
2418 case Intrinsic::mips_srari_h: Max = 15; break;
2419 case Intrinsic::mips_srari_w: Max = 31; break;
2420 case Intrinsic::mips_srari_d: Max = 63; break;
2421 default: llvm_unreachable("Unmatched intrinsic");
2422 }
2423 int64_t Value = cast<ConstantSDNode>(Val: Op->getOperand(Num: 2))->getSExtValue();
2424 if (Value < 0 || Value > Max)
2425 report_fatal_error(reason: "Immediate out of range");
2426 return SDValue();
2427 }
2428 case Intrinsic::mips_srl_b:
2429 case Intrinsic::mips_srl_h:
2430 case Intrinsic::mips_srl_w:
2431 case Intrinsic::mips_srl_d:
2432 return DAG.getNode(Opcode: ISD::SRL, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2433 N2: truncateVecElts(Op, DAG));
2434 case Intrinsic::mips_srli_b:
2435 case Intrinsic::mips_srli_h:
2436 case Intrinsic::mips_srli_w:
2437 case Intrinsic::mips_srli_d:
2438 return DAG.getNode(Opcode: ISD::SRL, DL, VT: Op->getValueType(ResNo: 0),
2439 N1: Op->getOperand(Num: 1), N2: lowerMSASplatImm(Op, ImmOp: 2, DAG));
2440 case Intrinsic::mips_srlri_b:
2441 case Intrinsic::mips_srlri_h:
2442 case Intrinsic::mips_srlri_w:
2443 case Intrinsic::mips_srlri_d: {
2444 // Report an error for out of range values.
2445 int64_t Max;
2446 switch (Intrinsic) {
2447 case Intrinsic::mips_srlri_b: Max = 7; break;
2448 case Intrinsic::mips_srlri_h: Max = 15; break;
2449 case Intrinsic::mips_srlri_w: Max = 31; break;
2450 case Intrinsic::mips_srlri_d: Max = 63; break;
2451 default: llvm_unreachable("Unmatched intrinsic");
2452 }
2453 int64_t Value = cast<ConstantSDNode>(Val: Op->getOperand(Num: 2))->getSExtValue();
2454 if (Value < 0 || Value > Max)
2455 report_fatal_error(reason: "Immediate out of range");
2456 return SDValue();
2457 }
2458 case Intrinsic::mips_subv_b:
2459 case Intrinsic::mips_subv_h:
2460 case Intrinsic::mips_subv_w:
2461 case Intrinsic::mips_subv_d:
2462 return DAG.getNode(Opcode: ISD::SUB, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2463 N2: Op->getOperand(Num: 2));
2464 case Intrinsic::mips_subvi_b:
2465 case Intrinsic::mips_subvi_h:
2466 case Intrinsic::mips_subvi_w:
2467 case Intrinsic::mips_subvi_d:
2468 return DAG.getNode(Opcode: ISD::SUB, DL, VT: Op->getValueType(ResNo: 0),
2469 N1: Op->getOperand(Num: 1), N2: lowerMSASplatImm(Op, ImmOp: 2, DAG));
2470 case Intrinsic::mips_vshf_b:
2471 case Intrinsic::mips_vshf_h:
2472 case Intrinsic::mips_vshf_w:
2473 case Intrinsic::mips_vshf_d:
2474 return DAG.getNode(Opcode: MipsISD::VSHF, DL, VT: Op->getValueType(ResNo: 0),
2475 N1: Op->getOperand(Num: 1), N2: Op->getOperand(Num: 2), N3: Op->getOperand(Num: 3));
2476 case Intrinsic::mips_xor_v:
2477 return DAG.getNode(Opcode: ISD::XOR, DL, VT: Op->getValueType(ResNo: 0), N1: Op->getOperand(Num: 1),
2478 N2: Op->getOperand(Num: 2));
2479 case Intrinsic::mips_xori_b:
2480 return DAG.getNode(Opcode: ISD::XOR, DL, VT: Op->getValueType(ResNo: 0),
2481 N1: Op->getOperand(Num: 1), N2: lowerMSASplatImm(Op, ImmOp: 2, DAG));
2482 case Intrinsic::thread_pointer: {
2483 EVT PtrVT = getPointerTy(DL: DAG.getDataLayout());
2484 return DAG.getNode(Opcode: MipsISD::ThreadPointer, DL, VT: PtrVT);
2485 }
2486 }
2487}
2488
2489static SDValue lowerMSALoadIntr(SDValue Op, SelectionDAG &DAG, unsigned Intr,
2490 const MipsSubtarget &Subtarget) {
2491 SDLoc DL(Op);
2492 SDValue ChainIn = Op->getOperand(Num: 0);
2493 SDValue Address = Op->getOperand(Num: 2);
2494 SDValue Offset = Op->getOperand(Num: 3);
2495 EVT ResTy = Op->getValueType(ResNo: 0);
2496 EVT PtrTy = Address->getValueType(ResNo: 0);
2497
2498 // For N64 addresses have the underlying type MVT::i64. This intrinsic
2499 // however takes an i32 signed constant offset. The actual type of the
2500 // intrinsic is a scaled signed i10.
2501 if (Subtarget.isABI_N64())
2502 Offset = DAG.getNode(Opcode: ISD::SIGN_EXTEND, DL, VT: PtrTy, Operand: Offset);
2503
2504 Address = DAG.getNode(Opcode: ISD::ADD, DL, VT: PtrTy, N1: Address, N2: Offset);
2505 return DAG.getLoad(VT: ResTy, dl: DL, Chain: ChainIn, Ptr: Address, PtrInfo: MachinePointerInfo(),
2506 Alignment: Align(16));
2507}
2508
2509SDValue MipsSETargetLowering::lowerINTRINSIC_W_CHAIN(SDValue Op,
2510 SelectionDAG &DAG) const {
2511 unsigned Intr = Op->getConstantOperandVal(Num: 1);
2512 switch (Intr) {
2513 default:
2514 return SDValue();
2515 case Intrinsic::mips_extp:
2516 return lowerDSPIntr(Op, DAG, Opc: MipsISD::EXTP);
2517 case Intrinsic::mips_extpdp:
2518 return lowerDSPIntr(Op, DAG, Opc: MipsISD::EXTPDP);
2519 case Intrinsic::mips_extr_w:
2520 return lowerDSPIntr(Op, DAG, Opc: MipsISD::EXTR_W);
2521 case Intrinsic::mips_extr_r_w:
2522 return lowerDSPIntr(Op, DAG, Opc: MipsISD::EXTR_R_W);
2523 case Intrinsic::mips_extr_rs_w:
2524 return lowerDSPIntr(Op, DAG, Opc: MipsISD::EXTR_RS_W);
2525 case Intrinsic::mips_extr_s_h:
2526 return lowerDSPIntr(Op, DAG, Opc: MipsISD::EXTR_S_H);
2527 case Intrinsic::mips_mthlip:
2528 return lowerDSPIntr(Op, DAG, Opc: MipsISD::MTHLIP);
2529 case Intrinsic::mips_mulsaq_s_w_ph:
2530 return lowerDSPIntr(Op, DAG, Opc: MipsISD::MULSAQ_S_W_PH);
2531 case Intrinsic::mips_maq_s_w_phl:
2532 return lowerDSPIntr(Op, DAG, Opc: MipsISD::MAQ_S_W_PHL);
2533 case Intrinsic::mips_maq_s_w_phr:
2534 return lowerDSPIntr(Op, DAG, Opc: MipsISD::MAQ_S_W_PHR);
2535 case Intrinsic::mips_maq_sa_w_phl:
2536 return lowerDSPIntr(Op, DAG, Opc: MipsISD::MAQ_SA_W_PHL);
2537 case Intrinsic::mips_maq_sa_w_phr:
2538 return lowerDSPIntr(Op, DAG, Opc: MipsISD::MAQ_SA_W_PHR);
2539 case Intrinsic::mips_dpaq_s_w_ph:
2540 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPAQ_S_W_PH);
2541 case Intrinsic::mips_dpsq_s_w_ph:
2542 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPSQ_S_W_PH);
2543 case Intrinsic::mips_dpaq_sa_l_w:
2544 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPAQ_SA_L_W);
2545 case Intrinsic::mips_dpsq_sa_l_w:
2546 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPSQ_SA_L_W);
2547 case Intrinsic::mips_dpaqx_s_w_ph:
2548 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPAQX_S_W_PH);
2549 case Intrinsic::mips_dpaqx_sa_w_ph:
2550 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPAQX_SA_W_PH);
2551 case Intrinsic::mips_dpsqx_s_w_ph:
2552 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPSQX_S_W_PH);
2553 case Intrinsic::mips_dpsqx_sa_w_ph:
2554 return lowerDSPIntr(Op, DAG, Opc: MipsISD::DPSQX_SA_W_PH);
2555 case Intrinsic::mips_ld_b:
2556 case Intrinsic::mips_ld_h:
2557 case Intrinsic::mips_ld_w:
2558 case Intrinsic::mips_ld_d:
2559 return lowerMSALoadIntr(Op, DAG, Intr, Subtarget);
2560 }
2561}
2562
2563static SDValue lowerMSAStoreIntr(SDValue Op, SelectionDAG &DAG, unsigned Intr,
2564 const MipsSubtarget &Subtarget) {
2565 SDLoc DL(Op);
2566 SDValue ChainIn = Op->getOperand(Num: 0);
2567 SDValue Value = Op->getOperand(Num: 2);
2568 SDValue Address = Op->getOperand(Num: 3);
2569 SDValue Offset = Op->getOperand(Num: 4);
2570 EVT PtrTy = Address->getValueType(ResNo: 0);
2571
2572 // For N64 addresses have the underlying type MVT::i64. This intrinsic
2573 // however takes an i32 signed constant offset. The actual type of the
2574 // intrinsic is a scaled signed i10.
2575 if (Subtarget.isABI_N64())
2576 Offset = DAG.getNode(Opcode: ISD::SIGN_EXTEND, DL, VT: PtrTy, Operand: Offset);
2577
2578 Address = DAG.getNode(Opcode: ISD::ADD, DL, VT: PtrTy, N1: Address, N2: Offset);
2579
2580 return DAG.getStore(Chain: ChainIn, dl: DL, Val: Value, Ptr: Address, PtrInfo: MachinePointerInfo(),
2581 Alignment: Align(16));
2582}
2583
2584SDValue MipsSETargetLowering::lowerINTRINSIC_VOID(SDValue Op,
2585 SelectionDAG &DAG) const {
2586 unsigned Intr = Op->getConstantOperandVal(Num: 1);
2587 switch (Intr) {
2588 default:
2589 return SDValue();
2590 case Intrinsic::mips_st_b:
2591 case Intrinsic::mips_st_h:
2592 case Intrinsic::mips_st_w:
2593 case Intrinsic::mips_st_d:
2594 return lowerMSAStoreIntr(Op, DAG, Intr, Subtarget);
2595 }
2596}
2597
2598// Lower ISD::EXTRACT_VECTOR_ELT into MipsISD::VEXTRACT_SEXT_ELT.
2599//
2600// The non-value bits resulting from ISD::EXTRACT_VECTOR_ELT are undefined. We
2601// choose to sign-extend but we could have equally chosen zero-extend. The
2602// DAGCombiner will fold any sign/zero extension of the ISD::EXTRACT_VECTOR_ELT
2603// result into this node later (possibly changing it to a zero-extend in the
2604// process).
2605SDValue MipsSETargetLowering::
2606lowerEXTRACT_VECTOR_ELT(SDValue Op, SelectionDAG &DAG) const {
2607 SDLoc DL(Op);
2608 EVT ResTy = Op->getValueType(ResNo: 0);
2609 SDValue Op0 = Op->getOperand(Num: 0);
2610 EVT VecTy = Op0->getValueType(ResNo: 0);
2611
2612 if (!VecTy.is128BitVector())
2613 return SDValue();
2614
2615 if (ResTy.isInteger()) {
2616 SDValue Op1 = Op->getOperand(Num: 1);
2617 EVT EltTy = VecTy.getVectorElementType();
2618 return DAG.getNode(Opcode: MipsISD::VEXTRACT_SEXT_ELT, DL, VT: ResTy, N1: Op0, N2: Op1,
2619 N3: DAG.getValueType(EltTy));
2620 }
2621
2622 return Op;
2623}
2624
2625static bool isConstantOrUndef(const SDValue Op) {
2626 if (Op->isUndef())
2627 return true;
2628 if (isa<ConstantSDNode>(Val: Op))
2629 return true;
2630 if (isa<ConstantFPSDNode>(Val: Op))
2631 return true;
2632 return false;
2633}
2634
2635static bool isConstantOrUndefBUILD_VECTOR(const BuildVectorSDNode *Op) {
2636 for (unsigned i = 0; i < Op->getNumOperands(); ++i)
2637 if (isConstantOrUndef(Op: Op->getOperand(Num: i)))
2638 return true;
2639 return false;
2640}
2641
2642// Lowers ISD::BUILD_VECTOR into appropriate SelectionDAG nodes for the
2643// backend.
2644//
2645// Lowers according to the following rules:
2646// - Constant splats are legal as-is as long as the SplatBitSize is a power of
2647// 2 less than or equal to 64 and the value fits into a signed 10-bit
2648// immediate
2649// - Constant splats are lowered to bitconverted BUILD_VECTORs if SplatBitSize
2650// is a power of 2 less than or equal to 64 and the value does not fit into a
2651// signed 10-bit immediate
2652// - Non-constant splats are legal as-is.
2653// - Non-constant non-splats are lowered to sequences of INSERT_VECTOR_ELT.
2654// - All others are illegal and must be expanded.
2655SDValue MipsSETargetLowering::lowerBUILD_VECTOR(SDValue Op,
2656 SelectionDAG &DAG) const {
2657 BuildVectorSDNode *Node = cast<BuildVectorSDNode>(Val&: Op);
2658 EVT ResTy = Op->getValueType(ResNo: 0);
2659 SDLoc DL(Op);
2660 APInt SplatValue, SplatUndef;
2661 unsigned SplatBitSize;
2662 bool HasAnyUndefs;
2663
2664 if (!Subtarget.hasMSA() || !ResTy.is128BitVector())
2665 return SDValue();
2666
2667 if (Node->isConstantSplat(SplatValue, SplatUndef, SplatBitSize,
2668 HasAnyUndefs, MinSplatBits: 8,
2669 isBigEndian: !Subtarget.isLittle()) && SplatBitSize <= 64) {
2670 // We can only cope with 8, 16, 32, or 64-bit elements
2671 if (SplatBitSize != 8 && SplatBitSize != 16 && SplatBitSize != 32 &&
2672 SplatBitSize != 64)
2673 return SDValue();
2674
2675 // If the value isn't an integer type we will have to bitcast
2676 // from an integer type first. Also, if there are any undefs, we must
2677 // lower them to defined values first.
2678 if (ResTy.isInteger() && !HasAnyUndefs)
2679 return Op;
2680
2681 EVT ViaVecTy;
2682
2683 switch (SplatBitSize) {
2684 default:
2685 return SDValue();
2686 case 8:
2687 ViaVecTy = MVT::v16i8;
2688 break;
2689 case 16:
2690 ViaVecTy = MVT::v8i16;
2691 break;
2692 case 32:
2693 ViaVecTy = MVT::v4i32;
2694 break;
2695 case 64:
2696 // There's no fill.d to fall back on for 64-bit values
2697 return SDValue();
2698 }
2699
2700 // SelectionDAG::getConstant will promote SplatValue appropriately.
2701 SDValue Result = DAG.getConstant(Val: SplatValue, DL, VT: ViaVecTy);
2702
2703 // Bitcast to the type we originally wanted
2704 if (ViaVecTy != ResTy)
2705 Result = DAG.getNode(Opcode: ISD::BITCAST, DL: SDLoc(Node), VT: ResTy, Operand: Result);
2706
2707 return Result;
2708 } else if (DAG.isSplatValue(V: Op, /* AllowUndefs */ false))
2709 return Op;
2710 else if (!isConstantOrUndefBUILD_VECTOR(Op: Node)) {
2711 // Use INSERT_VECTOR_ELT operations rather than expand to stores.
2712 // The resulting code is the same length as the expansion, but it doesn't
2713 // use memory operations
2714 EVT ResTy = Node->getValueType(ResNo: 0);
2715
2716 assert(ResTy.isVector());
2717
2718 unsigned NumElts = ResTy.getVectorNumElements();
2719 SDValue Vector = DAG.getUNDEF(VT: ResTy);
2720 for (unsigned i = 0; i < NumElts; ++i) {
2721 Vector = DAG.getNode(Opcode: ISD::INSERT_VECTOR_ELT, DL, VT: ResTy, N1: Vector,
2722 N2: Node->getOperand(Num: i),
2723 N3: DAG.getConstant(Val: i, DL, VT: MVT::i32));
2724 }
2725 return Vector;
2726 }
2727
2728 return SDValue();
2729}
2730
2731// Lower VECTOR_SHUFFLE into SHF (if possible).
2732//
2733// SHF splits the vector into blocks of four elements, then shuffles these
2734// elements according to a <4 x i2> constant (encoded as an integer immediate).
2735//
2736// It is therefore possible to lower into SHF when the mask takes the form:
2737// <a, b, c, d, a+4, b+4, c+4, d+4, a+8, b+8, c+8, d+8, ...>
2738// When undef's appear they are treated as if they were whatever value is
2739// necessary in order to fit the above forms.
2740//
2741// For example:
2742// %2 = shufflevector <8 x i16> %0, <8 x i16> undef,
2743// <8 x i32> <i32 3, i32 2, i32 1, i32 0,
2744// i32 7, i32 6, i32 5, i32 4>
2745// is lowered to:
2746// (SHF_H $w0, $w1, 27)
2747// where the 27 comes from:
2748// 3 + (2 << 2) + (1 << 4) + (0 << 6)
2749static SDValue lowerVECTOR_SHUFFLE_SHF(SDValue Op, EVT ResTy,
2750 SmallVector<int, 16> Indices,
2751 SelectionDAG &DAG) {
2752 int SHFIndices[4] = { -1, -1, -1, -1 };
2753
2754 if (Indices.size() < 4)
2755 return SDValue();
2756
2757 for (unsigned i = 0; i < 4; ++i) {
2758 for (unsigned j = i; j < Indices.size(); j += 4) {
2759 int Idx = Indices[j];
2760
2761 // Convert from vector index to 4-element subvector index
2762 // If an index refers to an element outside of the subvector then give up
2763 if (Idx != -1) {
2764 Idx -= 4 * (j / 4);
2765 if (Idx < 0 || Idx >= 4)
2766 return SDValue();
2767 }
2768
2769 // If the mask has an undef, replace it with the current index.
2770 // Note that it might still be undef if the current index is also undef
2771 if (SHFIndices[i] == -1)
2772 SHFIndices[i] = Idx;
2773
2774 // Check that non-undef values are the same as in the mask. If they
2775 // aren't then give up
2776 if (!(Idx == -1 || Idx == SHFIndices[i]))
2777 return SDValue();
2778 }
2779 }
2780
2781 // Calculate the immediate. Replace any remaining undefs with zero
2782 APInt Imm(32, 0);
2783 for (int i = 3; i >= 0; --i) {
2784 int Idx = SHFIndices[i];
2785
2786 if (Idx == -1)
2787 Idx = 0;
2788
2789 Imm <<= 2;
2790 Imm |= Idx & 0x3;
2791 }
2792
2793 SDLoc DL(Op);
2794 return DAG.getNode(Opcode: MipsISD::SHF, DL, VT: ResTy,
2795 N1: DAG.getTargetConstant(Val: Imm, DL, VT: MVT::i32),
2796 N2: Op->getOperand(Num: 0));
2797}
2798
2799/// Determine whether a range fits a regular pattern of values.
2800/// This function accounts for the possibility of jumping over the End iterator.
2801template <typename ValType>
2802static bool
2803fitsRegularPattern(typename SmallVectorImpl<ValType>::const_iterator Begin,
2804 unsigned CheckStride,
2805 typename SmallVectorImpl<ValType>::const_iterator End,
2806 ValType ExpectedIndex, unsigned ExpectedIndexStride) {
2807 auto &I = Begin;
2808
2809 while (I != End) {
2810 if (*I != -1 && *I != ExpectedIndex)
2811 return false;
2812 ExpectedIndex += ExpectedIndexStride;
2813
2814 // Incrementing past End is undefined behaviour so we must increment one
2815 // step at a time and check for End at each step.
2816 for (unsigned n = 0; n < CheckStride && I != End; ++n, ++I)
2817 ; // Empty loop body.
2818 }
2819 return true;
2820}
2821
2822// Determine whether VECTOR_SHUFFLE is a SPLATI.
2823//
2824// It is a SPLATI when the mask is:
2825// <x, x, x, ...>
2826// where x is any valid index.
2827//
2828// When undef's appear in the mask they are treated as if they were whatever
2829// value is necessary in order to fit the above form.
2830static bool isVECTOR_SHUFFLE_SPLATI(SDValue Op, EVT ResTy,
2831 SmallVector<int, 16> Indices,
2832 SelectionDAG &DAG) {
2833 assert((Indices.size() % 2) == 0);
2834
2835 int SplatIndex = -1;
2836 for (const auto &V : Indices) {
2837 if (V != -1) {
2838 SplatIndex = V;
2839 break;
2840 }
2841 }
2842
2843 return fitsRegularPattern<int>(Begin: Indices.begin(), CheckStride: 1, End: Indices.end(), ExpectedIndex: SplatIndex,
2844 ExpectedIndexStride: 0);
2845}
2846
2847// Lower VECTOR_SHUFFLE into ILVEV (if possible).
2848//
2849// ILVEV interleaves the even elements from each vector.
2850//
2851// It is possible to lower into ILVEV when the mask consists of two of the
2852// following forms interleaved:
2853// <0, 2, 4, ...>
2854// <n, n+2, n+4, ...>
2855// where n is the number of elements in the vector.
2856// For example:
2857// <0, 0, 2, 2, 4, 4, ...>
2858// <0, n, 2, n+2, 4, n+4, ...>
2859//
2860// When undef's appear in the mask they are treated as if they were whatever
2861// value is necessary in order to fit the above forms.
2862static SDValue lowerVECTOR_SHUFFLE_ILVEV(SDValue Op, EVT ResTy,
2863 SmallVector<int, 16> Indices,
2864 SelectionDAG &DAG) {
2865 assert((Indices.size() % 2) == 0);
2866
2867 SDValue Wt;
2868 SDValue Ws;
2869 const auto &Begin = Indices.begin();
2870 const auto &End = Indices.end();
2871
2872 // Check even elements are taken from the even elements of one half or the
2873 // other and pick an operand accordingly.
2874 if (fitsRegularPattern<int>(Begin, CheckStride: 2, End, ExpectedIndex: 0, ExpectedIndexStride: 2))
2875 Wt = Op->getOperand(Num: 0);
2876 else if (fitsRegularPattern<int>(Begin, CheckStride: 2, End, ExpectedIndex: Indices.size(), ExpectedIndexStride: 2))
2877 Wt = Op->getOperand(Num: 1);
2878 else
2879 return SDValue();
2880
2881 // Check odd elements are taken from the even elements of one half or the
2882 // other and pick an operand accordingly.
2883 if (fitsRegularPattern<int>(Begin: Begin + 1, CheckStride: 2, End, ExpectedIndex: 0, ExpectedIndexStride: 2))
2884 Ws = Op->getOperand(Num: 0);
2885 else if (fitsRegularPattern<int>(Begin: Begin + 1, CheckStride: 2, End, ExpectedIndex: Indices.size(), ExpectedIndexStride: 2))
2886 Ws = Op->getOperand(Num: 1);
2887 else
2888 return SDValue();
2889
2890 return DAG.getNode(Opcode: MipsISD::ILVEV, DL: SDLoc(Op), VT: ResTy, N1: Ws, N2: Wt);
2891}
2892
2893// Lower VECTOR_SHUFFLE into ILVOD (if possible).
2894//
2895// ILVOD interleaves the odd elements from each vector.
2896//
2897// It is possible to lower into ILVOD when the mask consists of two of the
2898// following forms interleaved:
2899// <1, 3, 5, ...>
2900// <n+1, n+3, n+5, ...>
2901// where n is the number of elements in the vector.
2902// For example:
2903// <1, 1, 3, 3, 5, 5, ...>
2904// <1, n+1, 3, n+3, 5, n+5, ...>
2905//
2906// When undef's appear in the mask they are treated as if they were whatever
2907// value is necessary in order to fit the above forms.
2908static SDValue lowerVECTOR_SHUFFLE_ILVOD(SDValue Op, EVT ResTy,
2909 SmallVector<int, 16> Indices,
2910 SelectionDAG &DAG) {
2911 assert((Indices.size() % 2) == 0);
2912
2913 SDValue Wt;
2914 SDValue Ws;
2915 const auto &Begin = Indices.begin();
2916 const auto &End = Indices.end();
2917
2918 // Check even elements are taken from the odd elements of one half or the
2919 // other and pick an operand accordingly.
2920 if (fitsRegularPattern<int>(Begin, CheckStride: 2, End, ExpectedIndex: 1, ExpectedIndexStride: 2))
2921 Wt = Op->getOperand(Num: 0);
2922 else if (fitsRegularPattern<int>(Begin, CheckStride: 2, End, ExpectedIndex: Indices.size() + 1, ExpectedIndexStride: 2))
2923 Wt = Op->getOperand(Num: 1);
2924 else
2925 return SDValue();
2926
2927 // Check odd elements are taken from the odd elements of one half or the
2928 // other and pick an operand accordingly.
2929 if (fitsRegularPattern<int>(Begin: Begin + 1, CheckStride: 2, End, ExpectedIndex: 1, ExpectedIndexStride: 2))
2930 Ws = Op->getOperand(Num: 0);
2931 else if (fitsRegularPattern<int>(Begin: Begin + 1, CheckStride: 2, End, ExpectedIndex: Indices.size() + 1, ExpectedIndexStride: 2))
2932 Ws = Op->getOperand(Num: 1);
2933 else
2934 return SDValue();
2935
2936 return DAG.getNode(Opcode: MipsISD::ILVOD, DL: SDLoc(Op), VT: ResTy, N1: Ws, N2: Wt);
2937}
2938
2939// Lower VECTOR_SHUFFLE into ILVR (if possible).
2940//
2941// ILVR interleaves consecutive elements from the right (lowest-indexed) half of
2942// each vector.
2943//
2944// It is possible to lower into ILVR when the mask consists of two of the
2945// following forms interleaved:
2946// <0, 1, 2, ...>
2947// <n, n+1, n+2, ...>
2948// where n is the number of elements in the vector.
2949// For example:
2950// <0, 0, 1, 1, 2, 2, ...>
2951// <0, n, 1, n+1, 2, n+2, ...>
2952//
2953// When undef's appear in the mask they are treated as if they were whatever
2954// value is necessary in order to fit the above forms.
2955static SDValue lowerVECTOR_SHUFFLE_ILVR(SDValue Op, EVT ResTy,
2956 SmallVector<int, 16> Indices,
2957 SelectionDAG &DAG) {
2958 assert((Indices.size() % 2) == 0);
2959
2960 SDValue Wt;
2961 SDValue Ws;
2962 const auto &Begin = Indices.begin();
2963 const auto &End = Indices.end();
2964
2965 // Check even elements are taken from the right (lowest-indexed) elements of
2966 // one half or the other and pick an operand accordingly.
2967 if (fitsRegularPattern<int>(Begin, CheckStride: 2, End, ExpectedIndex: 0, ExpectedIndexStride: 1))
2968 Wt = Op->getOperand(Num: 0);
2969 else if (fitsRegularPattern<int>(Begin, CheckStride: 2, End, ExpectedIndex: Indices.size(), ExpectedIndexStride: 1))
2970 Wt = Op->getOperand(Num: 1);
2971 else
2972 return SDValue();
2973
2974 // Check odd elements are taken from the right (lowest-indexed) elements of
2975 // one half or the other and pick an operand accordingly.
2976 if (fitsRegularPattern<int>(Begin: Begin + 1, CheckStride: 2, End, ExpectedIndex: 0, ExpectedIndexStride: 1))
2977 Ws = Op->getOperand(Num: 0);
2978 else if (fitsRegularPattern<int>(Begin: Begin + 1, CheckStride: 2, End, ExpectedIndex: Indices.size(), ExpectedIndexStride: 1))
2979 Ws = Op->getOperand(Num: 1);
2980 else
2981 return SDValue();
2982
2983 return DAG.getNode(Opcode: MipsISD::ILVR, DL: SDLoc(Op), VT: ResTy, N1: Ws, N2: Wt);
2984}
2985
2986// Lower VECTOR_SHUFFLE into ILVL (if possible).
2987//
2988// ILVL interleaves consecutive elements from the left (highest-indexed) half
2989// of each vector.
2990//
2991// It is possible to lower into ILVL when the mask consists of two of the
2992// following forms interleaved:
2993// <x, x+1, x+2, ...>
2994// <n+x, n+x+1, n+x+2, ...>
2995// where n is the number of elements in the vector and x is half n.
2996// For example:
2997// <x, x, x+1, x+1, x+2, x+2, ...>
2998// <x, n+x, x+1, n+x+1, x+2, n+x+2, ...>
2999//
3000// When undef's appear in the mask they are treated as if they were whatever
3001// value is necessary in order to fit the above forms.
3002static SDValue lowerVECTOR_SHUFFLE_ILVL(SDValue Op, EVT ResTy,
3003 SmallVector<int, 16> Indices,
3004 SelectionDAG &DAG) {
3005 assert((Indices.size() % 2) == 0);
3006
3007 unsigned HalfSize = Indices.size() / 2;
3008 SDValue Wt;
3009 SDValue Ws;
3010 const auto &Begin = Indices.begin();
3011 const auto &End = Indices.end();
3012
3013 // Check even elements are taken from the left (highest-indexed) elements of
3014 // one half or the other and pick an operand accordingly.
3015 if (fitsRegularPattern<int>(Begin, CheckStride: 2, End, ExpectedIndex: HalfSize, ExpectedIndexStride: 1))
3016 Wt = Op->getOperand(Num: 0);
3017 else if (fitsRegularPattern<int>(Begin, CheckStride: 2, End, ExpectedIndex: Indices.size() + HalfSize, ExpectedIndexStride: 1))
3018 Wt = Op->getOperand(Num: 1);
3019 else
3020 return SDValue();
3021
3022 // Check odd elements are taken from the left (highest-indexed) elements of
3023 // one half or the other and pick an operand accordingly.
3024 if (fitsRegularPattern<int>(Begin: Begin + 1, CheckStride: 2, End, ExpectedIndex: HalfSize, ExpectedIndexStride: 1))
3025 Ws = Op->getOperand(Num: 0);
3026 else if (fitsRegularPattern<int>(Begin: Begin + 1, CheckStride: 2, End, ExpectedIndex: Indices.size() + HalfSize,
3027 ExpectedIndexStride: 1))
3028 Ws = Op->getOperand(Num: 1);
3029 else
3030 return SDValue();
3031
3032 return DAG.getNode(Opcode: MipsISD::ILVL, DL: SDLoc(Op), VT: ResTy, N1: Ws, N2: Wt);
3033}
3034
3035// Lower VECTOR_SHUFFLE into PCKEV (if possible).
3036//
3037// PCKEV copies the even elements of each vector into the result vector.
3038//
3039// It is possible to lower into PCKEV when the mask consists of two of the
3040// following forms concatenated:
3041// <0, 2, 4, ...>
3042// <n, n+2, n+4, ...>
3043// where n is the number of elements in the vector.
3044// For example:
3045// <0, 2, 4, ..., 0, 2, 4, ...>
3046// <0, 2, 4, ..., n, n+2, n+4, ...>
3047//
3048// When undef's appear in the mask they are treated as if they were whatever
3049// value is necessary in order to fit the above forms.
3050static SDValue lowerVECTOR_SHUFFLE_PCKEV(SDValue Op, EVT ResTy,
3051 SmallVector<int, 16> Indices,
3052 SelectionDAG &DAG) {
3053 assert((Indices.size() % 2) == 0);
3054
3055 SDValue Wt;
3056 SDValue Ws;
3057 const auto &Begin = Indices.begin();
3058 const auto &Mid = Indices.begin() + Indices.size() / 2;
3059 const auto &End = Indices.end();
3060
3061 if (fitsRegularPattern<int>(Begin, CheckStride: 1, End: Mid, ExpectedIndex: 0, ExpectedIndexStride: 2))
3062 Wt = Op->getOperand(Num: 0);
3063 else if (fitsRegularPattern<int>(Begin, CheckStride: 1, End: Mid, ExpectedIndex: Indices.size(), ExpectedIndexStride: 2))
3064 Wt = Op->getOperand(Num: 1);
3065 else
3066 return SDValue();
3067
3068 if (fitsRegularPattern<int>(Begin: Mid, CheckStride: 1, End, ExpectedIndex: 0, ExpectedIndexStride: 2))
3069 Ws = Op->getOperand(Num: 0);
3070 else if (fitsRegularPattern<int>(Begin: Mid, CheckStride: 1, End, ExpectedIndex: Indices.size(), ExpectedIndexStride: 2))
3071 Ws = Op->getOperand(Num: 1);
3072 else
3073 return SDValue();
3074
3075 return DAG.getNode(Opcode: MipsISD::PCKEV, DL: SDLoc(Op), VT: ResTy, N1: Ws, N2: Wt);
3076}
3077
3078// Lower VECTOR_SHUFFLE into PCKOD (if possible).
3079//
3080// PCKOD copies the odd elements of each vector into the result vector.
3081//
3082// It is possible to lower into PCKOD when the mask consists of two of the
3083// following forms concatenated:
3084// <1, 3, 5, ...>
3085// <n+1, n+3, n+5, ...>
3086// where n is the number of elements in the vector.
3087// For example:
3088// <1, 3, 5, ..., 1, 3, 5, ...>
3089// <1, 3, 5, ..., n+1, n+3, n+5, ...>
3090//
3091// When undef's appear in the mask they are treated as if they were whatever
3092// value is necessary in order to fit the above forms.
3093static SDValue lowerVECTOR_SHUFFLE_PCKOD(SDValue Op, EVT ResTy,
3094 SmallVector<int, 16> Indices,
3095 SelectionDAG &DAG) {
3096 assert((Indices.size() % 2) == 0);
3097
3098 SDValue Wt;
3099 SDValue Ws;
3100 const auto &Begin = Indices.begin();
3101 const auto &Mid = Indices.begin() + Indices.size() / 2;
3102 const auto &End = Indices.end();
3103
3104 if (fitsRegularPattern<int>(Begin, CheckStride: 1, End: Mid, ExpectedIndex: 1, ExpectedIndexStride: 2))
3105 Wt = Op->getOperand(Num: 0);
3106 else if (fitsRegularPattern<int>(Begin, CheckStride: 1, End: Mid, ExpectedIndex: Indices.size() + 1, ExpectedIndexStride: 2))
3107 Wt = Op->getOperand(Num: 1);
3108 else
3109 return SDValue();
3110
3111 if (fitsRegularPattern<int>(Begin: Mid, CheckStride: 1, End, ExpectedIndex: 1, ExpectedIndexStride: 2))
3112 Ws = Op->getOperand(Num: 0);
3113 else if (fitsRegularPattern<int>(Begin: Mid, CheckStride: 1, End, ExpectedIndex: Indices.size() + 1, ExpectedIndexStride: 2))
3114 Ws = Op->getOperand(Num: 1);
3115 else
3116 return SDValue();
3117
3118 return DAG.getNode(Opcode: MipsISD::PCKOD, DL: SDLoc(Op), VT: ResTy, N1: Ws, N2: Wt);
3119}
3120
3121// Lower VECTOR_SHUFFLE into VSHF.
3122//
3123// This mostly consists of converting the shuffle indices in Indices into a
3124// BUILD_VECTOR and adding it as an operand to the resulting VSHF. There is
3125// also code to eliminate unused operands of the VECTOR_SHUFFLE. For example,
3126// if the type is v8i16 and all the indices are less than 8 then the second
3127// operand is unused and can be replaced with anything. We choose to replace it
3128// with the used operand since this reduces the number of instructions overall.
3129//
3130// NOTE: SPLATI shuffle masks may contain UNDEFs, since isSPLATI() treats
3131// UNDEFs as same as SPLATI index.
3132// For other instances we use the last valid index if UNDEF is
3133// encountered.
3134static SDValue lowerVECTOR_SHUFFLE_VSHF(SDValue Op, EVT ResTy,
3135 const SmallVector<int, 16> &Indices,
3136 const bool isSPLATI,
3137 SelectionDAG &DAG) {
3138 SmallVector<SDValue, 16> Ops;
3139 SDValue Op0;
3140 SDValue Op1;
3141 EVT MaskVecTy = ResTy.changeVectorElementTypeToInteger();
3142 EVT MaskEltTy = MaskVecTy.getVectorElementType();
3143 bool Using1stVec = false;
3144 bool Using2ndVec = false;
3145 SDLoc DL(Op);
3146 int ResTyNumElts = ResTy.getVectorNumElements();
3147
3148 for (int i = 0; i < ResTyNumElts; ++i) {
3149 // Idx == -1 means UNDEF/poison
3150 int Idx = Indices[i];
3151
3152 if (0 <= Idx && Idx < ResTyNumElts)
3153 Using1stVec = true;
3154 if (ResTyNumElts <= Idx && Idx < ResTyNumElts * 2)
3155 Using2ndVec = true;
3156 }
3157
3158 // Find the first non-undef index. This index is used as a default when there
3159 // is a leading UNDEF/poison.
3160 int SplatIndex = 0;
3161 for (int Idx : Indices)
3162 if (Idx >= 0) {
3163 SplatIndex = Idx;
3164 break;
3165 }
3166
3167 int LastValidIndex = SplatIndex;
3168 for (size_t i = 0; i < Indices.size(); i++) {
3169 int Idx = Indices[i];
3170 if (Idx < 0) {
3171 // Continue using splati index or use the last valid index.
3172 Idx = isSPLATI ? SplatIndex : LastValidIndex;
3173 } else {
3174 LastValidIndex = Idx;
3175 }
3176 Ops.push_back(Elt: DAG.getTargetConstant(Val: Idx, DL, VT: MaskEltTy));
3177 }
3178
3179 SDValue MaskVec = DAG.getBuildVector(VT: MaskVecTy, DL, Ops);
3180
3181 if (Using1stVec && Using2ndVec) {
3182 Op0 = Op->getOperand(Num: 0);
3183 Op1 = Op->getOperand(Num: 1);
3184 } else if (Using1stVec)
3185 Op0 = Op1 = Op->getOperand(Num: 0);
3186 else if (Using2ndVec)
3187 Op0 = Op1 = Op->getOperand(Num: 1);
3188 else
3189 llvm_unreachable("shuffle vector mask references neither vector operand?");
3190
3191 // VECTOR_SHUFFLE concatenates the vectors in an vectorwise fashion.
3192 // <0b00, 0b01> + <0b10, 0b11> -> <0b00, 0b01, 0b10, 0b11>
3193 // VSHF concatenates the vectors in a bitwise fashion:
3194 // <0b00, 0b01> + <0b10, 0b11> ->
3195 // 0b0100 + 0b1110 -> 0b01001110
3196 // <0b10, 0b11, 0b00, 0b01>
3197 // We must therefore swap the operands to get the correct result.
3198 return DAG.getNode(Opcode: MipsISD::VSHF, DL, VT: ResTy, N1: MaskVec, N2: Op1, N3: Op0);
3199}
3200
3201// Lower VECTOR_SHUFFLE into one of a number of instructions depending on the
3202// indices in the shuffle.
3203SDValue MipsSETargetLowering::lowerVECTOR_SHUFFLE(SDValue Op,
3204 SelectionDAG &DAG) const {
3205 ShuffleVectorSDNode *Node = cast<ShuffleVectorSDNode>(Val&: Op);
3206 EVT ResTy = Op->getValueType(ResNo: 0);
3207
3208 if (!ResTy.is128BitVector())
3209 return SDValue();
3210
3211 int ResTyNumElts = ResTy.getVectorNumElements();
3212 SmallVector<int, 16> Indices;
3213
3214 for (int i = 0; i < ResTyNumElts; ++i)
3215 Indices.push_back(Elt: Node->getMaskElt(Idx: i));
3216
3217 // splati.[bhwd] is preferable to the others but is matched from
3218 // MipsISD::VSHF.
3219 if (isVECTOR_SHUFFLE_SPLATI(Op, ResTy, Indices, DAG))
3220 return lowerVECTOR_SHUFFLE_VSHF(Op, ResTy, Indices, isSPLATI: true, DAG);
3221 SDValue Result;
3222 if ((Result = lowerVECTOR_SHUFFLE_ILVEV(Op, ResTy, Indices, DAG)))
3223 return Result;
3224 if ((Result = lowerVECTOR_SHUFFLE_ILVOD(Op, ResTy, Indices, DAG)))
3225 return Result;
3226 if ((Result = lowerVECTOR_SHUFFLE_ILVL(Op, ResTy, Indices, DAG)))
3227 return Result;
3228 if ((Result = lowerVECTOR_SHUFFLE_ILVR(Op, ResTy, Indices, DAG)))
3229 return Result;
3230 if ((Result = lowerVECTOR_SHUFFLE_PCKEV(Op, ResTy, Indices, DAG)))
3231 return Result;
3232 if ((Result = lowerVECTOR_SHUFFLE_PCKOD(Op, ResTy, Indices, DAG)))
3233 return Result;
3234 if ((Result = lowerVECTOR_SHUFFLE_SHF(Op, ResTy, Indices, DAG)))
3235 return Result;
3236 return lowerVECTOR_SHUFFLE_VSHF(Op, ResTy, Indices, isSPLATI: false, DAG);
3237}
3238
3239MachineBasicBlock *
3240MipsSETargetLowering::emitBPOSGE32(MachineInstr &MI,
3241 MachineBasicBlock *BB) const {
3242 // $bb:
3243 // bposge32_pseudo $vr0
3244 // =>
3245 // $bb:
3246 // bposge32 $tbb
3247 // $fbb:
3248 // li $vr2, 0
3249 // b $sink
3250 // $tbb:
3251 // li $vr1, 1
3252 // $sink:
3253 // $vr0 = phi($vr2, $fbb, $vr1, $tbb)
3254
3255 MachineRegisterInfo &RegInfo = BB->getParent()->getRegInfo();
3256 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
3257 const TargetRegisterClass *RC = &Mips::GPR32RegClass;
3258 DebugLoc DL = MI.getDebugLoc();
3259 const BasicBlock *LLVM_BB = BB->getBasicBlock();
3260 MachineFunction::iterator It = std::next(x: MachineFunction::iterator(BB));
3261 MachineFunction *F = BB->getParent();
3262 MachineBasicBlock *FBB = F->CreateMachineBasicBlock(BB: LLVM_BB);
3263 MachineBasicBlock *TBB = F->CreateMachineBasicBlock(BB: LLVM_BB);
3264 MachineBasicBlock *Sink = F->CreateMachineBasicBlock(BB: LLVM_BB);
3265 F->insert(MBBI: It, MBB: FBB);
3266 F->insert(MBBI: It, MBB: TBB);
3267 F->insert(MBBI: It, MBB: Sink);
3268
3269 // Transfer the remainder of BB and its successor edges to Sink.
3270 Sink->splice(Where: Sink->begin(), Other: BB, From: std::next(x: MachineBasicBlock::iterator(MI)),
3271 To: BB->end());
3272 Sink->transferSuccessorsAndUpdatePHIs(FromMBB: BB);
3273
3274 // Add successors.
3275 BB->addSuccessor(Succ: FBB);
3276 BB->addSuccessor(Succ: TBB);
3277 FBB->addSuccessor(Succ: Sink);
3278 TBB->addSuccessor(Succ: Sink);
3279
3280 // Insert the real bposge32 instruction to $BB.
3281 BuildMI(BB, MIMD: DL, MCID: TII->get(Opcode: Mips::BPOSGE32)).addMBB(MBB: TBB);
3282 // Insert the real bposge32c instruction to $BB.
3283 BuildMI(BB, MIMD: DL, MCID: TII->get(Opcode: Mips::BPOSGE32C_MMR3)).addMBB(MBB: TBB);
3284
3285 // Fill $FBB.
3286 Register VR2 = RegInfo.createVirtualRegister(RegClass: RC);
3287 BuildMI(BB&: *FBB, I: FBB->end(), MIMD: DL, MCID: TII->get(Opcode: Mips::ADDiu), DestReg: VR2)
3288 .addReg(RegNo: Mips::ZERO).addImm(Val: 0);
3289 BuildMI(BB&: *FBB, I: FBB->end(), MIMD: DL, MCID: TII->get(Opcode: Mips::B)).addMBB(MBB: Sink);
3290
3291 // Fill $TBB.
3292 Register VR1 = RegInfo.createVirtualRegister(RegClass: RC);
3293 BuildMI(BB&: *TBB, I: TBB->end(), MIMD: DL, MCID: TII->get(Opcode: Mips::ADDiu), DestReg: VR1)
3294 .addReg(RegNo: Mips::ZERO).addImm(Val: 1);
3295
3296 // Insert phi function to $Sink.
3297 BuildMI(BB&: *Sink, I: Sink->begin(), MIMD: DL, MCID: TII->get(Opcode: Mips::PHI),
3298 DestReg: MI.getOperand(i: 0).getReg())
3299 .addReg(RegNo: VR2)
3300 .addMBB(MBB: FBB)
3301 .addReg(RegNo: VR1)
3302 .addMBB(MBB: TBB);
3303
3304 MI.eraseFromParent(); // The pseudo instruction is gone now.
3305 return Sink;
3306}
3307
3308MachineBasicBlock *MipsSETargetLowering::emitMSACBranchPseudo(
3309 MachineInstr &MI, MachineBasicBlock *BB, unsigned BranchOp) const {
3310 // $bb:
3311 // vany_nonzero $rd, $ws
3312 // =>
3313 // $bb:
3314 // bnz.b $ws, $tbb
3315 // b $fbb
3316 // $fbb:
3317 // li $rd1, 0
3318 // b $sink
3319 // $tbb:
3320 // li $rd2, 1
3321 // $sink:
3322 // $rd = phi($rd1, $fbb, $rd2, $tbb)
3323
3324 MachineRegisterInfo &RegInfo = BB->getParent()->getRegInfo();
3325 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
3326 const TargetRegisterClass *RC = &Mips::GPR32RegClass;
3327 DebugLoc DL = MI.getDebugLoc();
3328 const BasicBlock *LLVM_BB = BB->getBasicBlock();
3329 MachineFunction::iterator It = std::next(x: MachineFunction::iterator(BB));
3330 MachineFunction *F = BB->getParent();
3331 MachineBasicBlock *FBB = F->CreateMachineBasicBlock(BB: LLVM_BB);
3332 MachineBasicBlock *TBB = F->CreateMachineBasicBlock(BB: LLVM_BB);
3333 MachineBasicBlock *Sink = F->CreateMachineBasicBlock(BB: LLVM_BB);
3334 F->insert(MBBI: It, MBB: FBB);
3335 F->insert(MBBI: It, MBB: TBB);
3336 F->insert(MBBI: It, MBB: Sink);
3337
3338 // Transfer the remainder of BB and its successor edges to Sink.
3339 Sink->splice(Where: Sink->begin(), Other: BB, From: std::next(x: MachineBasicBlock::iterator(MI)),
3340 To: BB->end());
3341 Sink->transferSuccessorsAndUpdatePHIs(FromMBB: BB);
3342
3343 // Add successors.
3344 BB->addSuccessor(Succ: FBB);
3345 BB->addSuccessor(Succ: TBB);
3346 FBB->addSuccessor(Succ: Sink);
3347 TBB->addSuccessor(Succ: Sink);
3348
3349 // Insert the real bnz.b instruction to $BB.
3350 BuildMI(BB, MIMD: DL, MCID: TII->get(Opcode: BranchOp))
3351 .addReg(RegNo: MI.getOperand(i: 1).getReg())
3352 .addMBB(MBB: TBB);
3353
3354 // Fill $FBB.
3355 Register RD1 = RegInfo.createVirtualRegister(RegClass: RC);
3356 BuildMI(BB&: *FBB, I: FBB->end(), MIMD: DL, MCID: TII->get(Opcode: Mips::ADDiu), DestReg: RD1)
3357 .addReg(RegNo: Mips::ZERO).addImm(Val: 0);
3358 BuildMI(BB&: *FBB, I: FBB->end(), MIMD: DL, MCID: TII->get(Opcode: Mips::B)).addMBB(MBB: Sink);
3359
3360 // Fill $TBB.
3361 Register RD2 = RegInfo.createVirtualRegister(RegClass: RC);
3362 BuildMI(BB&: *TBB, I: TBB->end(), MIMD: DL, MCID: TII->get(Opcode: Mips::ADDiu), DestReg: RD2)
3363 .addReg(RegNo: Mips::ZERO).addImm(Val: 1);
3364
3365 // Insert phi function to $Sink.
3366 BuildMI(BB&: *Sink, I: Sink->begin(), MIMD: DL, MCID: TII->get(Opcode: Mips::PHI),
3367 DestReg: MI.getOperand(i: 0).getReg())
3368 .addReg(RegNo: RD1)
3369 .addMBB(MBB: FBB)
3370 .addReg(RegNo: RD2)
3371 .addMBB(MBB: TBB);
3372
3373 MI.eraseFromParent(); // The pseudo instruction is gone now.
3374 return Sink;
3375}
3376
3377// Emit the COPY_FW pseudo instruction.
3378//
3379// copy_fw_pseudo $fd, $ws, n
3380// =>
3381// copy_u_w $rt, $ws, $n
3382// mtc1 $rt, $fd
3383//
3384// When n is zero, the equivalent operation can be performed with (potentially)
3385// zero instructions due to register overlaps. This optimization is never valid
3386// for lane 1 because it would require FR=0 mode which isn't supported by MSA.
3387MachineBasicBlock *
3388MipsSETargetLowering::emitCOPY_FW(MachineInstr &MI,
3389 MachineBasicBlock *BB) const {
3390 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
3391 MachineRegisterInfo &RegInfo = BB->getParent()->getRegInfo();
3392 DebugLoc DL = MI.getDebugLoc();
3393 Register Fd = MI.getOperand(i: 0).getReg();
3394 Register Ws = MI.getOperand(i: 1).getReg();
3395 unsigned Lane = MI.getOperand(i: 2).getImm();
3396
3397 if (Lane == 0) {
3398 unsigned Wt = Ws;
3399 if (!Subtarget.useOddSPReg()) {
3400 // We must copy to an even-numbered MSA register so that the
3401 // single-precision sub-register is also guaranteed to be even-numbered.
3402 Wt = RegInfo.createVirtualRegister(RegClass: &Mips::MSA128WEvensRegClass);
3403
3404 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::COPY), DestReg: Wt).addReg(RegNo: Ws);
3405 }
3406
3407 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::COPY), DestReg: Fd).addReg(RegNo: Wt, Flags: {}, SubReg: Mips::sub_lo);
3408 } else {
3409 Register Wt = RegInfo.createVirtualRegister(
3410 RegClass: Subtarget.useOddSPReg() ? &Mips::MSA128WRegClass
3411 : &Mips::MSA128WEvensRegClass);
3412
3413 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::SPLATI_W), DestReg: Wt).addReg(RegNo: Ws).addImm(Val: Lane);
3414 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::COPY), DestReg: Fd).addReg(RegNo: Wt, Flags: {}, SubReg: Mips::sub_lo);
3415 }
3416
3417 MI.eraseFromParent(); // The pseudo instruction is gone now.
3418 return BB;
3419}
3420
3421// Emit the COPY_FD pseudo instruction.
3422//
3423// copy_fd_pseudo $fd, $ws, n
3424// =>
3425// splati.d $wt, $ws, $n
3426// copy $fd, $wt:sub_64
3427//
3428// When n is zero, the equivalent operation can be performed with (potentially)
3429// zero instructions due to register overlaps. This optimization is always
3430// valid because FR=1 mode which is the only supported mode in MSA.
3431MachineBasicBlock *
3432MipsSETargetLowering::emitCOPY_FD(MachineInstr &MI,
3433 MachineBasicBlock *BB) const {
3434 assert(Subtarget.isFP64bit());
3435
3436 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
3437 MachineRegisterInfo &RegInfo = BB->getParent()->getRegInfo();
3438 Register Fd = MI.getOperand(i: 0).getReg();
3439 Register Ws = MI.getOperand(i: 1).getReg();
3440 unsigned Lane = MI.getOperand(i: 2).getImm() * 2;
3441 DebugLoc DL = MI.getDebugLoc();
3442
3443 if (Lane == 0)
3444 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::COPY), DestReg: Fd).addReg(RegNo: Ws, Flags: {}, SubReg: Mips::sub_64);
3445 else {
3446 Register Wt = RegInfo.createVirtualRegister(RegClass: &Mips::MSA128DRegClass);
3447
3448 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::SPLATI_D), DestReg: Wt).addReg(RegNo: Ws).addImm(Val: 1);
3449 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::COPY), DestReg: Fd).addReg(RegNo: Wt, Flags: {}, SubReg: Mips::sub_64);
3450 }
3451
3452 MI.eraseFromParent(); // The pseudo instruction is gone now.
3453 return BB;
3454}
3455
3456// Emit the INSERT_FW pseudo instruction.
3457//
3458// insert_fw_pseudo $wd, $wd_in, $n, $fs
3459// =>
3460// subreg_to_reg $wt:sub_lo, $fs
3461// insve_w $wd[$n], $wd_in, $wt[0]
3462MachineBasicBlock *
3463MipsSETargetLowering::emitINSERT_FW(MachineInstr &MI,
3464 MachineBasicBlock *BB) const {
3465 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
3466 MachineRegisterInfo &RegInfo = BB->getParent()->getRegInfo();
3467 DebugLoc DL = MI.getDebugLoc();
3468 Register Wd = MI.getOperand(i: 0).getReg();
3469 Register Wd_in = MI.getOperand(i: 1).getReg();
3470 unsigned Lane = MI.getOperand(i: 2).getImm();
3471 Register Fs = MI.getOperand(i: 3).getReg();
3472 Register Wt = RegInfo.createVirtualRegister(
3473 RegClass: Subtarget.useOddSPReg() ? &Mips::MSA128WRegClass
3474 : &Mips::MSA128WEvensRegClass);
3475
3476 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::SUBREG_TO_REG), DestReg: Wt)
3477 .addReg(RegNo: Fs)
3478 .addImm(Val: Mips::sub_lo);
3479 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::INSVE_W), DestReg: Wd)
3480 .addReg(RegNo: Wd_in)
3481 .addImm(Val: Lane)
3482 .addReg(RegNo: Wt)
3483 .addImm(Val: 0);
3484
3485 MI.eraseFromParent(); // The pseudo instruction is gone now.
3486 return BB;
3487}
3488
3489// Emit the INSERT_FD pseudo instruction.
3490//
3491// insert_fd_pseudo $wd, $fs, n
3492// =>
3493// subreg_to_reg $wt:sub_64, $fs
3494// insve_d $wd[$n], $wd_in, $wt[0]
3495MachineBasicBlock *
3496MipsSETargetLowering::emitINSERT_FD(MachineInstr &MI,
3497 MachineBasicBlock *BB) const {
3498 assert(Subtarget.isFP64bit());
3499
3500 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
3501 MachineRegisterInfo &RegInfo = BB->getParent()->getRegInfo();
3502 DebugLoc DL = MI.getDebugLoc();
3503 Register Wd = MI.getOperand(i: 0).getReg();
3504 Register Wd_in = MI.getOperand(i: 1).getReg();
3505 unsigned Lane = MI.getOperand(i: 2).getImm();
3506 Register Fs = MI.getOperand(i: 3).getReg();
3507 Register Wt = RegInfo.createVirtualRegister(RegClass: &Mips::MSA128DRegClass);
3508
3509 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::SUBREG_TO_REG), DestReg: Wt)
3510 .addReg(RegNo: Fs)
3511 .addImm(Val: Mips::sub_64);
3512 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::INSVE_D), DestReg: Wd)
3513 .addReg(RegNo: Wd_in)
3514 .addImm(Val: Lane)
3515 .addReg(RegNo: Wt)
3516 .addImm(Val: 0);
3517
3518 MI.eraseFromParent(); // The pseudo instruction is gone now.
3519 return BB;
3520}
3521
3522// Emit the INSERT_([BHWD]|F[WD])_VIDX pseudo instruction.
3523//
3524// For integer:
3525// (INSERT_([BHWD]|F[WD])_PSEUDO $wd, $wd_in, $n, $rs)
3526// =>
3527// (SLL $lanetmp1, $lane, <log2size)
3528// (SLD_B $wdtmp1, $wd_in, $wd_in, $lanetmp1)
3529// (INSERT_[BHWD], $wdtmp2, $wdtmp1, 0, $rs)
3530// (NEG $lanetmp2, $lanetmp1)
3531// (SLD_B $wd, $wdtmp2, $wdtmp2, $lanetmp2)
3532//
3533// For floating point:
3534// (INSERT_([BHWD]|F[WD])_PSEUDO $wd, $wd_in, $n, $fs)
3535// =>
3536// (SUBREG_TO_REG $wt, $fs, <subreg>)
3537// (SLL $lanetmp1, $lane, <log2size)
3538// (SLD_B $wdtmp1, $wd_in, $wd_in, $lanetmp1)
3539// (INSVE_[WD], $wdtmp2, 0, $wdtmp1, 0)
3540// (NEG $lanetmp2, $lanetmp1)
3541// (SLD_B $wd, $wdtmp2, $wdtmp2, $lanetmp2)
3542MachineBasicBlock *MipsSETargetLowering::emitINSERT_DF_VIDX(
3543 MachineInstr &MI, MachineBasicBlock *BB, unsigned EltSizeInBytes,
3544 bool IsFP) const {
3545 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
3546 MachineRegisterInfo &RegInfo = BB->getParent()->getRegInfo();
3547 DebugLoc DL = MI.getDebugLoc();
3548 Register Wd = MI.getOperand(i: 0).getReg();
3549 Register SrcVecReg = MI.getOperand(i: 1).getReg();
3550 Register LaneReg = MI.getOperand(i: 2).getReg();
3551 Register SrcValReg = MI.getOperand(i: 3).getReg();
3552
3553 const TargetRegisterClass *VecRC = nullptr;
3554 // FIXME: This should be true for N32 too.
3555 const TargetRegisterClass *GPRRC =
3556 Subtarget.isABI_N64() ? &Mips::GPR64RegClass : &Mips::GPR32RegClass;
3557 unsigned SubRegIdx = Subtarget.isABI_N64() ? Mips::sub_32 : 0;
3558 unsigned ShiftOp = Subtarget.isABI_N64() ? Mips::DSLL : Mips::SLL;
3559 unsigned EltLog2Size;
3560 unsigned InsertOp = 0;
3561 unsigned InsveOp = 0;
3562 switch (EltSizeInBytes) {
3563 default:
3564 llvm_unreachable("Unexpected size");
3565 case 1:
3566 EltLog2Size = 0;
3567 InsertOp = Mips::INSERT_B;
3568 InsveOp = Mips::INSVE_B;
3569 VecRC = &Mips::MSA128BRegClass;
3570 break;
3571 case 2:
3572 EltLog2Size = 1;
3573 InsertOp = Mips::INSERT_H;
3574 InsveOp = Mips::INSVE_H;
3575 VecRC = &Mips::MSA128HRegClass;
3576 break;
3577 case 4:
3578 EltLog2Size = 2;
3579 InsertOp = Mips::INSERT_W;
3580 InsveOp = Mips::INSVE_W;
3581 VecRC = &Mips::MSA128WRegClass;
3582 break;
3583 case 8:
3584 EltLog2Size = 3;
3585 InsertOp = Mips::INSERT_D;
3586 InsveOp = Mips::INSVE_D;
3587 VecRC = &Mips::MSA128DRegClass;
3588 break;
3589 }
3590
3591 if (IsFP) {
3592 Register Wt = RegInfo.createVirtualRegister(RegClass: VecRC);
3593 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::SUBREG_TO_REG), DestReg: Wt)
3594 .addReg(RegNo: SrcValReg)
3595 .addImm(Val: EltSizeInBytes == 8 ? Mips::sub_64 : Mips::sub_lo);
3596 SrcValReg = Wt;
3597 }
3598
3599 // Convert the lane index into a byte index
3600 if (EltSizeInBytes != 1) {
3601 Register LaneTmp1 = RegInfo.createVirtualRegister(RegClass: GPRRC);
3602 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: ShiftOp), DestReg: LaneTmp1)
3603 .addReg(RegNo: LaneReg)
3604 .addImm(Val: EltLog2Size);
3605 LaneReg = LaneTmp1;
3606 }
3607
3608 // Rotate bytes around so that the desired lane is element zero
3609 Register WdTmp1 = RegInfo.createVirtualRegister(RegClass: VecRC);
3610 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::SLD_B), DestReg: WdTmp1)
3611 .addReg(RegNo: SrcVecReg)
3612 .addReg(RegNo: SrcVecReg)
3613 .addReg(RegNo: LaneReg, Flags: {}, SubReg: SubRegIdx);
3614
3615 Register WdTmp2 = RegInfo.createVirtualRegister(RegClass: VecRC);
3616 if (IsFP) {
3617 // Use insve.df to insert to element zero
3618 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: InsveOp), DestReg: WdTmp2)
3619 .addReg(RegNo: WdTmp1)
3620 .addImm(Val: 0)
3621 .addReg(RegNo: SrcValReg)
3622 .addImm(Val: 0);
3623 } else {
3624 // Use insert.df to insert to element zero
3625 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: InsertOp), DestReg: WdTmp2)
3626 .addReg(RegNo: WdTmp1)
3627 .addReg(RegNo: SrcValReg)
3628 .addImm(Val: 0);
3629 }
3630
3631 // Rotate elements the rest of the way for a full rotation.
3632 // sld.df inteprets $rt modulo the number of columns so we only need to negate
3633 // the lane index to do this.
3634 Register LaneTmp2 = RegInfo.createVirtualRegister(RegClass: GPRRC);
3635 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Subtarget.isABI_N64() ? Mips::DSUB : Mips::SUB),
3636 DestReg: LaneTmp2)
3637 .addReg(RegNo: Subtarget.isABI_N64() ? Mips::ZERO_64 : Mips::ZERO)
3638 .addReg(RegNo: LaneReg);
3639 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::SLD_B), DestReg: Wd)
3640 .addReg(RegNo: WdTmp2)
3641 .addReg(RegNo: WdTmp2)
3642 .addReg(RegNo: LaneTmp2, Flags: {}, SubReg: SubRegIdx);
3643
3644 MI.eraseFromParent(); // The pseudo instruction is gone now.
3645 return BB;
3646}
3647
3648// Emit the FILL_FW pseudo instruction.
3649//
3650// fill_fw_pseudo $wd, $fs
3651// =>
3652// implicit_def $wt1
3653// insert_subreg $wt2:subreg_lo, $wt1, $fs
3654// splati.w $wd, $wt2[0]
3655MachineBasicBlock *
3656MipsSETargetLowering::emitFILL_FW(MachineInstr &MI,
3657 MachineBasicBlock *BB) const {
3658 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
3659 MachineRegisterInfo &RegInfo = BB->getParent()->getRegInfo();
3660 DebugLoc DL = MI.getDebugLoc();
3661 Register Wd = MI.getOperand(i: 0).getReg();
3662 Register Fs = MI.getOperand(i: 1).getReg();
3663 Register Wt1 = RegInfo.createVirtualRegister(
3664 RegClass: Subtarget.useOddSPReg() ? &Mips::MSA128WRegClass
3665 : &Mips::MSA128WEvensRegClass);
3666 Register Wt2 = RegInfo.createVirtualRegister(
3667 RegClass: Subtarget.useOddSPReg() ? &Mips::MSA128WRegClass
3668 : &Mips::MSA128WEvensRegClass);
3669
3670 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::IMPLICIT_DEF), DestReg: Wt1);
3671 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::INSERT_SUBREG), DestReg: Wt2)
3672 .addReg(RegNo: Wt1)
3673 .addReg(RegNo: Fs)
3674 .addImm(Val: Mips::sub_lo);
3675 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::SPLATI_W), DestReg: Wd).addReg(RegNo: Wt2).addImm(Val: 0);
3676
3677 MI.eraseFromParent(); // The pseudo instruction is gone now.
3678 return BB;
3679}
3680
3681// Emit the FILL_FD pseudo instruction.
3682//
3683// fill_fd_pseudo $wd, $fs
3684// =>
3685// implicit_def $wt1
3686// insert_subreg $wt2:subreg_64, $wt1, $fs
3687// splati.d $wd, $wt2[0]
3688MachineBasicBlock *
3689MipsSETargetLowering::emitFILL_FD(MachineInstr &MI,
3690 MachineBasicBlock *BB) const {
3691 assert(Subtarget.isFP64bit());
3692
3693 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
3694 MachineRegisterInfo &RegInfo = BB->getParent()->getRegInfo();
3695 DebugLoc DL = MI.getDebugLoc();
3696 Register Wd = MI.getOperand(i: 0).getReg();
3697 Register Fs = MI.getOperand(i: 1).getReg();
3698 Register Wt1 = RegInfo.createVirtualRegister(RegClass: &Mips::MSA128DRegClass);
3699 Register Wt2 = RegInfo.createVirtualRegister(RegClass: &Mips::MSA128DRegClass);
3700
3701 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::IMPLICIT_DEF), DestReg: Wt1);
3702 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::INSERT_SUBREG), DestReg: Wt2)
3703 .addReg(RegNo: Wt1)
3704 .addReg(RegNo: Fs)
3705 .addImm(Val: Mips::sub_64);
3706 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::SPLATI_D), DestReg: Wd).addReg(RegNo: Wt2).addImm(Val: 0);
3707
3708 MI.eraseFromParent(); // The pseudo instruction is gone now.
3709 return BB;
3710}
3711
3712// Emit the FEXP2_W_1 pseudo instructions.
3713//
3714// fexp2_w_1_pseudo $wd, $wt
3715// =>
3716// ldi.w $ws, 1
3717// fexp2.w $wd, $ws, $wt
3718MachineBasicBlock *
3719MipsSETargetLowering::emitFEXP2_W_1(MachineInstr &MI,
3720 MachineBasicBlock *BB) const {
3721 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
3722 MachineRegisterInfo &RegInfo = BB->getParent()->getRegInfo();
3723 const TargetRegisterClass *RC = &Mips::MSA128WRegClass;
3724 Register Ws1 = RegInfo.createVirtualRegister(RegClass: RC);
3725 Register Ws2 = RegInfo.createVirtualRegister(RegClass: RC);
3726 DebugLoc DL = MI.getDebugLoc();
3727
3728 // Splat 1.0 into a vector
3729 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::LDI_W), DestReg: Ws1).addImm(Val: 1);
3730 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::FFINT_U_W), DestReg: Ws2).addReg(RegNo: Ws1);
3731
3732 // Emit 1.0 * fexp2(Wt)
3733 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::FEXP2_W), DestReg: MI.getOperand(i: 0).getReg())
3734 .addReg(RegNo: Ws2)
3735 .addReg(RegNo: MI.getOperand(i: 1).getReg());
3736
3737 MI.eraseFromParent(); // The pseudo instruction is gone now.
3738 return BB;
3739}
3740
3741// Emit the FEXP2_D_1 pseudo instructions.
3742//
3743// fexp2_d_1_pseudo $wd, $wt
3744// =>
3745// ldi.d $ws, 1
3746// fexp2.d $wd, $ws, $wt
3747MachineBasicBlock *
3748MipsSETargetLowering::emitFEXP2_D_1(MachineInstr &MI,
3749 MachineBasicBlock *BB) const {
3750 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
3751 MachineRegisterInfo &RegInfo = BB->getParent()->getRegInfo();
3752 const TargetRegisterClass *RC = &Mips::MSA128DRegClass;
3753 Register Ws1 = RegInfo.createVirtualRegister(RegClass: RC);
3754 Register Ws2 = RegInfo.createVirtualRegister(RegClass: RC);
3755 DebugLoc DL = MI.getDebugLoc();
3756
3757 // Splat 1.0 into a vector
3758 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::LDI_D), DestReg: Ws1).addImm(Val: 1);
3759 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::FFINT_U_D), DestReg: Ws2).addReg(RegNo: Ws1);
3760
3761 // Emit 1.0 * fexp2(Wt)
3762 BuildMI(BB&: *BB, I&: MI, MIMD: DL, MCID: TII->get(Opcode: Mips::FEXP2_D), DestReg: MI.getOperand(i: 0).getReg())
3763 .addReg(RegNo: Ws2)
3764 .addReg(RegNo: MI.getOperand(i: 1).getReg());
3765
3766 MI.eraseFromParent(); // The pseudo instruction is gone now.
3767 return BB;
3768}
3769