1//===- ARMTargetTransformInfo.h - ARM specific TTI --------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This file a TargetTransformInfoImplBase conforming object specific to the
11/// ARM target machine. It uses the target's detailed information to
12/// provide more precise answers to certain TTI queries, while letting the
13/// target independent and default TTI implementations handle the rest.
14//
15//===----------------------------------------------------------------------===//
16
17#ifndef LLVM_LIB_TARGET_ARM_ARMTARGETTRANSFORMINFO_H
18#define LLVM_LIB_TARGET_ARM_ARMTARGETTRANSFORMINFO_H
19
20#include "ARM.h"
21#include "ARMSubtarget.h"
22#include "ARMTargetMachine.h"
23#include "llvm/ADT/ArrayRef.h"
24#include "llvm/Analysis/TargetTransformInfo.h"
25#include "llvm/CodeGen/BasicTTIImpl.h"
26#include "llvm/IR/Constant.h"
27#include "llvm/IR/Function.h"
28#include "llvm/TargetParser/SubtargetFeature.h"
29#include <optional>
30
31namespace llvm {
32
33class APInt;
34class ARMTargetLowering;
35class Instruction;
36class Loop;
37class SCEV;
38class ScalarEvolution;
39class Type;
40class Value;
41
42namespace TailPredication {
43enum Mode {
44 Disabled = 0,
45 EnabledNoReductions,
46 Enabled,
47 ForceEnabledNoReductions,
48 ForceEnabled
49};
50}
51
52// For controlling conversion of memcpy into Tail Predicated loop.
53namespace TPLoop {
54enum MemTransfer { ForceDisabled = 0, ForceEnabled, Allow };
55}
56
57class ARMTTIImpl final : public BasicTTIImplBase<ARMTTIImpl> {
58 using BaseT = BasicTTIImplBase<ARMTTIImpl>;
59 using TTI = TargetTransformInfo;
60
61 friend BaseT;
62
63 const ARMSubtarget *ST;
64 const ARMTargetLowering *TLI;
65
66 const ARMSubtarget *getST() const { return ST; }
67 const ARMTargetLowering *getTLI() const { return TLI; }
68
69public:
70 explicit ARMTTIImpl(const ARMBaseTargetMachine *TM, const Function &F)
71 : BaseT(TM, F.getDataLayout()), ST(TM->getSubtargetImpl(F)),
72 TLI(ST->getTargetLowering()) {}
73
74 bool enableInterleavedAccessVectorization() const override { return true; }
75
76 TTI::AddressingModeKind
77 getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const override;
78
79 /// Floating-point computation using ARMv8 AArch32 Advanced
80 /// SIMD instructions remains unchanged from ARMv7. Only AArch64 SIMD
81 /// and Arm MVE are IEEE-754 compliant.
82 bool isFPVectorizationPotentiallyUnsafe() const override {
83 return !ST->isTargetDarwin() && !ST->hasMVEFloatOps();
84 }
85
86 std::optional<Instruction *>
87 instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override;
88 std::optional<Value *> simplifyDemandedVectorEltsIntrinsic(
89 InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts,
90 APInt &UndefElts2, APInt &UndefElts3,
91 std::function<void(Instruction *, unsigned, APInt, APInt &)>
92 SimplifyAndSetOp) const override;
93
94 /// \name Scalar TTI Implementations
95 /// @{
96
97 InstructionCost getIntImmCodeSizeCost(unsigned Opcode, unsigned Idx,
98 const APInt &Imm,
99 Type *Ty) const override;
100
101 using BaseT::getIntImmCost;
102 InstructionCost getIntImmCost(const APInt &Imm, Type *Ty,
103 TTI::TargetCostKind CostKind) const override;
104
105 InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx,
106 const APInt &Imm, Type *Ty,
107 TTI::TargetCostKind CostKind,
108 Instruction *Inst = nullptr) const override;
109
110 /// @}
111
112 /// \name Vector TTI Implementations
113 /// @{
114
115 unsigned getNumberOfRegisters(unsigned ClassID) const override {
116 bool Vector = (ClassID == 1);
117 if (Vector) {
118 if (ST->hasNEON())
119 return 16;
120 if (ST->hasMVEIntegerOps())
121 return 8;
122 return 0;
123 }
124
125 if (ST->isThumb1Only())
126 return 8;
127 return 13;
128 }
129
130 TypeSize
131 getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override {
132 switch (K) {
133 case TargetTransformInfo::RGK_Scalar:
134 return TypeSize::getFixed(ExactSize: 32);
135 case TargetTransformInfo::RGK_FixedWidthVector:
136 if (ST->hasNEON())
137 return TypeSize::getFixed(ExactSize: 128);
138 if (ST->hasMVEIntegerOps())
139 return TypeSize::getFixed(ExactSize: 128);
140 return TypeSize::getFixed(ExactSize: 0);
141 case TargetTransformInfo::RGK_ScalableVector:
142 return TypeSize::getScalable(MinimumSize: 0);
143 }
144 llvm_unreachable("Unsupported register kind");
145 }
146
147 unsigned getMaxInterleaveFactor(ElementCount VF,
148 bool HasUnorderedReductions) const override {
149 return ST->getMaxInterleaveFactor();
150 }
151
152 bool isProfitableLSRChainElement(Instruction *I) const override;
153
154 bool
155 isLegalMaskedLoad(Type *DataTy, Align Alignment, unsigned AddressSpace,
156 TTI::MaskKind MaskKind =
157 TTI::MaskKind::VariableOrConstantMask) const override;
158
159 bool
160 isLegalMaskedStore(Type *DataTy, Align Alignment, unsigned AddressSpace,
161 TTI::MaskKind MaskKind =
162 TTI::MaskKind::VariableOrConstantMask) const override {
163 return isLegalMaskedLoad(DataTy, Alignment, AddressSpace, MaskKind);
164 }
165
166 bool forceScalarizeMaskedGather(VectorType *VTy,
167 Align Alignment) const override {
168 // For MVE, we have a custom lowering pass that will already have custom
169 // legalised any gathers that we can lower to MVE intrinsics, and want to
170 // expand all the rest. The pass runs before the masked intrinsic lowering
171 // pass.
172 return true;
173 }
174
175 bool forceScalarizeMaskedScatter(VectorType *VTy,
176 Align Alignment) const override {
177 return forceScalarizeMaskedGather(VTy, Alignment);
178 }
179
180 bool isLegalMaskedGather(Type *Ty, Align Alignment) const override;
181
182 bool isLegalMaskedScatter(Type *Ty, Align Alignment) const override {
183 return isLegalMaskedGather(Ty, Alignment);
184 }
185
186 InstructionCost getMemcpyCost(const Instruction *I) const override;
187
188 uint64_t getMaxMemIntrinsicInlineSizeThreshold() const override {
189 return ST->getMaxInlineSizeThreshold();
190 }
191
192 int getNumMemOps(const IntrinsicInst *I) const;
193
194 InstructionCost
195 getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy,
196 TTI::TargetCostKind CostKind, ArrayRef<int> Mask, int Index,
197 VectorType *SubTp, ArrayRef<const Value *> Args = {},
198 const Instruction *CtxI = nullptr,
199 TTI::VectorInstrContext VIC =
200 TTI::VectorInstrContext::None) const override;
201
202 bool preferInLoopReduction(RecurKind Kind, Type *Ty) const override;
203
204 bool preferPredicatedReductionSelect() const override;
205
206 bool shouldExpandReduction(const IntrinsicInst *II) const override {
207 return false;
208 }
209
210 InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind,
211 const Instruction *I = nullptr) const override;
212
213 InstructionCost
214 getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src,
215 TTI::CastContextHint CCH, TTI::TargetCostKind CostKind,
216 const Instruction *I = nullptr) const override;
217
218 InstructionCost getCmpSelInstrCost(
219 unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
220 TTI::TargetCostKind CostKind,
221 TTI::OperandValueInfo Op1Info = {.Kind: TTI::OK_AnyValue, .Properties: TTI::OP_None},
222 TTI::OperandValueInfo Op2Info = {.Kind: TTI::OK_AnyValue, .Properties: TTI::OP_None},
223 const Instruction *I = nullptr) const override;
224
225 using BaseT::getVectorInstrCost;
226 InstructionCost
227 getVectorInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
228 unsigned Index, const Value *Op0, const Value *Op1,
229 TTI::VectorInstrContext VIC =
230 TTI::VectorInstrContext::None) const override;
231
232 InstructionCost
233 getAddressComputationCost(Type *Ty, ScalarEvolution *SE, const SCEV *Ptr,
234 TTI::TargetCostKind CostKind) const override;
235
236 InstructionCost getArithmeticInstrCost(
237 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
238 TTI::OperandValueInfo Op1Info = {.Kind: TTI::OK_AnyValue, .Properties: TTI::OP_None},
239 TTI::OperandValueInfo Op2Info = {.Kind: TTI::OK_AnyValue, .Properties: TTI::OP_None},
240 ArrayRef<const Value *> Args = {},
241 const Instruction *CtxI = nullptr) const override;
242
243 InstructionCost getMemoryOpCost(
244 unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace,
245 TTI::TargetCostKind CostKind,
246 TTI::OperandValueInfo OpInfo = {.Kind: TTI::OK_AnyValue, .Properties: TTI::OP_None},
247 const Instruction *I = nullptr) const override;
248
249 InstructionCost
250 getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA,
251 TTI::TargetCostKind CostKind) const override;
252
253 InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA,
254 TTI::TargetCostKind CostKind) const;
255
256 InstructionCost getInterleavedMemoryOpCost(
257 unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices,
258 Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
259 bool UseMaskForCond = false, bool UseMaskForGaps = false) const override;
260
261 InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA,
262 TTI::TargetCostKind CostKind) const;
263
264 InstructionCost
265 getArithmeticReductionCost(unsigned Opcode, VectorType *ValTy,
266 std::optional<FastMathFlags> FMF,
267 TTI::TargetCostKind CostKind) const override;
268 InstructionCost
269 getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy,
270 VectorType *ValTy, std::optional<FastMathFlags> FMF,
271 TTI::TargetCostKind CostKind) const override;
272 InstructionCost
273 getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy,
274 VectorType *ValTy,
275 TTI::TargetCostKind CostKind) const override;
276
277 InstructionCost
278 getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF,
279 TTI::TargetCostKind CostKind) const override;
280
281 InstructionCost
282 getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA,
283 TTI::TargetCostKind CostKind) const override;
284
285 InstructionCost getPartialReductionCost(
286 unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType,
287 ElementCount VF, TTI::PartialReductionExtendKind OpAExtend,
288 TTI::PartialReductionExtendKind OpBExtend, std::optional<unsigned> BinOp,
289 TTI::TargetCostKind CostKind,
290 std::optional<FastMathFlags> FMF) const override {
291 return InstructionCost::getInvalid();
292 }
293
294 /// getScalingFactorCost - Return the cost of the scaling used in
295 /// addressing mode represented by AM.
296 /// If the AM is supported, the return value must be >= 0.
297 /// If the AM is not supported, the return value is an invalid cost.
298 InstructionCost getScalingFactorCost(Type *Ty, GlobalValue *BaseGV,
299 StackOffset BaseOffset, bool HasBaseReg,
300 int64_t Scale,
301 unsigned AddrSpace) const override;
302
303 bool maybeLoweredToCall(Instruction &I) const;
304 bool isLoweredToCall(const Function *F) const override;
305 bool isHardwareLoopProfitable(Loop *L, ScalarEvolution &SE,
306 AssumptionCache &AC, TargetLibraryInfo *LibInfo,
307 HardwareLoopInfo &HWLoopInfo) const override;
308 bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override;
309 void getUnrollingPreferences(Loop *L, ScalarEvolution &SE,
310 TTI::UnrollingPreferences &UP,
311 OptimizationRemarkEmitter *ORE) const override;
312
313 TailFoldingStyle getPreferredTailFoldingStyle() const override;
314
315 void getPeelingPreferences(Loop *L, ScalarEvolution &SE,
316 TTI::PeelingPreferences &PP) const override;
317 bool shouldBuildLookupTablesForConstant(Constant *C) const override {
318 // In the ROPI and RWPI relocation models we can't have pointers to global
319 // variables or functions in constant data, so don't convert switches to
320 // lookup tables if any of the values would need relocation.
321 if (ST->isROPI() || ST->isRWPI())
322 return !C->needsDynamicRelocation();
323
324 return true;
325 }
326
327 bool shouldConsiderVectorizationRegPressure() const override;
328
329 bool hasArmWideBranch(bool Thumb) const override;
330
331 bool isProfitableToSinkOperands(Instruction *I,
332 SmallVectorImpl<Use *> &Ops) const override;
333
334 unsigned getNumBytesToPadGlobalArray(unsigned Size,
335 Type *ArrayType) const override;
336
337 /// @}
338};
339
340/// isVREVMask - Check if a vector shuffle corresponds to a VREV
341/// instruction with the specified blocksize. (The order of the elements
342/// within each block of the vector is reversed.)
343inline bool isVREVMask(ArrayRef<int> M, EVT VT, unsigned BlockSize) {
344 assert((BlockSize == 16 || BlockSize == 32 || BlockSize == 64) &&
345 "Only possible block sizes for VREV are: 16, 32, 64");
346
347 unsigned EltSz = VT.getScalarSizeInBits();
348 if (EltSz != 8 && EltSz != 16 && EltSz != 32)
349 return false;
350
351 unsigned BlockElts = M[0] + 1;
352 // If the first shuffle index is UNDEF, be optimistic.
353 if (M[0] < 0)
354 BlockElts = BlockSize / EltSz;
355
356 if (BlockSize <= EltSz || BlockSize != BlockElts * EltSz)
357 return false;
358
359 for (unsigned i = 0, e = M.size(); i < e; ++i) {
360 if (M[i] < 0)
361 continue; // ignore UNDEF indices
362 if ((unsigned)M[i] != (i - i % BlockElts) + (BlockElts - 1 - i % BlockElts))
363 return false;
364 }
365
366 return true;
367}
368
369inline unsigned SelectPairHalf(unsigned Elements, ArrayRef<int> Mask,
370 unsigned Index) {
371 if (Mask.size() == Elements * 2)
372 return Index / Elements;
373 return Mask[Index] == 0 ? 0 : 1;
374}
375
376// Checks whether the shuffle mask represents a vector transpose (VTRN) by
377// checking that pairs of elements in the shuffle mask represent the same index
378// in each vector, incrementing the expected index by 2 at each step.
379// e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 4, 2, 6]
380// v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,e,c,g}
381// v2={e,f,g,h}
382// WhichResult gives the offset for each element in the mask based on which
383// of the two results it belongs to.
384//
385// The transpose can be represented either as:
386// result1 = shufflevector v1, v2, result1_shuffle_mask
387// result2 = shufflevector v1, v2, result2_shuffle_mask
388// where v1/v2 and the shuffle masks have the same number of elements
389// (here WhichResult (see below) indicates which result is being checked)
390//
391// or as:
392// results = shufflevector v1, v2, shuffle_mask
393// where both results are returned in one vector and the shuffle mask has twice
394// as many elements as v1/v2 (here WhichResult will always be 0 if true) here we
395// want to check the low half and high half of the shuffle mask as if it were
396// the other case
397inline bool isVTRNMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
398 unsigned EltSz = VT.getScalarSizeInBits();
399 if (EltSz == 64)
400 return false;
401
402 unsigned NumElts = VT.getVectorNumElements();
403 if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
404 return false;
405
406 // If the mask is twice as long as the input vector then we need to check the
407 // upper and lower parts of the mask with a matching value for WhichResult
408 // FIXME: A mask with only even values will be rejected in case the first
409 // element is undefined, e.g. [-1, 4, 2, 6] will be rejected, because only
410 // M[0] is used to determine WhichResult
411 for (unsigned i = 0; i < M.size(); i += NumElts) {
412 WhichResult = SelectPairHalf(Elements: NumElts, Mask: M, Index: i);
413 for (unsigned j = 0; j < NumElts; j += 2) {
414 if ((M[i + j] >= 0 && (unsigned)M[i + j] != j + WhichResult) ||
415 (M[i + j + 1] >= 0 &&
416 (unsigned)M[i + j + 1] != j + NumElts + WhichResult))
417 return false;
418 }
419 }
420
421 if (M.size() == NumElts * 2)
422 WhichResult = 0;
423
424 return true;
425}
426
427/// isVTRN_v_undef_Mask - Special case of isVTRNMask for canonical form of
428/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
429/// Mask is e.g., <0, 0, 2, 2> instead of <0, 4, 2, 6>.
430inline bool isVTRN_v_undef_Mask(ArrayRef<int> M, EVT VT,
431 unsigned &WhichResult) {
432 unsigned EltSz = VT.getScalarSizeInBits();
433 if (EltSz == 64)
434 return false;
435
436 unsigned NumElts = VT.getVectorNumElements();
437 if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
438 return false;
439
440 for (unsigned i = 0; i < M.size(); i += NumElts) {
441 WhichResult = SelectPairHalf(Elements: NumElts, Mask: M, Index: i);
442 for (unsigned j = 0; j < NumElts; j += 2) {
443 if ((M[i + j] >= 0 && (unsigned)M[i + j] != j + WhichResult) ||
444 (M[i + j + 1] >= 0 && (unsigned)M[i + j + 1] != j + WhichResult))
445 return false;
446 }
447 }
448
449 if (M.size() == NumElts * 2)
450 WhichResult = 0;
451
452 return true;
453}
454
455// Checks whether the shuffle mask represents a vector unzip (VUZP) by checking
456// that the mask elements are either all even and in steps of size 2 or all odd
457// and in steps of size 2.
458// e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 2, 4, 6]
459// v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,c,e,g}
460// v2={e,f,g,h}
461// Requires similar checks to that of isVTRNMask with
462// respect the how results are returned.
463inline bool isVUZPMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
464 unsigned EltSz = VT.getScalarSizeInBits();
465 if (EltSz == 64)
466 return false;
467
468 unsigned NumElts = VT.getVectorNumElements();
469 if (M.size() != NumElts && M.size() != NumElts * 2)
470 return false;
471
472 for (unsigned i = 0; i < M.size(); i += NumElts) {
473 WhichResult = SelectPairHalf(Elements: NumElts, Mask: M, Index: i);
474 for (unsigned j = 0; j < NumElts; ++j) {
475 if (M[i + j] >= 0 && (unsigned)M[i + j] != 2 * j + WhichResult)
476 return false;
477 }
478 }
479
480 if (M.size() == NumElts * 2)
481 WhichResult = 0;
482
483 // VUZP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
484 if (VT.is64BitVector() && EltSz == 32)
485 return false;
486
487 return true;
488}
489
490/// isVUZP_v_undef_Mask - Special case of isVUZPMask for canonical form of
491/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
492/// Mask is e.g., <0, 2, 0, 2> instead of <0, 2, 4, 6>,
493inline bool isVUZP_v_undef_Mask(ArrayRef<int> M, EVT VT,
494 unsigned &WhichResult) {
495 unsigned EltSz = VT.getScalarSizeInBits();
496 if (EltSz == 64)
497 return false;
498
499 unsigned NumElts = VT.getVectorNumElements();
500 if (M.size() != NumElts && M.size() != NumElts * 2)
501 return false;
502
503 unsigned Half = NumElts / 2;
504 for (unsigned i = 0; i < M.size(); i += NumElts) {
505 WhichResult = SelectPairHalf(Elements: NumElts, Mask: M, Index: i);
506 for (unsigned j = 0; j < NumElts; j += Half) {
507 unsigned Idx = WhichResult;
508 for (unsigned k = 0; k < Half; ++k) {
509 int MIdx = M[i + j + k];
510 if (MIdx >= 0 && (unsigned)MIdx != Idx)
511 return false;
512 Idx += 2;
513 }
514 }
515 }
516
517 if (M.size() == NumElts * 2)
518 WhichResult = 0;
519
520 // VUZP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
521 if (VT.is64BitVector() && EltSz == 32)
522 return false;
523
524 return true;
525}
526
527// Checks whether the shuffle mask represents a vector zip (VZIP) by checking
528// that pairs of elements of the shufflemask represent the same index in each
529// vector incrementing sequentially through the vectors.
530// e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 4, 1, 5]
531// v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,e,b,f}
532// v2={e,f,g,h}
533// Requires similar checks to that of isVTRNMask with respect the how results
534// are returned.
535inline bool isVZIPMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) {
536 unsigned EltSz = VT.getScalarSizeInBits();
537 if (EltSz == 64)
538 return false;
539
540 unsigned NumElts = VT.getVectorNumElements();
541 if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
542 return false;
543
544 for (unsigned i = 0; i < M.size(); i += NumElts) {
545 WhichResult = SelectPairHalf(Elements: NumElts, Mask: M, Index: i);
546 unsigned Idx = WhichResult * NumElts / 2;
547 for (unsigned j = 0; j < NumElts; j += 2) {
548 if ((M[i + j] >= 0 && (unsigned)M[i + j] != Idx) ||
549 (M[i + j + 1] >= 0 && (unsigned)M[i + j + 1] != Idx + NumElts))
550 return false;
551 Idx += 1;
552 }
553 }
554
555 if (M.size() == NumElts * 2)
556 WhichResult = 0;
557
558 // VZIP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
559 if (VT.is64BitVector() && EltSz == 32)
560 return false;
561
562 return true;
563}
564
565/// isVZIP_v_undef_Mask - Special case of isVZIPMask for canonical form of
566/// "vector_shuffle v, v", i.e., "vector_shuffle v, undef".
567/// Mask is e.g., <0, 0, 1, 1> instead of <0, 4, 1, 5>.
568inline bool isVZIP_v_undef_Mask(ArrayRef<int> M, EVT VT,
569 unsigned &WhichResult) {
570 unsigned EltSz = VT.getScalarSizeInBits();
571 if (EltSz == 64)
572 return false;
573
574 unsigned NumElts = VT.getVectorNumElements();
575 if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0)
576 return false;
577
578 for (unsigned i = 0; i < M.size(); i += NumElts) {
579 WhichResult = SelectPairHalf(Elements: NumElts, Mask: M, Index: i);
580 unsigned Idx = WhichResult * NumElts / 2;
581 for (unsigned j = 0; j < NumElts; j += 2) {
582 if ((M[i + j] >= 0 && (unsigned)M[i + j] != Idx) ||
583 (M[i + j + 1] >= 0 && (unsigned)M[i + j + 1] != Idx))
584 return false;
585 Idx += 1;
586 }
587 }
588
589 if (M.size() == NumElts * 2)
590 WhichResult = 0;
591
592 // VZIP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32.
593 if (VT.is64BitVector() && EltSz == 32)
594 return false;
595
596 return true;
597}
598
599} // end namespace llvm
600
601#endif // LLVM_LIB_TARGET_ARM_ARMTARGETTRANSFORMINFO_H
602