| 1 | //===- ARMTargetTransformInfo.h - ARM specific TTI --------------*- C++ -*-===// |
| 2 | // |
| 3 | // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. |
| 4 | // See https://llvm.org/LICENSE.txt for license information. |
| 5 | // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception |
| 6 | // |
| 7 | //===----------------------------------------------------------------------===// |
| 8 | // |
| 9 | /// \file |
| 10 | /// This file a TargetTransformInfoImplBase conforming object specific to the |
| 11 | /// ARM target machine. It uses the target's detailed information to |
| 12 | /// provide more precise answers to certain TTI queries, while letting the |
| 13 | /// target independent and default TTI implementations handle the rest. |
| 14 | // |
| 15 | //===----------------------------------------------------------------------===// |
| 16 | |
| 17 | #ifndef LLVM_LIB_TARGET_ARM_ARMTARGETTRANSFORMINFO_H |
| 18 | #define LLVM_LIB_TARGET_ARM_ARMTARGETTRANSFORMINFO_H |
| 19 | |
| 20 | #include "ARM.h" |
| 21 | #include "ARMSubtarget.h" |
| 22 | #include "ARMTargetMachine.h" |
| 23 | #include "llvm/ADT/ArrayRef.h" |
| 24 | #include "llvm/Analysis/TargetTransformInfo.h" |
| 25 | #include "llvm/CodeGen/BasicTTIImpl.h" |
| 26 | #include "llvm/IR/Constant.h" |
| 27 | #include "llvm/IR/Function.h" |
| 28 | #include "llvm/TargetParser/SubtargetFeature.h" |
| 29 | #include <optional> |
| 30 | |
| 31 | namespace llvm { |
| 32 | |
| 33 | class APInt; |
| 34 | class ARMTargetLowering; |
| 35 | class Instruction; |
| 36 | class Loop; |
| 37 | class SCEV; |
| 38 | class ScalarEvolution; |
| 39 | class Type; |
| 40 | class Value; |
| 41 | |
| 42 | namespace TailPredication { |
| 43 | enum Mode { |
| 44 | Disabled = 0, |
| 45 | EnabledNoReductions, |
| 46 | Enabled, |
| 47 | ForceEnabledNoReductions, |
| 48 | ForceEnabled |
| 49 | }; |
| 50 | } |
| 51 | |
| 52 | // For controlling conversion of memcpy into Tail Predicated loop. |
| 53 | namespace TPLoop { |
| 54 | enum MemTransfer { ForceDisabled = 0, ForceEnabled, Allow }; |
| 55 | } |
| 56 | |
| 57 | class ARMTTIImpl final : public BasicTTIImplBase<ARMTTIImpl> { |
| 58 | using BaseT = BasicTTIImplBase<ARMTTIImpl>; |
| 59 | using TTI = TargetTransformInfo; |
| 60 | |
| 61 | friend BaseT; |
| 62 | |
| 63 | const ARMSubtarget *ST; |
| 64 | const ARMTargetLowering *TLI; |
| 65 | |
| 66 | const ARMSubtarget *getST() const { return ST; } |
| 67 | const ARMTargetLowering *getTLI() const { return TLI; } |
| 68 | |
| 69 | public: |
| 70 | explicit ARMTTIImpl(const ARMBaseTargetMachine *TM, const Function &F) |
| 71 | : BaseT(TM, F.getDataLayout()), ST(TM->getSubtargetImpl(F)), |
| 72 | TLI(ST->getTargetLowering()) {} |
| 73 | |
| 74 | bool enableInterleavedAccessVectorization() const override { return true; } |
| 75 | |
| 76 | TTI::AddressingModeKind |
| 77 | getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const override; |
| 78 | |
| 79 | /// Floating-point computation using ARMv8 AArch32 Advanced |
| 80 | /// SIMD instructions remains unchanged from ARMv7. Only AArch64 SIMD |
| 81 | /// and Arm MVE are IEEE-754 compliant. |
| 82 | bool isFPVectorizationPotentiallyUnsafe() const override { |
| 83 | return !ST->isTargetDarwin() && !ST->hasMVEFloatOps(); |
| 84 | } |
| 85 | |
| 86 | std::optional<Instruction *> |
| 87 | instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override; |
| 88 | std::optional<Value *> simplifyDemandedVectorEltsIntrinsic( |
| 89 | InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, |
| 90 | APInt &UndefElts2, APInt &UndefElts3, |
| 91 | std::function<void(Instruction *, unsigned, APInt, APInt &)> |
| 92 | SimplifyAndSetOp) const override; |
| 93 | |
| 94 | /// \name Scalar TTI Implementations |
| 95 | /// @{ |
| 96 | |
| 97 | InstructionCost getIntImmCodeSizeCost(unsigned Opcode, unsigned Idx, |
| 98 | const APInt &Imm, |
| 99 | Type *Ty) const override; |
| 100 | |
| 101 | using BaseT::getIntImmCost; |
| 102 | InstructionCost getIntImmCost(const APInt &Imm, Type *Ty, |
| 103 | TTI::TargetCostKind CostKind) const override; |
| 104 | |
| 105 | InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, |
| 106 | const APInt &Imm, Type *Ty, |
| 107 | TTI::TargetCostKind CostKind, |
| 108 | Instruction *Inst = nullptr) const override; |
| 109 | |
| 110 | /// @} |
| 111 | |
| 112 | /// \name Vector TTI Implementations |
| 113 | /// @{ |
| 114 | |
| 115 | unsigned getNumberOfRegisters(unsigned ClassID) const override { |
| 116 | bool Vector = (ClassID == 1); |
| 117 | if (Vector) { |
| 118 | if (ST->hasNEON()) |
| 119 | return 16; |
| 120 | if (ST->hasMVEIntegerOps()) |
| 121 | return 8; |
| 122 | return 0; |
| 123 | } |
| 124 | |
| 125 | if (ST->isThumb1Only()) |
| 126 | return 8; |
| 127 | return 13; |
| 128 | } |
| 129 | |
| 130 | TypeSize |
| 131 | getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override { |
| 132 | switch (K) { |
| 133 | case TargetTransformInfo::RGK_Scalar: |
| 134 | return TypeSize::getFixed(ExactSize: 32); |
| 135 | case TargetTransformInfo::RGK_FixedWidthVector: |
| 136 | if (ST->hasNEON()) |
| 137 | return TypeSize::getFixed(ExactSize: 128); |
| 138 | if (ST->hasMVEIntegerOps()) |
| 139 | return TypeSize::getFixed(ExactSize: 128); |
| 140 | return TypeSize::getFixed(ExactSize: 0); |
| 141 | case TargetTransformInfo::RGK_ScalableVector: |
| 142 | return TypeSize::getScalable(MinimumSize: 0); |
| 143 | } |
| 144 | llvm_unreachable("Unsupported register kind" ); |
| 145 | } |
| 146 | |
| 147 | unsigned getMaxInterleaveFactor(ElementCount VF, |
| 148 | bool HasUnorderedReductions) const override { |
| 149 | return ST->getMaxInterleaveFactor(); |
| 150 | } |
| 151 | |
| 152 | bool isProfitableLSRChainElement(Instruction *I) const override; |
| 153 | |
| 154 | bool |
| 155 | isLegalMaskedLoad(Type *DataTy, Align Alignment, unsigned AddressSpace, |
| 156 | TTI::MaskKind MaskKind = |
| 157 | TTI::MaskKind::VariableOrConstantMask) const override; |
| 158 | |
| 159 | bool |
| 160 | isLegalMaskedStore(Type *DataTy, Align Alignment, unsigned AddressSpace, |
| 161 | TTI::MaskKind MaskKind = |
| 162 | TTI::MaskKind::VariableOrConstantMask) const override { |
| 163 | return isLegalMaskedLoad(DataTy, Alignment, AddressSpace, MaskKind); |
| 164 | } |
| 165 | |
| 166 | bool forceScalarizeMaskedGather(VectorType *VTy, |
| 167 | Align Alignment) const override { |
| 168 | // For MVE, we have a custom lowering pass that will already have custom |
| 169 | // legalised any gathers that we can lower to MVE intrinsics, and want to |
| 170 | // expand all the rest. The pass runs before the masked intrinsic lowering |
| 171 | // pass. |
| 172 | return true; |
| 173 | } |
| 174 | |
| 175 | bool forceScalarizeMaskedScatter(VectorType *VTy, |
| 176 | Align Alignment) const override { |
| 177 | return forceScalarizeMaskedGather(VTy, Alignment); |
| 178 | } |
| 179 | |
| 180 | bool isLegalMaskedGather(Type *Ty, Align Alignment) const override; |
| 181 | |
| 182 | bool isLegalMaskedScatter(Type *Ty, Align Alignment) const override { |
| 183 | return isLegalMaskedGather(Ty, Alignment); |
| 184 | } |
| 185 | |
| 186 | InstructionCost getMemcpyCost(const Instruction *I) const override; |
| 187 | |
| 188 | uint64_t getMaxMemIntrinsicInlineSizeThreshold() const override { |
| 189 | return ST->getMaxInlineSizeThreshold(); |
| 190 | } |
| 191 | |
| 192 | int getNumMemOps(const IntrinsicInst *I) const; |
| 193 | |
| 194 | InstructionCost |
| 195 | getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, |
| 196 | TTI::TargetCostKind CostKind, ArrayRef<int> Mask, int Index, |
| 197 | VectorType *SubTp, ArrayRef<const Value *> Args = {}, |
| 198 | const Instruction *CtxI = nullptr, |
| 199 | TTI::VectorInstrContext VIC = |
| 200 | TTI::VectorInstrContext::None) const override; |
| 201 | |
| 202 | bool preferInLoopReduction(RecurKind Kind, Type *Ty) const override; |
| 203 | |
| 204 | bool preferPredicatedReductionSelect() const override; |
| 205 | |
| 206 | bool shouldExpandReduction(const IntrinsicInst *II) const override { |
| 207 | return false; |
| 208 | } |
| 209 | |
| 210 | InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, |
| 211 | const Instruction *I = nullptr) const override; |
| 212 | |
| 213 | InstructionCost |
| 214 | getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, |
| 215 | TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, |
| 216 | const Instruction *I = nullptr) const override; |
| 217 | |
| 218 | InstructionCost getCmpSelInstrCost( |
| 219 | unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, |
| 220 | TTI::TargetCostKind CostKind, |
| 221 | TTI::OperandValueInfo Op1Info = {.Kind: TTI::OK_AnyValue, .Properties: TTI::OP_None}, |
| 222 | TTI::OperandValueInfo Op2Info = {.Kind: TTI::OK_AnyValue, .Properties: TTI::OP_None}, |
| 223 | const Instruction *I = nullptr) const override; |
| 224 | |
| 225 | using BaseT::getVectorInstrCost; |
| 226 | InstructionCost |
| 227 | getVectorInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, |
| 228 | unsigned Index, const Value *Op0, const Value *Op1, |
| 229 | TTI::VectorInstrContext VIC = |
| 230 | TTI::VectorInstrContext::None) const override; |
| 231 | |
| 232 | InstructionCost |
| 233 | getAddressComputationCost(Type *Ty, ScalarEvolution *SE, const SCEV *Ptr, |
| 234 | TTI::TargetCostKind CostKind) const override; |
| 235 | |
| 236 | InstructionCost getArithmeticInstrCost( |
| 237 | unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, |
| 238 | TTI::OperandValueInfo Op1Info = {.Kind: TTI::OK_AnyValue, .Properties: TTI::OP_None}, |
| 239 | TTI::OperandValueInfo Op2Info = {.Kind: TTI::OK_AnyValue, .Properties: TTI::OP_None}, |
| 240 | ArrayRef<const Value *> Args = {}, |
| 241 | const Instruction *CtxI = nullptr) const override; |
| 242 | |
| 243 | InstructionCost getMemoryOpCost( |
| 244 | unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, |
| 245 | TTI::TargetCostKind CostKind, |
| 246 | TTI::OperandValueInfo OpInfo = {.Kind: TTI::OK_AnyValue, .Properties: TTI::OP_None}, |
| 247 | const Instruction *I = nullptr) const override; |
| 248 | |
| 249 | InstructionCost |
| 250 | getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, |
| 251 | TTI::TargetCostKind CostKind) const override; |
| 252 | |
| 253 | InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, |
| 254 | TTI::TargetCostKind CostKind) const; |
| 255 | |
| 256 | InstructionCost getInterleavedMemoryOpCost( |
| 257 | unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices, |
| 258 | Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, |
| 259 | bool UseMaskForCond = false, bool UseMaskForGaps = false) const override; |
| 260 | |
| 261 | InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, |
| 262 | TTI::TargetCostKind CostKind) const; |
| 263 | |
| 264 | InstructionCost |
| 265 | getArithmeticReductionCost(unsigned Opcode, VectorType *ValTy, |
| 266 | std::optional<FastMathFlags> FMF, |
| 267 | TTI::TargetCostKind CostKind) const override; |
| 268 | InstructionCost |
| 269 | getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, |
| 270 | VectorType *ValTy, std::optional<FastMathFlags> FMF, |
| 271 | TTI::TargetCostKind CostKind) const override; |
| 272 | InstructionCost |
| 273 | getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, |
| 274 | VectorType *ValTy, |
| 275 | TTI::TargetCostKind CostKind) const override; |
| 276 | |
| 277 | InstructionCost |
| 278 | getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, |
| 279 | TTI::TargetCostKind CostKind) const override; |
| 280 | |
| 281 | InstructionCost |
| 282 | getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, |
| 283 | TTI::TargetCostKind CostKind) const override; |
| 284 | |
| 285 | InstructionCost getPartialReductionCost( |
| 286 | unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, |
| 287 | ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, |
| 288 | TTI::PartialReductionExtendKind OpBExtend, std::optional<unsigned> BinOp, |
| 289 | TTI::TargetCostKind CostKind, |
| 290 | std::optional<FastMathFlags> FMF) const override { |
| 291 | return InstructionCost::getInvalid(); |
| 292 | } |
| 293 | |
| 294 | /// getScalingFactorCost - Return the cost of the scaling used in |
| 295 | /// addressing mode represented by AM. |
| 296 | /// If the AM is supported, the return value must be >= 0. |
| 297 | /// If the AM is not supported, the return value is an invalid cost. |
| 298 | InstructionCost getScalingFactorCost(Type *Ty, GlobalValue *BaseGV, |
| 299 | StackOffset BaseOffset, bool HasBaseReg, |
| 300 | int64_t Scale, |
| 301 | unsigned AddrSpace) const override; |
| 302 | |
| 303 | bool maybeLoweredToCall(Instruction &I) const; |
| 304 | bool isLoweredToCall(const Function *F) const override; |
| 305 | bool isHardwareLoopProfitable(Loop *L, ScalarEvolution &SE, |
| 306 | AssumptionCache &AC, TargetLibraryInfo *LibInfo, |
| 307 | HardwareLoopInfo &HWLoopInfo) const override; |
| 308 | bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override; |
| 309 | void (Loop *L, ScalarEvolution &SE, |
| 310 | TTI::UnrollingPreferences &UP, |
| 311 | OptimizationRemarkEmitter *ORE) const override; |
| 312 | |
| 313 | TailFoldingStyle getPreferredTailFoldingStyle() const override; |
| 314 | |
| 315 | void getPeelingPreferences(Loop *L, ScalarEvolution &SE, |
| 316 | TTI::PeelingPreferences &PP) const override; |
| 317 | bool shouldBuildLookupTablesForConstant(Constant *C) const override { |
| 318 | // In the ROPI and RWPI relocation models we can't have pointers to global |
| 319 | // variables or functions in constant data, so don't convert switches to |
| 320 | // lookup tables if any of the values would need relocation. |
| 321 | if (ST->isROPI() || ST->isRWPI()) |
| 322 | return !C->needsDynamicRelocation(); |
| 323 | |
| 324 | return true; |
| 325 | } |
| 326 | |
| 327 | bool shouldConsiderVectorizationRegPressure() const override; |
| 328 | |
| 329 | bool hasArmWideBranch(bool Thumb) const override; |
| 330 | |
| 331 | bool isProfitableToSinkOperands(Instruction *I, |
| 332 | SmallVectorImpl<Use *> &Ops) const override; |
| 333 | |
| 334 | unsigned getNumBytesToPadGlobalArray(unsigned Size, |
| 335 | Type *ArrayType) const override; |
| 336 | |
| 337 | /// @} |
| 338 | }; |
| 339 | |
| 340 | /// isVREVMask - Check if a vector shuffle corresponds to a VREV |
| 341 | /// instruction with the specified blocksize. (The order of the elements |
| 342 | /// within each block of the vector is reversed.) |
| 343 | inline bool isVREVMask(ArrayRef<int> M, EVT VT, unsigned BlockSize) { |
| 344 | assert((BlockSize == 16 || BlockSize == 32 || BlockSize == 64) && |
| 345 | "Only possible block sizes for VREV are: 16, 32, 64" ); |
| 346 | |
| 347 | unsigned EltSz = VT.getScalarSizeInBits(); |
| 348 | if (EltSz != 8 && EltSz != 16 && EltSz != 32) |
| 349 | return false; |
| 350 | |
| 351 | unsigned BlockElts = M[0] + 1; |
| 352 | // If the first shuffle index is UNDEF, be optimistic. |
| 353 | if (M[0] < 0) |
| 354 | BlockElts = BlockSize / EltSz; |
| 355 | |
| 356 | if (BlockSize <= EltSz || BlockSize != BlockElts * EltSz) |
| 357 | return false; |
| 358 | |
| 359 | for (unsigned i = 0, e = M.size(); i < e; ++i) { |
| 360 | if (M[i] < 0) |
| 361 | continue; // ignore UNDEF indices |
| 362 | if ((unsigned)M[i] != (i - i % BlockElts) + (BlockElts - 1 - i % BlockElts)) |
| 363 | return false; |
| 364 | } |
| 365 | |
| 366 | return true; |
| 367 | } |
| 368 | |
| 369 | inline unsigned SelectPairHalf(unsigned Elements, ArrayRef<int> Mask, |
| 370 | unsigned Index) { |
| 371 | if (Mask.size() == Elements * 2) |
| 372 | return Index / Elements; |
| 373 | return Mask[Index] == 0 ? 0 : 1; |
| 374 | } |
| 375 | |
| 376 | // Checks whether the shuffle mask represents a vector transpose (VTRN) by |
| 377 | // checking that pairs of elements in the shuffle mask represent the same index |
| 378 | // in each vector, incrementing the expected index by 2 at each step. |
| 379 | // e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 4, 2, 6] |
| 380 | // v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,e,c,g} |
| 381 | // v2={e,f,g,h} |
| 382 | // WhichResult gives the offset for each element in the mask based on which |
| 383 | // of the two results it belongs to. |
| 384 | // |
| 385 | // The transpose can be represented either as: |
| 386 | // result1 = shufflevector v1, v2, result1_shuffle_mask |
| 387 | // result2 = shufflevector v1, v2, result2_shuffle_mask |
| 388 | // where v1/v2 and the shuffle masks have the same number of elements |
| 389 | // (here WhichResult (see below) indicates which result is being checked) |
| 390 | // |
| 391 | // or as: |
| 392 | // results = shufflevector v1, v2, shuffle_mask |
| 393 | // where both results are returned in one vector and the shuffle mask has twice |
| 394 | // as many elements as v1/v2 (here WhichResult will always be 0 if true) here we |
| 395 | // want to check the low half and high half of the shuffle mask as if it were |
| 396 | // the other case |
| 397 | inline bool isVTRNMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) { |
| 398 | unsigned EltSz = VT.getScalarSizeInBits(); |
| 399 | if (EltSz == 64) |
| 400 | return false; |
| 401 | |
| 402 | unsigned NumElts = VT.getVectorNumElements(); |
| 403 | if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0) |
| 404 | return false; |
| 405 | |
| 406 | // If the mask is twice as long as the input vector then we need to check the |
| 407 | // upper and lower parts of the mask with a matching value for WhichResult |
| 408 | // FIXME: A mask with only even values will be rejected in case the first |
| 409 | // element is undefined, e.g. [-1, 4, 2, 6] will be rejected, because only |
| 410 | // M[0] is used to determine WhichResult |
| 411 | for (unsigned i = 0; i < M.size(); i += NumElts) { |
| 412 | WhichResult = SelectPairHalf(Elements: NumElts, Mask: M, Index: i); |
| 413 | for (unsigned j = 0; j < NumElts; j += 2) { |
| 414 | if ((M[i + j] >= 0 && (unsigned)M[i + j] != j + WhichResult) || |
| 415 | (M[i + j + 1] >= 0 && |
| 416 | (unsigned)M[i + j + 1] != j + NumElts + WhichResult)) |
| 417 | return false; |
| 418 | } |
| 419 | } |
| 420 | |
| 421 | if (M.size() == NumElts * 2) |
| 422 | WhichResult = 0; |
| 423 | |
| 424 | return true; |
| 425 | } |
| 426 | |
| 427 | /// isVTRN_v_undef_Mask - Special case of isVTRNMask for canonical form of |
| 428 | /// "vector_shuffle v, v", i.e., "vector_shuffle v, undef". |
| 429 | /// Mask is e.g., <0, 0, 2, 2> instead of <0, 4, 2, 6>. |
| 430 | inline bool isVTRN_v_undef_Mask(ArrayRef<int> M, EVT VT, |
| 431 | unsigned &WhichResult) { |
| 432 | unsigned EltSz = VT.getScalarSizeInBits(); |
| 433 | if (EltSz == 64) |
| 434 | return false; |
| 435 | |
| 436 | unsigned NumElts = VT.getVectorNumElements(); |
| 437 | if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0) |
| 438 | return false; |
| 439 | |
| 440 | for (unsigned i = 0; i < M.size(); i += NumElts) { |
| 441 | WhichResult = SelectPairHalf(Elements: NumElts, Mask: M, Index: i); |
| 442 | for (unsigned j = 0; j < NumElts; j += 2) { |
| 443 | if ((M[i + j] >= 0 && (unsigned)M[i + j] != j + WhichResult) || |
| 444 | (M[i + j + 1] >= 0 && (unsigned)M[i + j + 1] != j + WhichResult)) |
| 445 | return false; |
| 446 | } |
| 447 | } |
| 448 | |
| 449 | if (M.size() == NumElts * 2) |
| 450 | WhichResult = 0; |
| 451 | |
| 452 | return true; |
| 453 | } |
| 454 | |
| 455 | // Checks whether the shuffle mask represents a vector unzip (VUZP) by checking |
| 456 | // that the mask elements are either all even and in steps of size 2 or all odd |
| 457 | // and in steps of size 2. |
| 458 | // e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 2, 4, 6] |
| 459 | // v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,c,e,g} |
| 460 | // v2={e,f,g,h} |
| 461 | // Requires similar checks to that of isVTRNMask with |
| 462 | // respect the how results are returned. |
| 463 | inline bool isVUZPMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) { |
| 464 | unsigned EltSz = VT.getScalarSizeInBits(); |
| 465 | if (EltSz == 64) |
| 466 | return false; |
| 467 | |
| 468 | unsigned NumElts = VT.getVectorNumElements(); |
| 469 | if (M.size() != NumElts && M.size() != NumElts * 2) |
| 470 | return false; |
| 471 | |
| 472 | for (unsigned i = 0; i < M.size(); i += NumElts) { |
| 473 | WhichResult = SelectPairHalf(Elements: NumElts, Mask: M, Index: i); |
| 474 | for (unsigned j = 0; j < NumElts; ++j) { |
| 475 | if (M[i + j] >= 0 && (unsigned)M[i + j] != 2 * j + WhichResult) |
| 476 | return false; |
| 477 | } |
| 478 | } |
| 479 | |
| 480 | if (M.size() == NumElts * 2) |
| 481 | WhichResult = 0; |
| 482 | |
| 483 | // VUZP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32. |
| 484 | if (VT.is64BitVector() && EltSz == 32) |
| 485 | return false; |
| 486 | |
| 487 | return true; |
| 488 | } |
| 489 | |
| 490 | /// isVUZP_v_undef_Mask - Special case of isVUZPMask for canonical form of |
| 491 | /// "vector_shuffle v, v", i.e., "vector_shuffle v, undef". |
| 492 | /// Mask is e.g., <0, 2, 0, 2> instead of <0, 2, 4, 6>, |
| 493 | inline bool isVUZP_v_undef_Mask(ArrayRef<int> M, EVT VT, |
| 494 | unsigned &WhichResult) { |
| 495 | unsigned EltSz = VT.getScalarSizeInBits(); |
| 496 | if (EltSz == 64) |
| 497 | return false; |
| 498 | |
| 499 | unsigned NumElts = VT.getVectorNumElements(); |
| 500 | if (M.size() != NumElts && M.size() != NumElts * 2) |
| 501 | return false; |
| 502 | |
| 503 | unsigned Half = NumElts / 2; |
| 504 | for (unsigned i = 0; i < M.size(); i += NumElts) { |
| 505 | WhichResult = SelectPairHalf(Elements: NumElts, Mask: M, Index: i); |
| 506 | for (unsigned j = 0; j < NumElts; j += Half) { |
| 507 | unsigned Idx = WhichResult; |
| 508 | for (unsigned k = 0; k < Half; ++k) { |
| 509 | int MIdx = M[i + j + k]; |
| 510 | if (MIdx >= 0 && (unsigned)MIdx != Idx) |
| 511 | return false; |
| 512 | Idx += 2; |
| 513 | } |
| 514 | } |
| 515 | } |
| 516 | |
| 517 | if (M.size() == NumElts * 2) |
| 518 | WhichResult = 0; |
| 519 | |
| 520 | // VUZP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32. |
| 521 | if (VT.is64BitVector() && EltSz == 32) |
| 522 | return false; |
| 523 | |
| 524 | return true; |
| 525 | } |
| 526 | |
| 527 | // Checks whether the shuffle mask represents a vector zip (VZIP) by checking |
| 528 | // that pairs of elements of the shufflemask represent the same index in each |
| 529 | // vector incrementing sequentially through the vectors. |
| 530 | // e.g. For v1,v2 of type v4i32 a valid shuffle mask is: [0, 4, 1, 5] |
| 531 | // v1={a,b,c,d} => x=shufflevector v1, v2 shufflemask => x={a,e,b,f} |
| 532 | // v2={e,f,g,h} |
| 533 | // Requires similar checks to that of isVTRNMask with respect the how results |
| 534 | // are returned. |
| 535 | inline bool isVZIPMask(ArrayRef<int> M, EVT VT, unsigned &WhichResult) { |
| 536 | unsigned EltSz = VT.getScalarSizeInBits(); |
| 537 | if (EltSz == 64) |
| 538 | return false; |
| 539 | |
| 540 | unsigned NumElts = VT.getVectorNumElements(); |
| 541 | if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0) |
| 542 | return false; |
| 543 | |
| 544 | for (unsigned i = 0; i < M.size(); i += NumElts) { |
| 545 | WhichResult = SelectPairHalf(Elements: NumElts, Mask: M, Index: i); |
| 546 | unsigned Idx = WhichResult * NumElts / 2; |
| 547 | for (unsigned j = 0; j < NumElts; j += 2) { |
| 548 | if ((M[i + j] >= 0 && (unsigned)M[i + j] != Idx) || |
| 549 | (M[i + j + 1] >= 0 && (unsigned)M[i + j + 1] != Idx + NumElts)) |
| 550 | return false; |
| 551 | Idx += 1; |
| 552 | } |
| 553 | } |
| 554 | |
| 555 | if (M.size() == NumElts * 2) |
| 556 | WhichResult = 0; |
| 557 | |
| 558 | // VZIP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32. |
| 559 | if (VT.is64BitVector() && EltSz == 32) |
| 560 | return false; |
| 561 | |
| 562 | return true; |
| 563 | } |
| 564 | |
| 565 | /// isVZIP_v_undef_Mask - Special case of isVZIPMask for canonical form of |
| 566 | /// "vector_shuffle v, v", i.e., "vector_shuffle v, undef". |
| 567 | /// Mask is e.g., <0, 0, 1, 1> instead of <0, 4, 1, 5>. |
| 568 | inline bool isVZIP_v_undef_Mask(ArrayRef<int> M, EVT VT, |
| 569 | unsigned &WhichResult) { |
| 570 | unsigned EltSz = VT.getScalarSizeInBits(); |
| 571 | if (EltSz == 64) |
| 572 | return false; |
| 573 | |
| 574 | unsigned NumElts = VT.getVectorNumElements(); |
| 575 | if ((M.size() != NumElts && M.size() != NumElts * 2) || NumElts % 2 != 0) |
| 576 | return false; |
| 577 | |
| 578 | for (unsigned i = 0; i < M.size(); i += NumElts) { |
| 579 | WhichResult = SelectPairHalf(Elements: NumElts, Mask: M, Index: i); |
| 580 | unsigned Idx = WhichResult * NumElts / 2; |
| 581 | for (unsigned j = 0; j < NumElts; j += 2) { |
| 582 | if ((M[i + j] >= 0 && (unsigned)M[i + j] != Idx) || |
| 583 | (M[i + j + 1] >= 0 && (unsigned)M[i + j + 1] != Idx)) |
| 584 | return false; |
| 585 | Idx += 1; |
| 586 | } |
| 587 | } |
| 588 | |
| 589 | if (M.size() == NumElts * 2) |
| 590 | WhichResult = 0; |
| 591 | |
| 592 | // VZIP.32 for 64-bit vectors is a pseudo-instruction alias for VTRN.32. |
| 593 | if (VT.is64BitVector() && EltSz == 32) |
| 594 | return false; |
| 595 | |
| 596 | return true; |
| 597 | } |
| 598 | |
| 599 | } // end namespace llvm |
| 600 | |
| 601 | #endif // LLVM_LIB_TARGET_ARM_ARMTARGETTRANSFORMINFO_H |
| 602 | |