| 1 | //===- SLPCostAnalysis.h - SLP Vectorizer free cost helpers ----*- C++ -*-===// |
| 2 | // |
| 3 | // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. |
| 4 | // See https://llvm.org/LICENSE.txt for license information. |
| 5 | // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception |
| 6 | // |
| 7 | //===----------------------------------------------------------------------===// |
| 8 | // |
| 9 | // Internal header used by SLPVectorizer.cpp. It declares free cost helpers |
| 10 | // that do not depend on BoUpSLP or any other SLP-private type. The bulk of |
| 11 | // the SLP cost model still lives in SLPVectorizer.cpp because it references |
| 12 | // BoUpSLP internals. |
| 13 | // |
| 14 | //===----------------------------------------------------------------------===// |
| 15 | |
| 16 | #ifndef LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPCOSTANALYSIS_H |
| 17 | #define LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPCOSTANALYSIS_H |
| 18 | |
| 19 | #include "SLPUtils.h" |
| 20 | #include "llvm/ADT/ArrayRef.h" |
| 21 | #include "llvm/ADT/DenseMap.h" |
| 22 | #include "llvm/Analysis/TargetTransformInfo.h" |
| 23 | #include "llvm/Support/InstructionCost.h" |
| 24 | |
| 25 | #include <tuple> |
| 26 | #include <utility> |
| 27 | |
| 28 | namespace llvm { |
| 29 | class APInt; |
| 30 | class DataLayout; |
| 31 | class FastMathFlags; |
| 32 | class FixedVectorType; |
| 33 | class Instruction; |
| 34 | class TargetLibraryInfo; |
| 35 | class Type; |
| 36 | class User; |
| 37 | class Value; |
| 38 | class VectorType; |
| 39 | enum class RecurKind; |
| 40 | } // namespace llvm |
| 41 | |
| 42 | namespace llvm::slpvectorizer { |
| 43 | |
| 44 | /// Returns the cost of the shuffle instructions with the given \p Kind, vector |
| 45 | /// type \p Tp and optional \p Mask. Adds SLP-specific cost estimation for |
| 46 | /// insert subvector pattern. |
| 47 | InstructionCost |
| 48 | getShuffleCost(const TargetTransformInfo &TTI, |
| 49 | TargetTransformInfo::ShuffleKind Kind, VectorType *Tp, |
| 50 | const TargetTransformInfo::TargetCostKind CostKind, |
| 51 | ArrayRef<int> Mask = {}, int Index = 0, |
| 52 | VectorType *SubTp = nullptr, ArrayRef<const Value *> Args = {}, |
| 53 | TargetTransformInfo::VectorInstrContext VIC = |
| 54 | TargetTransformInfo::VectorInstrContext::None); |
| 55 | |
| 56 | /// Calculate the scalar and the vector costs from vectorizing set of GEPs. |
| 57 | std::pair<InstructionCost, InstructionCost> |
| 58 | getGEPCosts(const TargetTransformInfo &TTI, ArrayRef<Value *> Ptrs, |
| 59 | Value *BasePtr, unsigned Opcode, |
| 60 | const TargetTransformInfo::TargetCostKind CostKind, Type *ScalarTy, |
| 61 | VectorType *VecTy); |
| 62 | |
| 63 | /// Returns the cost of a BlendedLoadVectorize node loading \p VecTy: two masked |
| 64 | /// loads (one per candidate base), a xor to negate the false-lane mask and a |
| 65 | /// select. The blend mask is a separate operand node, so its cost is counted |
| 66 | /// there, not here. |
| 67 | InstructionCost |
| 68 | getBlendedLoadCost(const TargetTransformInfo &TTI, Type *VecTy, Align Alignment, |
| 69 | unsigned AddressSpace, |
| 70 | const TargetTransformInfo::TargetCostKind CostKind); |
| 71 | |
| 72 | /// Returns the cost of the cast between the widened strided access type and |
| 73 | /// the entry vector type. |
| 74 | InstructionCost |
| 75 | getWidenedStridedCastCost(const TargetTransformInfo &TTI, Type *SrcTy, |
| 76 | Type *DstTy, const DataLayout &DL, |
| 77 | TargetTransformInfo::CastContextHint CCH, |
| 78 | TargetTransformInfo::TargetCostKind CostKind); |
| 79 | |
| 80 | /// For a non-power-of-2 \p NumElts-wide integer div/rem \p Opcode, checks if |
| 81 | /// padding to a full register and using the masked div/rem intrinsic is |
| 82 | /// cheaper than the direct vector op. Returns the cost of the masked |
| 83 | /// alternative, or an invalid cost if it is not applicable or not cheaper. |
| 84 | InstructionCost |
| 85 | getMaskedDivRemCost(const TargetTransformInfo &TTI, bool ReVec, unsigned Opcode, |
| 86 | Type *ScalarTy, unsigned NumElts, |
| 87 | const TargetTransformInfo::TargetCostKind CostKind, |
| 88 | FixedVectorType **PaddedTy = nullptr); |
| 89 | |
| 90 | /// Returns the cost of the booleanized logical and/or reduction of a vector |
| 91 | /// of type \p VecTy with the i1 root \p Root, emitted as the wide reduction |
| 92 | /// plus the result trunc. |
| 93 | InstructionCost |
| 94 | getBoolReduxWideRdxCost(const TargetTransformInfo &TTI, RecurKind RdxKind, |
| 95 | FixedVectorType *VecTy, const Value *Root, |
| 96 | FastMathFlags FMF, |
| 97 | TargetTransformInfo::TargetCostKind CostKind); |
| 98 | |
| 99 | /// Returns the cost of the booleanized logical and/or reduction of a vector |
| 100 | /// of type \p VecTy with the i1 root \p Root, emitted as trunc+bitcast+cmp, |
| 101 | /// estimated in the context of the replaced cast chain \p ChainInsts. |
| 102 | InstructionCost |
| 103 | getBoolReduxBitcastCmpCost(const TargetTransformInfo &TTI, RecurKind RdxKind, |
| 104 | FixedVectorType *VecTy, const Value *Root, |
| 105 | ArrayRef<Instruction *> ChainInsts, |
| 106 | TargetTransformInfo::TargetCostKind CostKind); |
| 107 | |
| 108 | /// Returns the cost of the boolean bitmask reduction of a vector of boolean |
| 109 | /// leaves of type \p NarrowScalarTy, emitted as [and] + [lane permutation |
| 110 | /// \p PermMask] + zero test + bitcast [+ zext] to \p WideTy. \p Root is the |
| 111 | /// reduction root, used as the context of the emitted instructions. |
| 112 | InstructionCost |
| 113 | getBoolBitmaskCost(const TargetTransformInfo &TTI, bool NeedMask, |
| 114 | Type *NarrowScalarTy, Type *WideTy, unsigned VF, |
| 115 | ArrayRef<int> PermMask, const Value *Root, |
| 116 | TargetTransformInfo::TargetCostKind CostKind); |
| 117 | |
| 118 | /// Returns the cost of the per-lane operations on the narrowed leaves |
| 119 | /// \p NarrowedLeafShifts: the shl in the wide vector type if any leaf is |
| 120 | /// shifted and the and in the narrow vector type if any leaf is masked. |
| 121 | InstructionCost getNarrowedLeafOpsCost( |
| 122 | const TargetTransformInfo &TTI, |
| 123 | const SmallDenseMap<Value *, NarrowedLeafInfo> &NarrowedLeafShifts, |
| 124 | VectorType *NarrowVecTy, VectorType *WideVecTy, const Instruction *CtxI, |
| 125 | TargetTransformInfo::TargetCostKind CostKind); |
| 126 | |
| 127 | /// This is similar to TargetTransformInfo::getScalarizationOverhead, but if |
| 128 | /// ScalarTy is a FixedVectorType, a vector will be inserted or extracted |
| 129 | /// instead of a scalar. |
| 130 | InstructionCost |
| 131 | (const TargetTransformInfo &TTI, bool ReVec, |
| 132 | Type *ScalarTy, VectorType *Ty, |
| 133 | const APInt &DemandedElts, bool Insert, bool , |
| 134 | const TargetTransformInfo::TargetCostKind CostKind, |
| 135 | bool ForPoisonSrc = true, ArrayRef<Value *> VL = {}, |
| 136 | TargetTransformInfo::VectorInstrContext VIC = |
| 137 | TargetTransformInfo::VectorInstrContext::None); |
| 138 | |
| 139 | /// This is similar to TargetTransformInfo::getVectorInstrCost, but if ScalarTy |
| 140 | /// is a FixedVectorType, a vector will be extracted instead of a scalar. |
| 141 | InstructionCost |
| 142 | getVectorInstrCost(const TargetTransformInfo &TTI, bool ReVec, Type *ScalarTy, |
| 143 | unsigned Opcode, Type *Val, |
| 144 | const TargetTransformInfo::TargetCostKind CostKind, |
| 145 | unsigned Index, Value *Scalar, |
| 146 | ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx, |
| 147 | TTI::VectorInstrContext VIC = TTI::VectorInstrContext::None); |
| 148 | |
| 149 | /// This is similar to TargetTransformInfo::getExtractWithExtendCost, but if Dst |
| 150 | /// is a FixedVectorType, a vector will be extracted instead of a scalar. |
| 151 | InstructionCost |
| 152 | (const TargetTransformInfo &TTI, bool ReVec, |
| 153 | unsigned Opcode, Type *Dst, VectorType *VecTy, |
| 154 | unsigned Index, |
| 155 | const TargetTransformInfo::TargetCostKind CostKind); |
| 156 | |
| 157 | /// Returns the cost of the bitfield packing of \p SrcTy into \p ResultTy, |
| 158 | /// picking the cheapest shift width. The packing is a trunc, an lshr, a byte |
| 159 | /// shuffle and a bitcast. \p ZExtSrcWidth is the source width of the lanes if |
| 160 | /// they are a plain zext (0 otherwise), so compacting them back to it is free. |
| 161 | /// \p CCH is the context of the pack's source operand. |
| 162 | InstructionCost getBitPackCost(const TargetTransformInfo &TTI, |
| 163 | FixedVectorType *SrcTy, Type *ResultTy, |
| 164 | const BitPackInfo &Info, unsigned ZExtSrcWidth, |
| 165 | TargetTransformInfo::CastContextHint CCH, |
| 166 | TargetTransformInfo::TargetCostKind CostKind, |
| 167 | const TargetLibraryInfo *TLI, |
| 168 | const Instruction *CtxI, unsigned &ShiftWidth); |
| 169 | |
| 170 | /// i1 reductions can be emitted as the plain target reduction or in the |
| 171 | /// bitcast-based form (bitcast to a scalar integer type plus a compare for |
| 172 | /// and/or, plus ctpop for add). Returns the cost of the cheaper form and |
| 173 | /// whether it is the bitcast-based one. Ties keep the historically default |
| 174 | /// form: plain for and/or, bitcast-based for add. |
| 175 | std::pair<InstructionCost, bool> |
| 176 | getI1ReductionCost(RecurKind Kind, const TargetTransformInfo &TTI, |
| 177 | FixedVectorType *VectorTy, Type *ScalarTy, |
| 178 | TargetTransformInfo::CastContextHint Ctx, |
| 179 | TargetTransformInfo::TargetCostKind CostKind); |
| 180 | |
| 181 | } // namespace llvm::slpvectorizer |
| 182 | |
| 183 | #endif // LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPCOSTANALYSIS_H |
| 184 | |