1//===- SLPUtils.h - SLP Vectorizer free utility helpers --------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// Internal header used by SLPVectorizer.cpp. It declares free helper
10// functions that do not depend on BoUpSLP, InstructionsState, or any other
11// SLP-private type. Splitting them out keeps SLPVectorizer.cpp focused on
12// the build / legality / cost / codegen pipeline.
13//
14//===----------------------------------------------------------------------===//
15
16#ifndef LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPUTILS_H
17#define LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPUTILS_H
18
19#include "llvm/ADT/APInt.h"
20#include "llvm/ADT/ArrayRef.h"
21#include "llvm/ADT/STLExtras.h"
22#include "llvm/ADT/STLFunctionalExtras.h"
23#include "llvm/ADT/SmallBitVector.h"
24#include "llvm/ADT/SmallVector.h"
25#include "llvm/Analysis/MemoryLocation.h"
26#include "llvm/Analysis/TargetTransformInfo.h"
27#include "llvm/IR/Intrinsics.h"
28
29#include <cstdint>
30#include <limits>
31#include <optional>
32#include <string>
33#include <tuple>
34
35namespace llvm {
36class AssumptionCache;
37class Constant;
38class DataLayout;
39class Instruction;
40class IRBuilderBase;
41class TargetLibraryInfo;
42class Type;
43class Value;
44} // namespace llvm
45
46namespace llvm::slpvectorizer {
47
48/// Limit of the number of uses for potentially transformed instructions/values,
49/// used in checks to avoid compile-time explode.
50inline constexpr int UsesLimit = 64;
51
52/// \returns True if the value is a constant (but not globals/constant
53/// expressions).
54bool isConstant(Value *V);
55
56/// \returns True if \p V is the integer identity constant for binary \p Opcode
57/// (e.g. 0 for add, 1 for mul, all-ones for and). Floating-point identities are
58/// excluded: a ConstantInt never matches the ConstantFP getBinOpIdentity()
59/// returns for FAdd/FMul, whose identity fast-math may break anyway.
60bool isBinOpIdentityConstant(const Value *V, unsigned Opcode);
61
62/// \returns the opcode of the combines emitted for a reassociated node:
63/// subtract chains regroup their positive and negative operand columns with
64/// plain adds.
65unsigned getReassocCombineOpcode(unsigned Opcode);
66
67/// \returns True if \p I can be a link of a flattenable binary chain:
68/// subtracts flatten as adds of a negated leaf, float subtracts need reassoc
69/// to allow the regrouping.
70bool isReassocChainLink(const Instruction *I);
71
72/// Checks if \p V is one of vector-like instructions, i.e. undef,
73/// insertelement/extractelement with constant indices for fixed vector type
74/// or extractvalue instruction.
75bool isVectorLikeInstWithConstOps(Value *V);
76
77/// \returns the number of elements for Ty.
78unsigned getNumElements(Type *Ty);
79
80/// Returns power-of-2 number of elements in a single register (part), given
81/// the total number of elements \p Size and number of registers (parts) \p
82/// NumParts.
83unsigned getPartNumElems(unsigned Size, unsigned NumParts);
84
85/// Returns correct remaining number of elements, considering total amount
86/// \p Size, (power-of-2 number) of elements in a single register
87/// \p PartNumElems and current register (part) \p Part.
88unsigned getNumElems(unsigned Size, unsigned PartNumElems, unsigned Part);
89
90#if !defined(NDEBUG)
91/// Print a short descriptor of the instruction bundle suitable for debug
92/// output.
93std::string shortBundleName(ArrayRef<Value *> VL, int Idx = -1);
94#endif
95
96/// \returns True if all of the instructions in \p VL are in the same block.
97bool allSameBlock(ArrayRef<Value *> VL);
98
99/// \returns True if all of the values in \p VL are constants (but not
100/// globals/constant expressions).
101bool allConstant(ArrayRef<Value *> VL);
102
103/// \returns True if all of the values in \p VL are identical or some of them
104/// are UndefValue.
105bool isSplat(ArrayRef<Value *> VL);
106
107/// Checks if \p LHS and \p RHS are the same intrinsic, or one is llvm.fma
108/// and the other is llvm.fmuladd, since both lower to the same fused
109/// vector operation.
110/// \returns the intrinsic ID to use for the pair (\p RHS if the IDs match,
111/// otherwise Intrinsic::fma), or Intrinsic::not_intrinsic if they are not
112/// equivalent.
113Intrinsic::ID isEquivalentIntrinsicID(Intrinsic::ID LHS, Intrinsic::ID RHS);
114
115/// \returns True if \p I is commutative, handles CmpInst and BinaryOperator.
116/// For BinaryOperator, it also checks if \p ValWithUses is used in specific
117/// patterns that make it effectively commutative (like equality comparisons
118/// with zero).
119/// In most cases, users should not call this function directly (since \p I and
120/// \p ValWithUses are the same). However, when analyzing interchangeable
121/// instructions, we need to use the converted opcode along with the original
122/// uses.
123/// \param I The instruction to check for commutativity
124/// \param ValWithUses The value whose uses are analyzed for special
125/// patterns
126bool isCommutative(const Instruction *I, const Value *ValWithUses,
127 bool IsCopyable = false);
128
129/// This is a helper function to check whether \p I is commutative.
130/// This is a convenience wrapper that calls the two-parameter version of
131/// isCommutative with the same instruction for both parameters. This is
132/// the common case where the instruction being checked for commutativity
133/// is the same as the instruction whose uses are analyzed for special
134/// patterns (see the two-parameter version above for details).
135/// \param I The instruction to check for commutativity
136/// \returns true if the instruction is commutative, false otherwise
137bool isCommutative(const Instruction *I);
138
139/// Checks if the operand is commutative. In commutative operations, not all
140/// operands might commutable, e.g. for fmuladd only 2 first operands are
141/// commutable.
142bool isCommutableOperand(const Instruction *I, Value *ValWithUses, unsigned Op,
143 bool IsCopyable = false);
144
145/// \returns number of operands of \p I, considering commutativity. Returns 2
146/// for commutative intrinsics.
147/// \param I The instruction to check for commutativity
148unsigned getNumberOfPotentiallyCommutativeOps(Instruction *I);
149
150/// \returns inserting or extracting index of InsertElement, ExtractElement
151/// or InsertValue instruction, using \p Offset as base offset for index.
152/// \returns std::nullopt if the index is not an immediate.
153std::optional<unsigned> getElementIndex(const Value *Inst, unsigned Offset = 0);
154
155/// \returns True if all of the values in \p VL use the same opcode.
156/// For comparison instructions, also checks if predicates match.
157/// PoisonValues are considered matching. Interchangeable instructions are
158/// not considered.
159bool allSameOpcode(ArrayRef<Value *> VL);
160
161/// \returns Optional element Idx for Extract{Value,Element} instructions.
162std::optional<unsigned> getExtractIndex(const Instruction *E);
163
164/// Compute the inverse permutation \p Mask of \p Indices.
165void inversePermutation(ArrayRef<unsigned> Indices, SmallVectorImpl<int> &Mask);
166
167/// Reorders the list of scalars in accordance with the given \p Mask.
168void reorderScalars(SmallVectorImpl<Value *> &Scalars, ArrayRef<int> Mask);
169
170/// Reorders the given \p Reuses mask according to the given \p Mask. \p Reuses
171/// contains original mask for the scalars reused in the node. Procedure
172/// transform this mask in accordance with the given \p Mask.
173void reorderReuses(SmallVectorImpl<int> &Reuses, ArrayRef<int> Mask);
174
175/// Reorders the given \p Order according to the given \p Mask. \p Order - is
176/// the original order of the scalars. Procedure transforms the provided order
177/// in accordance with the given \p Mask. If the resulting \p Order is just an
178/// identity order, \p Order is cleared.
179void reorderOrder(SmallVectorImpl<unsigned> &Order, ArrayRef<int> Mask,
180 bool BottomOrder = false);
181
182/// Check if \p Order represents reverse order.
183bool isReverseOrder(ArrayRef<unsigned> Order);
184
185/// Checks if the given mask is a "clustered" mask with the same clusters of
186/// size \p Sz, which are not identity submasks.
187bool isRepeatedNonIdentityClusteredMask(ArrayRef<int> Mask, unsigned Sz);
188
189/// Fills unset elements of \p Order (marked with the sentinel value equal to
190/// the order size) with the corresponding elements of \p SecondaryOrder,
191/// skipping already used indices, or with the identity order if
192/// \p SecondaryOrder is empty.
193void combineOrders(MutableArrayRef<unsigned> Order,
194 ArrayRef<unsigned> SecondaryOrder);
195
196/// \returns True iff every value in \p VL has the same Type as the first.
197bool allSameType(ArrayRef<Value *> VL);
198
199/// Checks if the provided value does not require scheduling. It does not
200/// require scheduling if this is not an instruction or it is an instruction
201/// that does not read/write memory and all operands are either not
202/// instructions or phi nodes or instructions from different blocks.
203bool areAllOperandsNonInsts(Value *V);
204
205/// Checks if the provided value does not require scheduling. It does not
206/// require scheduling if this is not an instruction or it is an instruction
207/// that does not read/write memory and all users are phi nodes or
208/// instructions from different blocks.
209bool isUsedOutsideBlock(Value *V);
210
211/// Checks if the specified value does not require scheduling. It does not
212/// require scheduling if all operands and all users do not need to be
213/// scheduled in the current basic block.
214bool doesNotNeedToBeScheduled(Value *V);
215
216/// Checks if the specified array of instructions does not require scheduling.
217/// It is so if all either instructions have operands that do not require
218/// scheduling or their users do not require scheduling since they are phis or
219/// in other basic blocks.
220bool doesNotNeedToSchedule(ArrayRef<Value *> VL);
221
222/// \returns inserting or extracting index of InsertElement / ExtractElement
223/// instruction, using \p Offset as base offset for index. Only instantiated
224/// for InsertElementInst and ExtractElementInst (see SLPUtils.cpp).
225template <typename T>
226std::optional<unsigned> getInsertExtractIndex(const Value *Inst,
227 unsigned Offset);
228
229void transformScalarShuffleIndiciesToVector(unsigned VecTyNumElements,
230 SmallVectorImpl<int> &Mask);
231
232/// \returns the number of groups of shufflevector
233/// A group has the following features
234/// 1. All of value in a group are shufflevector.
235/// 2. The mask of all shufflevector is isExtractSubvectorMask.
236/// 3. The mask of all shufflevector uses all of the elements of the source.
237/// e.g., it is 1 group (%0)
238/// %1 = shufflevector <16 x i8> %0, <16 x i8> poison,
239/// <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
240/// %2 = shufflevector <16 x i8> %0, <16 x i8> poison,
241/// <8 x i32> <i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
242/// it is 2 groups (%3 and %4)
243/// %5 = shufflevector <8 x i16> %3, <8 x i16> poison,
244/// <4 x i32> <i32 0, i32 1, i32 2, i32 3>
245/// %6 = shufflevector <8 x i16> %3, <8 x i16> poison,
246/// <4 x i32> <i32 4, i32 5, i32 6, i32 7>
247/// %7 = shufflevector <8 x i16> %4, <8 x i16> poison,
248/// <4 x i32> <i32 0, i32 1, i32 2, i32 3>
249/// %8 = shufflevector <8 x i16> %4, <8 x i16> poison,
250/// <4 x i32> <i32 4, i32 5, i32 6, i32 7>
251/// it is 0 group
252/// %12 = shufflevector <8 x i16> %10, <8 x i16> poison,
253/// <4 x i32> <i32 0, i32 1, i32 2, i32 3>
254/// %13 = shufflevector <8 x i16> %11, <8 x i16> poison,
255/// <4 x i32> <i32 0, i32 1, i32 2, i32 3>
256unsigned getShufflevectorNumGroups(ArrayRef<Value *> VL);
257
258/// \returns a shufflevector mask which is used to vectorize shufflevectors
259/// e.g.,
260/// %5 = shufflevector <8 x i16> %3, <8 x i16> poison,
261/// <4 x i32> <i32 0, i32 1, i32 2, i32 3>
262/// %6 = shufflevector <8 x i16> %3, <8 x i16> poison,
263/// <4 x i32> <i32 4, i32 5, i32 6, i32 7>
264/// %7 = shufflevector <8 x i16> %4, <8 x i16> poison,
265/// <4 x i32> <i32 0, i32 1, i32 2, i32 3>
266/// %8 = shufflevector <8 x i16> %4, <8 x i16> poison,
267/// <4 x i32> <i32 4, i32 5, i32 6, i32 7>
268/// the result is
269/// <0, 1, 2, 3, 12, 13, 14, 15, 16, 17, 18, 19, 28, 29, 30, 31>
270SmallVector<int> calculateShufflevectorMask(ArrayRef<Value *> VL);
271
272/// Checks if the values in \p VL can be represented as a shuffle of at most
273/// two vector operands (extractelement lanes). On success, \p Mask is the
274/// equivalent shuffle mask.
275std::optional<TargetTransformInfo::ShuffleKind>
276isFixedVectorShuffle(ArrayRef<Value *> VL, SmallVectorImpl<int> &Mask,
277 AssumptionCache *AC);
278
279/// Creates subvector insert. Generates shuffle using \p Generator or
280/// using default shuffle.
281Value *createInsertVector(
282 IRBuilderBase &Builder, Value *Vec, Value *V, unsigned Index,
283 function_ref<Value *(Value *, Value *, ArrayRef<int>)> Generator = {});
284
285/// Generates subvector extract.
286Value *createExtractVector(IRBuilderBase &Builder, Value *Vec,
287 unsigned SubVecVF, unsigned Index);
288
289/// Specifies the way the mask should be analyzed for undefs/poisonous elements
290/// in the shuffle mask.
291enum class UseMask {
292 FirstArg, ///< The mask is expected to be for permutation of 1-2 vectors,
293 ///< check for the mask elements for the first argument (mask
294 ///< indices are in range [0:VF)).
295 SecondArg, ///< The mask is expected to be for permutation of 2 vectors, check
296 ///< for the mask elements for the second argument (mask indices
297 ///< are in range [VF:2*VF))
298 UndefsAsMask ///< Consider undef mask elements (-1) as placeholders for
299 ///< future shuffle elements and mark them as ones as being used
300 ///< in future. Non-undef elements are considered as unused since
301 ///< they're already marked as used in the mask.
302};
303
304/// Prepares a use bitset for the given mask either for the first argument or
305/// for the second.
306SmallBitVector buildUseMask(int VF, ArrayRef<int> Mask, UseMask MaskArg);
307
308/// Checks if the given value is actually an undefined constant vector.
309/// Also, if the \p UseMask is not empty, tries to check if the non-masked
310/// elements actually mask the insertelement buildvector, if any.
311template <bool IsPoisonOnly = false>
312SmallBitVector isUndefVector(const Value *V,
313 const SmallBitVector &UseMask = {});
314
315/// \returns True if in-tree use also needs extract. This refers to
316/// possible scalar operand in vectorized instruction.
317bool doesInTreeUserNeedToExtract(Value *Scalar, Instruction *UserInst,
318 TargetLibraryInfo *TLI,
319 const TargetTransformInfo *TTI);
320
321/// \returns the AA location that is being access by the instruction.
322MemoryLocation getLocation(Instruction *I);
323
324/// \returns True if the instruction is not a volatile or atomic load/store.
325bool isSimple(Instruction *I);
326
327/// Checks if the loads with scalar type \p ScalarTy and pointer operands
328/// \p PointerOps are each (optionally via a constant-offset GEP) a
329/// `select Cond, A, B` picking between the same two base pointers A/B on
330/// every lane - the shape a fully unrolled `x = cond ? A[i] : B[i]` takes. On
331/// success \p TrueBase / \p FalseBase are the candidate bases and
332/// \p Conditions holds each lane's `select` condition, used to build the
333/// blend mask. Lane \p Idx must be at `Base + Idx * sizeof(ScalarTy)`; only
334/// dense, natural lane order starting at the base is recognized (reordered or
335/// partial groups fall back to Gather/Scatter).
336bool isSelectedBaseLoad(Type *ScalarTy, ArrayRef<Value *> PointerOps,
337 const DataLayout &DL, Value *&TrueBase,
338 Value *&FalseBase,
339 SmallVectorImpl<Value *> &Conditions);
340
341/// Returns the alignment of the shared base pointer of a blended load for
342/// the loads \p VL.
343Align computeBlendedLoadBaseAlignment(ArrayRef<Value *> VL,
344 const DataLayout &DL);
345
346/// Returns the common type for the indices of the single-index GEP lanes of
347/// a GEP node with the main op \p VL0, or nullptr if no such type exists.
348/// \p IsGEPLane tells which lanes of \p VL are matching GEPs, whose index is
349/// used as is; all other lanes (copyable, poison, non-GEP pointers) are
350/// modeled as gep V, 0 and just take a zero index of the common type.
351/// The common type is the index type of \p VL0 if all matching lanes share
352/// it. Otherwise the constant indices are cast: to the pointer index type if
353/// there are no non-constant indices (or the index type of \p VL0 already is
354/// the pointer index type), or to the index type of \p VL0 if all the
355/// non-constant indices have that type and all the constants are
356/// representable in it (GEP indices are sign-extended to the pointer index
357/// width, so the value must fit as a signed number). A non-constant index of
358/// a different type cannot be cast without a new instruction, so no common
359/// type exists.
360Type *getCommonGEPIndexType(ArrayRef<Value *> VL, Instruction *VL0,
361 function_ref<bool(Value *)> IsGEPLane,
362 const DataLayout &DL);
363
364/// Checks if the pointers \p PointerOps of the gathered loads, which are not
365/// compatible in the usual sense (some of them are constant-offset pointers,
366/// some have runtime indices), still form a cheap address vector as a
367/// copyable GEP node: a splat of the common base and a vector of indices.
368/// The constant-offset lanes are the base itself or single-index GEPs of the
369/// base with a constant index (modeled as gep V, 0 or as matching lanes with
370/// constant indices), the other lanes are single-index GEPs of the base of
371/// the same shape, whose indices are affine in a single runtime value (the
372/// stride): optionally cast (all with the same cast opcode) values, each of
373/// which is the stride itself or a binary operation of the stride and a
374/// constant, like b[i * S] or b[S + i]. Duplicate lanes are not accepted:
375/// the node would be a shuffled non-full vector, gathering the loads is
376/// cheaper then.
377bool isCopyableGEPAddressVector(ArrayRef<Value *> PointerOps);
378
379/// Shuffles \p Mask in accordance with the given \p SubMask.
380/// \param ExtendingManyInputs Supports reshuffling of the mask with not only
381/// one but two input vectors.
382void addMask(SmallVectorImpl<int> &Mask, ArrayRef<int> SubMask,
383 bool ExtendingManyInputs = false);
384
385/// Order may have elements assigned special value (size) which is out of
386/// bounds. Such indices only appear on places which correspond to undef values
387/// (see canReuseExtract for details) and used in order to avoid undef values
388/// have effect on operands ordering.
389/// The first loop below simply finds all unused indices and then the next loop
390/// nest assigns these indices for undef values positions.
391/// As an example below Order has two undef positions and they have assigned
392/// values 3 and 7 respectively:
393/// before: 6 9 5 4 9 2 1 0
394/// after: 6 3 5 4 7 2 1 0
395void fixupOrderingIndices(MutableArrayRef<unsigned> Order);
396
397/// \returns a bitset for selecting opcodes. false for Opcode0 and true for
398/// Opcode1.
399SmallBitVector getAltInstrMask(ArrayRef<Value *> VL, Type *ScalarTy,
400 unsigned Opcode0, unsigned Opcode1);
401
402/// Replicates the given \p Val \p VF times.
403SmallVector<Constant *> replicateMask(ArrayRef<Constant *> Val, unsigned VF);
404
405/// \returns the masked division/remainder intrinsic corresponding to \p
406/// Opcode. Disabled lanes of these intrinsics are poison rather than UB,
407/// unlike the plain opcode.
408Intrinsic::ID getMaskedDivRemIntrinsic(unsigned Opcode);
409
410/// Returns true if \p I forms a vectorizable bundle on its own and its single
411/// user does not tear the vector apart. Loads and addresses are excluded: the
412/// tree is built without the users, so it does not pay off the extracts. A
413/// cast, feeding a multi-used cast, is excluded for the same reason, such a
414/// user stays scalar. The fp-to-int conversions move the result to the other
415/// register domain, so the extracts are paid on top of the repacking. The
416/// values, feeding the inserts, are vectorized together with them by the
417/// dedicated attempt.
418bool isOnceUsedSeed(const Instruction *I);
419
420/// If \p V is a single-use fpext of a single-use fptrunc forming a round-trip
421/// back to the type of \p V, returns the fptrunc; the round-trip source is its
422/// operand, always an instruction of the same type as \p V. If
423/// \p MustBeElidable, matches only when the intermediate rounding may be
424/// removed: both casts must allow contraction and the widening cast cannot
425/// produce nan/inf.
426Instruction *lookThroughCastRoundTrip(Value *V, bool MustBeElidable);
427
428/// Narrow reduction leaf: the value, the shift applied after widening and
429/// the mask applied in the narrow type before widening, clearing the bits
430/// the absorbed narrow shls shift out and applying the absorbed narrow
431/// and-masks. Lossless narrow shls contribute their known-zero bits to the
432/// mask so matching lanes can form a splat. All-ones mask means nothing
433/// was absorbed and no 'and' is needed.
434struct NarrowedLeafInfo {
435 NarrowedLeafInfo(Value *V, unsigned Shift, APInt Mask)
436 : V(V), Shift(Shift), Mask(std::move(Mask)) {}
437
438 Value *V = nullptr;
439 unsigned Shift = 0;
440 APInt Mask;
441 /// Set to false, if the or chain is not disjoint
442 bool Disjoint = true;
443};
444
445/// Recursively collects the narrow leaves of the widened reduction value
446/// \p V. zext is looked through directly, same-kind binops per operand,
447/// shl of a zext - only if no bits are shifted out in the current type,
448/// shls in narrower types fold into the shift and ands with a constant into
449/// the mask applied in the narrow type. Also collects the looked-through
450/// instructions into \p ChainInsts.
451void collectNarrowedLeaves(Value *V, unsigned RdxOpcode, unsigned WideBW,
452 unsigned MaxDepth,
453 SmallVectorImpl<NarrowedLeafInfo> &Leaves,
454 SmallVectorImpl<Instruction *> &ChainInsts);
455
456TargetTransformInfo::TargetCostKind getSLPCostKind(const Function *F);
457
458/// Returns a saturating unsigned upper bound of the scalar V. The numeric
459/// bound keeps precision on arithmetic carries, where bit-wise analysis
460/// loses it.
461APInt getScalarMaxValue(const Value *V, unsigned Depth = 0);
462
463/// Checks if the values in \p VL are zero-extended sub-fields of the same
464/// wider integer scalar. Returns the source scalar, the field width and the
465/// field permutation mask. The extraction dual of the lane-packing layout.
466/// The field-to-lane mapping of the bitcast to the field vector is defined
467/// for little-endian targets only.
468std::optional<std::tuple<Value *, unsigned, SmallVector<int>>>
469matchGatheredExtractedFields(ArrayRef<Value *> VL, const DataLayout &DL);
470
471/// Description of a bitfield packing of vector lanes into a scalar value:
472/// every lane contributes a disjoint contiguous byte field of the result.
473struct BitPackInfo {
474 static constexpr unsigned NoLane = std::numeric_limits<unsigned>::max();
475 unsigned FieldWidth = 0;
476 /// Lane covering each field, NoLane if the field is always zero.
477 SmallVector<unsigned, 8> LaneOfField;
478 /// Per-lane right-shift amounts bringing the field content to the low bits.
479 SmallVector<uint64_t, 8> LShrAmts;
480
481 /// True if any lane needs a right shift to align its field content.
482 bool needsShift() const {
483 return any_of(Range: LShrAmts, P: [](uint64_t A) { return A != 0; });
484 }
485};
486
487/// Computes the bitfield packing layout from the per-lane possibly set bits
488/// of the source values, the per-lane left-shift amounts and the per-lane
489/// masks (all-ones for unmasked lanes).
490std::optional<BitPackInfo> computeBitPackInfo(unsigned BitWidth,
491 ArrayRef<APInt> PossibleBits,
492 ArrayRef<uint64_t> ShlAmts,
493 ArrayRef<APInt> Masks);
494
495/// Returns the byte shuffle mask packing the per-lane fields of the shifted
496/// lanes (BytesPerLane bytes each) into the packed scalar of NumBytes bytes.
497SmallVector<int> getBitPackMask(const BitPackInfo &Info, unsigned NumBytes,
498 unsigned NumElts, unsigned BytesPerLane);
499
500/// Builds the bitfield packing of X per the layout and the shift width.
501/// \p NumInsts returns the number of emitted instructions.
502Value *buildBitPack(IRBuilderBase &Builder, Value *X, const BitPackInfo &Info,
503 unsigned ShiftWidth, unsigned &NumInsts);
504
505/// The debug values of the erased \p From are kept on its replacement \p To.
506/// A record placed before \p To is cloned right after it, while the original
507/// one is killed together with the scalar, so the variable is undefined up to
508/// \p To. The clone is skipped if it would pass a record of the same variable,
509/// otherwise the variable would show a stale value.
510void redirectDbgValues(Instruction &From, Value &To);
511
512} // namespace llvm::slpvectorizer
513
514#endif // LLVM_LIB_TRANSFORMS_VECTORIZE_SLPVECTORIZER_SLPUTILS_H
515