1//===- LoopVectorize.cpp - A Loop Vectorizer ------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This is the LLVM loop vectorizer. This pass modifies 'vectorizable' loops
10// and generates target-independent LLVM-IR.
11// The vectorizer uses the TargetTransformInfo analysis to estimate the costs
12// of instructions in order to estimate the profitability of vectorization.
13//
14// The loop vectorizer combines consecutive loop iterations into a single
15// 'wide' iteration. After this transformation the index is incremented
16// by the SIMD vector width, and not by one.
17//
18// This pass has three parts:
19// 1. The main loop pass that drives the different parts.
20// 2. LoopVectorizationLegality - A unit that checks for the legality
21// of the vectorization.
22// 3. InnerLoopVectorizer - A unit that performs the actual
23// widening of instructions.
24// 4. LoopVectorizationCostModel - A unit that checks for the profitability
25// of vectorization. It decides on the optimal vector width, which
26// can be one, if vectorization is not profitable.
27//
28// There is a development effort going on to migrate loop vectorizer to the
29// VPlan infrastructure and to introduce outer loop vectorization support (see
30// docs/VectorizationPlan.rst and
31// http://lists.llvm.org/pipermail/llvm-dev/2017-December/119523.html). For this
32// purpose, we temporarily introduced the VPlan-native vectorization path: an
33// alternative vectorization path that is natively implemented on top of the
34// VPlan infrastructure. See EnableVPlanNativePath for enabling.
35//
36//===----------------------------------------------------------------------===//
37//
38// The reduction-variable vectorization is based on the paper:
39// D. Nuzman and R. Henderson. Multi-platform Auto-vectorization.
40//
41// Variable uniformity checks are inspired by:
42// Karrenberg, R. and Hack, S. Whole Function Vectorization.
43//
44// The interleaved access vectorization is based on the paper:
45// Dorit Nuzman, Ira Rosen and Ayal Zaks. Auto-Vectorization of Interleaved
46// Data for SIMD
47//
48// Other ideas/concepts are from:
49// A. Zaks and D. Nuzman. Autovectorization in GCC-two years later.
50//
51// S. Maleki, Y. Gao, M. Garzaran, T. Wong and D. Padua. An Evaluation of
52// Vectorizing Compilers.
53//
54//===----------------------------------------------------------------------===//
55
56#include "llvm/Transforms/Vectorize/LoopVectorize.h"
57#include "LoopVectorizationPlanner.h"
58#include "VPRecipeBuilder.h"
59#include "VPlan.h"
60#include "VPlanAnalysis.h"
61#include "VPlanCFG.h"
62#include "VPlanHelpers.h"
63#include "VPlanPatternMatch.h"
64#include "VPlanTransforms.h"
65#include "VPlanUtils.h"
66#include "VPlanVerifier.h"
67#include "llvm/ADT/APInt.h"
68#include "llvm/ADT/ArrayRef.h"
69#include "llvm/ADT/DenseMap.h"
70#include "llvm/ADT/Hashing.h"
71#include "llvm/ADT/MapVector.h"
72#include "llvm/ADT/STLExtras.h"
73#include "llvm/ADT/SmallPtrSet.h"
74#include "llvm/ADT/SmallVector.h"
75#include "llvm/ADT/Statistic.h"
76#include "llvm/ADT/StringRef.h"
77#include "llvm/ADT/Twine.h"
78#include "llvm/ADT/TypeSwitch.h"
79#include "llvm/ADT/iterator_range.h"
80#include "llvm/Analysis/AssumptionCache.h"
81#include "llvm/Analysis/BasicAliasAnalysis.h"
82#include "llvm/Analysis/BlockFrequencyInfo.h"
83#include "llvm/Analysis/BranchProbabilityInfo.h"
84#include "llvm/Analysis/CFG.h"
85#include "llvm/Analysis/CodeMetrics.h"
86#include "llvm/Analysis/CycleAnalysis.h"
87#include "llvm/Analysis/DemandedBits.h"
88#include "llvm/Analysis/GlobalsModRef.h"
89#include "llvm/Analysis/LoopAccessAnalysis.h"
90#include "llvm/Analysis/LoopAnalysisManager.h"
91#include "llvm/Analysis/LoopInfo.h"
92#include "llvm/Analysis/LoopIterator.h"
93#include "llvm/Analysis/OptimizationRemarkEmitter.h"
94#include "llvm/Analysis/ProfileSummaryInfo.h"
95#include "llvm/Analysis/ScalarEvolution.h"
96#include "llvm/Analysis/ScalarEvolutionExpressions.h"
97#include "llvm/Analysis/ScalarEvolutionPatternMatch.h"
98#include "llvm/Analysis/TargetLibraryInfo.h"
99#include "llvm/Analysis/TargetTransformInfo.h"
100#include "llvm/Analysis/ValueTracking.h"
101#include "llvm/Analysis/VectorUtils.h"
102#include "llvm/IR/Attributes.h"
103#include "llvm/IR/BasicBlock.h"
104#include "llvm/IR/CFG.h"
105#include "llvm/IR/Constant.h"
106#include "llvm/IR/Constants.h"
107#include "llvm/IR/DataLayout.h"
108#include "llvm/IR/DebugInfo.h"
109#include "llvm/IR/DebugLoc.h"
110#include "llvm/IR/DerivedTypes.h"
111#include "llvm/IR/DiagnosticInfo.h"
112#include "llvm/IR/Dominators.h"
113#include "llvm/IR/Function.h"
114#include "llvm/IR/IRBuilder.h"
115#include "llvm/IR/InstrTypes.h"
116#include "llvm/IR/Instruction.h"
117#include "llvm/IR/Instructions.h"
118#include "llvm/IR/IntrinsicInst.h"
119#include "llvm/IR/Intrinsics.h"
120#include "llvm/IR/MDBuilder.h"
121#include "llvm/IR/Metadata.h"
122#include "llvm/IR/Module.h"
123#include "llvm/IR/Operator.h"
124#include "llvm/IR/PatternMatch.h"
125#include "llvm/IR/ProfDataUtils.h"
126#include "llvm/IR/Type.h"
127#include "llvm/IR/Use.h"
128#include "llvm/IR/User.h"
129#include "llvm/IR/Value.h"
130#include "llvm/IR/Verifier.h"
131#include "llvm/Support/Casting.h"
132#include "llvm/Support/CommandLine.h"
133#include "llvm/Support/Debug.h"
134#include "llvm/Support/ErrorHandling.h"
135#include "llvm/Support/InstructionCost.h"
136#include "llvm/Support/MathExtras.h"
137#include "llvm/Support/NativeFormatting.h"
138#include "llvm/Support/raw_ostream.h"
139#include "llvm/Transforms/Utils/BasicBlockUtils.h"
140#include "llvm/Transforms/Utils/InjectTLIMappings.h"
141#include "llvm/Transforms/Utils/Local.h"
142#include "llvm/Transforms/Utils/LoopSimplify.h"
143#include "llvm/Transforms/Utils/LoopUtils.h"
144#include "llvm/Transforms/Utils/LoopVersioning.h"
145#include "llvm/Transforms/Utils/ScalarEvolutionExpander.h"
146#include "llvm/Transforms/Utils/SizeOpts.h"
147#include "llvm/Transforms/Vectorize/LoopVectorizationLegality.h"
148#include <algorithm>
149#include <cassert>
150#include <cmath>
151#include <cstdint>
152#include <functional>
153#include <iterator>
154#include <memory>
155#include <string>
156#include <tuple>
157#include <utility>
158
159using namespace llvm;
160using namespace SCEVPatternMatch;
161using namespace LoopVectorizationUtils;
162
163#define LV_NAME "loop-vectorize"
164#define DEBUG_TYPE LV_NAME
165
166#ifndef NDEBUG
167const char VerboseDebug[] = DEBUG_TYPE "-verbose";
168#endif
169
170STATISTIC(LoopsVectorized, "Number of loops vectorized");
171STATISTIC(LoopsAnalyzed, "Number of loops analyzed for vectorization");
172STATISTIC(LoopsEpilogueVectorized, "Number of epilogues vectorized");
173STATISTIC(LoopsEarlyExitVectorized, "Number of early exit loops vectorized");
174STATISTIC(LoopsPartialAliasVectorized,
175 "Number of partial aliasing loops vectorized");
176
177static cl::opt<bool> EnableEpilogueVectorization(
178 "enable-epilogue-vectorization", cl::init(Val: true), cl::Hidden,
179 cl::desc("Enable vectorization of epilogue loops."));
180
181static cl::opt<ElementCount> EpilogueVectorizationForceVF(
182 "epilogue-vectorization-force-VF", cl::init(Val: ElementCount::getFixed(MinVal: 1)),
183 cl::Hidden,
184 cl::desc("When epilogue vectorization is enabled, and a value greater than "
185 "1 is specified, forces the given VF for all applicable epilogue "
186 "loops. Note: This allows all scalable VFs >= vscale x 1."));
187
188static cl::opt<unsigned> EpilogueVectorizationMinVF(
189 "epilogue-vectorization-minimum-VF", cl::Hidden,
190 cl::desc("Only loops with vectorization factor equal to or larger than "
191 "the specified value are considered for epilogue vectorization."));
192
193/// Loops with a known constant trip count below this number are vectorized only
194/// if no scalar iteration overheads are incurred.
195static cl::opt<unsigned> TinyTripCountVectorThreshold(
196 "vectorizer-min-trip-count", cl::init(Val: 16), cl::Hidden,
197 cl::desc("Loops with a constant trip count that is smaller than this "
198 "value are vectorized only if no scalar iteration overheads "
199 "are incurred."));
200
201static cl::opt<bool> ForcePartialAliasingVectorization(
202 "force-partial-aliasing-vectorization", cl::init(Val: false), cl::Hidden,
203 cl::desc("Replace pointer diff checks with alias masks."));
204
205/// Option tail-folding-policy controls the tail-folding strategy and lists all
206/// available options. The vectorizer will attempt to fold the tail-loop into
207/// the vector loop (main/epilogue loops) and predicate the instructions
208/// accordingly. If tail-folding fails, there are different fallback strategies
209/// depending on these values:
210enum class TailFoldingPolicyTy { None = 0, PreferFoldTail, MustFoldTail };
211
212static cl::opt<TailFoldingPolicyTy> TailFoldingPolicy(
213 "tail-folding-policy", cl::init(Val: TailFoldingPolicyTy::None), cl::Hidden,
214 cl::desc("Tail-folding preferences over creating an epilogue loop."),
215 cl::values(
216 clEnumValN(TailFoldingPolicyTy::None, "dont-fold-tail",
217 "Don't tail-fold loops."),
218 clEnumValN(TailFoldingPolicyTy::PreferFoldTail, "prefer-fold-tail",
219 "prefer tail-folding, otherwise create an epilogue when "
220 "appropriate."),
221 clEnumValN(TailFoldingPolicyTy::MustFoldTail, "must-fold-tail",
222 "always tail-fold, don't attempt vectorization if "
223 "tail-folding fails.")));
224
225static cl::opt<TailFoldingPolicyTy> EpilogueTailFoldingPolicy(
226 "epilogue-tail-folding-policy", cl::Hidden,
227 cl::desc(
228 "Epilogue-tail-folding preferences over creating an epilogue loop."),
229 cl::values(
230 clEnumValN(TailFoldingPolicyTy::None, "dont-fold-tail",
231 "Don't tail-fold loops."),
232 clEnumValN(TailFoldingPolicyTy::PreferFoldTail, "prefer-fold-tail",
233 "prefer tail-folding, otherwise create an epilogue when "
234 "appropriate.")));
235
236static cl::opt<TailFoldingStyle> ForceTailFoldingStyle(
237 "force-tail-folding-style", cl::desc("Force the tail folding style"),
238 cl::init(Val: TailFoldingStyle::None),
239 cl::values(
240 clEnumValN(TailFoldingStyle::None, "none", "Disable tail folding"),
241 clEnumValN(
242 TailFoldingStyle::Data, "data",
243 "Create lane mask for data only, using active.lane.mask intrinsic"),
244 clEnumValN(TailFoldingStyle::DataWithoutLaneMask,
245 "data-without-lane-mask",
246 "Create lane mask with compare/stepvector"),
247 clEnumValN(TailFoldingStyle::DataAndControlFlow, "data-and-control",
248 "Create lane mask using active.lane.mask intrinsic, and use "
249 "it for both data and control flow"),
250 clEnumValN(TailFoldingStyle::DataWithEVL, "data-with-evl",
251 "Use predicated EVL instructions for tail folding. If EVL "
252 "is unsupported, fallback to data-without-lane-mask.")));
253
254static cl::opt<bool> EnableInterleavedMemAccesses(
255 "enable-interleaved-mem-accesses", cl::init(Val: false), cl::Hidden,
256 cl::desc("Enable vectorization on interleaved memory accesses in a loop"));
257
258/// An interleave-group may need masking if it resides in a block that needs
259/// predication, or in order to mask away gaps.
260static cl::opt<bool> EnableMaskedInterleavedMemAccesses(
261 "enable-masked-interleaved-mem-accesses", cl::init(Val: false), cl::Hidden,
262 cl::desc("Enable vectorization on masked interleaved memory accesses in a loop"));
263
264static cl::opt<unsigned> ForceTargetNumScalarRegs(
265 "force-target-num-scalar-regs", cl::init(Val: 0), cl::Hidden,
266 cl::desc("A flag that overrides the target's number of scalar registers."));
267
268static cl::opt<unsigned> ForceTargetNumVectorRegs(
269 "force-target-num-vector-regs", cl::init(Val: 0), cl::Hidden,
270 cl::desc("A flag that overrides the target's number of vector registers."));
271
272static cl::opt<unsigned> ForceTargetMaxScalarInterleaveFactor(
273 "force-target-max-scalar-interleave", cl::init(Val: 0), cl::Hidden,
274 cl::desc("A flag that overrides the target's max interleave factor for "
275 "scalar loops."));
276
277static cl::opt<unsigned> ForceTargetMaxVectorInterleaveFactor(
278 "force-target-max-vector-interleave", cl::init(Val: 0), cl::Hidden,
279 cl::desc("A flag that overrides the target's max interleave factor for "
280 "vectorized loops."));
281
282static cl::opt<unsigned> SmallLoopCost(
283 "small-loop-cost", cl::init(Val: 20), cl::Hidden,
284 cl::desc(
285 "The cost of a loop that is considered 'small' by the interleaver."));
286
287static cl::opt<bool> LoopVectorizeWithBlockFrequency(
288 "loop-vectorize-with-block-frequency", cl::init(Val: true), cl::Hidden,
289 cl::desc("Enable the use of the block frequency analysis to access PGO "
290 "heuristics minimizing code growth in cold regions and being more "
291 "aggressive in hot regions."));
292
293// Runtime interleave loops for load/store throughput.
294static cl::opt<bool> EnableLoadStoreRuntimeInterleave(
295 "enable-loadstore-runtime-interleave", cl::init(Val: true), cl::Hidden,
296 cl::desc(
297 "Enable runtime interleaving until load/store ports are saturated"));
298
299// TODO: Move size-based thresholds out of legality checking, make cost based
300// decisions instead of hard thresholds.
301static cl::opt<unsigned> VectorizeSCEVCheckThreshold(
302 "vectorize-scev-check-threshold", cl::init(Val: 16), cl::Hidden,
303 cl::desc("The maximum number of SCEV checks allowed."));
304
305static cl::opt<unsigned> PragmaVectorizeSCEVCheckThreshold(
306 "pragma-vectorize-scev-check-threshold", cl::init(Val: 128), cl::Hidden,
307 cl::desc("The maximum number of SCEV checks allowed with a "
308 "vectorize(enable) pragma"));
309
310static cl::opt<bool> EnableIndVarRegisterHeur(
311 "enable-ind-var-reg-heur", cl::init(Val: true), cl::Hidden,
312 cl::desc("Count the induction variable only once when interleaving"));
313
314static cl::opt<unsigned> MaxNestedScalarReductionIC(
315 "max-nested-scalar-reduction-interleave", cl::init(Val: 2), cl::Hidden,
316 cl::desc("The maximum interleave count to use when interleaving a scalar "
317 "reduction in a nested loop."));
318
319static cl::opt<bool> ForceOrderedReductions(
320 "force-ordered-reductions", cl::init(Val: false), cl::Hidden,
321 cl::desc("Enable the vectorisation of loops with in-order (strict) "
322 "FP reductions"));
323
324static cl::opt<bool> PreferPredicatedReductionSelect(
325 "prefer-predicated-reduction-select", cl::init(Val: false), cl::Hidden,
326 cl::desc(
327 "Prefer predicating a reduction operation over an after loop select."));
328
329static cl::opt<bool> EnableVPlanNativePath(
330 "enable-vplan-native-path", cl::Hidden,
331 cl::desc("Enable VPlan-native vectorization path with "
332 "support for outer loop vectorization."));
333
334cl::opt<bool>
335 llvm::VerifyEachVPlan("vplan-verify-each",
336#ifdef EXPENSIVE_CHECKS
337 cl::init(true),
338#else
339 cl::init(Val: false),
340#endif
341 cl::Hidden,
342 cl::desc("Verify VPlans after VPlan transforms."));
343
344#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
345cl::opt<bool> llvm::VPlanPrintBeforeAll(
346 "vplan-print-before-all", cl::init(false), cl::Hidden,
347 cl::desc("Print VPlans before all VPlan transformations."));
348
349cl::opt<bool> llvm::VPlanPrintAfterAll(
350 "vplan-print-after-all", cl::init(false), cl::Hidden,
351 cl::desc("Print VPlans after all VPlan transformations."));
352
353cl::list<std::string> llvm::VPlanPrintBeforePasses(
354 "vplan-print-before", cl::Hidden,
355 cl::desc("Print VPlans before specified VPlan transformations (regexp)."));
356
357cl::list<std::string> llvm::VPlanPrintAfterPasses(
358 "vplan-print-after", cl::Hidden,
359 cl::desc("Print VPlans after specified VPlan transformations (regexp)."));
360
361cl::opt<bool> llvm::VPlanPrintVectorRegionScope(
362 "vplan-print-vector-region-scope", cl::init(false), cl::Hidden,
363 cl::desc("Limit VPlan printing to vector loop region in "
364 "`-vplan-print-after*` if the plan has one."));
365#endif
366
367cl::opt<bool> llvm::EnableLoopInterleaving(
368 "interleave-loops", cl::init(Val: true), cl::Hidden,
369 cl::desc("Enable loop interleaving in Loop vectorization passes"));
370cl::opt<bool> llvm::EnableLoopVectorization(
371 "vectorize-loops", cl::init(Val: true), cl::Hidden,
372 cl::desc("Run the Loop vectorization passes"));
373
374namespace llvm {
375cl::opt<unsigned> ForceTargetInstructionCost(
376 "force-target-instruction-cost", cl::init(Val: 0), cl::Hidden,
377 cl::desc("A flag that overrides the target's expected cost for "
378 "an instruction to a single constant value. Mostly "
379 "useful for getting consistent testing."));
380
381/// The number of stores in a loop that are allowed to need predication.
382cl::opt<unsigned> NumberOfStoresToPredicate(
383 "vectorize-num-stores-pred", cl::init(Val: 1), cl::Hidden,
384 cl::desc("Max number of stores to be predicated behind an if."));
385
386// This flag enables the stress testing of the VPlan H-CFG construction in the
387// VPlan-native vectorization path. It must be used in conjuction with
388// -enable-vplan-native-path. -vplan-verify-hcfg can also be used to enable the
389// verification of the H-CFGs built.
390cl::opt<bool> VPlanBuildOuterloopStressTest(
391 "vplan-build-outerloop-stress-test", cl::init(Val: false), cl::Hidden,
392 cl::desc(
393 "Build VPlan for every supported loop nest in the function and bail "
394 "out right after the build (stress test the VPlan H-CFG construction "
395 "in the VPlan-native vectorization path)."));
396} // namespace llvm
397
398static cl::opt<cl::boolOrDefault>
399 ForceMaskedDivRem("force-widen-divrem-via-masked-intrinsic", cl::Hidden,
400 cl::desc("Override cost based masked intrinsic widening "
401 "for div/rem instructions"));
402
403static cl::opt<bool> EnableEarlyExitVectorization(
404 "enable-early-exit-vectorization", cl::init(Val: true), cl::Hidden,
405 cl::desc(
406 "Enable vectorization of early exit loops with uncountable exits."));
407
408static cl::opt<bool> EnableEarlyExitVectorizationWithSideEffects(
409 "enable-early-exit-vectorization-with-side-effects", cl::init(Val: false),
410 cl::Hidden,
411 cl::desc("Enable vectorization of early exit loops with uncountable exits "
412 "and side effects"));
413
414static cl::opt<unsigned> LowTripCountLoopBodySizeLimit(
415 "low-trip-count-loop-body-size-limit", cl::init(Val: 20), cl::Hidden,
416 cl::desc("Minimum number of instructions to vectorize loops with trip "
417 "counts below tail folding threshold"));
418
419// Returns true if the epilogue VF has been set to a non-zero value other than
420// VF=1 (scalar).
421static bool hasForcedEpilogueVF() {
422 return EpilogueVectorizationForceVF.isNonZero() &&
423 EpilogueVectorizationForceVF != ElementCount::getFixed(MinVal: 1);
424}
425
426// Likelyhood of bypassing the vectorized loop because there are zero trips left
427// after prolog. See `emitIterationCountCheck`.
428static constexpr uint32_t MinItersBypassWeights[] = {1, 127};
429
430/// A version of ScalarEvolution::getSmallConstantTripCount that returns an
431/// ElementCount to include loops whose trip count is a function of vscale.
432static ElementCount getSmallConstantTripCount(ScalarEvolution *SE,
433 const Loop *L) {
434 if (unsigned ExpectedTC = SE->getSmallConstantTripCount(L))
435 return ElementCount::getFixed(MinVal: ExpectedTC);
436
437 const SCEV *BTC = SE->getBackedgeTakenCount(L);
438 if (isa<SCEVCouldNotCompute>(Val: BTC))
439 return ElementCount::getFixed(MinVal: 0);
440
441 const SCEV *ExitCount = SE->getTripCountFromExitCount(ExitCount: BTC, EvalTy: BTC->getType(), L);
442 if (isa<SCEVVScale>(Val: ExitCount))
443 return ElementCount::getScalable(MinVal: 1);
444
445 const APInt *Scale;
446 if (match(S: ExitCount, P: m_scev_Mul(Op0: m_scev_APInt(C&: Scale), Op1: m_SCEVVScale())))
447 if (cast<SCEVMulExpr>(Val: ExitCount)->hasNoUnsignedWrap())
448 if (Scale->getActiveBits() <= 32)
449 return ElementCount::getScalable(MinVal: Scale->getZExtValue());
450
451 return ElementCount::getFixed(MinVal: 0);
452}
453
454/// Get the maximum trip count for \p L from the SCEV unsigned range, excluding
455/// zero from the range. Only valid when not folding the tail, as the minimum
456/// iteration count check guards against a zero trip count. Returns 0 if
457/// unknown.
458static unsigned getMaxTCFromNonZeroRange(PredicatedScalarEvolution &PSE,
459 Loop *L) {
460 const SCEV *BTC = PSE.getBackedgeTakenCount();
461 if (isa<SCEVCouldNotCompute>(Val: BTC))
462 return 0;
463 ScalarEvolution *SE = PSE.getSE();
464 const SCEV *TripCount = SE->getTripCountFromExitCount(ExitCount: BTC, EvalTy: BTC->getType(), L);
465 ConstantRange TCRange = SE->getUnsignedRange(S: TripCount);
466 APInt MaxTCFromRange = TCRange.getUnsignedMax();
467 if (!MaxTCFromRange.isZero() && MaxTCFromRange.getActiveBits() <= 32)
468 return MaxTCFromRange.getZExtValue();
469 return 0;
470}
471
472/// Returns "best known" trip count, which is either a valid positive trip count
473/// or std::nullopt when an estimate cannot be made (including when the trip
474/// count would overflow), for the specified loop \p L as defined by the
475/// following procedure:
476/// 1) Returns exact trip count if it is known.
477/// 2) Returns expected trip count according to profile data if any.
478/// 3) Returns upper bound estimate if known, if \p CanUseConstantMax, and
479/// if \p ComputeUpperBoundOnly is false.
480/// 4) Returns the maximum trip count from the SCEV range excluding zero,
481/// if \p CanUseConstantMax and \p CanExcludeZeroTrips.
482/// 5) Returns std::nullopt if all of the above failed.
483static std::optional<ElementCount> getSmallBestKnownTC(
484 PredicatedScalarEvolution &PSE, Loop *L, bool CanUseConstantMax = true,
485 bool CanExcludeZeroTrips = false, bool ComputeUpperBoundOnly = false) {
486 // Check if exact trip count is known.
487 if (auto ExpectedTC = getSmallConstantTripCount(SE: PSE.getSE(), L))
488 return ExpectedTC;
489
490 // Check if there is an expected trip count available from profile data.
491 // An estimate of zero means the loop is estimated not to be entered; it is
492 // not a usable trip count for the profitability decisions below (and would
493 // e.g. divide by zero when scaling runtime check cost), so treat it as
494 // unknown.
495 if (LoopVectorizeWithBlockFrequency && !ComputeUpperBoundOnly)
496 if (unsigned EstimatedTC = getLoopEstimatedTripCount(L).value_or(u: 0))
497 return ElementCount::getFixed(MinVal: EstimatedTC);
498
499 if (!CanUseConstantMax)
500 return std::nullopt;
501
502 // Check if upper bound estimate is known.
503 if (unsigned ExpectedTC = PSE.getSmallConstantMaxTripCount())
504 return ElementCount::getFixed(MinVal: ExpectedTC);
505
506 // Get the maximum trip count from the SCEV range excluding zero. This is
507 // only safe when not folding the tail, as the minimum iteration count check
508 // prevents entering the vector loop with a zero trip count.
509 if (CanUseConstantMax && CanExcludeZeroTrips)
510 if (unsigned RefinedTC = getMaxTCFromNonZeroRange(PSE, L))
511 return ElementCount::getFixed(MinVal: RefinedTC);
512
513 return std::nullopt;
514}
515
516namespace {
517// Forward declare GeneratedRTChecks.
518class GeneratedRTChecks;
519
520using SCEV2ValueTy = DenseMap<const SCEV *, Value *>;
521} // namespace
522
523namespace llvm {
524
525AnalysisKey ShouldRunExtraVectorPasses::Key;
526
527/// InnerLoopVectorizer vectorizes loops which contain only one basic
528/// block to a specified vectorization factor (VF).
529/// This class performs the widening of scalars into vectors, or multiple
530/// scalars. This class also implements the following features:
531/// * It inserts an epilogue loop for handling loops that don't have iteration
532/// counts that are known to be a multiple of the vectorization factor.
533/// * It handles the code generation for reduction variables.
534/// * Scalarization (implementation using scalars) of un-vectorizable
535/// instructions.
536/// InnerLoopVectorizer does not perform any vectorization-legality
537/// checks, and relies on the caller to check for the different legality
538/// aspects. The InnerLoopVectorizer relies on the
539/// LoopVectorizationLegality class to provide information about the induction
540/// and reduction variables that were found to a given vectorization factor.
541class InnerLoopVectorizer {
542public:
543 InnerLoopVectorizer(Loop *OrigLoop, PredicatedScalarEvolution &PSE,
544 LoopInfo *LI, DominatorTree *DT,
545 const TargetTransformInfo *TTI, AssumptionCache *AC,
546 ElementCount VecWidth, unsigned UnrollFactor,
547 GeneratedRTChecks &RTChecks, VPlan &Plan)
548 : OrigLoop(OrigLoop), PSE(PSE), LI(LI), DT(DT), TTI(TTI), AC(AC),
549 VF(VecWidth), UF(UnrollFactor), Builder(PSE.getSE()->getModule()),
550 RTChecks(RTChecks), Plan(Plan),
551 VectorPHVPBB(cast<VPBasicBlock>(
552 Val: Plan.getVectorLoopRegion()->getSinglePredecessor())) {}
553
554 virtual ~InnerLoopVectorizer() = default;
555
556 /// Creates a basic block for the scalar preheader.
557 /// EpilogueVectorizerEpilogueLoop overrides the method to create additional
558 /// blocks and checks needed for epilogue vectorization.
559 virtual BasicBlock *createVectorizedLoopSkeleton();
560
561 /// Fix the vectorized code, taking care of header phi's, and more.
562 void fixVectorizedLoop(VPTransformState &State);
563
564protected:
565 friend class LoopVectorizationPlanner;
566
567 /// Create and return a new IR basic block for the scalar preheader whose name
568 /// is prefixed with \p Prefix.
569 BasicBlock *createScalarPreheader(StringRef Prefix);
570
571 /// The original loop.
572 Loop *OrigLoop;
573
574 /// A wrapper around ScalarEvolution used to add runtime SCEV checks. Applies
575 /// dynamic knowledge to simplify SCEV expressions and converts them to a
576 /// more usable form.
577 PredicatedScalarEvolution &PSE;
578
579 /// Loop Info.
580 LoopInfo *LI;
581
582 /// Dominator Tree.
583 DominatorTree *DT;
584
585 /// Target Transform Info.
586 const TargetTransformInfo *TTI;
587
588 /// Assumption Cache.
589 AssumptionCache *AC;
590
591 /// The vectorization SIMD factor to use. Each vector will have this many
592 /// vector elements.
593 ElementCount VF;
594
595 /// The vectorization unroll factor to use. Each scalar is vectorized to this
596 /// many different vector instructions.
597 unsigned UF;
598
599 /// The builder that we use
600 IRBuilder<> Builder;
601
602 // --- Vectorization state ---
603
604 /// Structure to hold information about generated runtime checks, responsible
605 /// for cleaning the checks, if vectorization turns out unprofitable.
606 GeneratedRTChecks &RTChecks;
607
608 VPlan &Plan;
609
610 /// The vector preheader block of \p Plan, used as target for check blocks
611 /// introduced during skeleton creation.
612 VPBasicBlock *VectorPHVPBB;
613};
614
615/// A specialized derived class of inner loop vectorizer that performs
616/// vectorization of *epilogue* loops in the process of vectorizing loops and
617/// their epilogues. The idea is to run the vplan on a given loop twice, firstly
618/// to vectorize the main loop, and secondly to complete the skeleton from the
619/// first step and vectorize the epilogue. This helps us avoid regenerating and
620/// recomputing runtime safety checks, and shortens the iteration-count-check
621/// path length for loops whose iteration count is so small that the main vector
622/// loop is completely skipped.
623class EpilogueVectorizerEpilogueLoop : public InnerLoopVectorizer {
624 VPlan &MainPlan;
625
626public:
627 VPIRBasicBlock *VecEpilogueIterationCountCheck = nullptr;
628
629 EpilogueVectorizerEpilogueLoop(Loop *OrigLoop, PredicatedScalarEvolution &PSE,
630 LoopInfo *LI, DominatorTree *DT,
631 const TargetTransformInfo *TTI,
632 AssumptionCache *AC, ElementCount VecWidth,
633 unsigned UnrollFactor,
634 GeneratedRTChecks &Checks, VPlan &Plan,
635 VPlan &MainPlan)
636 : InnerLoopVectorizer(OrigLoop, PSE, LI, DT, TTI, AC, VecWidth,
637 UnrollFactor, Checks, Plan),
638 MainPlan(MainPlan) {}
639 /// Implements the interface for creating a vectorized skeleton using the
640 /// *epilogue loop* strategy (i.e., the second pass of VPlan execution).
641 BasicBlock *createVectorizedLoopSkeleton() final;
642};
643} // end namespace llvm
644
645/// Look for a meaningful debug location on the instruction or its operands.
646static DebugLoc getDebugLocFromInstOrOperands(Instruction *I) {
647 if (!I)
648 return DebugLoc::getUnknown();
649
650 DebugLoc Empty;
651 if (I->getDebugLoc() != Empty)
652 return I->getDebugLoc();
653
654 for (Use &Op : I->operands()) {
655 if (Instruction *OpInst = dyn_cast<Instruction>(Val&: Op))
656 if (OpInst->getDebugLoc() != Empty)
657 return OpInst->getDebugLoc();
658 }
659
660 return I->getDebugLoc();
661}
662
663namespace llvm {
664
665/// Return the runtime value for VF.
666Value *getRuntimeVF(IRBuilderBase &B, Type *Ty, ElementCount VF) {
667 return B.CreateElementCount(Ty, EC: VF);
668}
669
670} // end namespace llvm
671
672namespace llvm {
673
674// Loop vectorization cost-model hints how the epilogue/tail loop should be
675// lowered.
676enum EpilogueLowering {
677
678 // The default: allowing epilogues.
679 CM_EpilogueAllowed,
680
681 // Vectorization with OptForSize: don't allow epilogues.
682 CM_EpilogueNotAllowedOptSize,
683
684 // A special case of vectorisation with OptForSize: loops with a very small
685 // trip count are considered for vectorization under OptForSize, thereby
686 // making sure the cost of their loop body is dominant, free of runtime
687 // guards and scalar iteration overheads.
688 CM_EpilogueNotAllowedLowTripLoop,
689
690 // Loop hint indicating an epilogue is undesired, apply tail folding.
691 CM_EpilogueNotNeededFoldTail,
692
693 // Directive indicating we must either fold the epilogue/tail or not vectorize
694 CM_EpilogueNotAllowedFoldTail
695};
696
697enum class AliasMaskingStatus { NotDecided, Disabled, Enabled };
698
699/// LoopVectorizationCostModel - estimates the expected speedups due to
700/// vectorization.
701/// In many cases vectorization is not profitable. This can happen because of
702/// a number of reasons. In this class we mainly attempt to predict the
703/// expected speedup/slowdowns due to the supported instruction set. We use the
704/// TargetTransformInfo to query the different backends for the cost of
705/// different operations.
706class LoopVectorizationCostModel {
707 friend class LoopVectorizationPlanner;
708
709public:
710 LoopVectorizationCostModel(EpilogueLowering SEL, Loop *L,
711 PredicatedScalarEvolution &PSE, LoopInfo *LI,
712 LoopVectorizationLegality *Legal,
713 const TargetTransformInfo &TTI,
714 const TargetLibraryInfo *TLI, AssumptionCache *AC,
715 OptimizationRemarkEmitter *ORE,
716 std::function<BlockFrequencyInfo &()> GetBFI,
717 const Function *F, InterleavedAccessInfo &IAI,
718 VFSelectionContext &Config)
719 : Config(Config), EpilogueLoweringStatus(SEL), TheLoop(L), PSE(PSE),
720 LI(LI), Legal(Legal), TTI(TTI), TLI(TLI), AC(AC), ORE(ORE),
721 GetBFI(GetBFI), TheFunction(F), InterleaveInfo(IAI) {}
722
723 /// \return An upper bound for the vectorization factors (both fixed and
724 /// scalable). If the factors are 0, vectorization and interleaving should be
725 /// avoided up front.
726 FixedScalableVFPair computeMaxVF(ElementCount UserVF, unsigned UserIC);
727
728 /// Memory access instruction may be vectorized in more than one way.
729 /// Form of instruction after vectorization depends on cost.
730 /// This function takes cost-based decisions for Load/Store instructions
731 /// and collects them in a map. This decisions map is used for building
732 /// the lists of loop-uniform and loop-scalar instructions.
733 /// The calculated cost is saved with widening decision in order to
734 /// avoid redundant calculations.
735 void setCostBasedWideningDecision(ElementCount VF);
736
737 /// Collect values we want to ignore in the cost model.
738 void collectValuesToIgnore();
739
740 /// \returns True if it is more profitable to scalarize instruction \p I for
741 /// vectorization factor \p VF.
742 bool isProfitableToScalarize(Instruction *I, ElementCount VF) const {
743 assert(VF.isVector() &&
744 "Profitable to scalarize relevant only for VF > 1.");
745 assert(
746 TheLoop->isInnermost() &&
747 "cost-model should not be used for outer loops (in VPlan-native path)");
748
749 auto Scalars = InstsToScalarize.find(Key: VF);
750 assert(Scalars != InstsToScalarize.end() &&
751 "VF not yet analyzed for scalarization profitability");
752 return Scalars->second.contains(Key: I);
753 }
754
755 /// Returns true if \p I is known to be uniform after vectorization.
756 bool isUniformAfterVectorization(Instruction *I, ElementCount VF) const {
757 assert(
758 TheLoop->isInnermost() &&
759 "cost-model should not be used for outer loops (in VPlan-native path)");
760
761 // If VF is scalar, then all instructions are trivially uniform.
762 if (VF.isScalar())
763 return true;
764
765 // Pseudo probes must be duplicated per vector lane so that the
766 // profiled loop trip count is not undercounted.
767 if (isa<PseudoProbeInst>(Val: I))
768 return false;
769
770 auto UniformsPerVF = Uniforms.find(Val: VF);
771 assert(UniformsPerVF != Uniforms.end() &&
772 "VF not yet analyzed for uniformity");
773 return UniformsPerVF->second.count(Ptr: I);
774 }
775
776 /// Returns true if \p I is known to be scalar after vectorization.
777 bool isScalarAfterVectorization(Instruction *I, ElementCount VF) const {
778 assert(
779 TheLoop->isInnermost() &&
780 "cost-model should not be used for outer loops (in VPlan-native path)");
781 if (VF.isScalar())
782 return true;
783
784 auto ScalarsPerVF = Scalars.find(Val: VF);
785 assert(ScalarsPerVF != Scalars.end() &&
786 "Scalar values are not calculated for VF");
787 return ScalarsPerVF->second.count(Ptr: I);
788 }
789
790 /// \returns True if instruction \p I can be truncated to a smaller bitwidth
791 /// for vectorization factor \p VF.
792 bool canTruncateToMinimalBitwidth(Instruction *I, ElementCount VF) const {
793 const auto &MinBWs = Config.getMinimalBitwidths();
794 // Truncs must truncate at most to their destination type.
795 if (isa_and_nonnull<TruncInst>(Val: I) && MinBWs.contains(Key: I) &&
796 I->getType()->getScalarSizeInBits() < MinBWs.lookup(Key: I))
797 return false;
798 return VF.isVector() && MinBWs.contains(Key: I) &&
799 !isProfitableToScalarize(I, VF) &&
800 !isScalarAfterVectorization(I, VF);
801 }
802
803 /// Decision that was taken during cost calculation for memory instruction.
804 enum InstWidening {
805 CM_Unknown,
806 CM_Widen, // For consecutive accesses with stride +1.
807 CM_Widen_Reverse, // For consecutive accesses with stride -1.
808 CM_Interleave,
809 CM_GatherScatter,
810 CM_Scalarize,
811 /// A widening decision that has been invalidated after replacing the
812 /// corresponding recipe during VPlan transforms.
813 /// TODO: Remove once the legacy exit cost computation is retired.
814 CM_InvalidatedDecision
815 };
816
817#ifndef NDEBUG
818 static constexpr StringLiteral getInstWideningStr(InstWidening W) {
819 constexpr StringLiteral WideningStr[] = {
820 "Unknown", "Widen", "Widen_Reverse", "Interleave",
821 "GatherScatter", "Scalarize", "InvalidatedDecision"};
822 return WideningStr[W];
823 }
824#endif
825
826 /// Save vectorization decision \p W and \p Cost taken by the cost model for
827 /// instruction \p I and vector width \p VF.
828 void setWideningDecision(Instruction *I, ElementCount VF, InstWidening W,
829 InstructionCost Cost) {
830 assert(VF.isVector() && "Expected VF >=2");
831 LLVM_DEBUG(dbgs() << "LV: Setting widening decision to "
832 << getInstWideningStr(W) << " for VF " << VF
833 << " and instruction: " << *I << '\n');
834 WideningDecisions[{I, VF}] = {W, Cost};
835 }
836
837 /// Save vectorization decision \p W and \p Cost taken by the cost model for
838 /// interleaving group \p Grp and vector width \p VF.
839 void setWideningDecision(const InterleaveGroup<Instruction> *Grp,
840 ElementCount VF, InstWidening W,
841 InstructionCost Cost) {
842 assert(VF.isVector() && "Expected VF >=2");
843 /// Broadcast this decicion to all instructions inside the group.
844 /// When interleaving, the cost will only be assigned one instruction, the
845 /// insert position. For other cases, add the appropriate fraction of the
846 /// total cost to each instruction. This ensures accurate costs are used,
847 /// even if the insert position instruction is not used.
848 InstructionCost InsertPosCost = Cost;
849 InstructionCost OtherMemberCost = 0;
850 if (W != CM_Interleave)
851 OtherMemberCost = InsertPosCost = Cost / Grp->getNumMembers();
852 ;
853 for (auto *I : Grp->members()) {
854 LLVM_DEBUG(dbgs() << "LV: Setting widening decision to "
855 << getInstWideningStr(W) << " for VF " << VF
856 << " and instruction: " << *I << '\n');
857 if (Grp->getInsertPos() == I)
858 WideningDecisions[{I, VF}] = {W, InsertPosCost};
859 else
860 WideningDecisions[{I, VF}] = {W, OtherMemberCost};
861 }
862 }
863
864 /// Return the cost model decision for the given instruction \p I and vector
865 /// width \p VF. Return CM_Unknown if this instruction did not pass
866 /// through the cost modeling.
867 InstWidening getWideningDecision(Instruction *I, ElementCount VF) const {
868 assert(VF.isVector() && "Expected VF to be a vector VF");
869 assert(
870 TheLoop->isInnermost() &&
871 "cost-model should not be used for outer loops (in VPlan-native path)");
872
873 std::pair<Instruction *, ElementCount> InstOnVF(I, VF);
874 auto Itr = WideningDecisions.find(Val: InstOnVF);
875 if (Itr == WideningDecisions.end())
876 return CM_Unknown;
877 return Itr->second.first;
878 }
879
880 /// Return the vectorization cost for the given instruction \p I and vector
881 /// width \p VF.
882 InstructionCost getWideningCost(Instruction *I, ElementCount VF) {
883 assert(VF.isVector() && "Expected VF >=2");
884 std::pair<Instruction *, ElementCount> InstOnVF(I, VF);
885 assert(WideningDecisions.contains(InstOnVF) &&
886 "The cost is not calculated");
887 return WideningDecisions[InstOnVF].second;
888 }
889
890 /// Return True if instruction \p I is an optimizable truncate whose operand
891 /// is an induction variable. Such a truncate will be removed by adding a new
892 /// induction variable with the destination type.
893 bool isOptimizableIVTruncate(Instruction *I, ElementCount VF) {
894 // If the instruction is not a truncate, return false.
895 auto *Trunc = dyn_cast<TruncInst>(Val: I);
896 if (!Trunc)
897 return false;
898
899 // Get the source and destination types of the truncate.
900 Type *SrcTy = toVectorTy(Scalar: Trunc->getSrcTy(), EC: VF);
901 Type *DestTy = toVectorTy(Scalar: Trunc->getDestTy(), EC: VF);
902
903 // If the truncate is free for the given types, return false. Replacing a
904 // free truncate with an induction variable would add an induction variable
905 // update instruction to each iteration of the loop. We exclude from this
906 // check the primary induction variable since it will need an update
907 // instruction regardless.
908 Value *Op = Trunc->getOperand(i_nocapture: 0);
909 if (Op != Legal->getPrimaryInduction() && TTI.isTruncateFree(Ty1: SrcTy, Ty2: DestTy))
910 return false;
911
912 // If the truncated value is not an induction variable, return false.
913 return Legal->isInductionPhi(V: Op);
914 }
915
916 /// Collects the instructions to scalarize for each predicated instruction in
917 /// the loop.
918 void collectInstsToScalarize(ElementCount VF);
919
920 /// Collect values that will not be widened, including Uniforms, Scalars, and
921 /// Instructions to Scalarize for the given \p VF.
922 /// The sets depend on CM decision for Load/Store instructions
923 /// that may be vectorized as interleave, gather-scatter or scalarized.
924 /// Also make a decision on what to do about call instructions in the loop
925 /// at that VF -- scalarize, call a known vector routine, or call a
926 /// vector intrinsic.
927 void collectNonVectorizedAndSetWideningDecisions(ElementCount VF) {
928 // Do the analysis once.
929 if (VF.isScalar() || Uniforms.contains(Val: VF))
930 return;
931 setCostBasedWideningDecision(VF);
932 collectLoopUniforms(VF);
933 collectLoopScalars(VF);
934 collectInstsToScalarize(VF);
935 }
936
937 /// Given costs for both strategies, return true if the scalar predication
938 /// lowering should be used for div/rem. This incorporates an override
939 /// option so it is not simply a cost comparison.
940 bool isDivRemScalarWithPredication(InstructionCost ScalarCost,
941 InstructionCost MaskedCost) const {
942 switch (ForceMaskedDivRem) {
943 case cl::boolOrDefault::BOU_UNSET:
944 return ScalarCost < MaskedCost;
945 case cl::boolOrDefault::BOU_TRUE:
946 return false;
947 case cl::boolOrDefault::BOU_FALSE:
948 return true;
949 }
950 llvm_unreachable("impossible case value");
951 }
952
953 /// Returns true if \p I is an instruction which requires predication and
954 /// for which our chosen predication strategy is scalarization (i.e. we
955 /// don't have an alternate strategy such as masking available).
956 /// \p VF is the vectorization factor that will be used to vectorize \p I.
957 bool isScalarWithPredication(Instruction *I, ElementCount VF);
958
959 /// Wrapper function for LoopVectorizationLegality::isMaskRequired,
960 /// that passes the Instruction \p I and if we fold tail.
961 bool isMaskRequired(Instruction *I) const;
962
963 /// Returns true if \p I is an instruction that needs to be predicated
964 /// at runtime. The result is independent of the predication mechanism.
965 /// Superset of instructions that return true for isScalarWithPredication.
966 bool isPredicatedInst(Instruction *I) const;
967
968 /// A helper function that returns how much we should divide the cost of a
969 /// predicated block by. Typically this is the reciprocal of the block
970 /// probability, i.e. if we return X we are assuming the predicated block will
971 /// execute once for every X iterations of the loop header so the block should
972 /// only contribute 1/X of its cost to the total cost calculation, but when
973 /// optimizing for code size it will just be 1 as code size costs don't depend
974 /// on execution probabilities.
975 ///
976 /// Note that if a block wasn't originally predicated but was predicated due
977 /// to tail folding, the divisor will still be 1 because it will execute for
978 /// every iteration of the loop header.
979 inline uint64_t
980 getPredBlockCostDivisor(TargetTransformInfo::TargetCostKind CostKind,
981 const BasicBlock *BB);
982
983 /// Returns true if an artificially high cost for emulated masked memrefs
984 /// should be used.
985 bool useEmulatedMaskMemRefHack(Instruction *I, ElementCount VF) const;
986
987 /// Return the costs for our two available strategies for lowering a
988 /// div/rem operation which requires speculating at least one lane.
989 /// First result is for scalarization (will be invalid for scalable
990 /// vectors); second is for the masked intrinsic strategy.
991 std::pair<InstructionCost, InstructionCost>
992 getDivRemSpeculationCost(Instruction *I, ElementCount VF);
993
994 /// If \p I is a memory instruction with a consecutive pointer that can be
995 /// widened, returns the widening kind (CM_Widen or CM_Widen_Reverse) and
996 /// std::nullopt otherwise.
997 std::optional<InstWidening> memoryInstructionCanBeWidened(Instruction *I,
998 ElementCount VF);
999
1000 /// Returns true if \p I is a memory instruction in an interleaved-group
1001 /// of memory accesses that can be vectorized with wide vector loads/stores
1002 /// and shuffles.
1003 bool interleavedAccessCanBeWidened(Instruction *I, ElementCount VF) const;
1004
1005 /// Returns true if the target machine supports masked loads or stores
1006 /// for \p I's data type and alignment. The caller must ensure the access is
1007 /// consecutive or part of an interleave group.
1008 bool isLegalMaskedLoadOrStore(Instruction *I, ElementCount VF) const;
1009
1010 /// Returns true if the target machine supports gather or scatter for \p I's
1011 /// data type and alignment.
1012 bool isLegalGatherOrScatter(Instruction *I, ElementCount VF) const;
1013
1014 /// Check if \p Instr belongs to any interleaved access group.
1015 bool isAccessInterleaved(Instruction *Instr) const {
1016 return InterleaveInfo.isInterleaved(Instr);
1017 }
1018
1019 /// Get the interleaved access group that \p Instr belongs to.
1020 const InterleaveGroup<Instruction> *
1021 getInterleavedAccessGroup(Instruction *Instr) const {
1022 return InterleaveInfo.getInterleaveGroup(Instr);
1023 }
1024
1025 /// Returns true if we're required to use a scalar epilogue for at least
1026 /// the final iteration of the original loop.
1027 bool requiresScalarEpilogue(bool IsVectorizing) const {
1028 if (!isEpilogueAllowed()) {
1029 LLVM_DEBUG(dbgs() << "LV: Loop does not require scalar epilogue\n");
1030 return false;
1031 }
1032 // If we might exit from anywhere but the latch and early exit vectorization
1033 // is disabled, we must run the exiting iteration in scalar form.
1034 if (TheLoop->getExitingBlock() != TheLoop->getLoopLatch() &&
1035 !(EnableEarlyExitVectorization && Legal->hasUncountableEarlyExit())) {
1036 LLVM_DEBUG(dbgs() << "LV: Loop requires scalar epilogue: not exiting "
1037 "from latch block\n");
1038 return true;
1039 }
1040 if (IsVectorizing && InterleaveInfo.requiresScalarEpilogue()) {
1041 LLVM_DEBUG(dbgs() << "LV: Loop requires scalar epilogue: "
1042 "interleaved group requires scalar epilogue\n");
1043 return true;
1044 }
1045 LLVM_DEBUG(dbgs() << "LV: Loop does not require scalar epilogue\n");
1046 return false;
1047 }
1048
1049 /// Returns true if an epilogue is allowed (e.g., not prevented by
1050 /// optsize or a loop hint annotation).
1051 bool isEpilogueAllowed() const {
1052 return EpilogueLoweringStatus == CM_EpilogueAllowed;
1053 }
1054
1055 /// Returns the TailFoldingStyle that is best for the current loop.
1056 TailFoldingStyle getTailFoldingStyle() const {
1057 return ChosenTailFoldingStyle;
1058 }
1059
1060 /// Selects and saves TailFoldingStyle.
1061 /// \param IsScalableVF true if scalable vector factors enabled.
1062 /// \param UserIC User specific interleave count.
1063 void setTailFoldingStyle(bool IsScalableVF, unsigned UserIC) {
1064 assert(ChosenTailFoldingStyle == TailFoldingStyle::None &&
1065 "Tail folding must not be selected yet.");
1066 if (!Legal->canFoldTailByMasking()) {
1067 ChosenTailFoldingStyle = TailFoldingStyle::None;
1068 return;
1069 }
1070
1071 // Default to TTI preference, but allow command line override.
1072 ChosenTailFoldingStyle = TTI.getPreferredTailFoldingStyle();
1073 if (ForceTailFoldingStyle.getNumOccurrences())
1074 ChosenTailFoldingStyle = ForceTailFoldingStyle.getValue();
1075
1076 if (ChosenTailFoldingStyle != TailFoldingStyle::DataWithEVL)
1077 return;
1078 // Override EVL styles if needed.
1079 // FIXME: Investigate opportunity for fixed vector factor.
1080 bool EVLIsLegal = UserIC <= 1 && IsScalableVF &&
1081 TTI.hasActiveVectorLength() && !EnableVPlanNativePath;
1082 if (EVLIsLegal)
1083 return;
1084 // If for some reason EVL mode is unsupported, fallback to an epilogue
1085 // if it's allowed, or DataWithoutLaneMask otherwise.
1086 if (EpilogueLoweringStatus == CM_EpilogueAllowed ||
1087 EpilogueLoweringStatus == CM_EpilogueNotNeededFoldTail)
1088 ChosenTailFoldingStyle = TailFoldingStyle::None;
1089 else
1090 ChosenTailFoldingStyle = TailFoldingStyle::DataWithoutLaneMask;
1091
1092 LLVM_DEBUG(
1093 dbgs() << "LV: Preference for VP intrinsics indicated. Will "
1094 "not try to generate VP Intrinsics "
1095 << (UserIC > 1
1096 ? "since interleave count specified is greater than 1.\n"
1097 : "due to non-interleaving reasons.\n"));
1098 }
1099
1100 /// Returns true if all loop blocks should be masked to fold tail loop.
1101 bool foldTailByMasking() const {
1102 return getTailFoldingStyle() != TailFoldingStyle::None;
1103 }
1104
1105 void tryToEnablePartialAliasMasking() {
1106 assert(foldTailByMasking() && "Expected tail folding to be enabled!");
1107 assert(!foldTailWithEVL() &&
1108 "Did not expect to enable alias masking with EVL!");
1109 assert(PartialAliasMaskingStatus == AliasMaskingStatus::NotDecided);
1110
1111 // Assume we fail to enable alias masking (in case we early exit).
1112 PartialAliasMaskingStatus = AliasMaskingStatus::Disabled;
1113
1114 // Note: FixedOrderRecurrences are not supported yet as we cannot handle
1115 // the required `splice.right` with the alias-mask.
1116 if (!ForcePartialAliasingVectorization ||
1117 !Legal->getFixedOrderRecurrences().empty())
1118 return;
1119
1120 const RuntimePointerChecking *Checks = Legal->getRuntimePointerChecking();
1121 if (!Checks)
1122 return;
1123
1124 auto DiffChecks = Checks->getDiffChecks();
1125 if (!DiffChecks || DiffChecks->empty())
1126 return;
1127
1128 [[maybe_unused]] auto HasPointerArgs = [](CallBase *CB) {
1129 return any_of(Range: CB->args(), P: [](Value const *Arg) {
1130 return Arg->getType()->isPointerTy();
1131 });
1132 };
1133
1134 for (BasicBlock *BB : TheLoop->blocks()) {
1135 for (Instruction &I : *BB) {
1136 if (!isa<LoadInst, StoreInst>(Val: I)) {
1137 [[maybe_unused]] auto *Call = dyn_cast<CallInst>(Val: &I);
1138 assert(
1139 (!I.mayReadOrWriteMemory() || (Call && !HasPointerArgs(Call))) &&
1140 "Skipped unexpected memory access");
1141 continue;
1142 }
1143
1144 Type *ScalarTy = getLoadStoreType(I: &I);
1145 Value *Ptr = getLoadStorePointerOperand(V: &I);
1146
1147 // Currently, we can't handle alias masking in reverse. Reversing the
1148 // alias mask is not correct (or necessary). When combined with
1149 // tail-folding the active lane mask should only be reversed where the
1150 // alias-mask is true.
1151 if (Legal->isConsecutivePtr(AccessTy: ScalarTy, Ptr) == -1)
1152 return;
1153 }
1154 }
1155
1156 PartialAliasMaskingStatus = AliasMaskingStatus::Enabled;
1157 }
1158
1159 /// Returns true if all loop blocks should have partial aliases masked.
1160 bool maskPartialAliasing() const {
1161 return PartialAliasMaskingStatus == AliasMaskingStatus::Enabled;
1162 }
1163
1164 /// Returns true if the instructions in this block requires predication
1165 /// for any reason, e.g. because tail folding now requires a predicate
1166 /// or because the block in the original loop was predicated.
1167 bool blockNeedsPredicationForAnyReason(BasicBlock *BB) const {
1168 return foldTailByMasking() || Legal->blockNeedsPredication(BB);
1169 }
1170
1171 /// Returns true if VP intrinsics with explicit vector length support should
1172 /// be generated in the tail folded loop.
1173 bool foldTailWithEVL() const {
1174 return getTailFoldingStyle() == TailFoldingStyle::DataWithEVL;
1175 }
1176
1177 /// Returns true if the predicated reduction select should be used to set the
1178 /// incoming value for the reduction phi.
1179 bool usePredicatedReductionSelect(RecurKind RecurrenceKind,
1180 bool HasUsesOutsideReductionChain) const {
1181 // Force to use predicated reduction select since the EVL of the
1182 // second-to-last iteration might not be VF*UF.
1183 if (foldTailWithEVL())
1184 return true;
1185
1186 // Force a predicated select with alias-masking to avoid propagating poison
1187 // values to the header phi for lanes outside the alias-mask.
1188 if (maskPartialAliasing())
1189 return true;
1190
1191 // Note: For FindLast recurrences and multi-use reductions we prefer a
1192 // predicated select to simplify matching in handleFindLastReductions() and
1193 // handleMultiUseReductions() respectively, rather than handle multiple
1194 // cases.
1195 if (RecurrenceDescriptor::isFindLastRecurrenceKind(Kind: RecurrenceKind) ||
1196 HasUsesOutsideReductionChain)
1197 return true;
1198
1199 return PreferPredicatedReductionSelect ||
1200 TTI.preferPredicatedReductionSelect();
1201 }
1202
1203 /// Estimate cost of an intrinsic call instruction CI if it were vectorized
1204 /// with factor VF. Return the cost of the instruction, including
1205 /// scalarization overhead if it's needed.
1206 InstructionCost getVectorIntrinsicCost(CallInst *CI, ElementCount VF) const;
1207
1208 /// Estimate cost of a call instruction CI if it were vectorized with factor
1209 /// VF. Return the cost of the instruction, including scalarization overhead
1210 /// if it's needed.
1211 InstructionCost getVectorCallCost(CallInst *CI, ElementCount VF) const;
1212
1213 /// Invalidates decisions already taken by the cost model.
1214 void invalidateCostModelingDecisions() {
1215 WideningDecisions.clear();
1216 Uniforms.clear();
1217 Scalars.clear();
1218 }
1219
1220 /// Returns the expected execution cost. The unit of the cost does
1221 /// not matter because we use the 'cost' units to compare different
1222 /// vector widths. The cost that is returned is *not* normalized by
1223 /// the factor width.
1224 InstructionCost expectedCost(ElementCount VF);
1225
1226 /// Returns the execution time cost of an instruction for a given vector
1227 /// width. Vector width of one means scalar.
1228 InstructionCost getInstructionCost(Instruction *I, ElementCount VF);
1229
1230 /// Returns true if \p Op should be considered invariant and if it is
1231 /// trivially hoistable.
1232 bool shouldConsiderInvariant(Value *Op);
1233
1234 /// Returns true if \p I has been forced to be scalarized at \p VF.
1235 bool isForcedScalar(Instruction *I, ElementCount VF) const {
1236 auto FS = ForcedScalars.find(Val: VF);
1237 return FS != ForcedScalars.end() && FS->second.contains(key: I);
1238 }
1239
1240private:
1241 unsigned NumPredStores = 0;
1242
1243 /// VF selection state independent of cost-modeling decisions.
1244 VFSelectionContext &Config;
1245
1246 /// Wrapper around LoopVectorizationLegality::isUniform() that takes into
1247 /// account if alias-masking is enabled. We consider the VF to be unknown when
1248 /// alias masking.
1249 bool isUniform(Value *V, ElementCount VF) const {
1250 // With alias-masking our runtime VF is [2, VF] (and not necessarily a
1251 // power-of-two). Something that is uniform for VF may not be for the full
1252 // range.
1253 assert(PartialAliasMaskingStatus != AliasMaskingStatus::NotDecided &&
1254 "alias-mask status must be decided already");
1255 return Legal->isUniform(V, VF: PartialAliasMaskingStatus ==
1256 AliasMaskingStatus::Disabled
1257 ? std::optional(VF)
1258 : std::nullopt);
1259 }
1260
1261 /// Wrapper around LoopVectorizationLegality::isUniformMemOp() that takes into
1262 /// account if alias-masking is enabled. We consider the VF to be unknown when
1263 /// alias masking.
1264 bool isUniformMemOp(Instruction &I, ElementCount VF) const {
1265 assert(PartialAliasMaskingStatus != AliasMaskingStatus::NotDecided &&
1266 "alias-mask status must be decided already");
1267 return Legal->isUniformMemOp(I, VF: PartialAliasMaskingStatus ==
1268 AliasMaskingStatus::Disabled
1269 ? std::optional(VF)
1270 : std::nullopt);
1271 }
1272
1273 /// Calculate vectorization cost of memory instruction \p I.
1274 InstructionCost getMemoryInstructionCost(Instruction *I, ElementCount VF);
1275
1276 /// The cost computation for scalarized memory instruction.
1277 InstructionCost getMemInstScalarizationCost(Instruction *I, ElementCount VF);
1278
1279 /// The cost computation for interleaving group of memory instructions.
1280 InstructionCost getInterleaveGroupCost(Instruction *I, ElementCount VF) const;
1281
1282 /// The cost computation for Gather/Scatter instruction.
1283 InstructionCost getGatherScatterCost(Instruction *I, ElementCount VF) const;
1284
1285 /// The cost computation for widening instruction \p I with consecutive
1286 /// memory access.
1287 InstructionCost getConsecutiveMemOpCost(Instruction *I, ElementCount VF,
1288 InstWidening Kind);
1289
1290 /// The cost calculation for Load/Store instruction \p I with uniform pointer -
1291 /// Load: scalar load + broadcast.
1292 /// Store: scalar store + (loop invariant value stored? 0 : extract of last
1293 /// element)
1294 InstructionCost getUniformMemOpCost(Instruction *I, ElementCount VF) const;
1295
1296 /// Estimate the overhead of scalarizing an instruction. This is a
1297 /// convenience wrapper for the type-based getScalarizationOverhead API.
1298 InstructionCost getScalarizationOverhead(Instruction *I,
1299 ElementCount VF) const;
1300
1301 /// A type representing the costs for instructions if they were to be
1302 /// scalarized rather than vectorized. The entries are Instruction-Cost
1303 /// pairs.
1304 using ScalarCostsTy = MapVector<Instruction *, InstructionCost>;
1305
1306 /// A set containing all BasicBlocks that are known to present after
1307 /// vectorization as a predicated block.
1308 DenseMap<ElementCount, SmallPtrSet<BasicBlock *, 4>>
1309 PredicatedBBsAfterVectorization;
1310
1311 /// Records whether it is allowed to have the original scalar loop execute at
1312 /// least once. This may be needed as a fallback loop in case runtime
1313 /// aliasing/dependence checks fail, or to handle the tail/remainder
1314 /// iterations when the trip count is unknown or doesn't divide by the VF,
1315 /// or as a peel-loop to handle gaps in interleave-groups.
1316 /// Under optsize and when the trip count is very small we don't allow any
1317 /// iterations to execute in the scalar loop.
1318 EpilogueLowering EpilogueLoweringStatus = CM_EpilogueAllowed;
1319
1320 /// Control finally chosen tail folding style.
1321 TailFoldingStyle ChosenTailFoldingStyle = TailFoldingStyle::None;
1322
1323 /// If partial alias masking is enabled/disabled or not decided.
1324 AliasMaskingStatus PartialAliasMaskingStatus = AliasMaskingStatus::NotDecided;
1325
1326 /// A map holding scalar costs for different vectorization factors. The
1327 /// presence of a cost for an instruction in the mapping indicates that the
1328 /// instruction will be scalarized when vectorizing with the associated
1329 /// vectorization factor. The entries are VF-ScalarCostTy pairs.
1330 MapVector<ElementCount, ScalarCostsTy> InstsToScalarize;
1331
1332 /// Holds the instructions known to be uniform after vectorization.
1333 /// The data is collected per VF.
1334 DenseMap<ElementCount, SmallPtrSet<Instruction *, 4>> Uniforms;
1335
1336 /// Holds the instructions known to be scalar after vectorization.
1337 /// The data is collected per VF.
1338 DenseMap<ElementCount, SmallPtrSet<Instruction *, 4>> Scalars;
1339
1340 /// Holds the instructions (address computations) that are forced to be
1341 /// scalarized.
1342 DenseMap<ElementCount, SmallSetVector<Instruction *, 4>> ForcedScalars;
1343
1344 /// Returns the expected difference in cost from scalarizing the expression
1345 /// feeding a predicated instruction \p PredInst. The instructions to
1346 /// scalarize and their scalar costs are collected in \p ScalarCosts. A
1347 /// non-negative return value implies the expression will be scalarized.
1348 /// Currently, only single-use chains are considered for scalarization.
1349 InstructionCost computePredInstDiscount(Instruction *PredInst,
1350 ScalarCostsTy &ScalarCosts,
1351 ElementCount VF);
1352
1353 /// Collect the instructions that are uniform after vectorization. An
1354 /// instruction is uniform if we represent it with a single scalar value in
1355 /// the vectorized loop corresponding to each vector iteration. Examples of
1356 /// uniform instructions include pointer operands of consecutive or
1357 /// interleaved memory accesses. Note that although uniformity implies an
1358 /// instruction will be scalar, the reverse is not true. In general, a
1359 /// scalarized instruction will be represented by VF scalar values in the
1360 /// vectorized loop, each corresponding to an iteration of the original
1361 /// scalar loop.
1362 void collectLoopUniforms(ElementCount VF);
1363
1364 /// Collect the instructions that are scalar after vectorization. An
1365 /// instruction is scalar if it is known to be uniform or will be scalarized
1366 /// during vectorization. collectLoopScalars should only add non-uniform nodes
1367 /// to the list if they are used by a load/store instruction that is marked as
1368 /// CM_Scalarize. Non-uniform scalarized instructions will be represented by
1369 /// VF values in the vectorized loop, each corresponding to an iteration of
1370 /// the original scalar loop.
1371 void collectLoopScalars(ElementCount VF);
1372
1373 /// Keeps cost model vectorization decision and cost for instructions.
1374 /// Right now it is used for memory instructions only.
1375 using DecisionList = DenseMap<std::pair<Instruction *, ElementCount>,
1376 std::pair<InstWidening, InstructionCost>>;
1377
1378 DecisionList WideningDecisions;
1379
1380 /// Returns true if \p V is expected to be vectorized and it needs to be
1381 /// extracted.
1382 bool needsExtract(Value *V, ElementCount VF) const {
1383 Instruction *I = dyn_cast<Instruction>(Val: V);
1384 if (VF.isScalar() || !I || !TheLoop->contains(Inst: I) ||
1385 TheLoop->isLoopInvariant(V: I) ||
1386 getWideningDecision(I, VF) == CM_Scalarize)
1387 return false;
1388
1389 // Assume we can vectorize V (and hence we need extraction) if the
1390 // scalars are not computed yet. This can happen, because it is called
1391 // via getScalarizationOverhead from setCostBasedWideningDecision, before
1392 // the scalars are collected. That should be a safe assumption in most
1393 // cases, because we check if the operands have vectorizable types
1394 // beforehand in LoopVectorizationLegality.
1395 return !Scalars.contains(Val: VF) || !isScalarAfterVectorization(I, VF);
1396 };
1397
1398 /// Returns a range containing only operands needing to be extracted.
1399 SmallVector<Value *, 4> filterExtractingOperands(Instruction::op_range Ops,
1400 ElementCount VF) const {
1401
1402 SmallPtrSet<const Value *, 4> UniqueOperands;
1403 SmallVector<Value *, 4> Res;
1404 for (Value *Op : Ops) {
1405 if (isa<Constant>(Val: Op) || !UniqueOperands.insert(Ptr: Op).second ||
1406 !needsExtract(V: Op, VF))
1407 continue;
1408 Res.push_back(Elt: Op);
1409 }
1410 return Res;
1411 }
1412
1413public:
1414 /// The loop that we evaluate.
1415 Loop *TheLoop;
1416
1417 /// Predicated scalar evolution analysis.
1418 PredicatedScalarEvolution &PSE;
1419
1420 /// Loop Info analysis.
1421 LoopInfo *LI;
1422
1423 /// Vectorization legality.
1424 LoopVectorizationLegality *Legal;
1425
1426 /// Vector target information.
1427 const TargetTransformInfo &TTI;
1428
1429 /// Target Library Info.
1430 const TargetLibraryInfo *TLI;
1431
1432 /// Assumption cache.
1433 AssumptionCache *AC;
1434
1435 /// Interface to emit optimization remarks.
1436 OptimizationRemarkEmitter *ORE;
1437
1438 /// A function to lazily fetch BlockFrequencyInfo. This avoids computing it
1439 /// unless necessary, e.g. when the loop isn't legal to vectorize or when
1440 /// there is no predication.
1441 std::function<BlockFrequencyInfo &()> GetBFI;
1442 /// The BlockFrequencyInfo returned from GetBFI.
1443 BlockFrequencyInfo *BFI = nullptr;
1444 /// Returns the BlockFrequencyInfo for the function if cached, otherwise
1445 /// fetches it via GetBFI. Avoids an indirect call to the std::function.
1446 BlockFrequencyInfo &getBFI() {
1447 if (!BFI)
1448 BFI = &GetBFI();
1449 return *BFI;
1450 }
1451
1452 const Function *TheFunction;
1453
1454 /// The interleave access information contains groups of interleaved accesses
1455 /// with the same stride and close to each other.
1456 InterleavedAccessInfo &InterleaveInfo;
1457
1458 /// Values to ignore in the cost model.
1459 SmallPtrSet<const Value *, 16> ValuesToIgnore;
1460
1461 /// Values to ignore in the cost model when VF > 1.
1462 SmallPtrSet<const Value *, 16> VecValuesToIgnore;
1463};
1464} // end namespace llvm
1465
1466namespace {
1467/// Helper struct to manage generating runtime checks for vectorization.
1468///
1469/// The runtime checks are created up-front in temporary blocks to allow better
1470/// estimating the cost and un-linked from the existing IR. After deciding to
1471/// vectorize, the checks are attached to VPlan as IR or recipes. If deciding
1472/// not to vectorize, the temporary blocks are completely removed.
1473class GeneratedRTChecks {
1474 /// Basic block which contains the generated SCEV checks, if any.
1475 BasicBlock *SCEVCheckBlock = nullptr;
1476
1477 /// The value representing the result of the generated SCEV checks. If it is
1478 /// nullptr no SCEV checks have been generated.
1479 Value *SCEVCheckCond = nullptr;
1480
1481 /// Basic block which contains the generated memory runtime checks, if any.
1482 BasicBlock *MemCheckBlock = nullptr;
1483
1484 /// The value representing the result of the generated memory runtime checks.
1485 /// If it is nullptr no memory runtime checks have been generated.
1486 Value *MemRuntimeCheckCond = nullptr;
1487
1488 /// Whether checks were generated, retained after their IR is replaced or
1489 /// removed during VPlan execution.
1490 bool HasChecks = false;
1491
1492 DominatorTree *DT;
1493 LoopInfo *LI;
1494 TargetTransformInfo *TTI;
1495
1496 SCEVExpander SCEVExp;
1497 SCEVExpander MemCheckExp;
1498
1499 bool CostTooHigh = false;
1500
1501 Loop *OuterLoop = nullptr;
1502
1503 PredicatedScalarEvolution &PSE;
1504
1505 /// The kind of cost that we are calculating
1506 TTI::TargetCostKind CostKind;
1507
1508 /// True if the loop is alias-masked (which allows us to omit diff checks).
1509 bool LoopUsesPartialAliasMasking = false;
1510
1511public:
1512 GeneratedRTChecks(PredicatedScalarEvolution &PSE, DominatorTree *DT,
1513 LoopInfo *LI, TargetTransformInfo *TTI,
1514 TTI::TargetCostKind CostKind,
1515 bool LoopUsesPartialAliasMasking)
1516 : DT(DT), LI(LI), TTI(TTI),
1517 SCEVExp(*PSE.getSE(), "scev.check", /*PreserveLCSSA=*/false),
1518 MemCheckExp(*PSE.getSE(), "scev.check", /*PreserveLCSSA=*/false),
1519 PSE(PSE), CostKind(CostKind),
1520 LoopUsesPartialAliasMasking(LoopUsesPartialAliasMasking) {}
1521
1522 /// Generate runtime checks in SCEVCheckBlock and MemCheckBlock, so we can
1523 /// accurately estimate the cost of the runtime checks. The blocks are
1524 /// un-linked from the IR and attached to VPlan as IR or recipes if
1525 /// profitable. Otherwise, the check blocks are removed completely.
1526 void create(Loop *L, const LoopAccessInfo &LAI,
1527 const SCEVPredicate &UnionPred, ElementCount VF, unsigned IC,
1528 OptimizationRemarkEmitter &ORE) {
1529
1530 // Hard cutoff to limit compile-time increase in case a very large number of
1531 // runtime checks needs to be generated.
1532 // TODO: Skip cutoff if the loop is guaranteed to execute, e.g. due to
1533 // profile info.
1534 CostTooHigh = LAI.getNumRuntimePointerChecks() >
1535 VectorizerParams::VectorizeMemoryCheckThreshold;
1536 if (CostTooHigh) {
1537 // Mark runtime checks as never succeeding when they exceed the threshold.
1538 MemRuntimeCheckCond = ConstantInt::getTrue(Context&: L->getHeader()->getContext());
1539 SCEVCheckCond = ConstantInt::getTrue(Context&: L->getHeader()->getContext());
1540 ORE.emit(RemarkBuilder: [&]() {
1541 return OptimizationRemarkAnalysisAliasing(
1542 DEBUG_TYPE, "TooManyMemoryRuntimeChecks", L->getStartLoc(),
1543 L->getHeader())
1544 << "loop not vectorized: too many memory checks needed";
1545 });
1546 LLVM_DEBUG(dbgs() << "LV: Too many memory checks needed.\n");
1547 return;
1548 }
1549
1550 BasicBlock *LoopHeader = L->getHeader();
1551 BasicBlock *Preheader = L->getLoopPreheader();
1552
1553 // Use SplitBlock to create blocks for SCEV & memory runtime checks to
1554 // ensure the blocks are properly added to LoopInfo & DominatorTree. Those
1555 // may be used by SCEVExpander. The blocks will be un-linked from their
1556 // predecessors and removed from LI & DT at the end of the function.
1557 if (!UnionPred.isAlwaysTrue()) {
1558 SCEVCheckBlock = SplitBlock(Old: Preheader, SplitPt: Preheader->getTerminator(), DT, LI,
1559 MSSAU: nullptr, BBName: "vector.scevcheck");
1560
1561 SCEVCheckCond = SCEVExp.expandCodeForPredicate(
1562 Pred: &UnionPred, Loc: SCEVCheckBlock->getTerminator());
1563 if (isa<Constant>(Val: SCEVCheckCond)) {
1564 // Clean up directly after expanding the predicate to a constant, to
1565 // avoid further expansions re-using anything left over from SCEVExp.
1566 SCEVExpanderCleaner SCEVCleaner(SCEVExp);
1567 SCEVCleaner.cleanup();
1568 }
1569 }
1570
1571 const auto &RtPtrChecking = *LAI.getRuntimePointerChecking();
1572 // TODO: We need to estimate the cost of alias-masking in
1573 // GeneratedRTChecks::getCost(). We can't check the MemCheckBlock as the
1574 // alias-mask is generated later in VPlan.
1575 if (RtPtrChecking.Need && !LoopUsesPartialAliasMasking) {
1576 auto *Pred = SCEVCheckBlock ? SCEVCheckBlock : Preheader;
1577 MemCheckBlock = SplitBlock(Old: Pred, SplitPt: Pred->getTerminator(), DT, LI, MSSAU: nullptr,
1578 BBName: "vector.memcheck");
1579
1580 auto DiffChecks = RtPtrChecking.getDiffChecks();
1581 if (DiffChecks) {
1582 MemRuntimeCheckCond = addDiffRuntimeChecks(
1583 Loc: MemCheckBlock->getTerminator(), Checks: *DiffChecks, Expander&: MemCheckExp, VF, IC);
1584 } else {
1585 MemRuntimeCheckCond = addRuntimeChecks(
1586 Loc: MemCheckBlock->getTerminator(), TheLoop: L, PointerChecks: RtPtrChecking.getChecks(),
1587 Expander&: MemCheckExp, HoistRuntimeChecks: VectorizerParams::HoistRuntimeChecks);
1588 }
1589 assert(MemRuntimeCheckCond &&
1590 "no RT checks generated although RtPtrChecking "
1591 "claimed checks are required");
1592 }
1593
1594 SCEVExp.eraseDeadInstructions(Root: SCEVCheckCond);
1595 HasChecks = getSCEVChecks().first || getMemRuntimeChecks().first;
1596
1597 if (!MemCheckBlock && !SCEVCheckBlock)
1598 return;
1599
1600 // Unhook the temporary block with the checks, update various places
1601 // accordingly.
1602 if (SCEVCheckBlock)
1603 SCEVCheckBlock->replaceAllUsesWith(V: Preheader);
1604 if (MemCheckBlock)
1605 MemCheckBlock->replaceAllUsesWith(V: Preheader);
1606
1607 if (SCEVCheckBlock) {
1608 SCEVCheckBlock->getTerminator()->moveBefore(
1609 InsertPos: Preheader->getTerminator()->getIterator());
1610 auto *UI = new UnreachableInst(Preheader->getContext(), SCEVCheckBlock);
1611 UI->setDebugLoc(DebugLoc::getTemporary());
1612 Preheader->getTerminator()->eraseFromParent();
1613 }
1614 if (MemCheckBlock) {
1615 MemCheckBlock->getTerminator()->moveBefore(
1616 InsertPos: Preheader->getTerminator()->getIterator());
1617 auto *UI = new UnreachableInst(Preheader->getContext(), MemCheckBlock);
1618 UI->setDebugLoc(DebugLoc::getTemporary());
1619 Preheader->getTerminator()->eraseFromParent();
1620 }
1621
1622 DT->changeImmediateDominator(BB: LoopHeader, NewBB: Preheader);
1623 if (MemCheckBlock) {
1624 DT->eraseNode(BB: MemCheckBlock);
1625 LI->removeBlock(BB: MemCheckBlock);
1626 }
1627 if (SCEVCheckBlock) {
1628 DT->eraseNode(BB: SCEVCheckBlock);
1629 LI->removeBlock(BB: SCEVCheckBlock);
1630 }
1631
1632 // Outer loop is used as part of the later cost calculations.
1633 OuterLoop = L->getParentLoop();
1634 }
1635
1636 InstructionCost getCost() {
1637 if (SCEVCheckBlock || MemCheckBlock)
1638 LLVM_DEBUG(dbgs() << "Calculating cost of runtime checks:\n");
1639
1640 if (CostTooHigh) {
1641 InstructionCost Cost;
1642 Cost.setInvalid();
1643 LLVM_DEBUG(dbgs() << " number of checks exceeded threshold\n");
1644 return Cost;
1645 }
1646
1647 InstructionCost RTCheckCost = 0;
1648 if (SCEVCheckBlock)
1649 for (Instruction &I : *SCEVCheckBlock) {
1650 if (SCEVCheckBlock->getTerminator() == &I)
1651 continue;
1652 InstructionCost C = TTI->getInstructionCost(U: &I, CostKind);
1653 LLVM_DEBUG(dbgs() << " " << C << " for " << I << "\n");
1654 RTCheckCost += C;
1655 }
1656 if (MemCheckBlock) {
1657 InstructionCost MemCheckCost = 0;
1658 for (Instruction &I : *MemCheckBlock) {
1659 if (MemCheckBlock->getTerminator() == &I)
1660 continue;
1661 InstructionCost C = TTI->getInstructionCost(U: &I, CostKind);
1662 LLVM_DEBUG(dbgs() << " " << C << " for " << I << "\n");
1663 MemCheckCost += C;
1664 }
1665
1666 // If the runtime memory checks are being created inside an outer loop
1667 // we should find out if these checks are outer loop invariant. If so,
1668 // the checks will likely be hoisted out and so the effective cost will
1669 // reduce according to the outer loop trip count.
1670 if (OuterLoop) {
1671 ScalarEvolution *SE = MemCheckExp.getSE();
1672 // TODO: If profitable, we could refine this further by analysing every
1673 // individual memory check, since there could be a mixture of loop
1674 // variant and invariant checks that mean the final condition is
1675 // variant.
1676 const SCEV *Cond = SE->getSCEV(V: MemRuntimeCheckCond);
1677 if (SE->isLoopInvariant(S: Cond, L: OuterLoop)) {
1678 // It seems reasonable to assume that we can reduce the effective
1679 // cost of the checks even when we know nothing about the trip
1680 // count. Assume that the outer loop executes at least twice.
1681 unsigned BestTripCount = 2;
1682
1683 // Get the best known TC estimate.
1684 if (auto EstimatedTC = getSmallBestKnownTC(
1685 PSE, L: OuterLoop, /* CanUseConstantMax = */ false))
1686 if (EstimatedTC->isFixed())
1687 BestTripCount = EstimatedTC->getFixedValue();
1688
1689 InstructionCost NewMemCheckCost = MemCheckCost / BestTripCount;
1690
1691 // Let's ensure the cost is always at least 1.
1692 NewMemCheckCost = std::max(a: NewMemCheckCost.getValue(),
1693 b: (InstructionCost::CostType)1);
1694
1695 if (BestTripCount > 1)
1696 LLVM_DEBUG(dbgs()
1697 << "We expect runtime memory checks to be hoisted "
1698 << "out of the outer loop. Cost reduced from "
1699 << MemCheckCost << " to " << NewMemCheckCost << '\n');
1700
1701 MemCheckCost = NewMemCheckCost;
1702 }
1703 }
1704
1705 RTCheckCost += MemCheckCost;
1706 }
1707
1708 if (SCEVCheckBlock || MemCheckBlock)
1709 LLVM_DEBUG(dbgs() << "Total cost of runtime checks: " << RTCheckCost
1710 << "\n");
1711
1712 return RTCheckCost;
1713 }
1714
1715 /// Remove the created SCEV & memory runtime check blocks & instructions, if
1716 /// unused.
1717 ~GeneratedRTChecks() {
1718 SCEVExpanderCleaner SCEVCleaner(SCEVExp);
1719 bool SCEVChecksUsed = !SCEVCheckBlock || !pred_empty(BB: SCEVCheckBlock);
1720 if (SCEVChecksUsed)
1721 SCEVCleaner.markResultUsed();
1722
1723 if (MemCheckBlock && pred_empty(BB: MemCheckBlock))
1724 eraseMemCheckBlock();
1725
1726 SCEVCleaner.cleanup();
1727
1728 if (!SCEVChecksUsed)
1729 SCEVCheckBlock->eraseFromParent();
1730 }
1731
1732 /// Retrieves the SCEVCheckCond and SCEVCheckBlock that were generated as IR
1733 /// outside VPlan.
1734 std::pair<Value *, BasicBlock *> getSCEVChecks() const {
1735 using namespace llvm::PatternMatch;
1736 if (!SCEVCheckCond || match(V: SCEVCheckCond, P: m_ZeroInt()))
1737 return {nullptr, nullptr};
1738
1739 return {SCEVCheckCond, SCEVCheckBlock};
1740 }
1741
1742 /// Retrieves the MemCheckCond and MemCheckBlock that were generated as IR
1743 /// outside VPlan.
1744 std::pair<Value *, BasicBlock *> getMemRuntimeChecks() const {
1745 using namespace llvm::PatternMatch;
1746 if (MemRuntimeCheckCond && match(V: MemRuntimeCheckCond, P: m_ZeroInt()))
1747 return {nullptr, nullptr};
1748 return {MemRuntimeCheckCond, MemCheckBlock};
1749 }
1750
1751 /// Return true if any runtime checks have been added
1752 bool hasChecks() const { return HasChecks; }
1753
1754 /// Erase the memory check block, its instructions and their SCEV expansions.
1755 void eraseMemCheckBlock() {
1756 SCEVExpanderCleaner MemCheckCleaner(MemCheckExp);
1757 auto &SE = *MemCheckExp.getSE();
1758 // Memory runtime check generation creates compares that use expanded
1759 // values. Remove them before running the SCEVExpanderCleaner.
1760 for (auto &I : make_early_inc_range(Range: reverse(C&: *MemCheckBlock))) {
1761 if (MemCheckExp.isInsertedInstruction(I: &I))
1762 continue;
1763 SE.forgetValue(V: &I);
1764 I.eraseFromParent();
1765 }
1766 MemCheckCleaner.cleanup();
1767 MemCheckBlock->eraseFromParent();
1768 MemCheckBlock = nullptr;
1769 MemRuntimeCheckCond = nullptr;
1770 }
1771};
1772} // namespace
1773
1774static bool useActiveLaneMask(TailFoldingStyle Style) {
1775 return Style == TailFoldingStyle::Data ||
1776 Style == TailFoldingStyle::DataAndControlFlow;
1777}
1778
1779static bool useActiveLaneMaskForControlFlow(TailFoldingStyle Style) {
1780 return Style == TailFoldingStyle::DataAndControlFlow;
1781}
1782
1783// Return true if \p OuterLp is an outer loop annotated with hints for explicit
1784// vectorization. The loop needs to be annotated with #pragma omp simd
1785// simdlen(#) or #pragma clang vectorize(enable) vectorize_width(#). If the
1786// vector length information is not provided, vectorization is not considered
1787// explicit. Interleave hints are not allowed either. These limitations will be
1788// relaxed in the future.
1789// Please, note that we are currently forced to abuse the pragma 'clang
1790// vectorize' semantics. This pragma provides *auto-vectorization hints*
1791// (i.e., LV must check that vectorization is legal) whereas pragma 'omp simd'
1792// provides *explicit vectorization hints* (LV can bypass legal checks and
1793// assume that vectorization is legal). However, both hints are implemented
1794// using the same metadata (llvm.loop.vectorize, processed by
1795// LoopVectorizeHints). This will be fixed in the future when the native IR
1796// representation for pragma 'omp simd' is introduced.
1797static bool isExplicitVecOuterLoop(Loop *OuterLp,
1798 OptimizationRemarkEmitter *ORE) {
1799 assert(!OuterLp->isInnermost() && "This is not an outer loop");
1800 LoopVectorizeHints Hints(OuterLp, true /*DisableInterleaving*/, *ORE);
1801
1802 // Only outer loops with an explicit vectorization hint are supported.
1803 // Unannotated outer loops are ignored.
1804 if (Hints.getForce() == LoopVectorizeHints::FK_Undefined)
1805 return false;
1806
1807 Function *Fn = OuterLp->getHeader()->getParent();
1808 if (!Hints.allowVectorization(F: Fn, L: OuterLp,
1809 VectorizeOnlyWhenForced: true /*VectorizeOnlyWhenForced*/)) {
1810 LLVM_DEBUG(dbgs() << "LV: Loop hints prevent outer loop vectorization.\n");
1811 return false;
1812 }
1813
1814 if (Hints.getInterleave() > 1) {
1815 // TODO: Interleave support is future work.
1816 LLVM_DEBUG(dbgs() << "LV: Not vectorizing: Interleave is not supported for "
1817 "outer loops.\n");
1818 Hints.emitRemarkWithHints();
1819 return false;
1820 }
1821
1822 return true;
1823}
1824
1825static void collectSupportedLoops(Loop &L, LoopInfo *LI,
1826 OptimizationRemarkEmitter *ORE,
1827 SmallVectorImpl<Loop *> &V) {
1828 // Collect inner loops and outer loops without irreducible control flow. For
1829 // now, only collect outer loops that have explicit vectorization hints. If we
1830 // are stress testing the VPlan H-CFG construction, we collect the outermost
1831 // loop of every loop nest.
1832 if (L.isInnermost() || VPlanBuildOuterloopStressTest ||
1833 (EnableVPlanNativePath && isExplicitVecOuterLoop(OuterLp: &L, ORE))) {
1834 LoopBlocksRPO RPOT(&L);
1835 RPOT.perform(LI);
1836 if (!containsIrreducibleCFG<const BasicBlock *>(RPOTraversal&: RPOT, LI: *LI)) {
1837 V.push_back(Elt: &L);
1838 // TODO: Collect inner loops inside marked outer loops in case
1839 // vectorization fails for the outer loop. Do not invoke
1840 // 'containsIrreducibleCFG' again for inner loops when the outer loop is
1841 // already known to be reducible. We can use an inherited attribute for
1842 // that.
1843 return;
1844 }
1845 }
1846 for (Loop *InnerL : L)
1847 collectSupportedLoops(L&: *InnerL, LI, ORE, V);
1848}
1849
1850//===----------------------------------------------------------------------===//
1851// Implementation of LoopVectorizationLegality, InnerLoopVectorizer and
1852// LoopVectorizationCostModel and LoopVectorizationPlanner.
1853//===----------------------------------------------------------------------===//
1854
1855/// For the given VF and UF and maximum trip count computed for the loop, return
1856/// whether the induction variable might overflow in the vectorized loop. If not,
1857/// then we know a runtime overflow check always evaluates to false and can be
1858/// removed.
1859static bool isIndvarOverflowCheckKnownFalse(
1860 const LoopVectorizationCostModel *Cost,
1861 ElementCount VF, std::optional<unsigned> UF = std::nullopt) {
1862 // Always be conservative if we don't know the exact unroll factor.
1863 uint64_t MaxUF = UF ? *UF
1864 : std::max(a: Cost->TTI.getMaxInterleaveFactor(VF, HasUnorderedReductions: false),
1865 b: Cost->TTI.getMaxInterleaveFactor(VF, HasUnorderedReductions: true));
1866
1867 IntegerType *IdxTy = Cost->Legal->getWidestInductionType();
1868 APInt MaxUIntTripCount = IdxTy->getMask();
1869
1870 // We know the runtime overflow check is known false iff the (max) trip-count
1871 // is known and (max) trip-count + (VF * UF) does not overflow in the type of
1872 // the vector loop induction variable.
1873 if (std::optional<ElementCount> TC = getSmallBestKnownTC(
1874 PSE&: Cost->PSE, L: Cost->TheLoop,
1875 /*CanUseConstantMax=*/true, /*CanExcludeZeroTrips=*/false,
1876 /*ComputeUpperBoundOnly=*/true)) {
1877 // Compute the maximum runtime values of VF and the trip count.
1878 std::optional<uint64_t> MaxStep =
1879 getMaxRuntimeElementCount(EC: VF * MaxUF, F: *Cost->TheFunction);
1880 std::optional<uint64_t> MaxTC =
1881 getMaxRuntimeElementCount(EC: *TC, F: *Cost->TheFunction);
1882 if (!MaxStep || !MaxTC)
1883 return false;
1884
1885 // Bail out if the maximum trip count is not representable in the induction
1886 // variable's type.
1887 if (MaxUIntTripCount.ult(RHS: *MaxTC))
1888 return false;
1889
1890 return (MaxUIntTripCount - *MaxTC).ugt(RHS: *MaxStep);
1891 }
1892
1893 return false;
1894}
1895
1896// Return whether we allow using masked interleave-groups (for dealing with
1897// strided loads/stores that reside in predicated blocks, or for dealing
1898// with gaps).
1899static bool useMaskedInterleavedAccesses(const TargetTransformInfo &TTI) {
1900 // If an override option has been passed in for interleaved accesses, use it.
1901 if (EnableMaskedInterleavedMemAccesses.getNumOccurrences() > 0)
1902 return EnableMaskedInterleavedMemAccesses;
1903
1904 return TTI.enableMaskedInterleavedAccessVectorization();
1905}
1906
1907/// Replace \p VPBB with a VPIRBasicBlock wrapping \p IRBB. All recipes from \p
1908/// VPBB are moved to the end of the newly created VPIRBasicBlock. All
1909/// predecessors and successors of VPBB, if any, are rewired to the new
1910/// VPIRBasicBlock. If \p VPBB may be unreachable, \p Plan must be passed.
1911static VPIRBasicBlock *replaceVPBBWithIRVPBB(VPBasicBlock *VPBB,
1912 BasicBlock *IRBB,
1913 VPlan *Plan = nullptr) {
1914 if (!Plan)
1915 Plan = VPBB->getPlan();
1916 VPIRBasicBlock *IRVPBB = Plan->createEmptyVPIRBasicBlock(IRBB);
1917 auto IP = IRVPBB->begin();
1918 for (auto &R : make_early_inc_range(Range: VPBB->phis()))
1919 R.moveBefore(BB&: *IRVPBB, I: IP);
1920
1921 for (auto &R :
1922 make_early_inc_range(Range: make_range(x: VPBB->getFirstNonPhi(), y: VPBB->end())))
1923 R.moveBefore(BB&: *IRVPBB, I: IRVPBB->end());
1924
1925 VPBlockUtils::reassociateBlocks(Old: VPBB, New: IRVPBB);
1926 // VPBB is now dead and will be cleaned up when the plan gets destroyed.
1927 return IRVPBB;
1928}
1929
1930BasicBlock *InnerLoopVectorizer::createScalarPreheader(StringRef Prefix) {
1931 BasicBlock *VectorPH = OrigLoop->getLoopPreheader();
1932 assert(VectorPH && "Invalid loop structure");
1933
1934 // NOTE: The Plan's scalar preheader VPBB isn't replaced with a VPIRBasicBlock
1935 // wrapping the newly created scalar preheader here at the moment, because the
1936 // Plan's scalar preheader may be unreachable at this point. Instead it is
1937 // replaced in executePlan.
1938 return SplitBlock(Old: VectorPH, SplitPt: VectorPH->getTerminator(), DT, LI, MSSAU: nullptr,
1939 BBName: Twine(Prefix) + "scalar.ph");
1940}
1941
1942/// Knowing that loop \p L executes a single vector iteration, add instructions
1943/// that will get simplified and thus should not have any cost to \p
1944/// InstsToIgnore.
1945static void addFullyUnrolledInstructionsToIgnore(
1946 Loop *L, const LoopVectorizationLegality::InductionList &IL,
1947 SmallPtrSetImpl<Instruction *> &InstsToIgnore) {
1948 auto *Cmp = L->getLatchCmpInst();
1949 if (Cmp)
1950 InstsToIgnore.insert(Ptr: Cmp);
1951 for (PHINode *IV : IL.keys()) {
1952 // The induction is free: a widened induction generates a vector phi with
1953 // its start value and an increment that is dead without a backedge.
1954 InstsToIgnore.insert(Ptr: IV);
1955
1956 // Get next iteration value of the induction variable.
1957 Instruction *IVInst =
1958 cast<Instruction>(Val: IV->getIncomingValueForBlock(BB: L->getLoopLatch()));
1959 if (all_of(Range: IVInst->users(),
1960 P: [&](const User *U) { return U == IV || U == Cmp; }))
1961 InstsToIgnore.insert(Ptr: IVInst);
1962 }
1963}
1964
1965BasicBlock *InnerLoopVectorizer::createVectorizedLoopSkeleton() {
1966 // Create a new IR basic block for the scalar preheader.
1967 BasicBlock *ScalarPH = createScalarPreheader(Prefix: "");
1968 return ScalarPH->getSinglePredecessor();
1969}
1970
1971namespace {
1972
1973struct CSEDenseMapInfo {
1974 static bool canHandle(const Instruction *I) {
1975 return isa<InsertElementInst>(Val: I) || isa<ExtractElementInst>(Val: I) ||
1976 isa<ShuffleVectorInst>(Val: I) || isa<GetElementPtrInst>(Val: I);
1977 }
1978
1979 static unsigned getHashValue(const Instruction *I) {
1980 assert(canHandle(I) && "Unknown instruction!");
1981 return hash_combine(args: I->getOpcode(),
1982 args: hash_combine_range(R: I->operand_values()));
1983 }
1984
1985 static bool isEqual(const Instruction *LHS, const Instruction *RHS) {
1986 return LHS->isIdenticalTo(I: RHS);
1987 }
1988};
1989
1990} // end anonymous namespace
1991
1992/// FIXME: This legacy common-subexpression-elimination routine is scheduled for
1993/// removal, in favor of the VPlan-based one.
1994static void legacyCSE(BasicBlock *BB) {
1995 // Perform simple cse.
1996 SmallDenseMap<Instruction *, Instruction *, 4, CSEDenseMapInfo> CSEMap;
1997 for (Instruction &In : llvm::make_early_inc_range(Range&: *BB)) {
1998 if (!CSEDenseMapInfo::canHandle(I: &In))
1999 continue;
2000
2001 // Check if we can replace this instruction with any of the
2002 // visited instructions.
2003 if (Instruction *V = CSEMap.lookup(Val: &In)) {
2004 In.replaceAllUsesWith(V);
2005 In.eraseFromParent();
2006 continue;
2007 }
2008
2009 CSEMap[&In] = &In;
2010 }
2011}
2012
2013/// This function attempts to return a value that represents the ElementCount
2014/// at runtime. For fixed-width VFs we know this precisely at compile
2015/// time, but for scalable VFs we calculate it based on an estimate of the
2016/// vscale value.
2017static unsigned estimateElementCount(ElementCount VF,
2018 std::optional<unsigned> VScale) {
2019 unsigned EstimatedVF = VF.getKnownMinValue();
2020 if (VF.isScalable())
2021 if (VScale)
2022 EstimatedVF *= *VScale;
2023 assert(EstimatedVF >= 1 && "Estimated VF shouldn't be less than 1");
2024 return EstimatedVF;
2025}
2026
2027/// Returns the vector library variant function of \p CI usable at \p VF,
2028/// respecting \p MaskRequired, or nullptr if none is found: a mapping with
2029/// matching VF, masked if required, whose vector function is declared in the
2030/// module.
2031static Function *getVectorLibraryVariantFor(const CallInst &CI, ElementCount VF,
2032 bool MaskRequired,
2033 const TargetLibraryInfo *TLI) {
2034 if (!TLI || CI.isNoBuiltin())
2035 return nullptr;
2036 for (const VFInfo &Info : VFDatabase::getMappings(CI))
2037 if (Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()))
2038 if (Function *F = CI.getModule()->getFunction(Name: Info.VectorName))
2039 return F;
2040 return nullptr;
2041}
2042
2043/// Returns true iff \p CI has a library vector variant usable at \p VF.
2044static bool hasVectorLibraryVariantFor(const CallInst &CI, ElementCount VF,
2045 bool MaskRequired,
2046 const TargetLibraryInfo *TLI) {
2047 return getVectorLibraryVariantFor(CI, VF, MaskRequired, TLI) != nullptr;
2048}
2049
2050InstructionCost
2051LoopVectorizationCostModel::getVectorCallCost(CallInst *CI,
2052 ElementCount VF) const {
2053 Type *RetTy = CI->getType();
2054 SmallVector<Type *, 4> Tys;
2055 for (auto &ArgOp : CI->args())
2056 Tys.push_back(Elt: ArgOp->getType());
2057
2058 InstructionCost ScalarCallCost = TTI.getCallInstrCost(
2059 F: CI->getCalledFunction(), RetTy, Tys, CostKind: Config.CostKind);
2060
2061 // Cost of the scalar call (scalar VF) or its scalarization (vector VF). The
2062 // scalarization cost is only meaningful for fixed VFs.
2063 InstructionCost Cost = VF.isScalable()
2064 ? InstructionCost::getInvalid()
2065 : ScalarCallCost * VF.getKnownMinValue() +
2066 getScalarizationOverhead(I: CI, VF);
2067
2068 // The call may be vectorized at this VF, via a vector intrinsic or a vector
2069 // library variant.
2070 if (getVectorIntrinsicIDForCall(CI, TLI))
2071 Cost = std::min(a: Cost, b: getVectorIntrinsicCost(CI, VF));
2072
2073 if (Function *Variant =
2074 getVectorLibraryVariantFor(CI: *CI, VF, MaskRequired: isMaskRequired(I: CI), TLI))
2075 Cost = std::min(a: Cost,
2076 b: TTI.getCallInstrCost(
2077 /*F=*/nullptr, RetTy: Variant->getReturnType(),
2078 Tys: Variant->getFunctionType()->params(), CostKind: Config.CostKind));
2079
2080 return Cost;
2081}
2082
2083static Type *maybeVectorizeType(Type *Ty, ElementCount VF) {
2084 if (VF.isScalar() || !canVectorizeTy(Ty))
2085 return Ty;
2086 return toVectorizedTy(Ty, EC: VF);
2087}
2088
2089InstructionCost
2090LoopVectorizationCostModel::getVectorIntrinsicCost(CallInst *CI,
2091 ElementCount VF) const {
2092 Intrinsic::ID ID = getVectorIntrinsicIDForCall(CI, TLI);
2093 assert(ID && "Expected intrinsic call!");
2094 Type *RetTy = maybeVectorizeType(Ty: CI->getType(), VF);
2095 FastMathFlags FMF;
2096 if (auto *FPMO = dyn_cast<FPMathOperator>(Val: CI))
2097 FMF = FPMO->getFastMathFlags();
2098
2099 SmallVector<const Value *> Arguments(CI->args());
2100 FunctionType *FTy = CI->getCalledFunction()->getFunctionType();
2101 SmallVector<Type *> ParamTys;
2102 std::transform(first: FTy->param_begin(), last: FTy->param_end(),
2103 result: std::back_inserter(x&: ParamTys),
2104 unary_op: [&](Type *Ty) { return maybeVectorizeType(Ty, VF); });
2105
2106 IntrinsicCostAttributes CostAttrs(ID, RetTy, Arguments, ParamTys, FMF,
2107 dyn_cast<IntrinsicInst>(Val: CI),
2108 InstructionCost::getInvalid());
2109 return TTI.getIntrinsicInstrCost(ICA: CostAttrs, CostKind: Config.CostKind);
2110}
2111
2112void InnerLoopVectorizer::fixVectorizedLoop(VPTransformState &State) {
2113 // Don't apply optimizations below when no (vector) loop remains, as they all
2114 // require one at the moment.
2115 VPBasicBlock *HeaderVPBB =
2116 vputils::getFirstLoopHeader(Plan&: *State.Plan, VPDT&: State.VPDT);
2117 if (!HeaderVPBB)
2118 return;
2119
2120 BasicBlock *HeaderBB = State.CFG.VPBB2IRBB[HeaderVPBB];
2121
2122 // Remove redundant induction instructions.
2123 legacyCSE(BB: HeaderBB);
2124}
2125
2126void LoopVectorizationCostModel::collectLoopScalars(ElementCount VF) {
2127 // We should not collect Scalars more than once per VF. Right now, this
2128 // function is called from collectUniformsAndScalars(), which already does
2129 // this check. Collecting Scalars for VF=1 does not make any sense.
2130 assert(VF.isVector() && !Scalars.contains(VF) &&
2131 "This function should not be visited twice for the same VF");
2132
2133 // This avoids any chances of creating a REPLICATE recipe during planning
2134 // since that would result in generation of scalarized code during execution,
2135 // which is not supported for scalable vectors.
2136 if (VF.isScalable()) {
2137 Scalars[VF].insert_range(R&: Uniforms[VF]);
2138 return;
2139 }
2140
2141 SmallSetVector<Instruction *, 8> Worklist;
2142
2143 // These sets are used to seed the analysis with pointers used by memory
2144 // accesses that will remain scalar.
2145 SmallSetVector<Instruction *, 8> ScalarPtrs;
2146 SmallPtrSet<Instruction *, 8> PossibleNonScalarPtrs;
2147 auto *Latch = TheLoop->getLoopLatch();
2148
2149 // A helper that returns true if the use of Ptr by MemAccess will be scalar.
2150 // The pointer operands of loads and stores will be scalar as long as the
2151 // memory access is not a gather/scatter or histogram operation. The value
2152 // operand of a store will remain scalar if the store is scalarized.
2153 auto IsScalarUse = [&](Instruction *MemAccess, Value *Ptr) {
2154 InstWidening WideningDecision = getWideningDecision(I: MemAccess, VF);
2155 assert(WideningDecision != CM_Unknown &&
2156 "Widening decision should be ready at this moment");
2157 auto *Store = dyn_cast<StoreInst>(Val: MemAccess);
2158 if (Store && Ptr == Store->getValueOperand())
2159 return WideningDecision == CM_Scalarize;
2160 assert(Ptr == getLoadStorePointerOperand(MemAccess) &&
2161 "Ptr is neither a value or pointer operand");
2162 return WideningDecision != CM_GatherScatter &&
2163 !(Store && Legal->getHistogramInfo(I: Store));
2164 };
2165
2166 // A helper that returns true if the given value is a getelementptr
2167 // instruction contained in the loop.
2168 auto IsLoopVaryingGEP = [&](Value *V) {
2169 return isa<GetElementPtrInst>(Val: V) && !TheLoop->isLoopInvariant(V);
2170 };
2171
2172 // A helper that evaluates a memory access's use of a pointer. If the use will
2173 // be a scalar use and the pointer is only used by memory accesses, we place
2174 // the pointer in ScalarPtrs. Otherwise, the pointer is placed in
2175 // PossibleNonScalarPtrs.
2176 auto EvaluatePtrUse = [&](Instruction *MemAccess, Value *Ptr) {
2177 // We only care about bitcast and getelementptr instructions contained in
2178 // the loop.
2179 if (!IsLoopVaryingGEP(Ptr))
2180 return;
2181
2182 // If the pointer has already been identified as scalar (e.g., if it was
2183 // also identified as uniform), there's nothing to do.
2184 auto *I = cast<Instruction>(Val: Ptr);
2185 if (Worklist.count(key: I))
2186 return;
2187
2188 // If the use of the pointer will be a scalar use, and all users of the
2189 // pointer are memory accesses, place the pointer in ScalarPtrs. Otherwise,
2190 // place the pointer in PossibleNonScalarPtrs.
2191 if (IsScalarUse(MemAccess, Ptr) &&
2192 all_of(Range: I->users(), P: IsaPred<LoadInst, StoreInst>))
2193 ScalarPtrs.insert(X: I);
2194 else
2195 PossibleNonScalarPtrs.insert(Ptr: I);
2196 };
2197
2198 // We seed the scalars analysis with three classes of instructions: (1)
2199 // instructions marked uniform-after-vectorization and (2) bitcast,
2200 // getelementptr and (pointer) phi instructions used by memory accesses
2201 // requiring a scalar use.
2202 //
2203 // (1) Add to the worklist all instructions that have been identified as
2204 // uniform-after-vectorization.
2205 Worklist.insert_range(R&: Uniforms[VF]);
2206
2207 // (2) Add to the worklist all bitcast and getelementptr instructions used by
2208 // memory accesses requiring a scalar use. The pointer operands of loads and
2209 // stores will be scalar unless the operation is a gather or scatter.
2210 // The value operand of a store will remain scalar if the store is scalarized.
2211 for (auto *BB : TheLoop->blocks())
2212 for (auto &I : *BB) {
2213 if (auto *Load = dyn_cast<LoadInst>(Val: &I)) {
2214 EvaluatePtrUse(Load, Load->getPointerOperand());
2215 } else if (auto *Store = dyn_cast<StoreInst>(Val: &I)) {
2216 EvaluatePtrUse(Store, Store->getPointerOperand());
2217 EvaluatePtrUse(Store, Store->getValueOperand());
2218 }
2219 }
2220 for (auto *I : ScalarPtrs)
2221 if (!PossibleNonScalarPtrs.count(Ptr: I)) {
2222 LLVM_DEBUG(dbgs() << "LV: Found scalar instruction: " << *I << "\n");
2223 Worklist.insert(X: I);
2224 }
2225
2226 // Insert the forced scalars.
2227 // FIXME: Currently VPWidenPHIRecipe() often creates a dead vector
2228 // induction variable when the PHI user is scalarized.
2229 auto ForcedScalar = ForcedScalars.find(Val: VF);
2230 if (ForcedScalar != ForcedScalars.end())
2231 for (auto *I : ForcedScalar->second) {
2232 LLVM_DEBUG(dbgs() << "LV: Found (forced) scalar instruction: " << *I << "\n");
2233 Worklist.insert(X: I);
2234 }
2235
2236 // Expand the worklist by looking through any bitcasts and getelementptr
2237 // instructions we've already identified as scalar. This is similar to the
2238 // expansion step in collectLoopUniforms(); however, here we're only
2239 // expanding to include additional bitcasts and getelementptr instructions.
2240 unsigned Idx = 0;
2241 while (Idx != Worklist.size()) {
2242 Instruction *Dst = Worklist[Idx++];
2243 if (!IsLoopVaryingGEP(Dst->getOperand(i: 0)))
2244 continue;
2245 auto *Src = cast<Instruction>(Val: Dst->getOperand(i: 0));
2246 if (llvm::all_of(Range: Src->users(), P: [&](User *U) -> bool {
2247 auto *J = cast<Instruction>(Val: U);
2248 return !TheLoop->contains(Inst: J) || Worklist.count(key: J) ||
2249 ((isa<LoadInst>(Val: J) || isa<StoreInst>(Val: J)) &&
2250 IsScalarUse(J, Src));
2251 })) {
2252 Worklist.insert(X: Src);
2253 LLVM_DEBUG(dbgs() << "LV: Found scalar instruction: " << *Src << "\n");
2254 }
2255 }
2256
2257 // An induction variable will remain scalar if all users of the induction
2258 // variable and induction variable update remain scalar.
2259 for (const auto &Induction : Legal->getInductionVars()) {
2260 auto *Ind = Induction.first;
2261 auto *IndUpdate = cast<Instruction>(Val: Ind->getIncomingValueForBlock(BB: Latch));
2262
2263 // If tail-folding is applied, the primary induction variable will be used
2264 // to feed a vector compare.
2265 if (Ind == Legal->getPrimaryInduction() && foldTailByMasking())
2266 continue;
2267
2268 // Returns true if \p Indvar is a pointer induction that is used directly by
2269 // load/store instruction \p I.
2270 auto IsDirectLoadStoreFromPtrIndvar = [&](Instruction *Indvar,
2271 Instruction *I) {
2272 return Induction.second.getKind() ==
2273 InductionDescriptor::IK_PtrInduction &&
2274 (isa<LoadInst>(Val: I) || isa<StoreInst>(Val: I)) &&
2275 Indvar == getLoadStorePointerOperand(V: I) && IsScalarUse(I, Indvar);
2276 };
2277
2278 // Determine if all users of the induction variable are scalar after
2279 // vectorization.
2280 bool ScalarInd = all_of(Range: Ind->users(), P: [&](User *U) -> bool {
2281 auto *I = cast<Instruction>(Val: U);
2282 return I == IndUpdate || !TheLoop->contains(Inst: I) || Worklist.count(key: I) ||
2283 IsDirectLoadStoreFromPtrIndvar(Ind, I);
2284 });
2285 if (!ScalarInd)
2286 continue;
2287
2288 // If the induction variable update is a fixed-order recurrence, neither the
2289 // induction variable or its update should be marked scalar after
2290 // vectorization.
2291 auto *IndUpdatePhi = dyn_cast<PHINode>(Val: IndUpdate);
2292 if (IndUpdatePhi && Legal->isFixedOrderRecurrence(Phi: IndUpdatePhi))
2293 continue;
2294
2295 // Determine if all users of the induction variable update instruction are
2296 // scalar after vectorization.
2297 bool ScalarIndUpdate = all_of(Range: IndUpdate->users(), P: [&](User *U) -> bool {
2298 auto *I = cast<Instruction>(Val: U);
2299 return I == Ind || !TheLoop->contains(Inst: I) || Worklist.count(key: I) ||
2300 IsDirectLoadStoreFromPtrIndvar(IndUpdate, I);
2301 });
2302 if (!ScalarIndUpdate)
2303 continue;
2304
2305 // The induction variable and its update instruction will remain scalar.
2306 Worklist.insert(X: Ind);
2307 Worklist.insert(X: IndUpdate);
2308 LLVM_DEBUG(dbgs() << "LV: Found scalar instruction: " << *Ind << "\n");
2309 LLVM_DEBUG(dbgs() << "LV: Found scalar instruction: " << *IndUpdate
2310 << "\n");
2311 }
2312
2313 Scalars[VF].insert_range(R&: Worklist);
2314}
2315
2316bool LoopVectorizationCostModel::isLegalMaskedLoadOrStore(
2317 Instruction *I, ElementCount VF) const {
2318 assert(isa<LoadInst>(I) || isa<StoreInst>(I));
2319 return Config.isLegalMaskedLoadOrStore(IsLoad: isa<LoadInst>(Val: I), ScalarTy: getLoadStoreType(I),
2320 Alignment: getLoadStoreAlignment(I),
2321 AddressSpace: getLoadStoreAddressSpace(I));
2322}
2323
2324bool LoopVectorizationCostModel::isLegalGatherOrScatter(Instruction *I,
2325 ElementCount VF) const {
2326 assert((isa<LoadInst, StoreInst>(I)));
2327 return Config.isLegalGatherOrScatter(IsLoad: isa<LoadInst>(Val: I), ScalarTy: getLoadStoreType(I),
2328 Alignment: getLoadStoreAlignment(I), VF);
2329}
2330
2331bool LoopVectorizationCostModel::isScalarWithPredication(Instruction *I,
2332 ElementCount VF) {
2333 if (!isPredicatedInst(I))
2334 return false;
2335
2336 // Do we have a non-scalar lowering for this predicated
2337 // instruction? No - it is scalar with predication.
2338 switch(I->getOpcode()) {
2339 default:
2340 return true;
2341 case Instruction::Call: {
2342 if (VF.isScalar())
2343 return true;
2344 auto *CI = cast<CallInst>(Val: I);
2345 // A vector intrinsic or library variant lowering avoids scalarization.
2346 return !getVectorIntrinsicIDForCall(CI, TLI) &&
2347 !hasVectorLibraryVariantFor(CI: *CI, VF, MaskRequired: isMaskRequired(I: CI), TLI);
2348 }
2349 case Instruction::Load:
2350 case Instruction::Store: {
2351 bool IsConsecutive = Legal->isConsecutivePtr(AccessTy: getLoadStoreType(I),
2352 Ptr: getLoadStorePointerOperand(V: I));
2353 return !(IsConsecutive && isLegalMaskedLoadOrStore(I, VF)) &&
2354 !isLegalGatherOrScatter(I, VF);
2355 }
2356 case Instruction::UDiv:
2357 case Instruction::SDiv:
2358 case Instruction::SRem:
2359 case Instruction::URem: {
2360 // We have the option to use the llvm.masked.udiv intrinsics to avoid
2361 // predication. The cost based decision here will always select the masked
2362 // intrinsics for scalable vectors as scalarization isn't legal.
2363 const auto [ScalarCost, MaskedCost] = getDivRemSpeculationCost(I, VF);
2364 return isDivRemScalarWithPredication(ScalarCost, MaskedCost);
2365 }
2366 }
2367}
2368
2369bool LoopVectorizationCostModel::isMaskRequired(Instruction *I) const {
2370 return Legal->isMaskRequired(I, TailFolded: foldTailByMasking());
2371}
2372
2373// TODO: Fold into LoopVectorizationLegality::isMaskRequired.
2374bool LoopVectorizationCostModel::isPredicatedInst(Instruction *I) const {
2375 // TODO: We can use the loop-preheader as context point here and get
2376 // context sensitive reasoning for isSafeToSpeculativelyExecute.
2377 if (isSafeToSpeculativelyExecute(I) ||
2378 (isa<LoadInst, StoreInst, CallInst>(Val: I) && !isMaskRequired(I)) ||
2379 isa<UncondBrInst, CondBrInst, SwitchInst, PHINode, AllocaInst>(Val: I))
2380 return false;
2381
2382 // If the instruction was executed conditionally in the original scalar loop,
2383 // predication is needed with a mask whose lanes are all possibly inactive.
2384 if (Legal->blockNeedsPredication(BB: I->getParent()))
2385 return true;
2386
2387 // If we're not folding the tail by masking and not vectorizing a loop with
2388 // uncountable exits and side effects, predication is unnecessary.
2389 if (!foldTailByMasking() && !Legal->hasUncountableExitWithSideEffects())
2390 return false;
2391
2392 // All that remain are instructions with side-effects originally executed in
2393 // the loop unconditionally, but now execute under a tail-fold mask (only)
2394 // having at least one active lane (the first). If the side-effects of the
2395 // instruction are invariant, executing it w/o (the tail-folding) mask is safe
2396 // - it will cause the same side-effects as when masked.
2397 switch(I->getOpcode()) {
2398 default:
2399 llvm_unreachable(
2400 "instruction should have been considered by earlier checks");
2401 case Instruction::Call:
2402 // Side-effects of a Call are assumed to be non-invariant, needing a
2403 // (fold-tail) mask.
2404 assert(isMaskRequired(I) &&
2405 "should have returned earlier for calls not needing a mask");
2406 return true;
2407 case Instruction::Load:
2408 // If the address is loop invariant no predication is needed.
2409 return !Legal->isInvariant(V: getLoadStorePointerOperand(V: I));
2410 case Instruction::Store: {
2411 // For stores, we need to prove both speculation safety (which follows from
2412 // the same argument as loads), but also must prove the value being stored
2413 // is correct. The easiest form of the later is to require that all values
2414 // stored are the same.
2415 return !(Legal->isInvariant(V: getLoadStorePointerOperand(V: I)) &&
2416 TheLoop->isLoopInvariant(V: cast<StoreInst>(Val: I)->getValueOperand()));
2417 }
2418 case Instruction::UDiv:
2419 case Instruction::URem:
2420 // If the divisor is loop-invariant no predication is needed.
2421 return !Legal->isInvariant(V: I->getOperand(i: 1));
2422 case Instruction::SDiv:
2423 case Instruction::SRem:
2424 // Conservative for now, since masked-off lanes may be poison and could
2425 // trigger signed overflow.
2426 return true;
2427 }
2428}
2429
2430uint64_t LoopVectorizationCostModel::getPredBlockCostDivisor(
2431 TargetTransformInfo::TargetCostKind CostKind, const BasicBlock *BB) {
2432 if (CostKind == TTI::TCK_CodeSize)
2433 return 1;
2434 // If the block wasn't originally predicated then return early to avoid
2435 // computing BlockFrequencyInfo unnecessarily.
2436 if (!Legal->blockNeedsPredication(BB))
2437 return 1;
2438
2439 uint64_t HeaderFreq =
2440 getBFI().getBlockFreq(BB: TheLoop->getHeader()).getFrequency();
2441 uint64_t BBFreq = getBFI().getBlockFreq(BB).getFrequency();
2442 assert(HeaderFreq >= BBFreq &&
2443 "Header has smaller block freq than dominated BB?");
2444 return std::round(x: (double)HeaderFreq / BBFreq);
2445}
2446
2447static Intrinsic::ID getMaskedDivRemIntrinsic(unsigned Opcode) {
2448 switch (Opcode) {
2449 case Instruction::UDiv:
2450 return Intrinsic::masked_udiv;
2451 case Instruction::SDiv:
2452 return Intrinsic::masked_sdiv;
2453 case Instruction::URem:
2454 return Intrinsic::masked_urem;
2455 case Instruction::SRem:
2456 return Intrinsic::masked_srem;
2457 default:
2458 llvm_unreachable("Unexpected opcode");
2459 }
2460}
2461
2462std::pair<InstructionCost, InstructionCost>
2463LoopVectorizationCostModel::getDivRemSpeculationCost(Instruction *I,
2464 ElementCount VF) {
2465 assert(I->getOpcode() == Instruction::UDiv ||
2466 I->getOpcode() == Instruction::SDiv ||
2467 I->getOpcode() == Instruction::SRem ||
2468 I->getOpcode() == Instruction::URem);
2469 assert(!isSafeToSpeculativelyExecute(I));
2470
2471 // Scalarization isn't legal for scalable vector types
2472 InstructionCost ScalarizationCost = InstructionCost::getInvalid();
2473 if (!VF.isScalable()) {
2474 // Get the scalarization cost and scale this amount by the probability of
2475 // executing the predicated block. If the instruction is not predicated,
2476 // we fall through to the next case.
2477 ScalarizationCost = 0;
2478
2479 // These instructions have a non-void type, so account for the phi nodes
2480 // that we will create. This cost is likely to be zero. The phi node
2481 // cost, if any, should be scaled by the block probability because it
2482 // models a copy at the end of each predicated block.
2483 ScalarizationCost += VF.getFixedValue() *
2484 TTI.getCFInstrCost(Opcode: Instruction::PHI, CostKind: Config.CostKind);
2485
2486 // The cost of the non-predicated instruction.
2487 ScalarizationCost +=
2488 VF.getFixedValue() * TTI.getArithmeticInstrCost(
2489 Opcode: I->getOpcode(), Ty: I->getType(), CostKind: Config.CostKind);
2490
2491 // The cost of insertelement and extractelement instructions needed for
2492 // scalarization.
2493 ScalarizationCost += getScalarizationOverhead(I, VF);
2494
2495 // Scale the cost by the probability of executing the predicated blocks.
2496 // This assumes the predicated block for each vector lane is equally
2497 // likely.
2498 ScalarizationCost =
2499 ScalarizationCost /
2500 getPredBlockCostDivisor(CostKind: Config.CostKind, BB: I->getParent());
2501 }
2502
2503 auto *VecTy = toVectorTy(Scalar: I->getType(), EC: VF);
2504 auto *MaskTy = toVectorTy(Scalar: Type::getInt1Ty(C&: I->getContext()), EC: VF);
2505 IntrinsicCostAttributes ICA(getMaskedDivRemIntrinsic(Opcode: I->getOpcode()), VecTy,
2506 {VecTy, VecTy, MaskTy});
2507 InstructionCost MaskedCost = TTI.getIntrinsicInstrCost(ICA, CostKind: Config.CostKind);
2508 return {ScalarizationCost, MaskedCost};
2509}
2510
2511bool LoopVectorizationCostModel::interleavedAccessCanBeWidened(
2512 Instruction *I, ElementCount VF) const {
2513 assert(isAccessInterleaved(I) && "Expecting interleaved access.");
2514 assert(getWideningDecision(I, VF) == CM_Unknown &&
2515 "Decision should not be set yet.");
2516 auto *Group = getInterleavedAccessGroup(Instr: I);
2517 assert(Group && "Must have a group.");
2518 unsigned InterleaveFactor = Group->getFactor();
2519
2520 // If the instruction's allocated size doesn't equal its type size, it
2521 // requires padding and will be scalarized.
2522 auto &DL = I->getDataLayout();
2523 auto *ScalarTy = getLoadStoreType(I);
2524 if (hasIrregularType(Ty: ScalarTy, DL))
2525 return false;
2526
2527 // For scalable vectors, the interleave factors must be <= 8 since we require
2528 // the (de)interleaveN intrinsics instead of shufflevectors.
2529 if (VF.isScalable() && InterleaveFactor > 8)
2530 return false;
2531
2532 // If the group involves a non-integral pointer, we may not be able to
2533 // losslessly cast all values to a common type.
2534 bool ScalarNI = DL.isNonIntegralPointerType(Ty: ScalarTy);
2535 for (Instruction *Member : Group->members()) {
2536 auto *MemberTy = getLoadStoreType(I: Member);
2537 bool MemberNI = DL.isNonIntegralPointerType(Ty: MemberTy);
2538 // Don't coerce non-integral pointers to integers or vice versa.
2539 if (MemberNI != ScalarNI)
2540 // TODO: Consider adding special nullptr value case here
2541 return false;
2542 if (MemberNI && ScalarNI &&
2543 ScalarTy->getPointerAddressSpace() !=
2544 MemberTy->getPointerAddressSpace())
2545 return false;
2546 }
2547
2548 // Check if masking is required.
2549 // A Group may need masking for one of two reasons: it resides in a block that
2550 // needs predication, or it was decided to use masking to deal with gaps
2551 // (either a gap at the end of a load-access that may result in a speculative
2552 // load, or any gaps in a store-access).
2553 bool PredicatedAccessRequiresMasking =
2554 blockNeedsPredicationForAnyReason(BB: I->getParent()) && isMaskRequired(I);
2555 bool LoadAccessWithGapsRequiresEpilogMasking =
2556 isa<LoadInst>(Val: I) && Group->requiresScalarEpilogue() &&
2557 !isEpilogueAllowed();
2558 bool StoreAccessWithGapsRequiresMasking =
2559 isa<StoreInst>(Val: I) && !Group->isFull();
2560 if (!PredicatedAccessRequiresMasking &&
2561 !LoadAccessWithGapsRequiresEpilogMasking &&
2562 !StoreAccessWithGapsRequiresMasking)
2563 return true;
2564
2565 // If masked interleaving is required, we expect that the user/target had
2566 // enabled it, because otherwise it either wouldn't have been created or
2567 // it should have been invalidated by the CostModel.
2568 assert(useMaskedInterleavedAccesses(TTI) &&
2569 "Masked interleave-groups for predicated accesses are not enabled.");
2570
2571 if (Group->isReverse())
2572 return false;
2573
2574 // TODO: Support interleaved access that requires a gap mask for scalable VFs.
2575 bool NeedsMaskForGaps = LoadAccessWithGapsRequiresEpilogMasking ||
2576 StoreAccessWithGapsRequiresMasking;
2577 if (VF.isScalable() && NeedsMaskForGaps)
2578 return false;
2579
2580 return isLegalMaskedLoadOrStore(I, VF);
2581}
2582
2583std::optional<LoopVectorizationCostModel::InstWidening>
2584LoopVectorizationCostModel::memoryInstructionCanBeWidened(Instruction *I,
2585 ElementCount VF) {
2586 // Get and ensure we have a valid memory instruction.
2587 assert((isa<LoadInst, StoreInst>(I)) && "Invalid memory instruction");
2588
2589 auto *Ptr = getLoadStorePointerOperand(V: I);
2590 auto *ScalarTy = getLoadStoreType(I);
2591
2592 // In order to be widened, the pointer should be consecutive, first of all.
2593 int Stride = Legal->isConsecutivePtr(AccessTy: ScalarTy, Ptr);
2594 if (!Stride)
2595 return std::nullopt;
2596
2597 // If the instruction is a store located in a predicated block, it will be
2598 // scalarized.
2599 if (isScalarWithPredication(I, VF))
2600 return std::nullopt;
2601
2602 // If the instruction's allocated size doesn't equal it's type size, it
2603 // requires padding and will be scalarized.
2604 auto &DL = I->getDataLayout();
2605 if (hasIrregularType(Ty: ScalarTy, DL))
2606 return std::nullopt;
2607
2608 return Stride == 1 ? CM_Widen : CM_Widen_Reverse;
2609}
2610
2611void LoopVectorizationCostModel::collectLoopUniforms(ElementCount VF) {
2612 // We should not collect Uniforms more than once per VF. Right now,
2613 // this function is called from collectUniformsAndScalars(), which
2614 // already does this check. Collecting Uniforms for VF=1 does not make any
2615 // sense.
2616
2617 assert(VF.isVector() && !Uniforms.contains(VF) &&
2618 "This function should not be visited twice for the same VF");
2619
2620 // Visit the list of Uniforms. If we find no uniform value, we won't
2621 // analyze again. Uniforms.count(VF) will return 1.
2622 Uniforms[VF].clear();
2623
2624 // Now we know that the loop is vectorizable!
2625 // Collect instructions inside the loop that will remain uniform after
2626 // vectorization.
2627
2628 // Global values, params and instructions outside of current loop are out of
2629 // scope.
2630 auto IsOutOfScope = [&](Value *V) -> bool {
2631 Instruction *I = dyn_cast<Instruction>(Val: V);
2632 return (!I || !TheLoop->contains(Inst: I));
2633 };
2634
2635 // Worklist containing uniform instructions demanding lane 0.
2636 SetVector<Instruction *> Worklist;
2637
2638 // Add uniform instructions demanding lane 0 to the worklist. Instructions
2639 // that require predication must not be considered uniform after
2640 // vectorization, because that would create an erroneous replicating region
2641 // where only a single instance out of VF should be formed.
2642 auto AddToWorklistIfAllowed = [&](Instruction *I) -> void {
2643 if (IsOutOfScope(I)) {
2644 LLVM_DEBUG(dbgs() << "LV: Found not uniform due to scope: "
2645 << *I << "\n");
2646 return;
2647 }
2648 if (isPredicatedInst(I)) {
2649 LLVM_DEBUG(
2650 dbgs() << "LV: Found not uniform due to requiring predication: " << *I
2651 << "\n");
2652 return;
2653 }
2654 LLVM_DEBUG(dbgs() << "LV: Found uniform instruction: " << *I << "\n");
2655 Worklist.insert(X: I);
2656 };
2657
2658 // Start with the conditional branches exiting the loop. If the branch
2659 // condition is an instruction contained in the loop that is only used by the
2660 // branch, it is uniform. Note conditions from uncountable early exits are not
2661 // uniform.
2662 SmallVector<BasicBlock *> Exiting;
2663 TheLoop->getExitingBlocks(ExitingBlocks&: Exiting);
2664 for (BasicBlock *E : Exiting) {
2665 if (Legal->hasUncountableEarlyExit() && TheLoop->getLoopLatch() != E)
2666 continue;
2667 auto *Cmp = dyn_cast<Instruction>(Val: E->getTerminator()->getOperand(i: 0));
2668 if (!Cmp || !TheLoop->contains(Inst: Cmp) || !Cmp->hasOneUse())
2669 continue;
2670
2671 // If we have an exit condition that is actually two conditions (one
2672 // countable and the other uncountable) combined via an or, only add the
2673 // countable comparison as a uniform value.
2674 if (Legal->hasUncountableExitWithSideEffects() &&
2675 TheLoop->getLoopLatch() == E) {
2676 if (Instruction *Countable =
2677 Legal->findCountableComparisonInCombinedCondition(Cond: Cmp)) {
2678 if (Countable->hasOneUse())
2679 AddToWorklistIfAllowed(Countable);
2680 continue;
2681 }
2682 }
2683
2684 // Normal exit comparisons are uniform.
2685 AddToWorklistIfAllowed(Cmp);
2686 }
2687
2688 auto PrevVF = VF.divideCoefficientBy(RHS: 2);
2689 // Return true if all lanes perform the same memory operation, and we can
2690 // thus choose to execute only one.
2691 auto IsUniformMemOpUse = [&](Instruction *I) {
2692 // If the value was already known to not be uniform for the previous
2693 // (smaller VF), it cannot be uniform for the larger VF.
2694 if (PrevVF.isVector()) {
2695 auto Iter = Uniforms.find(Val: PrevVF);
2696 if (Iter != Uniforms.end() && !Iter->second.contains(Ptr: I))
2697 return false;
2698 }
2699 if (!isUniformMemOp(I&: *I, VF))
2700 return false;
2701 if (isa<LoadInst>(Val: I))
2702 // Loading the same address always produces the same result - at least
2703 // assuming aliasing and ordering which have already been checked.
2704 return true;
2705 // Storing the same value on every iteration.
2706 return TheLoop->isLoopInvariant(V: cast<StoreInst>(Val: I)->getValueOperand());
2707 };
2708
2709 auto IsUniformDecision = [&](Instruction *I, ElementCount VF) {
2710 InstWidening WideningDecision = getWideningDecision(I, VF);
2711 assert(WideningDecision != CM_Unknown &&
2712 "Widening decision should be ready at this moment");
2713
2714 if (IsUniformMemOpUse(I))
2715 return true;
2716
2717 return (WideningDecision == CM_Widen ||
2718 WideningDecision == CM_Widen_Reverse ||
2719 WideningDecision == CM_Interleave);
2720 };
2721
2722 // Returns true if Ptr is the pointer operand of a memory access instruction
2723 // I, I is known to not require scalarization, and the pointer is not also
2724 // stored.
2725 auto IsVectorizedMemAccessUse = [&](Instruction *I, Value *Ptr) -> bool {
2726 if (isa<StoreInst>(Val: I) && I->getOperand(i: 0) == Ptr)
2727 return false;
2728 return getLoadStorePointerOperand(V: I) == Ptr &&
2729 (IsUniformDecision(I, VF) || Legal->isInvariant(V: Ptr));
2730 };
2731
2732 // Holds a list of values which are known to have at least one uniform use.
2733 // Note that there may be other uses which aren't uniform. A "uniform use"
2734 // here is something which only demands lane 0 of the unrolled iterations;
2735 // it does not imply that all lanes produce the same value (e.g. this is not
2736 // the usual meaning of uniform)
2737 SetVector<Value *> HasUniformUse;
2738
2739 // Scan the loop for instructions which are either a) known to have only
2740 // lane 0 demanded or b) are uses which demand only lane 0 of their operand.
2741 for (auto *BB : TheLoop->blocks())
2742 for (auto &I : *BB) {
2743 if (IntrinsicInst *II = dyn_cast<IntrinsicInst>(Val: &I)) {
2744 switch (II->getIntrinsicID()) {
2745 case Intrinsic::sideeffect:
2746 case Intrinsic::experimental_noalias_scope_decl:
2747 case Intrinsic::assume:
2748 case Intrinsic::lifetime_start:
2749 case Intrinsic::lifetime_end:
2750 if (TheLoop->hasLoopInvariantOperands(I: &I))
2751 AddToWorklistIfAllowed(&I);
2752 break;
2753 default:
2754 break;
2755 }
2756 }
2757
2758 if (auto *EVI = dyn_cast<ExtractValueInst>(Val: &I)) {
2759 if (IsOutOfScope(EVI->getAggregateOperand())) {
2760 AddToWorklistIfAllowed(EVI);
2761 continue;
2762 }
2763 // Only ExtractValue instructions where the aggregate value comes from a
2764 // call are allowed to be non-uniform.
2765 assert(isa<CallInst>(EVI->getAggregateOperand()) &&
2766 "Expected aggregate value to be call return value");
2767 }
2768
2769 // If there's no pointer operand, there's nothing to do.
2770 auto *Ptr = getLoadStorePointerOperand(V: &I);
2771 if (!Ptr)
2772 continue;
2773
2774 // If the pointer can be proven to be uniform, always add it to the
2775 // worklist.
2776 if (isa<Instruction>(Val: Ptr) && isUniform(V: Ptr, VF))
2777 AddToWorklistIfAllowed(cast<Instruction>(Val: Ptr));
2778
2779 if (IsUniformMemOpUse(&I))
2780 AddToWorklistIfAllowed(&I);
2781
2782 if (IsVectorizedMemAccessUse(&I, Ptr))
2783 HasUniformUse.insert(X: Ptr);
2784 }
2785
2786 // Add to the worklist any operands which have *only* uniform (e.g. lane 0
2787 // demanding) users. Since loops are assumed to be in LCSSA form, this
2788 // disallows uses outside the loop as well.
2789 for (auto *V : HasUniformUse) {
2790 if (IsOutOfScope(V))
2791 continue;
2792 auto *I = cast<Instruction>(Val: V);
2793 bool UsersAreMemAccesses = all_of(Range: I->users(), P: [&](User *U) -> bool {
2794 auto *UI = cast<Instruction>(Val: U);
2795 return TheLoop->contains(Inst: UI) && IsVectorizedMemAccessUse(UI, V);
2796 });
2797 if (UsersAreMemAccesses)
2798 AddToWorklistIfAllowed(I);
2799 }
2800
2801 // Expand Worklist in topological order: whenever a new instruction
2802 // is added , its users should be already inside Worklist. It ensures
2803 // a uniform instruction will only be used by uniform instructions.
2804 unsigned Idx = 0;
2805 while (Idx != Worklist.size()) {
2806 Instruction *I = Worklist[Idx++];
2807
2808 for (auto *OV : I->operand_values()) {
2809 // isOutOfScope operands cannot be uniform instructions.
2810 if (IsOutOfScope(OV))
2811 continue;
2812 // First order recurrence Phi's should typically be considered
2813 // non-uniform.
2814 auto *OP = dyn_cast<PHINode>(Val: OV);
2815 if (OP && Legal->isFixedOrderRecurrence(Phi: OP))
2816 continue;
2817 // If all the users of the operand are uniform, then add the
2818 // operand into the uniform worklist.
2819 auto *OI = cast<Instruction>(Val: OV);
2820 if (llvm::all_of(Range: OI->users(), P: [&](User *U) -> bool {
2821 auto *J = cast<Instruction>(Val: U);
2822 return Worklist.count(key: J) || IsVectorizedMemAccessUse(J, OI);
2823 }))
2824 AddToWorklistIfAllowed(OI);
2825 }
2826 }
2827
2828 // For an instruction to be added into Worklist above, all its users inside
2829 // the loop should also be in Worklist. However, this condition cannot be
2830 // true for phi nodes that form a cyclic dependence. We must process phi
2831 // nodes separately. An induction variable will remain uniform if all users
2832 // of the induction variable and induction variable update remain uniform.
2833 // The code below handles both pointer and non-pointer induction variables.
2834 BasicBlock *Latch = TheLoop->getLoopLatch();
2835 for (PHINode *Ind : Legal->getInductionVars().keys()) {
2836 auto *IndUpdate = cast<Instruction>(Val: Ind->getIncomingValueForBlock(BB: Latch));
2837
2838 // Determine if all users of the induction variable are uniform after
2839 // vectorization.
2840 bool UniformInd = all_of(Range: Ind->users(), P: [&](User *U) -> bool {
2841 auto *I = cast<Instruction>(Val: U);
2842 return I == IndUpdate || !TheLoop->contains(Inst: I) || Worklist.count(key: I) ||
2843 IsVectorizedMemAccessUse(I, Ind);
2844 });
2845 if (!UniformInd)
2846 continue;
2847
2848 // Determine if all users of the induction variable update instruction are
2849 // uniform after vectorization.
2850 bool UniformIndUpdate = all_of(Range: IndUpdate->users(), P: [&](User *U) -> bool {
2851 auto *I = cast<Instruction>(Val: U);
2852 return I == Ind || Worklist.count(key: I) ||
2853 IsVectorizedMemAccessUse(I, IndUpdate);
2854 });
2855 if (!UniformIndUpdate)
2856 continue;
2857
2858 // The induction variable and its update instruction will remain uniform.
2859 AddToWorklistIfAllowed(Ind);
2860 AddToWorklistIfAllowed(IndUpdate);
2861 }
2862
2863 Uniforms[VF].insert_range(R&: Worklist);
2864}
2865
2866FixedScalableVFPair
2867LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
2868 // Make sure once we return PartialAliasMaskingStatus is not "NotDecided".
2869 scope_exit EnsureAliasMaskingStatusIsDecidedOnReturn([this] {
2870 if (PartialAliasMaskingStatus == AliasMaskingStatus::NotDecided)
2871 PartialAliasMaskingStatus = AliasMaskingStatus::Disabled;
2872 });
2873
2874 // For outer loops, use simple type-based heuristic VF. No cost model or
2875 // memory dependence analysis is available.
2876 if (!TheLoop->isInnermost()) {
2877 return Config.computeVPlanOuterloopVF(UserVF);
2878 }
2879
2880 if (Legal->getRuntimePointerChecking()->Need && TTI.hasBranchDivergence()) {
2881 // TODO: It may be useful to do since it's still likely to be dynamically
2882 // uniform if the target can skip.
2883 reportVectorizationFailure(
2884 DebugMsg: "Not inserting runtime ptr check for divergent target",
2885 OREMsg: "runtime pointer checks needed. Not enabled for divergent target",
2886 ORETag: "CantVersionLoopWithDivergentTarget", ORE, TheLoop);
2887 return FixedScalableVFPair::getNone();
2888 }
2889
2890 ScalarEvolution *SE = PSE.getSE();
2891 ElementCount TC = getSmallConstantTripCount(SE, L: TheLoop);
2892 unsigned MaxTC = PSE.getSmallConstantMaxTripCount();
2893 if (!MaxTC && EpilogueLoweringStatus == CM_EpilogueAllowed)
2894 MaxTC = getMaxTCFromNonZeroRange(PSE, L: TheLoop);
2895 LLVM_DEBUG(dbgs() << "LV: Found trip count: " << TC << '\n');
2896 if (TC != ElementCount::getFixed(MinVal: MaxTC))
2897 LLVM_DEBUG(dbgs() << "LV: Found maximum trip count: " << MaxTC << '\n');
2898 if (TC.isScalar()) {
2899 reportVectorizationFailure(
2900 DebugMsg: "Single iteration (non) loop",
2901 OREMsg: "loop trip count is one, irrelevant for vectorization",
2902 ORETag: "SingleIterationLoop", ORE, TheLoop);
2903 return FixedScalableVFPair::getNone();
2904 }
2905
2906 // If BTC matches the widest induction type and is -1 then the trip count
2907 // computation will wrap to 0 and the vector trip count will be 0. Do not try
2908 // to vectorize.
2909 const SCEV *BTC = SE->getBackedgeTakenCount(L: TheLoop);
2910 if (!isa<SCEVCouldNotCompute>(Val: BTC) &&
2911 BTC->getType()->getScalarSizeInBits() >=
2912 Legal->getWidestInductionType()->getScalarSizeInBits() &&
2913 SE->isKnownPredicate(Pred: CmpInst::ICMP_EQ, LHS: BTC,
2914 RHS: SE->getMinusOne(Ty: BTC->getType()))) {
2915 reportVectorizationFailure(
2916 DebugMsg: "Trip count computation wrapped",
2917 OREMsg: "backedge-taken count is -1, loop trip count wrapped to 0",
2918 ORETag: "TripCountWrapped", ORE, TheLoop);
2919 return FixedScalableVFPair::getNone();
2920 }
2921
2922 assert(WideningDecisions.empty() && Uniforms.empty() && Scalars.empty() &&
2923 "No cost-modeling decisions should have been taken at this point");
2924
2925 switch (EpilogueLoweringStatus) {
2926 case CM_EpilogueAllowed:
2927 return Config.computeFeasibleMaxVF(MaxTripCount: MaxTC, UserVF, UserIC, FoldTailByMasking: false,
2928 RequiresScalarEpilogue: requiresScalarEpilogue(IsVectorizing: true));
2929 case CM_EpilogueNotAllowedFoldTail:
2930 [[fallthrough]];
2931 case CM_EpilogueNotNeededFoldTail:
2932 LLVM_DEBUG(dbgs() << "LV: tail-folding hint/switch found.\n"
2933 << "LV: Not allowing epilogue, creating tail-folded "
2934 << "vector loop.\n");
2935 break;
2936 case CM_EpilogueNotAllowedLowTripLoop:
2937 // fallthrough as a special case of OptForSize
2938 case CM_EpilogueNotAllowedOptSize:
2939 if (EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize)
2940 LLVM_DEBUG(dbgs() << "LV: Not allowing epilogue due to -Os/-Oz.\n");
2941 else
2942 LLVM_DEBUG(dbgs() << "LV: Not allowing epilogue due to low trip "
2943 << "count.\n");
2944
2945 // Bail if runtime checks are required, which are not good when optimising
2946 // for size.
2947 if (Config.runtimeChecksRequired())
2948 return FixedScalableVFPair::getNone();
2949
2950 break;
2951 }
2952
2953 // Now try the tail folding
2954
2955 // Invalidate interleave groups that require an epilogue if we can't mask
2956 // the interleave-group.
2957 if (!useMaskedInterleavedAccesses(TTI)) {
2958 // Note: There is no need to invalidate any cost modeling decisions here, as
2959 // none were taken so far (see assertion above).
2960 InterleaveInfo.invalidateGroupsRequiringScalarEpilogue();
2961 }
2962
2963 FixedScalableVFPair MaxFactors = Config.computeFeasibleMaxVF(
2964 MaxTripCount: MaxTC, UserVF, UserIC, FoldTailByMasking: true, RequiresScalarEpilogue: requiresScalarEpilogue(IsVectorizing: true));
2965
2966 // Avoid tail folding if the trip count is known to be a multiple of any VF
2967 // we choose.
2968 std::optional<uint64_t> MaxPowerOf2RuntimeVF =
2969 MaxFactors.FixedVF.getFixedValue();
2970 if (MaxFactors.ScalableVF) {
2971 if (std::optional<uint64_t> MaxRuntimeScalableVF =
2972 getMaxRuntimeElementCount(EC: MaxFactors.ScalableVF, F: *TheFunction))
2973 MaxPowerOf2RuntimeVF =
2974 std::max(a: *MaxPowerOf2RuntimeVF, b: *MaxRuntimeScalableVF);
2975 else
2976 MaxPowerOf2RuntimeVF = std::nullopt; // Stick with tail-folding for now.
2977 }
2978
2979 auto NoScalarEpilogueNeeded = [this, &UserIC](uint64_t MaxRuntimeVF) {
2980 // Return false if the loop is neither a single-latch-exit loop nor an
2981 // early-exit loop as tail-folding is not supported in that case.
2982 if (TheLoop->getExitingBlock() != TheLoop->getLoopLatch() &&
2983 !Legal->hasUncountableEarlyExit())
2984 return false;
2985 uint64_t MaxVFtimesIC = MaxRuntimeVF * std::max<uint64_t>(a: UserIC, b: 1);
2986 ScalarEvolution *SE = PSE.getSE();
2987 // Calling getSymbolicMaxBackedgeTakenCount enables support for loops
2988 // with uncountable exits. For countable loops, the symbolic maximum must
2989 // remain identical to the known back-edge taken count.
2990 const SCEV *BackedgeTakenCount = PSE.getSymbolicMaxBackedgeTakenCount();
2991 assert((Legal->hasUncountableEarlyExit() ||
2992 BackedgeTakenCount == PSE.getBackedgeTakenCount()) &&
2993 "Invalid loop count");
2994 const SCEV *ExitCount = SE->getAddExpr(
2995 LHS: BackedgeTakenCount, RHS: SE->getOne(Ty: BackedgeTakenCount->getType()));
2996 const SCEV *Rem = SE->getURemExpr(
2997 LHS: SE->applyLoopGuards(Expr: ExitCount, L: TheLoop),
2998 RHS: SE->getConstant(Ty: BackedgeTakenCount->getType(), V: MaxVFtimesIC));
2999 return Rem->isZero();
3000 };
3001
3002 if (MaxPowerOf2RuntimeVF > 0u) {
3003 assert((UserVF.isNonZero() || isPowerOf2_64(*MaxPowerOf2RuntimeVF)) &&
3004 "MaxFixedVF must be a power of 2");
3005 if (NoScalarEpilogueNeeded(*MaxPowerOf2RuntimeVF)) {
3006 // Accept MaxFixedVF if we do not have a tail.
3007 LLVM_DEBUG(dbgs() << "LV: No tail will remain for any chosen VF.\n");
3008 return MaxFactors;
3009 }
3010 }
3011
3012 auto ExpectedTC = getSmallBestKnownTC(PSE, L: TheLoop);
3013 if (ExpectedTC && ExpectedTC->isFixed() &&
3014 ExpectedTC->getFixedValue() <=
3015 TTI.getMinTripCountTailFoldingThreshold()) {
3016 // If we have a low-trip-count, and the fixed-width VF is known to divide
3017 // the trip count the fixed-width factor in preference to allow the
3018 // generation of a non-predicated loop.
3019 if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
3020 NoScalarEpilogueNeeded(MaxFactors.FixedVF.getFixedValue())) {
3021 LLVM_DEBUG(dbgs() << "LV: Picking a fixed-width so that no tail will "
3022 "remain for any chosen VF.\n");
3023 MaxFactors.ScalableVF = ElementCount::getScalable(MinVal: 0);
3024 return MaxFactors;
3025 }
3026
3027 // Allow cases where the ExactTC == (VF * IC) + 1.
3028 //
3029 // This produces 1 vector iteration, and 1 scalar iteration with no
3030 // remainder. Later passes will eliminate the loop and leave straight-line
3031 // code as the both iteration counts are statically known.
3032 //
3033 // If a function is marked as minsize/optsize or OptForSize is set, do not
3034 // allow this form of transformation as this will increase CodeSize.
3035 //
3036 // For loops with small bodies, the cost model is not currently reliable
3037 // enough to accurately determine if vectorization is beneficial.
3038 unsigned EffectiveIC = UserIC > 0 ? UserIC : 1;
3039 unsigned MaxVFForTC = llvm::bit_floor(Value: TC.getFixedValue());
3040 if (TC.getFixedValue() - MaxVFForTC == 1 && MaxVFForTC / EffectiveIC > 1 &&
3041 MaxVFForTC <= (MaxFactors.FixedVF.getFixedValue() * EffectiveIC) &&
3042 !Config.OptForSize) {
3043 unsigned NumOfInstructions = llvm::sum_of(
3044 Range: llvm::map_range(C: TheLoop->blocks(),
3045 F: [](BasicBlock *BB) { return BB->size(); }),
3046 Init: unsigned(0));
3047 if (NumOfInstructions > LowTripCountLoopBodySizeLimit) {
3048 unsigned VF = MaxVFForTC / EffectiveIC;
3049 LLVM_DEBUG(dbgs() << "LV: Picking MaxVF=" << VF
3050 << " with 1 scalar iteration remaining.\n");
3051 MaxFactors.FixedVF = ElementCount::getFixed(MinVal: VF);
3052 MaxFactors.ScalableVF = ElementCount::getScalable(MinVal: 0);
3053 return MaxFactors;
3054 }
3055 }
3056
3057 reportVectorizationFailure(
3058 DebugMsg: "The trip count is below the minial threshold value.",
3059 OREMsg: "loop trip count is too low, avoiding vectorization", ORETag: "LowTripCount",
3060 ORE, TheLoop);
3061 return FixedScalableVFPair::getNone();
3062 }
3063
3064 // If we don't know the precise trip count, or if the trip count that we
3065 // found modulo the vectorization factor is not zero, try to fold the tail
3066 // by masking.
3067 // FIXME: look for a smaller MaxVF that does divide TC rather than masking.
3068 bool ContainsScalableVF = MaxFactors.ScalableVF.isNonZero();
3069 setTailFoldingStyle(IsScalableVF: ContainsScalableVF, UserIC);
3070 if (foldTailByMasking()) {
3071 if (foldTailWithEVL()) {
3072 LLVM_DEBUG(
3073 dbgs()
3074 << "LV: tail is folded with EVL, forcing unroll factor to be 1. Will "
3075 "try to generate VP Intrinsics with scalable vector "
3076 "factors only.\n");
3077 // Tail folded loop using VP intrinsics restricts the VF to be scalable
3078 // for now.
3079 // TODO: extend it for fixed vectors, if required.
3080 assert(ContainsScalableVF && "Expected scalable vector factor.");
3081
3082 MaxFactors.FixedVF = ElementCount::getFixed(MinVal: 1);
3083 } else {
3084 tryToEnablePartialAliasMasking();
3085 }
3086 return MaxFactors;
3087 }
3088
3089 // If there was a tail-folding hint/switch, but we can't fold the tail by
3090 // masking, fallback to a vectorization with an epilogue.
3091 if (EpilogueLoweringStatus == CM_EpilogueNotNeededFoldTail) {
3092 LLVM_DEBUG(dbgs() << "LV: Cannot fold tail by masking: vectorize with an "
3093 "epilogue instead.\n");
3094 EpilogueLoweringStatus = CM_EpilogueAllowed;
3095 return MaxFactors;
3096 }
3097
3098 if (EpilogueLoweringStatus == CM_EpilogueNotAllowedFoldTail) {
3099 LLVM_DEBUG(dbgs() << "LV: Can't fold tail by masking: don't vectorize\n");
3100 return FixedScalableVFPair::getNone();
3101 }
3102
3103 if (TC.isZero()) {
3104 reportVectorizationFailure(
3105 DebugMsg: "unable to calculate the loop count due to complex control flow",
3106 ORETag: "UnknownLoopCountComplexCFG", ORE, TheLoop);
3107 return FixedScalableVFPair::getNone();
3108 }
3109
3110 reportVectorizationFailure(
3111 DebugMsg: "Cannot optimize for size and vectorize at the same time.",
3112 OREMsg: "cannot optimize for size and vectorize at the same time. "
3113 "Enable vectorization of this loop with '#pragma clang loop "
3114 "vectorize(enable)' when compiling with -Os/-Oz",
3115 ORETag: "NoTailLoopWithOptForSize", ORE, TheLoop);
3116 return FixedScalableVFPair::getNone();
3117}
3118
3119void LoopVectorizationPlanner::emitInvalidCostRemarks(
3120 OptimizationRemarkEmitter *ORE) {
3121 using RecipeVFPair = std::pair<VPRecipeBase *, ElementCount>;
3122 SmallVector<RecipeVFPair> InvalidCosts;
3123 for (const auto &Plan : VPlans) {
3124 for (ElementCount VF : Plan->vectorFactors()) {
3125 // The VPlan-based cost model is designed for computing vector cost.
3126 // Querying VPlan-based cost model with a scarlar VF will cause some
3127 // errors because we expect the VF is vector for most of the widen
3128 // recipes.
3129 if (VF.isScalar())
3130 continue;
3131
3132 VPCostContext CostCtx(*TLI, *Plan, *CM, Config,
3133 /*ReusePrintingSlotTracker=*/true);
3134 precomputeCosts(Plan&: *Plan, VF, CostCtx);
3135 auto Iter = vp_depth_first_deep(G: Plan->getVectorLoopRegion()->getEntry());
3136 for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(Range&: Iter)) {
3137 for (auto &R : *VPBB) {
3138 if (!R.cost(VF, Ctx&: CostCtx).isValid())
3139 InvalidCosts.emplace_back(Args: &R, Args&: VF);
3140 }
3141 }
3142 }
3143 }
3144 if (InvalidCosts.empty())
3145 return;
3146
3147 // Emit a report of VFs with invalid costs in the loop.
3148
3149 // Group the remarks per recipe, keeping the recipe order from InvalidCosts.
3150 DenseMap<VPRecipeBase *, unsigned> Numbering;
3151 unsigned I = 0;
3152 for (auto &Pair : InvalidCosts)
3153 if (Numbering.try_emplace(Key: Pair.first, Args&: I).second)
3154 ++I;
3155
3156 // Sort the list, first on recipe(number) then on VF.
3157 sort(C&: InvalidCosts, Comp: [&Numbering](RecipeVFPair &A, RecipeVFPair &B) {
3158 unsigned NA = Numbering[A.first];
3159 unsigned NB = Numbering[B.first];
3160 if (NA != NB)
3161 return NA < NB;
3162 return ElementCount::isKnownLT(LHS: A.second, RHS: B.second);
3163 });
3164
3165 // For a list of ordered recipe-VF pairs:
3166 // [(load, VF1), (load, VF2), (store, VF1)]
3167 // group the recipes together to emit separate remarks for:
3168 // load (VF1, VF2)
3169 // store (VF1)
3170 auto Tail = ArrayRef<RecipeVFPair>(InvalidCosts);
3171 auto Subset = ArrayRef<RecipeVFPair>();
3172 do {
3173 if (Subset.empty())
3174 Subset = Tail.take_front(N: 1);
3175
3176 VPRecipeBase *R = Subset.front().first;
3177
3178 unsigned Opcode =
3179 TypeSwitch<const VPRecipeBase *, unsigned>(R)
3180 .Case(caseFn: [](const VPHeaderPHIRecipe *R) { return Instruction::PHI; })
3181 .Case(
3182 caseFn: [](const VPWidenStoreRecipe *R) { return Instruction::Store; })
3183 .Case(caseFn: [](const VPWidenLoadRecipe *R) { return Instruction::Load; })
3184 .Case<VPWidenCallRecipe, VPWidenIntrinsicRecipe>(
3185 caseFn: [](const auto *R) { return Instruction::Call; })
3186 .Case<VPInstruction, VPWidenRecipe, VPReplicateRecipe,
3187 VPWidenCastRecipe>(
3188 caseFn: [](const auto *R) { return R->getOpcode(); })
3189 .Case(caseFn: [](const VPInterleaveRecipe *R) {
3190 return R->getStoredValues().empty() ? Instruction::Load
3191 : Instruction::Store;
3192 })
3193 .Case(caseFn: [](const VPReductionRecipe *R) {
3194 return RecurrenceDescriptor::getOpcode(Kind: R->getRecurrenceKind());
3195 });
3196
3197 // If the next recipe is different, or if there are no other pairs,
3198 // emit a remark for the collated subset. e.g.
3199 // [(load, VF1), (load, VF2))]
3200 // to emit:
3201 // remark: invalid costs for 'load' at VF=(VF1, VF2)
3202 if (Subset == Tail || Tail[Subset.size()].first != R) {
3203 std::string OutString;
3204 raw_string_ostream OS(OutString);
3205 assert(!Subset.empty() && "Unexpected empty range");
3206 OS << "Recipe with invalid costs prevented vectorization at VF=(";
3207 for (const auto &Pair : Subset)
3208 OS << (Pair.second == Subset.front().second ? "" : ", ") << Pair.second;
3209 OS << "):";
3210 if (Opcode == Instruction::Call) {
3211 StringRef Name = "";
3212 if (auto *Int = dyn_cast<VPWidenIntrinsicRecipe>(Val: R)) {
3213 Name = Int->getIntrinsicName();
3214 } else {
3215 auto *WidenCall = dyn_cast<VPWidenCallRecipe>(Val: R);
3216 Function *CalledFn =
3217 WidenCall
3218 ? WidenCall->getCalledScalarFunction()
3219 : cast<Function>(Val: R->getLastOperand()->getLiveInIRValue());
3220 Name = CalledFn->getName();
3221 }
3222 OS << " call to " << Name;
3223 } else
3224 OS << " " << Instruction::getOpcodeName(Opcode);
3225 reportVectorizationInfo(Msg: OutString, ORETag: "InvalidCost", ORE, TheLoop: OrigLoop, I: nullptr,
3226 DL: R->getDebugLoc());
3227 Tail = Tail.drop_front(N: Subset.size());
3228 Subset = {};
3229 } else
3230 // Grow the subset by one element
3231 Subset = Tail.take_front(N: Subset.size() + 1);
3232 } while (!Tail.empty());
3233}
3234
3235/// Check if any recipe of \p Plan will generate a vector value, which will be
3236/// assigned a vector register.
3237static bool willGenerateVectors(VPlan &Plan, ElementCount VF,
3238 const TargetTransformInfo &TTI) {
3239 assert(VF.isVector() && "Checking a scalar VF?");
3240 DenseSet<VPRecipeBase *> EphemeralRecipes;
3241 collectEphemeralRecipesForVPlan(Plan, EphRecipes&: EphemeralRecipes);
3242 // Set of already visited types.
3243 DenseSet<Type *> Visited;
3244 for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
3245 Range: vp_depth_first_shallow(G: Plan.getVectorLoopRegion()->getEntry()))) {
3246 for (VPRecipeBase &R : *VPBB) {
3247 if (EphemeralRecipes.contains(V: &R))
3248 continue;
3249 // Continue early if the recipe is considered to not produce a vector
3250 // result. Note that this includes VPInstruction where some opcodes may
3251 // produce a vector, to preserve existing behavior as VPInstructions model
3252 // aspects not directly mapped to existing IR instructions.
3253 switch (R.getVPRecipeID()) {
3254 case VPRecipeBase::VPDerivedIVSC:
3255 case VPRecipeBase::VPScalarIVStepsSC:
3256 case VPRecipeBase::VPReplicateSC:
3257 case VPRecipeBase::VPInstructionSC:
3258 case VPRecipeBase::VPCurrentIterationPHISC:
3259 case VPRecipeBase::VPVectorPointerSC:
3260 case VPRecipeBase::VPVectorEndPointerSC:
3261 case VPRecipeBase::VPExpandSCEVSC:
3262 case VPRecipeBase::VPPredInstPHISC:
3263 case VPRecipeBase::VPBranchOnMaskSC:
3264 continue;
3265 case VPRecipeBase::VPReductionSC:
3266 case VPRecipeBase::VPActiveLaneMaskPHISC:
3267 case VPRecipeBase::VPWidenCallSC:
3268 case VPRecipeBase::VPWidenCanonicalIVSC:
3269 case VPRecipeBase::VPWidenCastSC:
3270 case VPRecipeBase::VPWidenGEPSC:
3271 case VPRecipeBase::VPWidenIntrinsicSC:
3272 case VPRecipeBase::VPWidenMemIntrinsicSC:
3273 case VPRecipeBase::VPWidenSC:
3274 case VPRecipeBase::VPBlendSC:
3275 case VPRecipeBase::VPFirstOrderRecurrencePHISC:
3276 case VPRecipeBase::VPHistogramSC:
3277 case VPRecipeBase::VPWidenPHISC:
3278 case VPRecipeBase::VPWidenIntOrFpInductionSC:
3279 case VPRecipeBase::VPWidenPointerInductionSC:
3280 case VPRecipeBase::VPReductionPHISC:
3281 case VPRecipeBase::VPInterleaveEVLSC:
3282 case VPRecipeBase::VPInterleaveSC:
3283 case VPRecipeBase::VPWidenLoadEVLSC:
3284 case VPRecipeBase::VPWidenLoadSC:
3285 case VPRecipeBase::VPWidenStoreEVLSC:
3286 case VPRecipeBase::VPWidenStoreSC:
3287 break;
3288 default:
3289 llvm_unreachable("unhandled recipe");
3290 }
3291
3292 auto WillGenerateTargetVectors = [&TTI, VF](Type *VectorTy) {
3293 unsigned NumLegalParts = TTI.getNumberOfParts(Tp: VectorTy);
3294 if (!NumLegalParts)
3295 return false;
3296 if (VF.isScalable()) {
3297 // <vscale x 1 x iN> is assumed to be profitable over iN because
3298 // scalable registers are a distinct register class from scalar
3299 // ones. If we ever find a target which wants to lower scalable
3300 // vectors back to scalars, we'll need to update this code to
3301 // explicitly ask TTI about the register class uses for each part.
3302 return NumLegalParts <= VF.getKnownMinValue();
3303 }
3304 // Two or more elements that share a register - are vectorized.
3305 return NumLegalParts < VF.getFixedValue();
3306 };
3307
3308 // If no def nor is a store, e.g., branches, continue - no value to check.
3309 if (R.getNumDefinedValues() == 0 &&
3310 !isa<VPWidenStoreRecipe, VPWidenStoreEVLRecipe, VPInterleaveBase>(Val: &R))
3311 continue;
3312 // For multi-def recipes, currently only interleaved loads, suffice to
3313 // check first def only.
3314 // For stores check their stored value; for interleaved stores suffice
3315 // the check first stored value only. In all cases this is the second
3316 // operand.
3317 VPValue *ToCheck =
3318 R.getNumDefinedValues() >= 1 ? R.getVPValue(I: 0) : R.getOperand(N: 1);
3319 Type *ScalarTy = ToCheck->getScalarType();
3320 if (!Visited.insert(V: {ScalarTy}).second)
3321 continue;
3322 Type *WideTy = toVectorizedTy(Ty: ScalarTy, EC: VF);
3323 if (any_of(Range: getContainedTypes(Ty: WideTy), P: WillGenerateTargetVectors))
3324 return true;
3325 }
3326 }
3327
3328 return false;
3329}
3330
3331static bool hasReplicatorRegion(VPlan &Plan) {
3332 return any_of(Range: VPBlockUtils::blocksOnly<VPRegionBlock>(Range: vp_depth_first_shallow(
3333 G: Plan.getVectorLoopRegion()->getEntry())),
3334 P: [](auto *VPRB) { return VPRB->isReplicator(); });
3335}
3336
3337/// Returns true if the VPlan contains a VPReductionPHIRecipe with
3338/// FindLast recurrence kind.
3339static bool hasFindLastReductionPhi(VPlan &Plan) {
3340 return any_of(Range: make_isa_range<VPReductionPHIRecipe>(
3341 Range: Plan.getVectorLoopRegion()->getEntryBasicBlock()->phis()),
3342 P: [](VPReductionPHIRecipe &RedPhi) {
3343 return RecurrenceDescriptor::isFindLastRecurrenceKind(
3344 Kind: RedPhi.getRecurrenceKind());
3345 });
3346}
3347
3348/// Determine how to lower the epilogue for the vector epilogue loop.
3349/// Check if there are any conflicts that prevent tail-folding the epilogue.
3350/// \return CM_EpilogueNotNeededFoldTail if epilogue tail-folding is possible,
3351/// otherwise CM_EpilogueAllowed.
3352static EpilogueLowering getEpilogueTailLowering(
3353 const LoopVectorizationCostModel &MainCM, const Loop *L,
3354 OptimizationRemarkEmitter *ORE, LoopVectorizationLegality &LVL,
3355 const LoopVectorizeHints &Hints, TargetTransformInfo *TTI) {
3356 // Epilogue TF is only enabled when explicitly requested via command line.
3357 if (!EpilogueTailFoldingPolicy.getNumOccurrences() ||
3358 EpilogueTailFoldingPolicy != TailFoldingPolicyTy::PreferFoldTail)
3359 return CM_EpilogueAllowed;
3360
3361 if (!EnableEpilogueVectorization) {
3362 reportVectorizationInfo(
3363 Msg: "Options conflict, epilogue vectorization is disallowed while "
3364 "epilogue tail-folding allowed!",
3365 ORETag: "UnsupportedEpilogueTailFoldingPolicy", ORE, TheLoop: L);
3366 return CM_EpilogueAllowed;
3367 }
3368
3369 if (!Hints.getWidth() || !hasForcedEpilogueVF()) {
3370 reportVectorizationInfo(Msg: "For now, epilogue tail-folding can't be "
3371 "applied without forced main/epilogue loop VF",
3372 ORETag: "UnsupportedEpilogueTailFoldingPolicy", ORE, TheLoop: L);
3373 return CM_EpilogueAllowed;
3374 }
3375
3376 if (ElementCount::isKnownLE(LHS: Hints.getWidth(), RHS: EpilogueVectorizationForceVF)) {
3377 reportVectorizationInfo(Msg: "For now, epilogue tail-folding can't be applied "
3378 "when VF of the main loop <= VF of the epilogue",
3379 ORETag: "UnsupportedEpilogueTailFoldingPolicy", ORE, TheLoop: L);
3380 return CM_EpilogueAllowed;
3381 }
3382
3383 if (!L->isInnermost()) {
3384 reportVectorizationInfo(
3385 Msg: "Epilogue tail-folding is not supported for outer loop",
3386 ORETag: "InvalidTailFoldedEpilogue", ORE, TheLoop: L);
3387 return CM_EpilogueAllowed;
3388 }
3389
3390 // If scalar epilogue is explicitly required, we can't apply TF.
3391 if (MainCM.requiresScalarEpilogue(/*IsVectorizing*/ true)) {
3392 reportVectorizationInfo(
3393 Msg: "Epilogue tail-folding can't be applied because scalar epilogue is "
3394 "required. Fall back to a normal epilogue",
3395 ORETag: "InvalidTailFoldedEpilogue", ORE, TheLoop: L);
3396 return CM_EpilogueAllowed;
3397 }
3398
3399 // If having epilogue is NOT allowed, then no epilogue to apply TF for.
3400 if (!MainCM.isEpilogueAllowed()) {
3401 reportVectorizationInfo(Msg: "Not applying tail-folding to the epilogue, since "
3402 "no epilogue is allowed.",
3403 ORETag: "InvalidTailFoldedEpilogue", ORE, TheLoop: L);
3404 return CM_EpilogueAllowed;
3405 }
3406
3407 if (L->getExitingBlock() != L->getLoopLatch() ||
3408 LVL.hasUncountableEarlyExit()) {
3409 reportVectorizationInfo(
3410 Msg: "Epilogue tail-folding is not supported yet for early-exit loops",
3411 ORETag: "InvalidTailFoldedEpilogue", ORE, TheLoop: L);
3412 return CM_EpilogueAllowed;
3413 }
3414
3415 // The epilogue reuses the main loop's interleave groups, so it can't be
3416 // tail-folded if the target can't mask interleaved accesses.
3417 // TODO: Add support once the epilogue has its own IAI, separate from the main
3418 // loop's.
3419 if (MainCM.InterleaveInfo.hasGroups() &&
3420 !useMaskedInterleavedAccesses(TTI: *TTI)) {
3421 reportVectorizationInfo(
3422 Msg: "Epilogue tail-folding is not supported with interleaved accesses "
3423 "when masking them isn't supported",
3424 ORETag: "InvalidTailFoldedEpilogue", ORE, TheLoop: L);
3425 return CM_EpilogueAllowed;
3426 }
3427
3428 if (ForcePartialAliasingVectorization) {
3429 reportVectorizationInfo(
3430 Msg: "Epilogue tail-folding is not supported with alias masking",
3431 ORETag: "InvalidTailFoldedEpilogue", ORE, TheLoop: L);
3432 return CM_EpilogueAllowed;
3433 }
3434
3435 if (!LVL.getReductionVars().empty()) {
3436 reportVectorizationInfo(
3437 Msg: "Epilogue tail-folding is not supported with reductions",
3438 ORETag: "InvalidTailFoldedEpilogue", ORE, TheLoop: L);
3439 return CM_EpilogueAllowed;
3440 }
3441
3442 if (!LVL.getFixedOrderRecurrences().empty()) {
3443 reportVectorizationInfo(
3444 Msg: "Epilogue tail-folding is not supported with fixed-order recurrence",
3445 ORETag: "InvalidTailFoldedEpilogue", ORE, TheLoop: L);
3446 return CM_EpilogueAllowed;
3447 }
3448
3449 // TODO: This is conservative: it rejects any target that prefers EVL, even
3450 // for fixed-width epilogue VFs where EVL won't be chosen. Move this check to
3451 // where the epilogue's TF style is known once epilogue TF is supported.
3452 TailFoldingStyle TFStyle = TTI->getPreferredTailFoldingStyle();
3453 if (ForceTailFoldingStyle.getNumOccurrences())
3454 TFStyle = ForceTailFoldingStyle.getValue();
3455 // TODO: Remove once EVL recipes support cloning.
3456 if (TFStyle == TailFoldingStyle::DataWithEVL) {
3457 reportVectorizationInfo(Msg: "Epilogue tail-folding is not supported yet with "
3458 "EVL-based tail-folding",
3459 ORETag: "UnsupportedEpilogueTailFoldingPolicy", ORE, TheLoop: L);
3460 return CM_EpilogueAllowed;
3461 }
3462
3463 // We can apply tail-folding on the vectorized epilogue loop.
3464 return CM_EpilogueNotNeededFoldTail;
3465}
3466
3467bool VFSelectionContext::isEpilogueVectorizationProfitable(
3468 const ElementCount VF, const unsigned IC) const {
3469 // FIXME: We need a much better cost-model to take different parameters such
3470 // as register pressure, code size increase and cost of extra branches into
3471 // account. For now we apply a very crude heuristic and only consider loops
3472 // with vectorization factors larger than a certain value.
3473
3474 // Allow the target to opt out.
3475 if (!TTI.preferEpilogueVectorization(Iters: VF * IC))
3476 return false;
3477
3478 unsigned MinVFThreshold = EpilogueVectorizationMinVF.getNumOccurrences() > 0
3479 ? EpilogueVectorizationMinVF
3480 : TTI.getEpilogueVectorizationMinVF();
3481 return estimateElementCount(VF: VF * IC, VScale: getVScaleForTuning()) >= MinVFThreshold;
3482}
3483
3484std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
3485 VPlan &MainPlan, ElementCount MainLoopVF, unsigned IC,
3486 bool ScalarEpilogueAllowed) {
3487 if (!EnableEpilogueVectorization) {
3488 LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization is disabled.\n");
3489 return nullptr;
3490 }
3491
3492 if (!ScalarEpilogueAllowed) {
3493 LLVM_DEBUG(dbgs() << "LEV: Unable to vectorize epilogue because no "
3494 "epilogue is allowed.\n");
3495 return nullptr;
3496 }
3497
3498 if (vputils::findIncomingAliasMask(Plan: MainPlan)) {
3499 LLVM_DEBUG(
3500 dbgs()
3501 << "LEV: Epilogue vectorization not supported with alias masking.\n");
3502 return nullptr;
3503 }
3504
3505 // Not really a cost consideration, but check for unsupported cases here to
3506 // simplify the logic.
3507 if (!isCandidateForEpilogueVectorization(MainPlan)) {
3508 LLVM_DEBUG(dbgs() << "LEV: Unable to vectorize epilogue because the loop "
3509 "is not a supported candidate.\n");
3510 return nullptr;
3511 }
3512
3513 if (hasForcedEpilogueVF()) {
3514 if (estimateElementCount(VF: EpilogueVectorizationForceVF,
3515 VScale: Config.getVScaleForTuning()) >=
3516 IC * estimateElementCount(VF: MainLoopVF, VScale: Config.getVScaleForTuning())) {
3517 // Note that the main loop leaves IC * MainLoopVF iterations iff a scalar
3518 // epilogue is required, but then the epilogue loop also requires a scalar
3519 // epilogue.
3520 LLVM_DEBUG(dbgs() << "LEV: Forced epilogue VF results in dead epilogue "
3521 "vector loop, skipping vectorizing epilogue.\n");
3522 return nullptr;
3523 }
3524
3525 LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization factor is forced.\n");
3526 if (hasPlanWithVF(VF: EpilogueVectorizationForceVF)) {
3527 std::unique_ptr<VPlan> Clone(
3528 getPlanFor(VF: EpilogueVectorizationForceVF).duplicate());
3529 Clone->setVF(EpilogueVectorizationForceVF);
3530 return Clone;
3531 }
3532
3533 LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization forced factor is not "
3534 "viable.\n");
3535 return nullptr;
3536 }
3537
3538 if (OrigLoop->getHeader()->getParent()->hasOptSize()) {
3539 LLVM_DEBUG(
3540 dbgs() << "LEV: Epilogue vectorization skipped due to opt for size.\n");
3541 return nullptr;
3542 }
3543
3544 if (!Config.isEpilogueVectorizationProfitable(VF: MainLoopVF, IC)) {
3545 LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization is not profitable for "
3546 "this loop\n");
3547 return nullptr;
3548 }
3549
3550 // Check if a plan's vector loop processes fewer iterations than VF (e.g. when
3551 // interleave groups have been narrowed) narrowInterleaveGroups) and return
3552 // the adjusted, effective VF.
3553 using namespace VPlanPatternMatch;
3554 auto GetEffectiveVF = [](VPlan &Plan, ElementCount VF) -> ElementCount {
3555 auto *Exiting = Plan.getVectorLoopRegion()->getExitingBasicBlock();
3556 if (match(V: &Exiting->back(),
3557 P: m_BranchOnCount(Op0: m_Add(Op0: m_CanonicalIV(), Op1: m_Specific(VPV: &Plan.getUF())),
3558 Op1: m_VPValue())))
3559 return ElementCount::get(MinVal: 1, Scalable: VF.isScalable());
3560 return VF;
3561 };
3562
3563 // Check if the main loop processes fewer than MainLoopVF elements per
3564 // iteration (e.g. due to narrowing interleave groups). Adjust MainLoopVF
3565 // as needed.
3566 MainLoopVF = GetEffectiveVF(MainPlan, MainLoopVF);
3567
3568 // If MainLoopVF = vscale x 2, and vscale is expected to be 4, then we know
3569 // the main loop handles 8 lanes per iteration. We could still benefit from
3570 // vectorizing the epilogue loop with VF=4.
3571 ElementCount EstimatedRuntimeVF = ElementCount::getFixed(
3572 MinVal: estimateElementCount(VF: MainLoopVF, VScale: Config.getVScaleForTuning()));
3573
3574 Type *TCType = Legal->getWidestInductionType();
3575 const SCEV *RemainingIterations = nullptr;
3576 unsigned MaxTripCount = 0;
3577 const SCEV *TC = vputils::getSCEVExprForVPValue(V: MainPlan.getTripCount(), PSE);
3578 assert(!isa<SCEVCouldNotCompute>(TC) && "Trip count SCEV must be computable");
3579 const SCEV *KnownMinTC;
3580 bool ScalableTC = match(S: TC, P: m_scev_c_Mul(Op0: m_SCEV(V&: KnownMinTC), Op1: m_SCEVVScale()));
3581 bool ScalableRemIter = false;
3582 ScalarEvolution &SE = *PSE.getSE();
3583 // Use versions of TC and VF in which both are either scalable or fixed.
3584 if (ScalableTC == MainLoopVF.isScalable()) {
3585 ScalableRemIter = ScalableTC;
3586 RemainingIterations =
3587 SE.getURemExpr(LHS: TC, RHS: SE.getElementCount(Ty: TCType, EC: MainLoopVF * IC));
3588 } else if (ScalableTC) {
3589 const SCEV *EstimatedTC = SE.getMulExpr(
3590 LHS: KnownMinTC,
3591 RHS: SE.getConstant(Ty: TCType, V: Config.getVScaleForTuning().value_or(u: 1)));
3592 RemainingIterations = SE.getURemExpr(
3593 LHS: EstimatedTC, RHS: SE.getElementCount(Ty: TCType, EC: MainLoopVF * IC));
3594 } else
3595 RemainingIterations =
3596 SE.getURemExpr(LHS: TC, RHS: SE.getElementCount(Ty: TCType, EC: EstimatedRuntimeVF * IC));
3597
3598 // No iterations left to process in the epilogue.
3599 if (RemainingIterations->isZero())
3600 return nullptr;
3601
3602 if (MainLoopVF.isFixed()) {
3603 MaxTripCount = MainLoopVF.getFixedValue() * IC - 1;
3604 if (SE.isKnownPredicate(Pred: CmpInst::ICMP_ULT, LHS: RemainingIterations,
3605 RHS: SE.getConstant(Ty: TCType, V: MaxTripCount))) {
3606 MaxTripCount = SE.getUnsignedRangeMax(S: RemainingIterations).getZExtValue();
3607 }
3608 LLVM_DEBUG(dbgs() << "LEV: Maximum Trip Count for Epilogue: "
3609 << MaxTripCount << "\n");
3610 }
3611
3612 auto SkipVF = [&](const SCEV *VF, const SCEV *RemIter) -> bool {
3613 return SE.isKnownPredicate(Pred: CmpInst::ICMP_UGT, LHS: VF, RHS: RemIter);
3614 };
3615 VectorizationFactor Result = VectorizationFactor::Disabled();
3616 VPlan *BestPlan = nullptr;
3617 for (auto &NextVF : ProfitableVFs) {
3618 // Skip candidate VFs without a corresponding VPlan.
3619 if (!hasPlanWithVF(VF: NextVF.Width))
3620 continue;
3621
3622 VPlan &CurrentPlan = getPlanFor(VF: NextVF.Width);
3623 ElementCount EffectiveVF = GetEffectiveVF(CurrentPlan, NextVF.Width);
3624 // Skip fixed vector VFs > than the estimated runtime VF, or any VF > than
3625 // the VF of the main loop.
3626 if ((!EffectiveVF.isScalable() && MainLoopVF.isScalable() &&
3627 ElementCount::isKnownGT(LHS: EffectiveVF, RHS: EstimatedRuntimeVF)) ||
3628 ElementCount::isKnownGT(LHS: EffectiveVF, RHS: MainLoopVF))
3629 continue;
3630
3631 // If EffectiveVF is greater than the number of remaining iterations, the
3632 // epilogue loop would be dead. Skip such factors. If the epilogue plan
3633 // also has narrowed interleave groups, use the effective VF since
3634 // the epilogue step will be reduced to its IC.
3635 // TODO: We should also consider comparing against a scalable
3636 // RemainingIterations when SCEV be able to evaluate non-canonical
3637 // vscale-based expressions.
3638 if (!ScalableRemIter) {
3639 // Handle the case where EffectiveVF and RemainingIterations are in
3640 // different numerical spaces.
3641 if (EffectiveVF.isScalable())
3642 EffectiveVF = ElementCount::getFixed(
3643 MinVal: estimateElementCount(VF: EffectiveVF, VScale: Config.getVScaleForTuning()));
3644 if (SkipVF(SE.getElementCount(Ty: TCType, EC: EffectiveVF), RemainingIterations))
3645 continue;
3646 }
3647
3648 if (Result.Width.isScalar() ||
3649 isMoreProfitable(A: NextVF, B: Result, MaxTripCount,
3650 HasTail: !MainPlan.hasTailFolded(),
3651 /*IsEpilogue*/ true)) {
3652 Result = NextVF;
3653 BestPlan = &CurrentPlan;
3654 }
3655 }
3656
3657 if (!BestPlan)
3658 return nullptr;
3659
3660 LLVM_DEBUG(dbgs() << "LEV: Vectorizing epilogue loop with VF = "
3661 << Result.Width << "\n");
3662 std::unique_ptr<VPlan> Clone(BestPlan->duplicate());
3663 Clone->setVF(Result.Width);
3664 return Clone;
3665}
3666
3667unsigned
3668LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
3669 InstructionCost LoopCost) {
3670 // -- The interleave heuristics --
3671 // We interleave the loop in order to expose ILP and reduce the loop overhead.
3672 // There are many micro-architectural considerations that we can't predict
3673 // at this level. For example, frontend pressure (on decode or fetch) due to
3674 // code size, or the number and capabilities of the execution ports.
3675 //
3676 // We use the following heuristics to select the interleave count:
3677 // 1. If the code has reductions, then we interleave to break the cross
3678 // iteration dependency.
3679 // 2. If the loop is really small, then we interleave to reduce the loop
3680 // overhead.
3681 // 3. We don't interleave if we think that we will spill registers to memory
3682 // due to the increased register pressure.
3683
3684 // Do not interleave tail-folded loops, as the overhead of multiple
3685 // instructions to calculate the predicate is likely not beneficial.
3686 // If an epilogue is not allowed for any other reason, do not interleave.
3687 if (!CM->isEpilogueAllowed())
3688 return 1;
3689
3690 if (any_of(Range: Plan.getVectorLoopRegion()->getEntryBasicBlock()->phis(),
3691 P: IsaPred<VPCurrentIterationPHIRecipe>)) {
3692 LLVM_DEBUG(dbgs() << "LV: Loop requires variable-length step. "
3693 "Unroll factor forced to be 1.\n");
3694 return 1;
3695 }
3696
3697 // We used the distance for the interleave count.
3698 if (!Legal->isSafeForAnyVectorWidth())
3699 return 1;
3700
3701 // We don't attempt to perform interleaving for loops with uncountable early
3702 // exits because the VPInstruction::AnyOf code cannot currently handle
3703 // multiple parts.
3704 if (Plan.hasEarlyExit())
3705 return 1;
3706
3707 const bool HasReductions =
3708 any_of(Range: Plan.getVectorLoopRegion()->getEntryBasicBlock()->phis(),
3709 P: IsaPred<VPReductionPHIRecipe>);
3710
3711 // FIXME: implement interleaving for FindLast transform correctly.
3712 if (hasFindLastReductionPhi(Plan))
3713 return 1;
3714
3715 VPRegisterUsage R = calculateRegisterUsageForPlan(Plan, VFs: {VF}, TTI)[0];
3716
3717 // If we did not calculate the cost for VF (because the user selected the VF)
3718 // then we calculate the cost of VF here.
3719 if (LoopCost == 0) {
3720 if (VF.isScalar())
3721 LoopCost = CM->expectedCost(VF);
3722 else
3723 LoopCost = cost(Plan, VF, RU: &R);
3724 assert(LoopCost.isValid() && "Expected to have chosen a VF with valid cost");
3725
3726 // Loop body is free and there is no need for interleaving.
3727 if (LoopCost == 0)
3728 return 1;
3729 }
3730
3731 // We divide by these constants so assume that we have at least one
3732 // instruction that uses at least one register.
3733 for (auto &Pair : R.MaxLocalUsers) {
3734 Pair.second = std::max(a: Pair.second, b: 1U);
3735 }
3736
3737 // We calculate the interleave count using the following formula.
3738 // Subtract the number of loop invariants from the number of available
3739 // registers. These registers are used by all of the interleaved instances.
3740 // Next, divide the remaining registers by the number of registers that is
3741 // required by the loop, in order to estimate how many parallel instances
3742 // fit without causing spills. All of this is rounded down if necessary to be
3743 // a power of two. We want power of two interleave count to simplify any
3744 // addressing operations or alignment considerations.
3745 // We also want power of two interleave counts to ensure that the induction
3746 // variable of the vector loop wraps to zero, when tail is folded by masking;
3747 // this currently happens when OptForSize, in which case IC is set to 1 above.
3748 unsigned IC = UINT_MAX;
3749
3750 for (const auto &Pair : R.MaxLocalUsers) {
3751 unsigned TargetNumRegisters = TTI.getNumberOfRegisters(ClassID: Pair.first);
3752 LLVM_DEBUG(dbgs() << "LV: The target has " << TargetNumRegisters
3753 << " registers of "
3754 << TTI.getRegisterClassName(Pair.first)
3755 << " register class\n");
3756 if (VF.isScalar()) {
3757 if (ForceTargetNumScalarRegs.getNumOccurrences() > 0)
3758 TargetNumRegisters = ForceTargetNumScalarRegs;
3759 } else {
3760 if (ForceTargetNumVectorRegs.getNumOccurrences() > 0)
3761 TargetNumRegisters = ForceTargetNumVectorRegs;
3762 }
3763 unsigned MaxLocalUsers = Pair.second;
3764 unsigned LoopInvariantRegs = 0;
3765 if (R.LoopInvariantRegs.contains(Key: Pair.first))
3766 LoopInvariantRegs = R.LoopInvariantRegs[Pair.first];
3767
3768 unsigned TmpIC = llvm::bit_floor(Value: (TargetNumRegisters - LoopInvariantRegs) /
3769 MaxLocalUsers);
3770 // Don't count the induction variable as interleaved.
3771 if (EnableIndVarRegisterHeur) {
3772 TmpIC = llvm::bit_floor(Value: (TargetNumRegisters - LoopInvariantRegs - 1) /
3773 std::max(a: 1U, b: (MaxLocalUsers - 1)));
3774 }
3775
3776 IC = std::min(a: IC, b: TmpIC);
3777 }
3778
3779 // Clamp the interleave ranges to reasonable counts.
3780 bool HasUnorderedReductions =
3781 HasReductions &&
3782 !any_of(Range: make_isa_range<VPReductionPHIRecipe>(
3783 Range: Plan.getVectorLoopRegion()->getEntryBasicBlock()->phis()),
3784 P: [](VPReductionPHIRecipe &RedR) { return RedR.isOrdered(); });
3785 unsigned MaxInterleaveCount =
3786 TTI.getMaxInterleaveFactor(VF, HasUnorderedReductions);
3787 LLVM_DEBUG(dbgs() << "LV: MaxInterleaveFactor for the target is "
3788 << MaxInterleaveCount << "\n");
3789
3790 // Check if the user has overridden the max.
3791 if (VF.isScalar()) {
3792 if (ForceTargetMaxScalarInterleaveFactor.getNumOccurrences() > 0)
3793 MaxInterleaveCount = ForceTargetMaxScalarInterleaveFactor;
3794 } else {
3795 if (ForceTargetMaxVectorInterleaveFactor.getNumOccurrences() > 0)
3796 MaxInterleaveCount = ForceTargetMaxVectorInterleaveFactor;
3797 }
3798
3799 // Try to get the exact trip count, or an estimate based on profiling data or
3800 // ConstantMax from PSE, failing that.
3801 auto BestKnownTC =
3802 getSmallBestKnownTC(PSE, L: OrigLoop,
3803 /*CanUseConstantMax=*/true,
3804 /*CanExcludeZeroTrips=*/CM->isEpilogueAllowed());
3805
3806 // For fixed length VFs treat a scalable trip count as unknown.
3807 if (BestKnownTC && (BestKnownTC->isFixed() || VF.isScalable())) {
3808 // Re-evaluate trip counts and VFs to be in the same numerical space.
3809 unsigned AvailableTC =
3810 estimateElementCount(VF: *BestKnownTC, VScale: Config.getVScaleForTuning());
3811 unsigned EstimatedVF =
3812 estimateElementCount(VF, VScale: Config.getVScaleForTuning());
3813
3814 // At least one iteration must be scalar when this constraint holds. So the
3815 // maximum available iterations for interleaving is one less.
3816 if (Plan.requiresScalarEpilogue())
3817 --AvailableTC;
3818
3819 unsigned InterleaveCountLB = bit_floor(Value: std::max(
3820 a: 1u, b: std::min(a: AvailableTC / (EstimatedVF * 2), b: MaxInterleaveCount)));
3821
3822 if (getSmallConstantTripCount(SE: PSE.getSE(), L: OrigLoop).isNonZero()) {
3823 // If the best known trip count is exact, we select between two
3824 // prospective ICs, where
3825 //
3826 // 1) the aggressive IC is capped by the trip count divided by VF
3827 // 2) the conservative IC is capped by the trip count divided by (VF * 2)
3828 //
3829 // The final IC is selected in a way that the epilogue loop trip count is
3830 // minimized while maximizing the IC itself, so that we either run the
3831 // vector loop at least once if it generates a small epilogue loop, or
3832 // else we run the vector loop at least twice.
3833
3834 unsigned InterleaveCountUB = bit_floor(Value: std::max(
3835 a: 1u, b: std::min(a: AvailableTC / EstimatedVF, b: MaxInterleaveCount)));
3836 MaxInterleaveCount = InterleaveCountLB;
3837
3838 if (InterleaveCountUB != InterleaveCountLB) {
3839 unsigned TailTripCountUB =
3840 (AvailableTC % (EstimatedVF * InterleaveCountUB));
3841 unsigned TailTripCountLB =
3842 (AvailableTC % (EstimatedVF * InterleaveCountLB));
3843 // If both produce same scalar tail, maximize the IC to do the same work
3844 // in fewer vector loop iterations
3845 if (TailTripCountUB == TailTripCountLB)
3846 MaxInterleaveCount = InterleaveCountUB;
3847 }
3848 } else {
3849 // If trip count is an estimated compile time constant, limit the
3850 // IC to be capped by the trip count divided by VF * 2, such that the
3851 // vector loop runs at least twice to make interleaving seem profitable
3852 // when there is an epilogue loop present. Since exact Trip count is not
3853 // known we choose to be conservative in our IC estimate.
3854 MaxInterleaveCount = InterleaveCountLB;
3855 }
3856 }
3857
3858 assert(MaxInterleaveCount > 0 &&
3859 "Maximum interleave count must be greater than 0");
3860
3861 // Clamp the calculated IC to be between the 1 and the max interleave count
3862 // that the target and trip count allows.
3863 if (IC > MaxInterleaveCount)
3864 IC = MaxInterleaveCount;
3865 else
3866 // Make sure IC is greater than 0.
3867 IC = std::max(a: 1u, b: IC);
3868
3869 assert(IC > 0 && "Interleave count must be greater than 0.");
3870
3871 // Interleave if we vectorized this loop and there is a reduction that could
3872 // benefit from interleaving.
3873 if (VF.isVector() && HasReductions) {
3874 LLVM_DEBUG(dbgs() << "LV: Interleaving because of reductions.\n");
3875 return IC;
3876 }
3877
3878 // For any scalar loop that either requires runtime checks or tail-folding we
3879 // are better off leaving this to the unroller. Note that if we've already
3880 // vectorized the loop we will have done the runtime check and so interleaving
3881 // won't require further checks.
3882 bool ScalarInterleavingRequiresPredication =
3883 (VF.isScalar() && any_of(Range: OrigLoop->blocks(), P: [this](BasicBlock *BB) {
3884 return Legal->blockNeedsPredication(BB);
3885 }));
3886 bool ScalarInterleavingRequiresRuntimePointerCheck =
3887 (VF.isScalar() && Legal->getRuntimePointerChecking()->Need);
3888
3889 // We want to interleave small loops in order to reduce the loop overhead and
3890 // potentially expose ILP opportunities.
3891 LLVM_DEBUG(dbgs() << "LV: Loop cost is " << LoopCost << '\n'
3892 << "LV: IC is " << IC << '\n'
3893 << "LV: VF is " << VF << '\n');
3894 const bool AggressivelyInterleave =
3895 TTI.enableAggressiveInterleaving(LoopHasReductions: HasReductions);
3896 if (!ScalarInterleavingRequiresRuntimePointerCheck &&
3897 !ScalarInterleavingRequiresPredication && LoopCost < SmallLoopCost) {
3898 // We assume that the cost overhead is 1 and we use the cost model
3899 // to estimate the cost of the loop and interleave until the cost of the
3900 // loop overhead is about 5% of the cost of the loop.
3901 unsigned SmallIC = std::min(a: IC, b: (unsigned)llvm::bit_floor<uint64_t>(
3902 Value: SmallLoopCost / LoopCost.getValue()));
3903
3904 // Interleave until store/load ports (estimated by max interleave count) are
3905 // saturated.
3906 unsigned NumStores = 0;
3907 unsigned NumLoads = 0;
3908 for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
3909 Range: vp_depth_first_deep(G: Plan.getVectorLoopRegion()->getEntry()))) {
3910 for (VPRecipeBase &R : *VPBB) {
3911 if (isa<VPWidenLoadRecipe, VPWidenLoadEVLRecipe>(Val: &R)) {
3912 NumLoads++;
3913 continue;
3914 }
3915 if (isa<VPWidenStoreRecipe, VPWidenStoreEVLRecipe>(Val: &R)) {
3916 NumStores++;
3917 continue;
3918 }
3919
3920 if (auto *InterleaveR = dyn_cast<VPInterleaveRecipe>(Val: &R)) {
3921 if (unsigned StoreOps = InterleaveR->getNumStoreOperands())
3922 NumStores += StoreOps;
3923 else
3924 NumLoads += InterleaveR->getNumDefinedValues();
3925 continue;
3926 }
3927 if (auto *RepR = dyn_cast<VPReplicateRecipe>(Val: &R)) {
3928 NumLoads += isa<LoadInst>(Val: RepR->getUnderlyingInstr());
3929 NumStores += isa<StoreInst>(Val: RepR->getUnderlyingInstr());
3930 continue;
3931 }
3932 if (isa<VPHistogramRecipe>(Val: &R)) {
3933 NumLoads++;
3934 NumStores++;
3935 continue;
3936 }
3937 }
3938 }
3939 unsigned StoresIC = IC / (NumStores ? NumStores : 1);
3940 unsigned LoadsIC = IC / (NumLoads ? NumLoads : 1);
3941
3942 // There is little point in interleaving for reductions containing selects
3943 // and compares when VF=1 since it may just create more overhead than it's
3944 // worth for loops with small trip counts. This is because we still have to
3945 // do the final reduction after the loop.
3946 bool HasSelectCmpReductions =
3947 HasReductions &&
3948 any_of(Range: make_isa_range<VPReductionPHIRecipe>(
3949 Range: Plan.getVectorLoopRegion()->getEntryBasicBlock()->phis()),
3950 P: [](VPReductionPHIRecipe &RedR) {
3951 return RecurrenceDescriptor::isAnyOfRecurrenceKind(
3952 Kind: RedR.getRecurrenceKind()) ||
3953 RecurrenceDescriptor::isFindIVRecurrenceKind(
3954 Kind: RedR.getRecurrenceKind());
3955 });
3956 if (HasSelectCmpReductions) {
3957 LLVM_DEBUG(dbgs() << "LV: Not interleaving select-cmp reductions.\n");
3958 return 1;
3959 }
3960
3961 // If we have a scalar reduction (vector reductions are already dealt with
3962 // by this point), we can increase the critical path length if the loop
3963 // we're interleaving is inside another loop. For tree-wise reductions
3964 // set the limit to 2, and for ordered reductions it's best to disable
3965 // interleaving entirely.
3966 if (HasReductions && OrigLoop->getLoopDepth() > 1) {
3967 bool HasOrderedReductions =
3968 any_of(Range: make_isa_range<VPReductionPHIRecipe>(
3969 Range: Plan.getVectorLoopRegion()->getEntryBasicBlock()->phis()),
3970 P: [](VPReductionPHIRecipe &RedR) { return RedR.isOrdered(); });
3971 if (HasOrderedReductions) {
3972 LLVM_DEBUG(
3973 dbgs() << "LV: Not interleaving scalar ordered reductions.\n");
3974 return 1;
3975 }
3976
3977 unsigned F = MaxNestedScalarReductionIC;
3978 SmallIC = std::min(a: SmallIC, b: F);
3979 StoresIC = std::min(a: StoresIC, b: F);
3980 LoadsIC = std::min(a: LoadsIC, b: F);
3981 }
3982
3983 if (EnableLoadStoreRuntimeInterleave &&
3984 std::max(a: StoresIC, b: LoadsIC) > SmallIC) {
3985 LLVM_DEBUG(
3986 dbgs() << "LV: Interleaving to saturate store or load ports.\n");
3987 return std::max(a: StoresIC, b: LoadsIC);
3988 }
3989
3990 // If there are scalar reductions and TTI has enabled aggressive
3991 // interleaving for reductions, we will interleave to expose ILP.
3992 if (VF.isScalar() && AggressivelyInterleave) {
3993 LLVM_DEBUG(dbgs() << "LV: Interleaving to expose ILP.\n");
3994 // Interleave no less than SmallIC but not as aggressive as the normal IC
3995 // to satisfy the rare situation when resources are too limited.
3996 return std::max(a: IC / 2, b: SmallIC);
3997 }
3998
3999 LLVM_DEBUG(dbgs() << "LV: Interleaving to reduce branch cost.\n");
4000 return SmallIC;
4001 }
4002
4003 // Interleave if this is a large loop (small loops are already dealt with by
4004 // this point) that could benefit from interleaving.
4005 if (AggressivelyInterleave) {
4006 LLVM_DEBUG(dbgs() << "LV: Interleaving to expose ILP.\n");
4007 return IC;
4008 }
4009
4010 LLVM_DEBUG(dbgs() << "LV: Not Interleaving.\n");
4011 return 1;
4012}
4013
4014bool LoopVectorizationCostModel::useEmulatedMaskMemRefHack(
4015 Instruction *I, ElementCount VF) const {
4016 // TODO: Cost model for emulated masked load/store is completely
4017 // broken. This hack guides the cost model to use an artificially
4018 // high enough value to practically disable vectorization with such
4019 // operations, except where previously deployed legality hack allowed
4020 // using very low cost values. This is to avoid regressions coming simply
4021 // from moving "masked load/store" check from legality to cost model.
4022 // Masked Load/Gather emulation was previously never allowed.
4023 // Limited number of Masked Store/Scatter emulation was allowed.
4024 assert((isPredicatedInst(I)) &&
4025 "Expecting a scalar emulated instruction");
4026 return isa<LoadInst>(Val: I) ||
4027 (isa<StoreInst>(Val: I) &&
4028 NumPredStores > NumberOfStoresToPredicate);
4029}
4030
4031void LoopVectorizationCostModel::collectInstsToScalarize(ElementCount VF) {
4032 assert(VF.isVector() && "Expected VF >= 2");
4033
4034 // If we've already collected the instructions to scalarize or the predicated
4035 // BBs after vectorization, there's nothing to do. Collection may already have
4036 // occurred if we have a user-selected VF and are now computing the expected
4037 // cost for interleaving.
4038 if (InstsToScalarize.contains(Key: VF) ||
4039 PredicatedBBsAfterVectorization.contains(Val: VF))
4040 return;
4041
4042 // Initialize a mapping for VF in InstsToScalalarize. If we find that it's
4043 // not profitable to scalarize any instructions, the presence of VF in the
4044 // map will indicate that we've analyzed it already.
4045 ScalarCostsTy &ScalarCostsVF = InstsToScalarize[VF];
4046
4047 // Find all the instructions that are scalar with predication in the loop and
4048 // determine if it would be better to not if-convert the blocks they are in.
4049 // If so, we also record the instructions to scalarize.
4050 for (BasicBlock *BB : TheLoop->blocks()) {
4051 if (!blockNeedsPredicationForAnyReason(BB))
4052 continue;
4053 for (Instruction &I : *BB)
4054 if (isScalarWithPredication(I: &I, VF)) {
4055 ScalarCostsTy ScalarCosts;
4056 // Do not apply discount logic for:
4057 // 1. Scalars after vectorization, as there will only be a single copy
4058 // of the instruction.
4059 // 2. Scalable VF, as that would lead to invalid scalarization costs.
4060 // 3. Emulated masked memrefs, if a hacked cost is needed.
4061 if (!isScalarAfterVectorization(I: &I, VF) && !VF.isScalable() &&
4062 !useEmulatedMaskMemRefHack(I: &I, VF) &&
4063 computePredInstDiscount(PredInst: &I, ScalarCosts, VF) >= 0) {
4064 for (const auto &[I, IC] : ScalarCosts)
4065 ScalarCostsVF.insert(KV: {I, IC});
4066 }
4067 // Remember that BB will remain after vectorization.
4068 PredicatedBBsAfterVectorization[VF].insert(Ptr: BB);
4069 for (auto *Pred : predecessors(BB)) {
4070 if (Pred->getSingleSuccessor() == BB)
4071 PredicatedBBsAfterVectorization[VF].insert(Ptr: Pred);
4072 }
4073 }
4074 }
4075}
4076
4077InstructionCost LoopVectorizationCostModel::computePredInstDiscount(
4078 Instruction *PredInst, ScalarCostsTy &ScalarCosts, ElementCount VF) {
4079 assert(!isUniformAfterVectorization(PredInst, VF) &&
4080 "Instruction marked uniform-after-vectorization will be predicated");
4081
4082 // Initialize the discount to zero, meaning that the scalar version and the
4083 // vector version cost the same.
4084 InstructionCost Discount = 0;
4085
4086 // Holds instructions to analyze. The instructions we visit are mapped in
4087 // ScalarCosts. Those instructions are the ones that would be scalarized if
4088 // we find that the scalar version costs less.
4089 SmallVector<Instruction *, 8> Worklist;
4090
4091 // Returns true if the given instruction can be scalarized.
4092 auto CanBeScalarized = [&](Instruction *I) -> bool {
4093 // We only attempt to scalarize instructions forming a single-use chain
4094 // from the original predicated block that would otherwise be vectorized.
4095 // Although not strictly necessary, we give up on instructions we know will
4096 // already be scalar to avoid traversing chains that are unlikely to be
4097 // beneficial.
4098 if (!I->hasOneUse() || PredInst->getParent() != I->getParent() ||
4099 isScalarAfterVectorization(I, VF))
4100 return false;
4101
4102 // If the instruction is scalar with predication, it will be analyzed
4103 // separately. We ignore it within the context of PredInst.
4104 if (isScalarWithPredication(I, VF))
4105 return false;
4106
4107 // If any of the instruction's operands are uniform after vectorization,
4108 // the instruction cannot be scalarized. This prevents, for example, a
4109 // masked load from being scalarized.
4110 //
4111 // We assume we will only emit a value for lane zero of an instruction
4112 // marked uniform after vectorization, rather than VF identical values.
4113 // Thus, if we scalarize an instruction that uses a uniform, we would
4114 // create uses of values corresponding to the lanes we aren't emitting code
4115 // for. This behavior can be changed by allowing getScalarValue to clone
4116 // the lane zero values for uniforms rather than asserting.
4117 for (Use &U : I->operands())
4118 if (auto *J = dyn_cast<Instruction>(Val: U.get()))
4119 if (isUniformAfterVectorization(I: J, VF))
4120 return false;
4121
4122 // Otherwise, we can scalarize the instruction.
4123 return true;
4124 };
4125
4126 // Compute the expected cost discount from scalarizing the entire expression
4127 // feeding the predicated instruction. We currently only consider expressions
4128 // that are single-use instruction chains.
4129 Worklist.push_back(Elt: PredInst);
4130 while (!Worklist.empty()) {
4131 Instruction *I = Worklist.pop_back_val();
4132
4133 // If we've already analyzed the instruction, there's nothing to do.
4134 if (ScalarCosts.contains(Key: I))
4135 continue;
4136
4137 // Cannot scalarize fixed-order recurrence phis at the moment.
4138 if (isa<PHINode>(Val: I) && Legal->isFixedOrderRecurrence(Phi: cast<PHINode>(Val: I)))
4139 continue;
4140
4141 // Compute the cost of the vector instruction. Note that this cost already
4142 // includes the scalarization overhead of the predicated instruction.
4143 InstructionCost VectorCost = getInstructionCost(I, VF);
4144
4145 // Compute the cost of the scalarized instruction. This cost is the cost of
4146 // the instruction as if it wasn't if-converted and instead remained in the
4147 // predicated block. We will scale this cost by block probability after
4148 // computing the scalarization overhead.
4149 InstructionCost ScalarCost =
4150 VF.getFixedValue() * getInstructionCost(I, VF: ElementCount::getFixed(MinVal: 1));
4151
4152 // Compute the scalarization overhead of needed insertelement instructions
4153 // and phi nodes.
4154 if (isScalarWithPredication(I, VF) && !I->getType()->isVoidTy()) {
4155 Type *WideTy = toVectorizedTy(Ty: I->getType(), EC: VF);
4156 for (Type *VectorTy : getContainedTypes(Ty: WideTy)) {
4157 ScalarCost += TTI.getScalarizationOverhead(
4158 Ty: cast<VectorType>(Val: VectorTy), DemandedElts: APInt::getAllOnes(numBits: VF.getFixedValue()),
4159 /*Insert=*/true,
4160 /*Extract=*/false, CostKind: Config.CostKind);
4161 }
4162 ScalarCost += VF.getFixedValue() *
4163 TTI.getCFInstrCost(Opcode: Instruction::PHI, CostKind: Config.CostKind);
4164 }
4165
4166 // Compute the scalarization overhead of needed extractelement
4167 // instructions. For each of the instruction's operands, if the operand can
4168 // be scalarized, add it to the worklist; otherwise, account for the
4169 // overhead.
4170 for (Use &U : I->operands())
4171 if (auto *J = dyn_cast<Instruction>(Val: U.get())) {
4172 assert(canVectorizeTy(J->getType()) &&
4173 "Instruction has non-scalar type");
4174 if (CanBeScalarized(J))
4175 Worklist.push_back(Elt: J);
4176 else if (needsExtract(V: J, VF)) {
4177 Type *WideTy = toVectorizedTy(Ty: J->getType(), EC: VF);
4178 for (Type *VectorTy : getContainedTypes(Ty: WideTy)) {
4179 ScalarCost += TTI.getScalarizationOverhead(
4180 Ty: cast<VectorType>(Val: VectorTy),
4181 DemandedElts: APInt::getAllOnes(numBits: VF.getFixedValue()), /*Insert*/ false,
4182 /*Extract*/ true, CostKind: Config.CostKind);
4183 }
4184 }
4185 }
4186
4187 // Scale the total scalar cost by block probability.
4188 ScalarCost /= getPredBlockCostDivisor(CostKind: Config.CostKind, BB: I->getParent());
4189
4190 // Compute the discount. A non-negative discount means the vector version
4191 // of the instruction costs more, and scalarizing would be beneficial.
4192 Discount += VectorCost - ScalarCost;
4193 ScalarCosts[I] = ScalarCost;
4194 }
4195
4196 return Discount;
4197}
4198
4199InstructionCost LoopVectorizationCostModel::expectedCost(ElementCount VF) {
4200 InstructionCost Cost;
4201 assert(VF.isScalar() && "must only be called for scalar VFs");
4202
4203 // For each block.
4204 for (BasicBlock *BB : TheLoop->blocks()) {
4205 InstructionCost BlockCost;
4206
4207 // For each instruction in the old loop.
4208 for (Instruction &I : *BB) {
4209 // Skip ignored values.
4210 if (ValuesToIgnore.count(Ptr: &I) ||
4211 (VF.isVector() && VecValuesToIgnore.count(Ptr: &I)))
4212 continue;
4213
4214 InstructionCost C = getInstructionCost(I: &I, VF);
4215
4216 // Check if we should override the cost.
4217 if (C.isValid() && ForceTargetInstructionCost.getNumOccurrences() > 0)
4218 C = InstructionCost(ForceTargetInstructionCost);
4219
4220 BlockCost += C;
4221 LLVM_DEBUG(dbgs() << "LV: Found an estimated cost of " << C << " for VF "
4222 << VF << " For instruction: " << I << '\n');
4223 }
4224
4225 // In the scalar loop, we may not always execute the predicated block, if it
4226 // is an if-else block. Thus, scale the block's cost by the probability of
4227 // executing it. getPredBlockCostDivisor will return 1 for blocks that are
4228 // only predicated by the header mask when folding the tail.
4229 Cost += BlockCost / getPredBlockCostDivisor(CostKind: Config.CostKind, BB);
4230 }
4231
4232 return Cost;
4233}
4234
4235/// Gets the address access SCEV for Ptr, if it should be used for cost modeling
4236/// according to isAddressSCEVForCost.
4237///
4238/// This SCEV can be sent to the Target in order to estimate the address
4239/// calculation cost.
4240static const SCEV *getAddressAccessSCEV(
4241 Value *Ptr,
4242 PredicatedScalarEvolution &PSE,
4243 const Loop *TheLoop) {
4244 const SCEV *Addr = PSE.getSCEV(V: Ptr);
4245 return vputils::isAddressSCEVForCost(Addr, SE&: *PSE.getSE(), L: TheLoop) ? Addr
4246 : nullptr;
4247}
4248
4249InstructionCost
4250LoopVectorizationCostModel::getMemInstScalarizationCost(Instruction *I,
4251 ElementCount VF) {
4252 assert(VF.isVector() &&
4253 "Scalarization cost of instruction implies vectorization.");
4254 if (VF.isScalable())
4255 return InstructionCost::getInvalid();
4256
4257 Type *ValTy = getLoadStoreType(I);
4258 auto *SE = PSE.getSE();
4259
4260 unsigned AS = getLoadStoreAddressSpace(I);
4261 Value *Ptr = getLoadStorePointerOperand(V: I);
4262 Type *PtrTy = toVectorTy(Scalar: Ptr->getType(), EC: VF);
4263 // NOTE: PtrTy is a vector to signal `TTI::getAddressComputationCost`
4264 // that it is being called from this specific place.
4265
4266 // Figure out whether the access is strided and get the stride value
4267 // if it's known in compile time
4268 const SCEV *PtrSCEV = getAddressAccessSCEV(Ptr, PSE, TheLoop);
4269
4270 // Get the cost of the scalar memory instruction and address computation.
4271 InstructionCost Cost =
4272 VF.getFixedValue() *
4273 TTI.getAddressComputationCost(PtrTy, SE, Ptr: PtrSCEV, CostKind: Config.CostKind);
4274
4275 // Don't pass *I here, since it is scalar but will actually be part of a
4276 // vectorized loop where the user of it is a vectorized instruction.
4277 const Align Alignment = getLoadStoreAlignment(I);
4278 TTI::OperandValueInfo OpInfo = TTI::getOperandInfo(V: I->getOperand(i: 0));
4279 Cost += VF.getFixedValue() *
4280 TTI.getMemoryOpCost(Opcode: I->getOpcode(), Src: ValTy->getScalarType(), Alignment,
4281 AddressSpace: AS, CostKind: Config.CostKind, OpdInfo: OpInfo);
4282
4283 // Get the overhead of the extractelement and insertelement instructions
4284 // we might create due to scalarization.
4285 Cost += getScalarizationOverhead(I, VF);
4286
4287 // If we have a predicated load/store, it will need extra i1 extracts and
4288 // conditional branches, but may not be executed for each vector lane. Scale
4289 // the cost by the probability of executing the predicated block.
4290 if (isPredicatedInst(I)) {
4291 Cost /= getPredBlockCostDivisor(CostKind: Config.CostKind, BB: I->getParent());
4292
4293 // Add the cost of an i1 extract and a branch
4294 auto *VecI1Ty =
4295 VectorType::get(ElementType: IntegerType::getInt1Ty(C&: ValTy->getContext()), EC: VF);
4296 Cost += TTI.getScalarizationOverhead(
4297 Ty: VecI1Ty, DemandedElts: APInt::getAllOnes(numBits: VF.getFixedValue()),
4298 /*Insert=*/false, /*Extract=*/true, CostKind: Config.CostKind);
4299 Cost += TTI.getCFInstrCost(Opcode: Instruction::CondBr, CostKind: Config.CostKind);
4300
4301 if (useEmulatedMaskMemRefHack(I, VF))
4302 // Artificially setting to a high enough value to practically disable
4303 // vectorization with such operations.
4304 Cost = 3000000;
4305 }
4306
4307 return Cost;
4308}
4309
4310InstructionCost LoopVectorizationCostModel::getConsecutiveMemOpCost(
4311 Instruction *I, ElementCount VF, InstWidening Kind) {
4312 assert((Kind == CM_Widen || Kind == CM_Widen_Reverse) &&
4313 "Expected a consecutive widening decision");
4314 Type *ValTy = getLoadStoreType(I);
4315 auto *VectorTy = cast<VectorType>(Val: toVectorTy(Scalar: ValTy, EC: VF));
4316 unsigned AS = getLoadStoreAddressSpace(I);
4317
4318 const Align Alignment = getLoadStoreAlignment(I);
4319 InstructionCost Cost = 0;
4320 if (isMaskRequired(I)) {
4321 unsigned IID = I->getOpcode() == Instruction::Load
4322 ? Intrinsic::masked_load
4323 : Intrinsic::masked_store;
4324 Cost += TTI.getMemIntrinsicInstrCost(
4325 MICA: MemIntrinsicCostAttributes(IID, VectorTy, Alignment, AS),
4326 CostKind: Config.CostKind);
4327 } else {
4328 TTI::OperandValueInfo OpInfo = TTI::getOperandInfo(V: I->getOperand(i: 0));
4329 Cost += TTI.getMemoryOpCost(Opcode: I->getOpcode(), Src: VectorTy, Alignment, AddressSpace: AS,
4330 CostKind: Config.CostKind, OpdInfo: OpInfo, I);
4331 }
4332
4333 if (Kind == CM_Widen_Reverse)
4334 Cost += TTI.getShuffleCost(Kind: TargetTransformInfo::SK_Reverse, DstTy: VectorTy,
4335 SrcTy: VectorTy, CostKind: Config.CostKind, Mask: {}, Index: 0);
4336 return Cost;
4337}
4338
4339InstructionCost
4340LoopVectorizationCostModel::getUniformMemOpCost(Instruction *I,
4341 ElementCount VF) const {
4342 assert(isUniformMemOp(*I, VF));
4343
4344 Type *ValTy = getLoadStoreType(I);
4345 Type *PtrTy = getLoadStorePointerOperand(V: I)->getType();
4346 auto *VectorTy = cast<VectorType>(Val: toVectorTy(Scalar: ValTy, EC: VF));
4347 const Align Alignment = getLoadStoreAlignment(I);
4348 unsigned AS = getLoadStoreAddressSpace(I);
4349 if (isa<LoadInst>(Val: I)) {
4350 return TTI.getAddressComputationCost(PtrTy, SE: nullptr, Ptr: nullptr,
4351 CostKind: Config.CostKind) +
4352 TTI.getMemoryOpCost(Opcode: Instruction::Load, Src: ValTy, Alignment, AddressSpace: AS,
4353 CostKind: Config.CostKind) +
4354 TTI.getShuffleCost(Kind: TargetTransformInfo::SK_Broadcast, DstTy: VectorTy,
4355 SrcTy: VectorTy, CostKind: Config.CostKind);
4356 }
4357 StoreInst *SI = cast<StoreInst>(Val: I);
4358
4359 bool IsLoopInvariantStoreValue = Legal->isInvariant(V: SI->getValueOperand());
4360 // TODO: We have existing tests that request the cost of extracting element
4361 // VF.getKnownMinValue() - 1 from a scalable vector. This does not represent
4362 // the actual generated code, which involves extracting the last element of
4363 // a scalable vector where the lane to extract is unknown at compile time.
4364 InstructionCost Cost =
4365 TTI.getAddressComputationCost(PtrTy, SE: nullptr, Ptr: nullptr, CostKind: Config.CostKind) +
4366 TTI.getMemoryOpCost(Opcode: Instruction::Store, Src: ValTy, Alignment, AddressSpace: AS,
4367 CostKind: Config.CostKind);
4368 if (!IsLoopInvariantStoreValue)
4369 Cost += TTI.getIndexedVectorInstrCostFromEnd(Opcode: Instruction::ExtractElement,
4370 Val: VectorTy, CostKind: Config.CostKind, Index: 0);
4371 return Cost;
4372}
4373
4374InstructionCost
4375LoopVectorizationCostModel::getGatherScatterCost(Instruction *I,
4376 ElementCount VF) const {
4377 Type *ValTy = getLoadStoreType(I);
4378 auto *VectorTy = cast<VectorType>(Val: toVectorTy(Scalar: ValTy, EC: VF));
4379 const Align Alignment = getLoadStoreAlignment(I);
4380 Value *Ptr = getLoadStorePointerOperand(V: I);
4381 Type *PtrTy = Ptr->getType();
4382
4383 if (!isUniform(V: Ptr, VF))
4384 PtrTy = toVectorTy(Scalar: PtrTy, EC: VF);
4385
4386 unsigned IID = I->getOpcode() == Instruction::Load
4387 ? Intrinsic::masked_gather
4388 : Intrinsic::masked_scatter;
4389 return TTI.getAddressComputationCost(PtrTy, SE: nullptr, Ptr: nullptr,
4390 CostKind: Config.CostKind) +
4391 TTI.getMemIntrinsicInstrCost(
4392 MICA: MemIntrinsicCostAttributes(IID, VectorTy, Ptr, isMaskRequired(I),
4393 Alignment, I),
4394 CostKind: Config.CostKind);
4395}
4396
4397InstructionCost
4398LoopVectorizationCostModel::getInterleaveGroupCost(Instruction *I,
4399 ElementCount VF) const {
4400 const auto *Group = getInterleavedAccessGroup(Instr: I);
4401 assert(Group && "Fail to get an interleaved access group.");
4402
4403 Instruction *InsertPos = Group->getInsertPos();
4404 Type *ValTy = getLoadStoreType(I: InsertPos);
4405 auto *VectorTy = cast<VectorType>(Val: toVectorTy(Scalar: ValTy, EC: VF));
4406 unsigned AS = getLoadStoreAddressSpace(I: InsertPos);
4407
4408 unsigned InterleaveFactor = Group->getFactor();
4409 auto *WideVecTy = VectorType::get(ElementType: ValTy, EC: VF * InterleaveFactor);
4410
4411 // Holds the indices of existing members in the interleaved group.
4412 SmallVector<unsigned, 4> Indices;
4413 for (unsigned IF = 0; IF < InterleaveFactor; IF++)
4414 if (Group->getMember(Index: IF))
4415 Indices.push_back(Elt: IF);
4416
4417 // Calculate the cost of the whole interleaved group.
4418 bool UseMaskForGaps =
4419 (Group->requiresScalarEpilogue() && !isEpilogueAllowed()) ||
4420 (isa<StoreInst>(Val: I) && !Group->isFull());
4421 InstructionCost Cost = TTI.getInterleavedMemoryOpCost(
4422 Opcode: InsertPos->getOpcode(), VecTy: WideVecTy, Factor: Group->getFactor(), Indices,
4423 Alignment: Group->getAlign(), AddressSpace: AS, CostKind: Config.CostKind, UseMaskForCond: isMaskRequired(I),
4424 UseMaskForGaps);
4425
4426 if (Group->isReverse()) {
4427 // TODO: Add support for reversed masked interleaved access.
4428 assert(!isMaskRequired(I) &&
4429 "Reverse masked interleaved access not supported.");
4430 Cost += Group->getNumMembers() *
4431 TTI.getShuffleCost(Kind: TargetTransformInfo::SK_Reverse, DstTy: VectorTy,
4432 SrcTy: VectorTy, CostKind: Config.CostKind, Mask: {}, Index: 0);
4433 }
4434 return Cost;
4435}
4436
4437InstructionCost
4438LoopVectorizationCostModel::getMemoryInstructionCost(Instruction *I,
4439 ElementCount VF) {
4440 // Calculate scalar cost only. Vectorization cost should be ready at this
4441 // moment.
4442 if (VF.isScalar()) {
4443 Type *ValTy = getLoadStoreType(I);
4444 Type *PtrTy = getLoadStorePointerOperand(V: I)->getType();
4445 const Align Alignment = getLoadStoreAlignment(I);
4446 unsigned AS = getLoadStoreAddressSpace(I);
4447
4448 TTI::OperandValueInfo OpInfo = TTI::getOperandInfo(V: I->getOperand(i: 0));
4449 return TTI.getAddressComputationCost(PtrTy, SE: nullptr, Ptr: nullptr,
4450 CostKind: Config.CostKind) +
4451 TTI.getMemoryOpCost(Opcode: I->getOpcode(), Src: ValTy, Alignment, AddressSpace: AS,
4452 CostKind: Config.CostKind, OpdInfo: OpInfo, I);
4453 }
4454 return getWideningCost(I, VF);
4455}
4456
4457InstructionCost
4458LoopVectorizationCostModel::getScalarizationOverhead(Instruction *I,
4459 ElementCount VF) const {
4460
4461 // There is no mechanism yet to create a scalable scalarization loop,
4462 // so this is currently Invalid.
4463 if (VF.isScalable())
4464 return InstructionCost::getInvalid();
4465
4466 if (VF.isScalar())
4467 return 0;
4468
4469 InstructionCost Cost = 0;
4470 Type *RetTy = toVectorizedTy(Ty: I->getType(), EC: VF);
4471 if (!RetTy->isVoidTy() &&
4472 (!isa<LoadInst>(Val: I) || !TTI.supportsEfficientVectorElementLoadStore())) {
4473
4474 TTI::VectorInstrContext VIC = TTI::VectorInstrContext::None;
4475 if (isa<LoadInst>(Val: I))
4476 VIC = TTI::VectorInstrContext::Load;
4477 else if (isa<StoreInst>(Val: I))
4478 VIC = TTI::VectorInstrContext::Store;
4479
4480 for (Type *VectorTy : getContainedTypes(Ty: RetTy)) {
4481 Cost += TTI.getScalarizationOverhead(
4482 Ty: cast<VectorType>(Val: VectorTy), DemandedElts: APInt::getAllOnes(numBits: VF.getFixedValue()),
4483 /*Insert=*/true, /*Extract=*/false, CostKind: Config.CostKind,
4484 /*ForPoisonSrc=*/true, VL: {}, VIC);
4485 }
4486 }
4487
4488 // Some targets keep addresses scalar.
4489 if (isa<LoadInst>(Val: I) && !TTI.prefersVectorizedAddressing())
4490 return Cost;
4491
4492 // Some targets support efficient element stores.
4493 if (isa<StoreInst>(Val: I) && TTI.supportsEfficientVectorElementLoadStore())
4494 return Cost;
4495
4496 // Collect operands to consider.
4497 CallInst *CI = dyn_cast<CallInst>(Val: I);
4498 Instruction::op_range Ops = CI ? CI->args() : I->operands();
4499
4500 // Skip operands that do not require extraction/scalarization and do not incur
4501 // any overhead.
4502 SmallVector<Type *> Tys;
4503 for (auto *V : filterExtractingOperands(Ops, VF))
4504 Tys.push_back(Elt: maybeVectorizeType(Ty: V->getType(), VF));
4505
4506 TTI::VectorInstrContext OperandVIC = isa<StoreInst>(Val: I)
4507 ? TTI::VectorInstrContext::Store
4508 : TTI::VectorInstrContext::None;
4509 return Cost +
4510 TTI.getOperandsScalarizationOverhead(Tys, CostKind: Config.CostKind, VIC: OperandVIC);
4511}
4512
4513void LoopVectorizationCostModel::setCostBasedWideningDecision(ElementCount VF) {
4514 if (VF.isScalar())
4515 return;
4516
4517 // TODO: We should generate better code and update the cost model for
4518 // predicated uniform stores. Today they are treated as any other
4519 // predicated store (see added test cases in
4520 // invariant-store-vectorization.ll).
4521 NumPredStores = 0;
4522 for (BasicBlock *BB : TheLoop->blocks())
4523 for (Instruction &I : *BB)
4524 if (isa<StoreInst>(Val: &I) && isScalarWithPredication(I: &I, VF))
4525 ++NumPredStores;
4526
4527 for (BasicBlock *BB : TheLoop->blocks()) {
4528 // For each instruction in the old loop.
4529 for (Instruction &I : *BB) {
4530 Value *Ptr = getLoadStorePointerOperand(V: &I);
4531 if (!Ptr)
4532 continue;
4533
4534 LLVM_DEBUG(dbgs() << "LV: Memory widening: calculating best strategy for "
4535 << I << '\n');
4536 if (isUniformMemOp(I, VF)) {
4537 auto IsLegalToScalarize = [&]() {
4538 if (!VF.isScalable())
4539 // Scalarization of fixed length vectors "just works".
4540 return true;
4541
4542 // We have dedicated lowering for unpredicated uniform loads and
4543 // stores. Note that even with tail folding we know that at least
4544 // one lane is active (i.e. generalized predication is not possible
4545 // here), and the logic below depends on this fact.
4546 if (!foldTailByMasking())
4547 return true;
4548
4549 // For scalable vectors, a uniform memop load is always
4550 // uniform-by-parts and we know how to scalarize that.
4551 if (isa<LoadInst>(Val: I))
4552 return true;
4553
4554 // A uniform store isn't neccessarily uniform-by-part
4555 // and we can't assume scalarization.
4556 auto &SI = cast<StoreInst>(Val&: I);
4557 return TheLoop->isLoopInvariant(V: SI.getValueOperand());
4558 };
4559
4560 const InstructionCost GatherScatterCost =
4561 isLegalGatherOrScatter(I: &I, VF) ? getGatherScatterCost(I: &I, VF)
4562 : InstructionCost::getInvalid();
4563
4564 // Load: Scalar load + broadcast
4565 // Store: Scalar store + isLoopInvariantStoreValue ? 0 : extract
4566 // FIXME: This cost is a significant under-estimate for tail folded
4567 // memory ops.
4568 const InstructionCost ScalarizationCost =
4569 IsLegalToScalarize() ? getUniformMemOpCost(I: &I, VF)
4570 : InstructionCost::getInvalid();
4571
4572 // Choose better solution for the current VF, Note that Invalid
4573 // costs compare as maximumal large. If both are invalid, we get
4574 // scalable invalid which signals a failure and a vectorization abort.
4575 LLVM_DEBUG(dbgs() << "LV: Memory widening: uniform memory op has "
4576 "GatherScatterCost = "
4577 << GatherScatterCost << ", ScalarizationCost = "
4578 << ScalarizationCost << '\n');
4579 if (GatherScatterCost < ScalarizationCost)
4580 setWideningDecision(I: &I, VF, W: CM_GatherScatter, Cost: GatherScatterCost);
4581 else
4582 setWideningDecision(I: &I, VF, W: CM_Scalarize, Cost: ScalarizationCost);
4583 continue;
4584 }
4585
4586 // We assume that widening is the best solution when possible.
4587 if (std::optional<InstWidening> Decision =
4588 memoryInstructionCanBeWidened(I: &I, VF)) {
4589 InstructionCost WidenCost = getConsecutiveMemOpCost(I: &I, VF, Kind: *Decision);
4590 LLVM_DEBUG(
4591 dbgs() << "LV: Memory widening: can be widened normally with cost "
4592 << WidenCost << '\n');
4593 setWideningDecision(I: &I, VF, W: *Decision, Cost: WidenCost);
4594 continue;
4595 }
4596
4597 // Choose between Interleaving, Gather/Scatter or Scalarization.
4598 InstructionCost InterleaveCost = InstructionCost::getInvalid();
4599 unsigned NumAccesses = 1;
4600 if (isAccessInterleaved(Instr: &I)) {
4601 const auto *Group = getInterleavedAccessGroup(Instr: &I);
4602 assert(Group && "Fail to get an interleaved access group.");
4603
4604 // Make one decision for the whole group.
4605 if (getWideningDecision(I: &I, VF) != CM_Unknown)
4606 continue;
4607
4608 NumAccesses = Group->getNumMembers();
4609 if (interleavedAccessCanBeWidened(I: &I, VF))
4610 InterleaveCost = getInterleaveGroupCost(I: &I, VF);
4611 }
4612
4613 InstructionCost GatherScatterCost =
4614 isLegalGatherOrScatter(I: &I, VF)
4615 ? getGatherScatterCost(I: &I, VF) * NumAccesses
4616 : InstructionCost::getInvalid();
4617
4618 InstructionCost ScalarizationCost =
4619 getMemInstScalarizationCost(I: &I, VF) * NumAccesses;
4620
4621 // Choose better solution for the current VF,
4622 // write down this decision and use it during vectorization.
4623 InstructionCost Cost;
4624 InstWidening Decision;
4625 if (InterleaveCost <= GatherScatterCost &&
4626 InterleaveCost < ScalarizationCost) {
4627 Decision = CM_Interleave;
4628 Cost = InterleaveCost;
4629 } else if (GatherScatterCost < ScalarizationCost) {
4630 Decision = CM_GatherScatter;
4631 Cost = GatherScatterCost;
4632 } else {
4633 Decision = CM_Scalarize;
4634 Cost = ScalarizationCost;
4635 }
4636 LLVM_DEBUG(
4637 dbgs() << "LV: Memory widening: InterleaveCost = " << InterleaveCost
4638 << ", GatherScatterCost = " << GatherScatterCost
4639 << ", ScalarizationCost = " << ScalarizationCost << '\n');
4640
4641 // If the instructions belongs to an interleave group, the whole group
4642 // receives the same decision. The whole group receives the cost, but
4643 // the cost will actually be assigned to one instruction.
4644 if (const auto *Group = getInterleavedAccessGroup(Instr: &I)) {
4645 if (Decision == CM_Scalarize) {
4646 for (Instruction *I : Group->members())
4647 setWideningDecision(I, VF, W: Decision,
4648 Cost: getMemInstScalarizationCost(I, VF));
4649 } else {
4650 setWideningDecision(Grp: Group, VF, W: Decision, Cost);
4651 }
4652 } else
4653 setWideningDecision(I: &I, VF, W: Decision, Cost);
4654 }
4655 }
4656
4657 // Make sure that any load of address and any other address computation
4658 // remains scalar unless there is gather/scatter support. This avoids
4659 // inevitable extracts into address registers, and also has the benefit of
4660 // activating LSR more, since that pass can't optimize vectorized
4661 // addresses.
4662 if (TTI.prefersVectorizedAddressing())
4663 return;
4664
4665 // Start with all scalar pointer uses.
4666 SmallSetVector<Instruction *, 8> AddrDefs;
4667 for (BasicBlock *BB : TheLoop->blocks())
4668 for (Instruction &I : *BB) {
4669 Instruction *PtrDef =
4670 dyn_cast_or_null<Instruction>(Val: getLoadStorePointerOperand(V: &I));
4671 if (PtrDef && TheLoop->contains(Inst: PtrDef) &&
4672 getWideningDecision(I: &I, VF) != CM_GatherScatter)
4673 AddrDefs.insert(X: PtrDef);
4674 }
4675
4676 // Add all instructions used to generate the addresses.
4677 SmallVector<Instruction *, 4> Worklist;
4678 append_range(C&: Worklist, R&: AddrDefs);
4679 while (!Worklist.empty()) {
4680 Instruction *I = Worklist.pop_back_val();
4681 for (auto &Op : I->operands())
4682 if (auto *InstOp = dyn_cast<Instruction>(Val&: Op))
4683 if (TheLoop->contains(Inst: InstOp) && !isa<PHINode>(Val: InstOp) &&
4684 AddrDefs.insert(X: InstOp))
4685 Worklist.push_back(Elt: InstOp);
4686 }
4687
4688 auto UpdateMemOpUserCost = [this, VF](LoadInst *LI) {
4689 // If there are direct memory op users of the newly scalarized load,
4690 // their cost may have changed because there's no scalarization
4691 // overhead for the operand. Update it.
4692 for (User *U : LI->users()) {
4693 if (!isa<LoadInst, StoreInst>(Val: U))
4694 continue;
4695 if (getWideningDecision(I: cast<Instruction>(Val: U), VF) != CM_Scalarize)
4696 continue;
4697 auto UI = cast<Instruction>(Val: U);
4698 LLVM_DEBUG(
4699 dbgs() << "LV: Memory widening: updating decision for load user "
4700 << *UI << '\n');
4701 setWideningDecision(
4702 I: UI, VF, W: CM_Scalarize,
4703 Cost: getMemInstScalarizationCost(I: cast<Instruction>(Val: U), VF));
4704 }
4705 };
4706 for (auto *I : AddrDefs) {
4707 if (isa<LoadInst>(Val: I)) {
4708 // Setting the desired widening decision should ideally be handled in
4709 // by cost functions, but since this involves the task of finding out
4710 // if the loaded register is involved in an address computation, it is
4711 // instead changed here when we know this is the case.
4712 InstWidening Decision = getWideningDecision(I, VF);
4713 if (!isPredicatedInst(I) &&
4714 (Decision == CM_Widen || Decision == CM_Widen_Reverse ||
4715 (!isUniformMemOp(I&: *I, VF) && Decision == CM_Scalarize))) {
4716 // Scalarize a widened load of address or update the cost of a scalar
4717 // load of an address.
4718 LLVM_DEBUG(dbgs() << "LV: Memory widening: updating decision for load "
4719 << *I << '\n');
4720 setWideningDecision(
4721 I, VF, W: CM_Scalarize,
4722 Cost: (VF.getKnownMinValue() *
4723 getMemoryInstructionCost(I, VF: ElementCount::getFixed(MinVal: 1))));
4724 UpdateMemOpUserCost(cast<LoadInst>(Val: I));
4725 } else if (const auto *Group = getInterleavedAccessGroup(Instr: I)) {
4726 // Scalarize all members of this interleaved group when any member
4727 // is used as an address. The address-used load skips scalarization
4728 // overhead, other members include it.
4729 for (Instruction *Member : Group->members()) {
4730 InstructionCost Cost = AddrDefs.contains(key: Member)
4731 ? (VF.getKnownMinValue() *
4732 getMemoryInstructionCost(
4733 I: Member, VF: ElementCount::getFixed(MinVal: 1)))
4734 : getMemInstScalarizationCost(I: Member, VF);
4735 LLVM_DEBUG(
4736 dbgs()
4737 << "LV: Memory widening: updating decision for interleave member "
4738 << *Member << '\n');
4739 setWideningDecision(I: Member, VF, W: CM_Scalarize, Cost);
4740 UpdateMemOpUserCost(cast<LoadInst>(Val: Member));
4741 }
4742 }
4743 } else {
4744 // Cannot scalarize fixed-order recurrence phis at the moment.
4745 if (isa<PHINode>(Val: I) && Legal->isFixedOrderRecurrence(Phi: cast<PHINode>(Val: I)))
4746 continue;
4747
4748 // Make sure I gets scalarized and a cost estimate without
4749 // scalarization overhead.
4750 ForcedScalars[VF].insert(X: I);
4751 }
4752 }
4753}
4754
4755bool LoopVectorizationCostModel::shouldConsiderInvariant(Value *Op) {
4756 if (!Legal->isInvariant(V: Op))
4757 return false;
4758 // Consider Op invariant, if it or its operands aren't predicated
4759 // instruction in the loop. In that case, it is not trivially hoistable.
4760 auto *OpI = dyn_cast<Instruction>(Val: Op);
4761 return !OpI || !TheLoop->contains(Inst: OpI) ||
4762 (!isPredicatedInst(I: OpI) &&
4763 (!isa<PHINode>(Val: OpI) || OpI->getParent() != TheLoop->getHeader()) &&
4764 all_of(Range: OpI->operands(),
4765 P: [this](Value *Op) { return shouldConsiderInvariant(Op); }));
4766}
4767
4768InstructionCost
4769LoopVectorizationCostModel::getInstructionCost(Instruction *I,
4770 ElementCount VF) {
4771 // If we know that this instruction will remain uniform, check the cost of
4772 // the scalar version.
4773 if (isUniformAfterVectorization(I, VF))
4774 VF = ElementCount::getFixed(MinVal: 1);
4775
4776 if (VF.isVector() && isProfitableToScalarize(I, VF))
4777 return InstsToScalarize[VF][I];
4778
4779 // Forced scalars do not have any scalarization overhead.
4780 auto ForcedScalar = ForcedScalars.find(Val: VF);
4781 if (VF.isVector() && ForcedScalar != ForcedScalars.end()) {
4782 auto InstSet = ForcedScalar->second;
4783 if (InstSet.count(key: I))
4784 return getInstructionCost(I, VF: ElementCount::getFixed(MinVal: 1)) *
4785 VF.getKnownMinValue();
4786 }
4787
4788 const auto &MinBWs = Config.getMinimalBitwidths();
4789 uint64_t InstrMinBWs = MinBWs.lookup(Key: I);
4790 Type *RetTy = I->getType();
4791 if (canTruncateToMinimalBitwidth(I, VF))
4792 RetTy = IntegerType::get(C&: RetTy->getContext(), NumBits: InstrMinBWs);
4793 auto *SE = PSE.getSE();
4794
4795 Type *VectorTy;
4796 if (isScalarAfterVectorization(I, VF)) {
4797 [[maybe_unused]] auto HasSingleCopyAfterVectorization =
4798 [this](Instruction *I, ElementCount VF) -> bool {
4799 if (VF.isScalar())
4800 return true;
4801
4802 auto Scalarized = InstsToScalarize.find(Key: VF);
4803 assert(Scalarized != InstsToScalarize.end() &&
4804 "VF not yet analyzed for scalarization profitability");
4805 return !Scalarized->second.count(Key: I) &&
4806 llvm::all_of(Range: I->users(), P: [&](User *U) {
4807 auto *UI = cast<Instruction>(Val: U);
4808 return !Scalarized->second.count(Key: UI);
4809 });
4810 };
4811
4812 // With the exception of GEPs and PHIs, after scalarization there should
4813 // only be one copy of the instruction generated in the loop. This is
4814 // because the VF is either 1, or any instructions that need scalarizing
4815 // have already been dealt with by the time we get here. As a result,
4816 // it means we don't have to multiply the instruction cost by VF.
4817 assert(I->getOpcode() == Instruction::GetElementPtr ||
4818 I->getOpcode() == Instruction::PHI ||
4819 (I->getOpcode() == Instruction::BitCast &&
4820 I->getType()->isPointerTy()) ||
4821 HasSingleCopyAfterVectorization(I, VF));
4822 VectorTy = RetTy;
4823 } else
4824 VectorTy = toVectorizedTy(Ty: RetTy, EC: VF);
4825
4826 if (VF.isVector() && VectorTy->isVectorTy() &&
4827 !TTI.getNumberOfParts(Tp: VectorTy))
4828 return InstructionCost::getInvalid();
4829
4830 // TODO: We need to estimate the cost of intrinsic calls.
4831 switch (I->getOpcode()) {
4832 case Instruction::GetElementPtr:
4833 // We mark this instruction as zero-cost because the cost of GEPs in
4834 // vectorized code depends on whether the corresponding memory instruction
4835 // is scalarized or not. Therefore, we handle GEPs with the memory
4836 // instruction cost.
4837 return 0;
4838 case Instruction::UncondBr:
4839 case Instruction::CondBr: {
4840 // In cases of scalarized and predicated instructions, there will be VF
4841 // predicated blocks in the vectorized loop. Each branch around these
4842 // blocks requires also an extract of its vector compare i1 element.
4843 // Note that the conditional branch from the loop latch will be replaced by
4844 // a single branch controlling the loop, so there is no extra overhead from
4845 // scalarization.
4846 bool ScalarPredicatedBB = false;
4847 CondBrInst *BI = dyn_cast<CondBrInst>(Val: I);
4848 if (VF.isVector() && BI &&
4849 (PredicatedBBsAfterVectorization[VF].count(Ptr: BI->getSuccessor(i: 0)) ||
4850 PredicatedBBsAfterVectorization[VF].count(Ptr: BI->getSuccessor(i: 1))) &&
4851 BI->getParent() != TheLoop->getLoopLatch())
4852 ScalarPredicatedBB = true;
4853
4854 if (ScalarPredicatedBB) {
4855 // Not possible to scalarize scalable vector with predicated instructions.
4856 if (VF.isScalable())
4857 return InstructionCost::getInvalid();
4858 // Return cost for branches around scalarized and predicated blocks.
4859 auto *VecI1Ty =
4860 VectorType::get(ElementType: IntegerType::getInt1Ty(C&: RetTy->getContext()), EC: VF);
4861 return (TTI.getScalarizationOverhead(
4862 Ty: VecI1Ty, DemandedElts: APInt::getAllOnes(numBits: VF.getFixedValue()),
4863 /*Insert*/ false, /*Extract*/ true, CostKind: Config.CostKind) +
4864 (TTI.getCFInstrCost(Opcode: Instruction::CondBr, CostKind: Config.CostKind) *
4865 VF.getFixedValue()));
4866 }
4867
4868 if (I->getParent() == TheLoop->getLoopLatch() || VF.isScalar())
4869 // The back-edge branch will remain, as will all scalar branches.
4870 return TTI.getCFInstrCost(Opcode: Instruction::UncondBr, CostKind: Config.CostKind);
4871
4872 // This branch will be eliminated by if-conversion.
4873 return 0;
4874 // Note: We currently assume zero cost for an unconditional branch inside
4875 // a predicated block since it will become a fall-through, although we
4876 // may decide in the future to call TTI for all branches.
4877 }
4878 case Instruction::Switch: {
4879 if (VF.isScalar())
4880 return TTI.getCFInstrCost(Opcode: Instruction::Switch, CostKind: Config.CostKind);
4881 auto *Switch = cast<SwitchInst>(Val: I);
4882 return Switch->getNumCases() *
4883 TTI.getCmpSelInstrCost(
4884 Opcode: Instruction::ICmp,
4885 ValTy: toVectorTy(Scalar: Switch->getCondition()->getType(), EC: VF),
4886 CondTy: toVectorTy(Scalar: Type::getInt1Ty(C&: I->getContext()), EC: VF),
4887 VecPred: CmpInst::ICMP_EQ, CostKind: Config.CostKind);
4888 }
4889 case Instruction::PHI: {
4890 auto *Phi = cast<PHINode>(Val: I);
4891
4892 // First-order recurrences are replaced by vector shuffles inside the loop.
4893 if (VF.isVector() && Legal->isFixedOrderRecurrence(Phi)) {
4894 return TTI.getShuffleCost(
4895 Kind: TargetTransformInfo::SK_Splice, DstTy: cast<VectorType>(Val: VectorTy),
4896 SrcTy: cast<VectorType>(Val: VectorTy), CostKind: Config.CostKind, Mask: {}, Index: -1);
4897 }
4898
4899 // Phi nodes in non-header blocks (not inductions, reductions, etc.) are
4900 // converted into select instructions. We require N - 1 selects per phi
4901 // node, where N is the number of incoming values.
4902 if (VF.isVector() && Phi->getParent() != TheLoop->getHeader()) {
4903 Type *ResultTy = Phi->getType();
4904
4905 // All instructions in an Any-of reduction chain are narrowed to bool.
4906 // Check if that is the case for this phi node.
4907 auto *HeaderUser = cast_if_present<PHINode>(
4908 Val: find_singleton<User>(Range: Phi->users(), P: [this](User *U, bool) -> User * {
4909 auto *Phi = dyn_cast<PHINode>(Val: U);
4910 if (Phi && Phi->getParent() == TheLoop->getHeader())
4911 return Phi;
4912 return nullptr;
4913 }));
4914 if (HeaderUser) {
4915 auto &ReductionVars = Legal->getReductionVars();
4916 auto Iter = ReductionVars.find(Key: HeaderUser);
4917 if (Iter != ReductionVars.end() &&
4918 RecurrenceDescriptor::isAnyOfRecurrenceKind(
4919 Kind: Iter->second.getRecurrenceKind()))
4920 ResultTy = Type::getInt1Ty(C&: Phi->getContext());
4921 }
4922 return (Phi->getNumIncomingValues() - 1) *
4923 TTI.getCmpSelInstrCost(
4924 Opcode: Instruction::Select, ValTy: toVectorTy(Scalar: ResultTy, EC: VF),
4925 CondTy: toVectorTy(Scalar: Type::getInt1Ty(C&: Phi->getContext()), EC: VF),
4926 VecPred: CmpInst::BAD_ICMP_PREDICATE, CostKind: Config.CostKind);
4927 }
4928
4929 // When tail folding with EVL, if the phi is part of an out of loop
4930 // reduction then it will be transformed into a wide vp_merge.
4931 if (VF.isVector() && foldTailWithEVL() &&
4932 Legal->getReductionVars().contains(Key: Phi) &&
4933 !Config.isInLoopReduction(Phi)) {
4934 IntrinsicCostAttributes ICA(
4935 Intrinsic::vp_merge, toVectorTy(Scalar: Phi->getType(), EC: VF),
4936 {toVectorTy(Scalar: Type::getInt1Ty(C&: Phi->getContext()), EC: VF)});
4937 return TTI.getIntrinsicInstrCost(ICA, CostKind: Config.CostKind);
4938 }
4939
4940 return TTI.getCFInstrCost(Opcode: Instruction::PHI, CostKind: Config.CostKind);
4941 }
4942 case Instruction::UDiv:
4943 case Instruction::SDiv:
4944 case Instruction::URem:
4945 case Instruction::SRem:
4946 if (VF.isVector() && isPredicatedInst(I)) {
4947 const auto [ScalarCost, MaskedCost] = getDivRemSpeculationCost(I, VF);
4948 return isDivRemScalarWithPredication(ScalarCost, MaskedCost) ? ScalarCost
4949 : MaskedCost;
4950 }
4951 // We've proven all lanes safe to speculate, fall through.
4952 [[fallthrough]];
4953 case Instruction::Add:
4954 case Instruction::Sub: {
4955 auto Info = Legal->getHistogramInfo(I);
4956 if (Info && VF.isVector()) {
4957 const HistogramInfo *HGram = Info.value();
4958 // Assume that a non-constant update value (or a constant != 1) requires
4959 // a multiply, and add that into the cost.
4960 InstructionCost MulCost = TTI::TCC_Free;
4961 ConstantInt *RHS = dyn_cast<ConstantInt>(Val: I->getOperand(i: 1));
4962 if (!RHS || RHS->getZExtValue() != 1)
4963 MulCost = TTI.getArithmeticInstrCost(Opcode: Instruction::Mul, Ty: VectorTy,
4964 CostKind: Config.CostKind);
4965
4966 // Find the cost of the histogram operation itself.
4967 Type *PtrTy = VectorType::get(ElementType: HGram->Load->getPointerOperandType(), EC: VF);
4968 Type *ScalarTy = I->getType();
4969 Type *MaskTy = VectorType::get(ElementType: Type::getInt1Ty(C&: I->getContext()), EC: VF);
4970 IntrinsicCostAttributes ICA(Intrinsic::experimental_vector_histogram_add,
4971 Type::getVoidTy(C&: I->getContext()),
4972 {PtrTy, ScalarTy, MaskTy});
4973
4974 // Add the costs together with the add/sub operation.
4975 return TTI.getIntrinsicInstrCost(ICA, CostKind: Config.CostKind) + MulCost +
4976 TTI.getArithmeticInstrCost(Opcode: I->getOpcode(), Ty: VectorTy,
4977 CostKind: Config.CostKind);
4978 }
4979 [[fallthrough]];
4980 }
4981 case Instruction::FAdd:
4982 case Instruction::FSub:
4983 case Instruction::Mul:
4984 case Instruction::FMul:
4985 case Instruction::FDiv:
4986 case Instruction::FRem:
4987 case Instruction::Shl:
4988 case Instruction::LShr:
4989 case Instruction::AShr:
4990 case Instruction::And:
4991 case Instruction::Or:
4992 case Instruction::Xor: {
4993 // If we're speculating on the stride being 1, the multiplication may
4994 // fold away. We can generalize this for all operations using the notion
4995 // of neutral elements. (TODO)
4996 if (I->getOpcode() == Instruction::Mul &&
4997 ((TheLoop->isLoopInvariant(V: I->getOperand(i: 0)) &&
4998 PSE.getSCEV(V: I->getOperand(i: 0))->isOne()) ||
4999 (TheLoop->isLoopInvariant(V: I->getOperand(i: 1)) &&
5000 PSE.getSCEV(V: I->getOperand(i: 1))->isOne())))
5001 return 0;
5002
5003 // Certain instructions can be cheaper to vectorize if they have a constant
5004 // second vector operand. One example of this are shifts on x86.
5005 Value *Op2 = I->getOperand(i: 1);
5006 if (!isa<Constant>(Val: Op2) && TheLoop->isLoopInvariant(V: Op2) &&
5007 PSE.getSE()->isSCEVable(Ty: Op2->getType()) &&
5008 isa<SCEVConstant>(Val: PSE.getSCEV(V: Op2))) {
5009 Op2 = cast<SCEVConstant>(Val: PSE.getSCEV(V: Op2))->getValue();
5010 }
5011 auto Op2Info = TTI.getOperandInfo(V: Op2);
5012 if (Op2Info.Kind == TargetTransformInfo::OK_AnyValue &&
5013 shouldConsiderInvariant(Op: Op2))
5014 Op2Info.Kind = TargetTransformInfo::OK_UniformValue;
5015
5016 SmallVector<const Value *, 4> Operands(I->operand_values());
5017 return TTI.getArithmeticInstrCost(
5018 Opcode: I->getOpcode(), Ty: VectorTy, CostKind: Config.CostKind,
5019 Opd1Info: {.Kind: TargetTransformInfo::OK_AnyValue, .Properties: TargetTransformInfo::OP_None},
5020 Opd2Info: Op2Info, Args: Operands, CtxI: I, TLibInfo: TLI);
5021 }
5022 case Instruction::FNeg: {
5023 return TTI.getArithmeticInstrCost(
5024 Opcode: I->getOpcode(), Ty: VectorTy, CostKind: Config.CostKind,
5025 Opd1Info: {.Kind: TargetTransformInfo::OK_AnyValue, .Properties: TargetTransformInfo::OP_None},
5026 Opd2Info: {.Kind: TargetTransformInfo::OK_AnyValue, .Properties: TargetTransformInfo::OP_None},
5027 Args: I->getOperand(i: 0), CtxI: I);
5028 }
5029 case Instruction::Select: {
5030 SelectInst *SI = cast<SelectInst>(Val: I);
5031 const SCEV *CondSCEV = SE->getSCEV(V: SI->getCondition());
5032 bool ScalarCond = (SE->isLoopInvariant(S: CondSCEV, L: TheLoop));
5033
5034 const Value *Op0, *Op1;
5035 using namespace llvm::PatternMatch;
5036 if (!ScalarCond && (match(V: I, P: m_LogicalAnd(L: m_Value(V&: Op0), R: m_Value(V&: Op1))) ||
5037 match(V: I, P: m_LogicalOr(L: m_Value(V&: Op0), R: m_Value(V&: Op1))))) {
5038 // select x, y, false --> x & y
5039 // select x, true, y --> x | y
5040 const auto [Op1VK, Op1VP] = TTI::getOperandInfo(V: Op0);
5041 const auto [Op2VK, Op2VP] = TTI::getOperandInfo(V: Op1);
5042 assert(Op0->getType()->getScalarSizeInBits() == 1 &&
5043 Op1->getType()->getScalarSizeInBits() == 1);
5044
5045 return TTI.getArithmeticInstrCost(
5046 Opcode: match(V: I, P: m_LogicalOr()) ? Instruction::Or : Instruction::And,
5047 Ty: VectorTy, CostKind: Config.CostKind, Opd1Info: {.Kind: Op1VK, .Properties: Op1VP}, Opd2Info: {.Kind: Op2VK, .Properties: Op2VP}, Args: {Op0, Op1},
5048 CtxI: I);
5049 }
5050
5051 Type *CondTy = SI->getCondition()->getType();
5052 if (!ScalarCond)
5053 CondTy = VectorType::get(ElementType: CondTy, EC: VF);
5054
5055 CmpInst::Predicate Pred = CmpInst::BAD_ICMP_PREDICATE;
5056 if (auto *Cmp = dyn_cast<CmpInst>(Val: SI->getCondition()))
5057 Pred = Cmp->getPredicate();
5058 return TTI.getCmpSelInstrCost(
5059 Opcode: I->getOpcode(), ValTy: VectorTy, CondTy, VecPred: Pred, CostKind: Config.CostKind,
5060 Op1Info: {.Kind: TTI::OK_AnyValue, .Properties: TTI::OP_None}, Op2Info: {.Kind: TTI::OK_AnyValue, .Properties: TTI::OP_None}, I);
5061 }
5062 case Instruction::ICmp:
5063 case Instruction::FCmp: {
5064 Type *ValTy = I->getOperand(i: 0)->getType();
5065
5066 if (canTruncateToMinimalBitwidth(I, VF)) {
5067 [[maybe_unused]] Instruction *Op0AsInstruction =
5068 dyn_cast<Instruction>(Val: I->getOperand(i: 0));
5069 assert((!canTruncateToMinimalBitwidth(Op0AsInstruction, VF) ||
5070 InstrMinBWs == MinBWs.lookup(Op0AsInstruction)) &&
5071 "if both the operand and the compare are marked for "
5072 "truncation, they must have the same bitwidth");
5073 ValTy = IntegerType::get(C&: ValTy->getContext(), NumBits: InstrMinBWs);
5074 }
5075
5076 VectorTy = toVectorTy(Scalar: ValTy, EC: VF);
5077 return TTI.getCmpSelInstrCost(
5078 Opcode: I->getOpcode(), ValTy: VectorTy, CondTy: CmpInst::makeCmpResultType(opnd_type: VectorTy),
5079 VecPred: cast<CmpInst>(Val: I)->getPredicate(), CostKind: Config.CostKind,
5080 Op1Info: {.Kind: TTI::OK_AnyValue, .Properties: TTI::OP_None}, Op2Info: {.Kind: TTI::OK_AnyValue, .Properties: TTI::OP_None}, I);
5081 }
5082 case Instruction::Store:
5083 case Instruction::Load: {
5084 ElementCount Width = VF;
5085 if (Width.isVector()) {
5086 InstWidening Decision = getWideningDecision(I, VF: Width);
5087 assert(Decision != CM_Unknown &&
5088 "CM decision should be taken at this point");
5089 if (getWideningCost(I, VF) == InstructionCost::getInvalid())
5090 return InstructionCost::getInvalid();
5091 if (Decision == CM_Scalarize)
5092 Width = ElementCount::getFixed(MinVal: 1);
5093 }
5094 VectorTy = toVectorTy(Scalar: getLoadStoreType(I), EC: Width);
5095 return getMemoryInstructionCost(I, VF);
5096 }
5097 case Instruction::BitCast:
5098 if (I->getType()->isPointerTy())
5099 return 0;
5100 [[fallthrough]];
5101 case Instruction::ZExt:
5102 case Instruction::SExt:
5103 case Instruction::FPToUI:
5104 case Instruction::FPToSI:
5105 case Instruction::FPExt:
5106 case Instruction::PtrToInt:
5107 case Instruction::IntToPtr:
5108 case Instruction::SIToFP:
5109 case Instruction::UIToFP:
5110 case Instruction::Trunc:
5111 case Instruction::FPTrunc: {
5112 // Computes the CastContextHint from a Load/Store instruction.
5113 auto ComputeCCH = [&](Instruction *I) -> TTI::CastContextHint {
5114 assert((isa<LoadInst>(I) || isa<StoreInst>(I)) &&
5115 "Expected a load or a store!");
5116
5117 if (VF.isScalar() || !TheLoop->contains(Inst: I))
5118 return TTI::CastContextHint::Normal;
5119
5120 switch (getWideningDecision(I, VF)) {
5121 case LoopVectorizationCostModel::CM_GatherScatter:
5122 return TTI::CastContextHint::GatherScatter;
5123 case LoopVectorizationCostModel::CM_Interleave:
5124 return TTI::CastContextHint::Interleave;
5125 case LoopVectorizationCostModel::CM_Scalarize:
5126 case LoopVectorizationCostModel::CM_Widen:
5127 return isPredicatedInst(I) ? TTI::CastContextHint::Masked
5128 : TTI::CastContextHint::Normal;
5129 case LoopVectorizationCostModel::CM_Widen_Reverse:
5130 return TTI::CastContextHint::Reversed;
5131 case LoopVectorizationCostModel::CM_Unknown:
5132 llvm_unreachable("Instr did not go through cost modelling?");
5133 case LoopVectorizationCostModel::CM_InvalidatedDecision:
5134 return TTI::CastContextHint::None;
5135 }
5136
5137 llvm_unreachable("Unhandled case!");
5138 };
5139
5140 unsigned Opcode = I->getOpcode();
5141 TTI::CastContextHint CCH = TTI::CastContextHint::None;
5142 // For Trunc, the context is the only user, which must be a StoreInst.
5143 if (Opcode == Instruction::Trunc || Opcode == Instruction::FPTrunc) {
5144 if (I->hasOneUse())
5145 if (StoreInst *Store = dyn_cast<StoreInst>(Val: *I->user_begin()))
5146 CCH = ComputeCCH(Store);
5147 }
5148 // For Z/Sext, the context is the operand, which must be a LoadInst.
5149 else if (Opcode == Instruction::ZExt || Opcode == Instruction::SExt ||
5150 Opcode == Instruction::FPExt) {
5151 if (LoadInst *Load = dyn_cast<LoadInst>(Val: I->getOperand(i: 0)))
5152 CCH = ComputeCCH(Load);
5153 }
5154
5155 // We optimize the truncation of induction variables having constant
5156 // integer steps. The cost of these truncations is the same as the scalar
5157 // operation.
5158 if (isOptimizableIVTruncate(I, VF)) {
5159 auto *Trunc = cast<TruncInst>(Val: I);
5160 return TTI.getCastInstrCost(Opcode: Instruction::Trunc, Dst: Trunc->getDestTy(),
5161 Src: Trunc->getSrcTy(), CCH, CostKind: Config.CostKind,
5162 I: Trunc);
5163 }
5164
5165 Type *SrcScalarTy = I->getOperand(i: 0)->getType();
5166 Instruction *Op0AsInstruction = dyn_cast<Instruction>(Val: I->getOperand(i: 0));
5167 if (canTruncateToMinimalBitwidth(I: Op0AsInstruction, VF))
5168 SrcScalarTy = IntegerType::get(C&: SrcScalarTy->getContext(),
5169 NumBits: MinBWs.lookup(Key: Op0AsInstruction));
5170 Type *SrcVecTy =
5171 VectorTy->isVectorTy() ? toVectorTy(Scalar: SrcScalarTy, EC: VF) : SrcScalarTy;
5172
5173 if (canTruncateToMinimalBitwidth(I, VF)) {
5174 // If the result type is <= the source type, there will be no extend
5175 // after truncating the users to the minimal required bitwidth.
5176 if (VectorTy->getScalarSizeInBits() <= SrcVecTy->getScalarSizeInBits() &&
5177 (I->getOpcode() == Instruction::ZExt ||
5178 I->getOpcode() == Instruction::SExt))
5179 return 0;
5180 }
5181
5182 return TTI.getCastInstrCost(Opcode, Dst: VectorTy, Src: SrcVecTy, CCH,
5183 CostKind: Config.CostKind, I);
5184 }
5185 case Instruction::Call:
5186 return getVectorCallCost(CI: cast<CallInst>(Val: I), VF);
5187 case Instruction::ExtractValue:
5188 return TTI.getInstructionCost(U: I, CostKind: Config.CostKind);
5189 case Instruction::Alloca:
5190 // We cannot easily widen alloca to a scalable alloca, as
5191 // the result would need to be a vector of pointers.
5192 if (VF.isScalable())
5193 return InstructionCost::getInvalid();
5194 return TTI.getArithmeticInstrCost(Opcode: Instruction::Mul, Ty: RetTy, CostKind: Config.CostKind);
5195 case Instruction::Freeze:
5196 return TTI::TCC_Free;
5197 default:
5198 // This opcode is unknown. Assume that it is the same as 'mul'.
5199 return TTI.getArithmeticInstrCost(Opcode: Instruction::Mul, Ty: VectorTy,
5200 CostKind: Config.CostKind);
5201 } // end of switch.
5202}
5203
5204void LoopVectorizationCostModel::collectValuesToIgnore() {
5205 // Ignore ephemeral values.
5206 CodeMetrics::collectEphemeralValues(L: TheLoop, AC, EphValues&: ValuesToIgnore);
5207
5208 SmallVector<Value *, 4> DeadInterleavePointerOps;
5209 SmallVector<Value *, 4> DeadOps;
5210
5211 // If a scalar epilogue is required, users outside the loop won't use
5212 // live-outs from the vector loop but from the scalar epilogue. Ignore them if
5213 // that is the case.
5214 bool RequiresScalarEpilogue = requiresScalarEpilogue(IsVectorizing: true);
5215 auto IsLiveOutDead = [this, RequiresScalarEpilogue](User *U) {
5216 return RequiresScalarEpilogue &&
5217 !TheLoop->contains(BB: cast<Instruction>(Val: U)->getParent());
5218 };
5219
5220 LoopBlocksDFS DFS(TheLoop);
5221 DFS.perform(LI);
5222 for (BasicBlock *BB : reverse(C: make_range(x: DFS.beginRPO(), y: DFS.endRPO())))
5223 for (Instruction &I : reverse(C&: *BB)) {
5224 if (VecValuesToIgnore.contains(Ptr: &I) || ValuesToIgnore.contains(Ptr: &I))
5225 continue;
5226
5227 // Add instructions that would be trivially dead and are only used by
5228 // values already ignored to DeadOps to seed worklist.
5229 if (wouldInstructionBeTriviallyDead(I: &I, TLI) &&
5230 all_of(Range: I.users(), P: [this, IsLiveOutDead](User *U) {
5231 return VecValuesToIgnore.contains(Ptr: U) ||
5232 ValuesToIgnore.contains(Ptr: U) || IsLiveOutDead(U);
5233 }))
5234 DeadOps.push_back(Elt: &I);
5235
5236 // For interleave groups, we only create a pointer for the start of the
5237 // interleave group. Queue up addresses of group members except the insert
5238 // position for further processing.
5239 if (isAccessInterleaved(Instr: &I)) {
5240 auto *Group = getInterleavedAccessGroup(Instr: &I);
5241 if (Group->getInsertPos() == &I)
5242 continue;
5243 Value *PointerOp = getLoadStorePointerOperand(V: &I);
5244 DeadInterleavePointerOps.push_back(Elt: PointerOp);
5245 }
5246
5247 // Queue branches for analysis. They are dead, if their successors only
5248 // contain dead instructions.
5249 if (isa<CondBrInst>(Val: &I))
5250 DeadOps.push_back(Elt: &I);
5251 }
5252
5253 // Mark ops feeding interleave group members as free, if they are only used
5254 // by other dead computations.
5255 for (unsigned I = 0; I != DeadInterleavePointerOps.size(); ++I) {
5256 auto *Op = dyn_cast<Instruction>(Val: DeadInterleavePointerOps[I]);
5257 if (!Op || !TheLoop->contains(Inst: Op) || any_of(Range: Op->users(), P: [this](User *U) {
5258 Instruction *UI = cast<Instruction>(Val: U);
5259 return !VecValuesToIgnore.contains(Ptr: U) &&
5260 (!isAccessInterleaved(Instr: UI) ||
5261 getInterleavedAccessGroup(Instr: UI)->getInsertPos() == UI);
5262 }))
5263 continue;
5264 VecValuesToIgnore.insert(Ptr: Op);
5265 append_range(C&: DeadInterleavePointerOps, R: Op->operands());
5266 }
5267
5268 // Mark ops that would be trivially dead and are only used by ignored
5269 // instructions as free.
5270 BasicBlock *Header = TheLoop->getHeader();
5271
5272 // Returns true if the block contains only dead instructions. Such blocks will
5273 // be removed by VPlan-to-VPlan transforms and won't be considered by the
5274 // VPlan-based cost model, so skip them in the legacy cost-model as well.
5275 auto IsEmptyBlock = [this](BasicBlock *BB) {
5276 return all_of(Range&: *BB, P: [this](Instruction &I) {
5277 return ValuesToIgnore.contains(Ptr: &I) || VecValuesToIgnore.contains(Ptr: &I) ||
5278 isa<UncondBrInst>(Val: &I);
5279 });
5280 };
5281 SmallPtrSet<Instruction *, 16> ProcessedDeadOps;
5282 for (unsigned I = 0; I != DeadOps.size(); ++I) {
5283 auto *Op = dyn_cast<Instruction>(Val: DeadOps[I]);
5284
5285 // Check if the branch should be considered dead.
5286 if (auto *Br = dyn_cast_or_null<CondBrInst>(Val: Op)) {
5287 BasicBlock *ThenBB = Br->getSuccessor(i: 0);
5288 BasicBlock *ElseBB = Br->getSuccessor(i: 1);
5289 // Don't considers branches leaving the loop for simplification.
5290 if (!TheLoop->contains(BB: ThenBB) || !TheLoop->contains(BB: ElseBB))
5291 continue;
5292 bool ThenEmpty = IsEmptyBlock(ThenBB);
5293 bool ElseEmpty = IsEmptyBlock(ElseBB);
5294 if ((ThenEmpty && ElseEmpty) ||
5295 (ThenEmpty && ThenBB->getSingleSuccessor() == ElseBB &&
5296 ElseBB->phis().empty()) ||
5297 (ElseEmpty && ElseBB->getSingleSuccessor() == ThenBB &&
5298 ThenBB->phis().empty())) {
5299 VecValuesToIgnore.insert(Ptr: Br);
5300 DeadOps.push_back(Elt: Br->getCondition());
5301 }
5302 continue;
5303 }
5304
5305 // Skip any op that shouldn't be considered dead.
5306 if (!Op || !TheLoop->contains(Inst: Op) ||
5307 (isa<PHINode>(Val: Op) && Op->getParent() == Header) ||
5308 !wouldInstructionBeTriviallyDead(I: Op, TLI) ||
5309 any_of(Range: Op->users(), P: [this, IsLiveOutDead](User *U) {
5310 return !VecValuesToIgnore.contains(Ptr: U) &&
5311 !ValuesToIgnore.contains(Ptr: U) && !IsLiveOutDead(U);
5312 }))
5313 continue;
5314
5315 // If all of Op's users are in ValuesToIgnore, add it to ValuesToIgnore
5316 // which applies for both scalar and vector versions. Otherwise it is only
5317 // dead in vector versions, so only add it to VecValuesToIgnore.
5318 bool BecameScalarDead = false;
5319 if (all_of(Range: Op->users(),
5320 P: [this](User *U) { return ValuesToIgnore.contains(Ptr: U); }))
5321 BecameScalarDead = ValuesToIgnore.insert(Ptr: Op).second;
5322
5323 VecValuesToIgnore.insert(Ptr: Op);
5324 // Shared operands may be queued more than once. Propagate deadness once,
5325 // and again if an instruction previously dead only in the vector loop
5326 // becomes dead in the scalar loop too.
5327 if (ProcessedDeadOps.insert(Ptr: Op).second || BecameScalarDead)
5328 append_range(C&: DeadOps, R: Op->operands());
5329 }
5330
5331 // Ignore type-promoting instructions we identified during reduction
5332 // detection.
5333 for (const RecurrenceDescriptor &RedDes :
5334 Legal->getReductionVars().values()) {
5335 VecValuesToIgnore.insert_range(R: RedDes.getCastInsts());
5336 }
5337 // Ignore type-casting instructions we identified during induction
5338 // detection.
5339 for (const InductionDescriptor &IndDes : Legal->getInductionVars().values())
5340 VecValuesToIgnore.insert_range(R: IndDes.getCastInsts());
5341}
5342
5343void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
5344 CM->collectValuesToIgnore();
5345 Config.collectElementTypesForWidening(ValuesToIgnore: &CM->ValuesToIgnore);
5346
5347 FixedScalableVFPair MaxFactors = CM->computeMaxVF(UserVF, UserIC);
5348 if (!MaxFactors) // Cases that should not to be vectorized nor interleaved.
5349 return;
5350
5351 Config.collectInLoopReductions();
5352 // Cases that may be vectorized may be optimized by unit stride predicates.
5353 // TODO: Currently unit stride predicates are added unconditionally, even if
5354 // they are not used for the selected VF (e.g. when only interleaving).
5355 if (MaxFactors.FixedVF.isVector() || MaxFactors.ScalableVF.isVector())
5356 Legal->collectUnitStridePredicates();
5357
5358 auto VPlan1 = tryToBuildVPlan1();
5359 if (!VPlan1)
5360 return;
5361
5362 LLVM_DEBUG(dbgs() << "LV: VPlan created successfully. Loop can be "
5363 "vectorized.\n");
5364
5365 if (!OrigLoop->isInnermost()) {
5366 // For outer loops, computeMaxVF returns a single non-scalar VF; build a
5367 // plan for that VF only.
5368 ElementCount VF =
5369 MaxFactors.FixedVF ? MaxFactors.FixedVF : MaxFactors.ScalableVF;
5370 buildVPlans(VPlan1&: *VPlan1, MinVF: VF, MaxVF: VF);
5371 LLVM_DEBUG(printPlans(dbgs()));
5372 return;
5373 }
5374
5375 // Compute the minimal bitwidths required for integer operations in the loop
5376 // for later use by the cost model.
5377 Config.computeMinimalBitwidths();
5378
5379 // Invalidate interleave groups if all blocks of loop will be predicated.
5380 if (CM->blockNeedsPredicationForAnyReason(BB: OrigLoop->getHeader()) &&
5381 !useMaskedInterleavedAccesses(TTI)) {
5382 LLVM_DEBUG(
5383 dbgs()
5384 << "LV: Invalidate all interleaved groups due to fold-tail by masking "
5385 "which requires masked-interleaved support.\n");
5386 if (CM->InterleaveInfo.invalidateGroups())
5387 // Invalidating interleave groups also requires invalidating all decisions
5388 // based on them, which includes widening decisions and uniform and scalar
5389 // values.
5390 CM->invalidateCostModelingDecisions();
5391 }
5392
5393 if (CM->foldTailByMasking())
5394 Legal->prepareToFoldTailByMasking();
5395
5396 ElementCount MaxUserVF =
5397 UserVF.isScalable() ? MaxFactors.ScalableVF : MaxFactors.FixedVF;
5398 if (UserVF) {
5399 if (!ElementCount::isKnownLE(LHS: UserVF, RHS: MaxUserVF)) {
5400 reportVectorizationInfo(
5401 Msg: "UserVF ignored because it may be larger than the maximal safe VF",
5402 ORETag: "InvalidUserVF", ORE, TheLoop: OrigLoop);
5403 } else {
5404 assert(isPowerOf2_32(UserVF.getKnownMinValue()) &&
5405 "VF needs to be a power of two");
5406 // Collect the instructions (and their associated costs) that will be more
5407 // profitable to scalarize.
5408 CM->collectNonVectorizedAndSetWideningDecisions(VF: UserVF);
5409 buildVPlans(VPlan1&: *VPlan1, MinVF: UserVF, MaxVF: UserVF);
5410 ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
5411 if (EpilogueUserVF.isVector() &&
5412 ElementCount::isKnownLT(LHS: EpilogueUserVF, RHS: UserVF)) {
5413 CM->collectNonVectorizedAndSetWideningDecisions(VF: EpilogueUserVF);
5414 buildVPlans(VPlan1&: *VPlan1, MinVF: EpilogueUserVF, MaxVF: EpilogueUserVF);
5415 }
5416 if (!VPlans.empty() && VPlans.front()->getSingleVF() == UserVF) {
5417 // For scalar VF, skip VPlan cost check as VPlan cost is designed for
5418 // vector VFs only.
5419 if (UserVF.isScalar() ||
5420 cost(Plan&: *VPlans.front(), VF: UserVF, /*RU=*/nullptr).isValid()) {
5421 LLVM_DEBUG(dbgs() << "LV: Using user VF " << UserVF << ".\n");
5422 LLVM_DEBUG(printPlans(dbgs()));
5423 return;
5424 }
5425 }
5426 VPlans.clear();
5427 reportVectorizationInfo(Msg: "UserVF ignored because of invalid costs.",
5428 ORETag: "InvalidCost", ORE, TheLoop: OrigLoop);
5429 }
5430 }
5431
5432 // Collect the Vectorization Factor Candidates.
5433 SmallVector<ElementCount> VFCandidates;
5434 for (auto VF = ElementCount::getFixed(MinVal: 1);
5435 ElementCount::isKnownLE(LHS: VF, RHS: MaxFactors.FixedVF); VF *= 2)
5436 VFCandidates.push_back(Elt: VF);
5437 for (auto VF = ElementCount::getScalable(MinVal: 1);
5438 ElementCount::isKnownLE(LHS: VF, RHS: MaxFactors.ScalableVF); VF *= 2)
5439 VFCandidates.push_back(Elt: VF);
5440
5441 for (const auto &VF : VFCandidates) {
5442 // Collect Uniform and Scalar instructions after vectorization with VF.
5443 CM->collectNonVectorizedAndSetWideningDecisions(VF);
5444 }
5445
5446 buildVPlans(VPlan1&: *VPlan1, MinVF: ElementCount::getFixed(MinVal: 1), MaxVF: MaxFactors.FixedVF);
5447 buildVPlans(VPlan1&: *VPlan1, MinVF: ElementCount::getScalable(MinVal: 1), MaxVF: MaxFactors.ScalableVF);
5448
5449 LLVM_DEBUG(printPlans(dbgs()));
5450}
5451
5452VPCostContext::VPCostContext(const TargetLibraryInfo &TLI, const VPlan &Plan,
5453 LoopVectorizationCostModel &CM,
5454 VFSelectionContext &Config,
5455 bool ReusePrintingSlotTracker)
5456 : TTI(Config.getTTI()), TLI(TLI), LLVMCtx(Plan.getContext()), CM(CM),
5457 Config(Config), CostKind(Config.CostKind), PSE(Config.getPSE()),
5458 L(Config.getLoop()) {
5459#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5460 if (ReusePrintingSlotTracker)
5461 PlanForSlotTracker = &Plan;
5462#endif
5463}
5464
5465InstructionCost VPCostContext::getLegacyCost(Instruction *UI,
5466 ElementCount VF) const {
5467 InstructionCost Cost = CM.getInstructionCost(I: UI, VF);
5468 if (Cost.isValid() && ForceTargetInstructionCost.getNumOccurrences())
5469 return InstructionCost(ForceTargetInstructionCost);
5470 return Cost;
5471}
5472
5473bool VPCostContext::skipCostComputation(Instruction *UI, bool IsVector) const {
5474 return CM.ValuesToIgnore.contains(Ptr: UI) ||
5475 (IsVector && CM.VecValuesToIgnore.contains(Ptr: UI)) ||
5476 SkipCostComputation.contains(Ptr: UI);
5477}
5478
5479void VPCostContext::invalidateWideningDecision(Instruction *I,
5480 ElementCount VF) {
5481 CM.setWideningDecision(I, VF,
5482 W: LoopVectorizationCostModel::CM_InvalidatedDecision, Cost: 0);
5483}
5484
5485bool VPCostContext::willBeScalarized(Instruction *I, ElementCount VF) const {
5486 return CM.isScalarWithPredication(I, VF) ||
5487 CM.isUniformAfterVectorization(I, VF) || CM.isForcedScalar(I, VF) ||
5488 (VF.isVector() && CM.isProfitableToScalarize(I, VF));
5489}
5490
5491bool VPCostContext::isMaskRequired(Instruction *I) const {
5492 return CM.isMaskRequired(I);
5493}
5494
5495bool VPCostContext::executesAtMostOnce(const VPlan &Plan, ElementCount VF) {
5496 auto *TC = dyn_cast_if_present<ConstantInt>(
5497 Val: Plan.getTripCount()->getUnderlyingValue());
5498 return TC && TC->getValue().ule(RHS: VF.getKnownMinValue());
5499}
5500
5501InstructionCost
5502LoopVectorizationPlanner::precomputeCosts(VPlan &Plan, ElementCount VF,
5503 VPCostContext &CostCtx) const {
5504 InstructionCost Cost;
5505
5506 // If the vector loop gets executed exactly once with the given VF, ignore the
5507 // costs of comparison and induction instructions, as they'll get simplified
5508 // away.
5509 // TODO: Remove this code after stepping away from the legacy cost model and
5510 // adding code to simplify VPlans before calculating their costs.
5511 auto TC = getSmallConstantTripCount(SE: PSE.getSE(), L: OrigLoop);
5512 if (TC == VF && !Plan.hasTailFolded())
5513 addFullyUnrolledInstructionsToIgnore(L: OrigLoop, IL: Legal->getInductionVars(),
5514 InstsToIgnore&: CostCtx.SkipCostComputation);
5515
5516 // Pre-compute the costs for branches except for the backedge, as the number
5517 // of replicate regions in a VPlan may not directly match the number of
5518 // branches, which would lead to different decisions.
5519 // TODO: Compute cost of branches for each replicate region in the VPlan,
5520 // which is more accurate than the legacy cost model.
5521 for (BasicBlock *BB : OrigLoop->blocks()) {
5522 if (CostCtx.skipCostComputation(UI: BB->getTerminator(), IsVector: VF.isVector()))
5523 continue;
5524 CostCtx.SkipCostComputation.insert(Ptr: BB->getTerminator());
5525 if (BB == OrigLoop->getLoopLatch())
5526 continue;
5527 auto BranchCost = CostCtx.getLegacyCost(UI: BB->getTerminator(), VF);
5528 Cost += BranchCost;
5529 }
5530
5531 // Don't apply special costs when instruction cost is forced to make sure the
5532 // forced cost is used for each recipe.
5533 if (ForceTargetInstructionCost.getNumOccurrences())
5534 return Cost;
5535
5536 // Pre-compute costs for instructions that are forced-scalar or profitable to
5537 // scalarize. For most such instructions, their scalarization costs are
5538 // accounted for here using the legacy cost model. However, some opcodes
5539 // are excluded from these precomputed scalarization costs and are instead
5540 // modeled later by the VPlan cost model (see UseVPlanCostModel below).
5541 for (Instruction *ForcedScalar : CostCtx.CM.ForcedScalars[VF]) {
5542 if (CostCtx.skipCostComputation(UI: ForcedScalar, IsVector: VF.isVector()))
5543 continue;
5544 CostCtx.SkipCostComputation.insert(Ptr: ForcedScalar);
5545 InstructionCost ForcedCost = CostCtx.getLegacyCost(UI: ForcedScalar, VF);
5546 LLVM_DEBUG({
5547 dbgs() << "Cost of " << ForcedCost << " for VF " << VF
5548 << ": forced scalar " << *ForcedScalar << "\n";
5549 });
5550 Cost += ForcedCost;
5551 }
5552
5553 // Don't apply legacy scalarization costs if nothing remains scalar &
5554 // predicated.
5555 if (!hasReplicatorRegion(Plan))
5556 return Cost;
5557
5558 auto UseVPlanCostModel = [](Instruction *I) -> bool {
5559 switch (I->getOpcode()) {
5560 case Instruction::SDiv:
5561 case Instruction::UDiv:
5562 case Instruction::SRem:
5563 case Instruction::URem:
5564 return true;
5565 default:
5566 return false;
5567 }
5568 };
5569 for (const auto &[Scalarized, ScalarCost] : CostCtx.CM.InstsToScalarize[VF]) {
5570 if (UseVPlanCostModel(Scalarized) ||
5571 CostCtx.skipCostComputation(UI: Scalarized, IsVector: VF.isVector()))
5572 continue;
5573 CostCtx.SkipCostComputation.insert(Ptr: Scalarized);
5574 LLVM_DEBUG({
5575 dbgs() << "Cost of " << ScalarCost << " for VF " << VF
5576 << ": profitable to scalarize " << *Scalarized << "\n";
5577 });
5578 Cost += ScalarCost;
5579 }
5580
5581 return Cost;
5582}
5583
5584#ifndef NDEBUG
5585/// Returns the frequency with which \p VPBB executes, as recorded on its
5586/// recipes. All recipes of a block share the same frequency.
5587static std::optional<VPExecutionFrequency>
5588getRecordedExecutionFrequency(const VPBasicBlock *VPBB) {
5589 if (VPBB->empty())
5590 return std::nullopt;
5591 return cast<VPInstruction>(&VPBB->front())->getExecutionFrequency();
5592}
5593#endif
5594
5595InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
5596 VPRegisterUsage *RU) const {
5597 VPCostContext CostCtx(*TLI, Plan, *CM, Config,
5598 /*ReusePrintingSlotTracker=*/true);
5599 InstructionCost Cost = precomputeCosts(Plan, VF, CostCtx);
5600 LLVM_DEBUG(dbgs() << "Precomputed costs for VF " << VF << ": " << Cost
5601 << '\n');
5602
5603 // Now compute and add the VPlan-based cost.
5604 Cost += Plan.cost(VF, Ctx&: CostCtx);
5605
5606 // Add the cost of spills due to excess register usage
5607 if (RU && Config.shouldConsiderRegPressureForVF(VF)) {
5608 InstructionCost SpillCost =
5609 RU->spillCost(TTI, CostKind: Config.CostKind, OverrideMaxNumRegs: ForceTargetNumVectorRegs);
5610 LLVM_DEBUG(dbgs() << "Spill costs for VF " << VF << ": " << SpillCost
5611 << '\n');
5612 Cost += SpillCost;
5613 }
5614
5615#ifndef NDEBUG
5616 unsigned EstimatedWidth =
5617 estimateElementCount(VF, Config.getVScaleForTuning());
5618 LLVM_DEBUG(dbgs() << "Cost for VF " << VF << ": " << Cost
5619 << " (Estimated cost per lane: ");
5620 if (Cost.isValid()) {
5621 APFloat CostPerLane(APFloat::IEEEdouble());
5622 APFloat EstimatedWidthAsAPFloat(APFloat::IEEEdouble());
5623 (void)CostPerLane.convertFromAPInt(APInt(64, (uint64_t)Cost.getValue()),
5624 false, APFloat::rmTowardZero);
5625 (void)EstimatedWidthAsAPFloat.convertFromAPInt(
5626 APInt(64, (uint64_t)EstimatedWidth), false, APFloat::rmTowardZero);
5627 (void)CostPerLane.divide(EstimatedWidthAsAPFloat, APFloat::rmTowardZero);
5628
5629 SmallString<16> Str;
5630 CostPerLane.toString(Str, 3);
5631 LLVM_DEBUG(dbgs() << Str);
5632 } else /* No point dividing an invalid cost - it will still be invalid */
5633 LLVM_DEBUG(dbgs() << "Invalid");
5634 LLVM_DEBUG(dbgs() << ")\n");
5635#endif
5636 return Cost;
5637}
5638
5639std::pair<VectorizationFactor, VPlan *>
5640LoopVectorizationPlanner::computeBestVF() {
5641 if (VPlans.empty())
5642 return {VectorizationFactor::Disabled(), nullptr};
5643 // If there is a single VPlan with a single VF, return it directly.
5644 VPlan &FirstPlan = *VPlans[0];
5645
5646 ElementCount UserVF = Config.getHints().getWidth();
5647 if (VPlans.size() == 1) {
5648 // For outer loops, the plan has a single vector VF determined by the
5649 // heuristic.
5650 assert((FirstPlan.hasScalarVFOnly() || hasPlanWithVF(UserVF) ||
5651 FirstPlan.isOuterLoop()) &&
5652 "must have a single scalar VF, UserVF or an outer loop");
5653 return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
5654 }
5655
5656 if (hasPlanWithVF(VF: UserVF) && hasForcedEpilogueVF() && VPlans.size() == 2) {
5657 assert(VPlans[0]->getSingleVF() == UserVF &&
5658 "expected second plan to be for the forced UserVF");
5659 assert(VPlans[1]->getSingleVF() == EpilogueVectorizationForceVF &&
5660 "expected first plan to be for the forced epilogue VF");
5661 return {VectorizationFactor(UserVF, 0, 0), VPlans[0].get()};
5662 }
5663
5664 LLVM_DEBUG(dbgs() << "LV: Computing best VF using cost kind: "
5665 << (Config.CostKind == TTI::TCK_RecipThroughput
5666 ? "Reciprocal Throughput\n"
5667 : Config.CostKind == TTI::TCK_Latency
5668 ? "Instruction Latency\n"
5669 : Config.CostKind == TTI::TCK_CodeSize ? "Code Size\n"
5670 : Config.CostKind == TTI::TCK_SizeAndLatency
5671 ? "Code Size and Latency\n"
5672 : "Unknown\n"));
5673
5674 ElementCount ScalarVF = ElementCount::getFixed(MinVal: 1);
5675 assert(FirstPlan.hasVF(ScalarVF) &&
5676 "More than a single plan/VF w/o any plan having scalar VF");
5677
5678 // TODO: Compute scalar cost using VPlan-based cost model.
5679 InstructionCost ScalarCost = CM->expectedCost(VF: ScalarVF);
5680 LLVM_DEBUG(dbgs() << "LV: Scalar loop costs: " << ScalarCost << ".\n");
5681 VectorizationFactor ScalarFactor(ScalarVF, ScalarCost, ScalarCost);
5682 VectorizationFactor BestFactor = ScalarFactor;
5683
5684 bool ForceVectorization =
5685 Config.getHints().getForce() == LoopVectorizeHints::FK_Enabled;
5686 if (ForceVectorization) {
5687 // Ignore scalar width, because the user explicitly wants vectorization.
5688 // Initialize cost to max so that VF = 2 is, at least, chosen during cost
5689 // evaluation.
5690 BestFactor.Cost = InstructionCost::getMax();
5691 }
5692
5693 VPlan *PlanForBestVF = &FirstPlan;
5694 ElementCount ExactTC = getSmallConstantTripCount(SE: PSE.getSE(), L: OrigLoop);
5695
5696 for (auto &P : VPlans) {
5697 ArrayRef<ElementCount> VFs(P->vectorFactors().begin(),
5698 P->vectorFactors().end());
5699
5700 // For loops where the Trip Count is below the TailFoldingThreshold, only
5701 // consider the largest VF to result in at most one vector iteration, and at
5702 // most one scalar iteration.
5703 // FIXME: Encode this decision directly in LVPlanner.
5704 if (!ForceVectorization && P->hasScalarTail() && ExactTC.isFixed() &&
5705 ExactTC.getFixedValue() > 0 &&
5706 ExactTC.getFixedValue() <= TTI.getMinTripCountTailFoldingThreshold()) {
5707 VFs = VFs.take_back(N: 1);
5708 }
5709
5710 SmallVector<VPRegisterUsage, 8> RUs;
5711 bool ConsiderRegPressure = any_of(Range&: VFs, P: [this](ElementCount VF) {
5712 return Config.shouldConsiderRegPressureForVF(VF);
5713 });
5714 if (ConsiderRegPressure)
5715 RUs = calculateRegisterUsageForPlan(Plan&: *P, VFs, TTI);
5716
5717 for (unsigned I = 0; I < VFs.size(); I++) {
5718 ElementCount VF = VFs[I];
5719 if (VF.isScalar())
5720 continue;
5721 if (!ForceVectorization && !willGenerateVectors(Plan&: *P, VF, TTI)) {
5722 LLVM_DEBUG(
5723 dbgs()
5724 << "LV: Not considering vector loop of width " << VF
5725 << " because it will not generate any vector instructions.\n");
5726 continue;
5727 }
5728 if (Config.OptForSize && !ForceVectorization && hasReplicatorRegion(Plan&: *P)) {
5729 LLVM_DEBUG(
5730 dbgs()
5731 << "LV: Not considering vector loop of width " << VF
5732 << " because it would cause replicated blocks to be generated,"
5733 << " which isn't allowed when optimizing for size.\n");
5734 continue;
5735 }
5736
5737 InstructionCost Cost =
5738 cost(Plan&: *P, VF, RU: ConsiderRegPressure ? &RUs[I] : nullptr);
5739 VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
5740
5741 if (isMoreProfitable(A: CurrentFactor, B: BestFactor, HasTail: P->hasScalarTail())) {
5742 BestFactor = CurrentFactor;
5743 PlanForBestVF = P.get();
5744 }
5745
5746 // If profitable add it to ProfitableVF list.
5747 if (isMoreProfitable(A: CurrentFactor, B: ScalarFactor, HasTail: P->hasScalarTail()))
5748 ProfitableVFs.push_back(Elt: CurrentFactor);
5749 }
5750 }
5751
5752 VPlan &BestPlan = *PlanForBestVF;
5753
5754 assert((BestFactor.Width.isScalar() || BestFactor.ScalarCost > 0) &&
5755 "when vectorizing, the scalar cost must be computed.");
5756
5757 LLVM_DEBUG(dbgs() << "LV: Selecting VF: " << BestFactor.Width << ".\n");
5758 return {BestFactor, &BestPlan};
5759}
5760
5761LoopVectorizationPlanner::LoopVectorizationPlanner(
5762 Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
5763 const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
5764 std::unique_ptr<LoopVectorizationCostModel> CM, VFSelectionContext &Config,
5765 InterleavedAccessInfo &IAI, PredicatedScalarEvolution &PSE,
5766 OptimizationRemarkEmitter *ORE,
5767 std::function<const BranchProbabilityInfo &()> GetBPI)
5768 : OrigLoop(L), LI(LI), DT(DT), TLI(TLI), TTI(TTI), Legal(Legal),
5769 CM(std::move(CM)), Config(Config), IAI(IAI), PSE(PSE), ORE(ORE),
5770 GetBPI(GetBPI) {}
5771
5772LoopVectorizationPlanner::~LoopVectorizationPlanner() = default;
5773
5774void LoopVectorizationPlanner::clearCostModel() { CM.reset(); }
5775
5776DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
5777 ElementCount BestVF, unsigned BestUF, VPlan &BestVPlan,
5778 InnerLoopVectorizer &ILV, DominatorTree *DT,
5779 EpilogueVectorizationKind EpilogueVecKind) {
5780 assert(BestVPlan.hasVF(BestVF) &&
5781 "Trying to execute plan with unsupported VF");
5782 assert(BestVPlan.hasUF(BestUF) &&
5783 "Trying to execute plan with unsupported UF");
5784 if (BestVPlan.hasEarlyExit())
5785 ++LoopsEarlyExitVectorized;
5786
5787 RUN_VPLAN_PASS(VPlanTransforms::replaceWideCanonicalIVWithWideIV, BestVPlan,
5788 *PSE.getSE(), TTI, Config.CostKind, BestVF, BestUF);
5789 if (!BestVPlan.hasTailFolded() &&
5790 TTI.hasMultiVectorLoadStore(NumVectors: BestUF,
5791 Mask: TargetTransformInfo::MaskSource::None))
5792 RUN_VPLAN_PASS(VPlanTransforms::widenMemoryAccessesByUF, BestVPlan, BestVF,
5793 BestUF, TTI);
5794 // TODO: Move to VPlan transform stage once the transition to the VPlan-based
5795 // cost model is complete for better cost estimates.
5796 RUN_VPLAN_PASS(VPlanTransforms::unrollByUF, BestVPlan, BestUF);
5797 RUN_VPLAN_PASS(VPlanTransforms::materializePacksAndUnpacks, BestVPlan);
5798 RUN_VPLAN_PASS(VPlanTransforms::materializeBroadcasts, BestVPlan);
5799 RUN_VPLAN_PASS(VPlanTransforms::replicateByVF, BestVPlan, BestVF);
5800 bool HasBranchWeights =
5801 hasBranchWeightMD(I: *OrigLoop->getLoopLatch()->getTerminator());
5802 if (HasBranchWeights) {
5803 std::optional<unsigned> VScale = Config.getVScaleForTuning();
5804 RUN_VPLAN_PASS(VPlanTransforms::addBranchWeightToMiddleTerminator,
5805 BestVPlan, BestVF, VScale);
5806 }
5807
5808 if (vputils::findIncomingAliasMask(Plan: BestVPlan)) {
5809 assert(BestVPlan.hasTailFolded() && "Expected tail folding to be enabled");
5810 RUN_VPLAN_PASS(VPlanTransforms::materializeAliasMaskCheckBlock, BestVPlan,
5811 *Legal->getRuntimePointerChecking()->getDiffChecks(),
5812 HasBranchWeights);
5813 ++LoopsPartialAliasVectorized;
5814 }
5815
5816 // Retrieving VectorPH now when it's easier while VPlan still has Regions.
5817 VPBasicBlock *VectorPH = cast<VPBasicBlock>(Val: BestVPlan.getVectorPreheader());
5818
5819 RUN_VPLAN_PASS(VPlanTransforms::materializeConstantVectorTripCount, BestVPlan,
5820 BestVF, BestUF, PSE);
5821 RUN_VPLAN_PASS(VPlanTransforms::optimizeForVFAndUF, BestVPlan, BestVF, BestUF,
5822 PSE);
5823 RUN_VPLAN_PASS(VPlanTransforms::combineRecipes, BestVPlan);
5824 // Check if scalar epilogue is required, before simplifying constant branches.
5825 const bool RequiresScalarEpilogue = BestVPlan.requiresScalarEpilogue();
5826 if (EpilogueVecKind == EpilogueVectorizationKind::None)
5827 RUN_VPLAN_PASS(VPlanTransforms::removeBranchOnConst, BestVPlan,
5828 /*OnlyLatches=*/false);
5829 if (BestVPlan.getEntry()->getSingleSuccessor() ==
5830 BestVPlan.getScalarPreheader()) {
5831 // TODO: The vector loop would be dead, should not even try to vectorize.
5832 ORE->emit(RemarkBuilder: [&]() {
5833 return OptimizationRemarkAnalysis(DEBUG_TYPE, "VectorizationDead",
5834 OrigLoop->getStartLoc(),
5835 OrigLoop->getHeader())
5836 << "Created vector loop never executes due to insufficient trip "
5837 "count.";
5838 });
5839 return DenseMap<const SCEV *, Value *>();
5840 }
5841
5842 RUN_VPLAN_PASS(VPlanTransforms::removeDeadRecipes, BestVPlan);
5843
5844 RUN_VPLAN_PASS(VPlanTransforms::convertToConcreteRecipes, BestVPlan);
5845 // Convert the exit condition to AVLNext == 0 for EVL tail folded loops.
5846 RUN_VPLAN_PASS(VPlanTransforms::convertEVLExitCond, BestVPlan);
5847 // Regions are dissolved after optimizing for VF and UF, which completely
5848 // removes unneeded loop regions first.
5849 const bool HasTailFolded = BestVPlan.hasTailFolded();
5850 RUN_VPLAN_PASS(VPlanTransforms::dissolveLoopRegions, BestVPlan);
5851 // Expand BranchOnTwoConds after dissolution, when latch has direct access to
5852 // its successors.
5853 RUN_VPLAN_PASS(VPlanTransforms::expandBranchOnTwoConds, BestVPlan);
5854 // Convert loops with variable-length stepping after regions are dissolved.
5855 RUN_VPLAN_PASS(VPlanTransforms::convertToVariableLengthStep, BestVPlan);
5856 // Remove dead back-edges for single-iteration loops with BranchOnCond(true).
5857 // Only process loop latches to avoid removing edges from the middle block,
5858 // which may be needed for epilogue vectorization.
5859 RUN_VPLAN_PASS(VPlanTransforms::removeBranchOnConst, BestVPlan,
5860 /*OnlyLatches=*/true);
5861 RUN_VPLAN_PASS(VPlanTransforms::materializeBackedgeTakenCount, BestVPlan,
5862 VectorPH);
5863 std::optional<uint64_t> MaxRuntimeStep = getMaxRuntimeElementCount(
5864 EC: BestVF * BestUF, F: *OrigLoop->getHeader()->getParent());
5865
5866 assert((LI->getUniqueLatchExitBlock(*OrigLoop) || RequiresScalarEpilogue) &&
5867 "loops not exiting via the latch without required epilogue?");
5868 RUN_VPLAN_PASS(VPlanTransforms::materializeVectorTripCount, BestVPlan,
5869 VectorPH, HasTailFolded, RequiresScalarEpilogue,
5870 &BestVPlan.getVFxUF(), MaxRuntimeStep);
5871 RUN_VPLAN_PASS(VPlanTransforms::materializeFactors, BestVPlan, VectorPH,
5872 BestVF);
5873 // Limit expansions to VPInstruction to when not vectorizing the epilogue.
5874 // Currently this code path still relies on code re-using SCEVs expanded
5875 // directly to IR instructions.
5876 if (EpilogueVecKind == EpilogueVectorizationKind::None)
5877 RUN_VPLAN_PASS(VPlanTransforms::expandSCEVsToVPInstructions, BestVPlan,
5878 *PSE.getSE());
5879 RUN_VPLAN_PASS(VPlanTransforms::cse, BestVPlan);
5880 RUN_VPLAN_PASS(VPlanTransforms::combineRecipes, BestVPlan);
5881 // Removing branches and incoming values may expose additional simplification
5882 // opportunities.
5883 if (RUN_VPLAN_PASS(VPlanTransforms::removeBranchOnConst, BestVPlan,
5884 /*OnlyLatches=*/EpilogueVecKind !=
5885 EpilogueVectorizationKind::None))
5886 RUN_VPLAN_PASS(VPlanTransforms::combineRecipes, BestVPlan);
5887 RUN_VPLAN_PASS(VPlanTransforms::simplifyKnownEVL, BestVPlan, BestVF, PSE);
5888
5889 // 0. Generate SCEV-dependent code in the entry, including TripCount, before
5890 // making any changes to the CFG.
5891 DenseMap<const SCEV *, Value *> ExpandedSCEVs =
5892 RUN_VPLAN_PASS(VPlanTransforms::expandSCEVs, BestVPlan, *PSE.getSE());
5893
5894 // Perform the actual loop transformation.
5895 VPTransformState State(&TTI, BestVF, LI, DT, ILV.AC, ILV.Builder, &BestVPlan,
5896 OrigLoop->getParentLoop());
5897
5898#ifdef EXPENSIVE_CHECKS
5899 assert(DT->verify(DominatorTree::VerificationLevel::Fast));
5900#endif
5901
5902 // 1. Set up the skeleton for vectorization, including vector pre-header and
5903 // middle block. The vector loop is created during VPlan execution.
5904 State.CFG.PrevBB = ILV.createVectorizedLoopSkeleton();
5905 if (VPBasicBlock *ScalarPH = BestVPlan.getScalarPreheader())
5906 replaceVPBBWithIRVPBB(VPBB: ScalarPH, IRBB: State.CFG.PrevBB->getSingleSuccessor(),
5907 Plan: &BestVPlan);
5908 RUN_VPLAN_PASS(VPlanTransforms::removeDeadRecipes, BestVPlan);
5909
5910 assert(verifyVPlanIsValid(BestVPlan) && "final VPlan is invalid");
5911
5912 // After vectorization, the exit blocks of the original loop will have
5913 // additional predecessors. Invalidate SCEVs for the exit phis in case SE
5914 // looked through single-entry phis.
5915 ScalarEvolution &SE = *PSE.getSE();
5916 for (VPIRBasicBlock *Exit : BestVPlan.getExitBlocks()) {
5917 if (!Exit->hasPredecessors())
5918 continue;
5919 for (VPRecipeBase &PhiR : Exit->phis())
5920 SE.forgetLcssaPhiWithNewPredecessor(L: OrigLoop,
5921 V: &cast<VPIRPhi>(Val&: PhiR).getIRPhi());
5922 }
5923
5924 // Query whether the target wants loops it vectorizes to remain eligible for
5925 // runtime unrolling. Do this here, on the original loop and before its SCEV
5926 // is forgotten below.
5927 TargetTransformInfo::UnrollingPreferences UP;
5928 TTI.getUnrollingPreferences(L: OrigLoop, SE, UP, ORE);
5929 bool UnrollVectorizedLoop = UP.UnrollVectorizedLoop;
5930
5931 // Forget the original loop and block dispositions.
5932 SE.forgetLoop(L: OrigLoop);
5933 SE.forgetBlockAndLoopDispositions();
5934
5935 //===------------------------------------------------===//
5936 //
5937 // Notice: any optimization or new instruction that go
5938 // into the code below should also be implemented in
5939 // the cost-model.
5940 //
5941 //===------------------------------------------------===//
5942
5943 // Retrieve loop information before executing the plan, which may remove the
5944 // original loop, if it becomes unreachable.
5945 MDNode *LID = OrigLoop->getLoopID();
5946 unsigned OrigLoopInvocationWeight = 0;
5947 std::optional<unsigned> OrigAverageTripCount =
5948 getLoopEstimatedTripCount(L: OrigLoop, EstimatedLoopInvocationWeight: &OrigLoopInvocationWeight);
5949
5950 BestVPlan.execute(State: &State);
5951
5952 // 2.6. Maintain Loop Hints
5953 // Keep all loop hints from the original loop on the vector loop (we'll
5954 // replace the vectorizer-specific hints below).
5955 VPBasicBlock *HeaderVPBB = vputils::getFirstLoopHeader(Plan&: BestVPlan, VPDT&: State.VPDT);
5956 // Add metadata to disable runtime unrolling a scalar loop when there
5957 // are no runtime checks about strides and memory. A scalar loop that is
5958 // rarely used is not worth unrolling.
5959 bool DisableRuntimeUnroll = !ILV.RTChecks.hasChecks() && !BestVF.isScalar();
5960 updateLoopMetadataAndProfileInfo(
5961 VectorLoop: HeaderVPBB ? LI->getLoopFor(BB: State.CFG.VPBB2IRBB.lookup(Val: HeaderVPBB))
5962 : nullptr,
5963 HeaderVPBB, Plan: BestVPlan,
5964 VectorizingEpilogue: EpilogueVecKind == EpilogueVectorizationKind::Epilogue, OrigLoopID: LID,
5965 OrigAverageTripCount, OrigLoopInvocationWeight,
5966 EstimatedVFxUF: estimateElementCount(VF: BestVF * BestUF, VScale: Config.getVScaleForTuning()),
5967 DisableRuntimeUnroll, UnrollVectorizedLoop);
5968
5969 // 3. Fix the vectorized code: take care of header phi's, live-outs,
5970 // predication, updating analyses.
5971 ILV.fixVectorizedLoop(State);
5972
5973 // Wrap the generated blocks in VPIRBasicBlocks, so they can be used in the
5974 // epilogue plan.
5975 if (EpilogueVecKind == EpilogueVectorizationKind::MainLoop)
5976 for (VPBasicBlock *VPBB : to_vector(Range: VPBlockUtils::blocksAs<VPBasicBlock>(
5977 Range: vp_depth_first_shallow(G: BestVPlan.getEntry()))))
5978 if (!isa<VPIRBasicBlock>(Val: VPBB))
5979 replaceVPBBWithIRVPBB(VPBB, IRBB: State.CFG.VPBB2IRBB.at(Val: VPBB), Plan: &BestVPlan);
5980
5981 return ExpandedSCEVs;
5982}
5983
5984//===--------------------------------------------------------------------===//
5985// EpilogueVectorizerEpilogueLoop
5986//===--------------------------------------------------------------------===//
5987
5988/// This function creates a new scalar preheader, using the previous one as
5989/// entry block to the epilogue VPlan. The minimum iteration check is being
5990/// represented in VPlan.
5991BasicBlock *EpilogueVectorizerEpilogueLoop::createVectorizedLoopSkeleton() {
5992 BasicBlock *NewScalarPH = createScalarPreheader(Prefix: "vec.epilog.");
5993 BasicBlock *OriginalScalarPH = NewScalarPH->getSinglePredecessor();
5994 OriginalScalarPH->setName("vec.epilog.iter.check");
5995 VPIRBasicBlock *NewEntry = Plan.createVPIRBasicBlock(IRBB: OriginalScalarPH);
5996 VPBasicBlock *OldEntry = Plan.getEntry();
5997 for (auto &R : make_early_inc_range(Range&: *OldEntry)) {
5998 // Skip moving VPIRInstructions (including VPIRPhis), which are unmovable by
5999 // defining.
6000 if (isa<VPIRInstruction>(Val: &R))
6001 continue;
6002 R.moveBefore(BB&: *NewEntry, I: NewEntry->end());
6003 }
6004
6005 VPBlockUtils::reassociateBlocks(Old: OldEntry, New: NewEntry);
6006
6007 VecEpilogueIterationCountCheck = NewEntry;
6008
6009 // Model the skeleton from the main vector loop in the epilogue plan.
6010 RUN_VPLAN_PASS(VPlanTransforms::modelGeneratedMainLoopBlocks, Plan, MainPlan,
6011 NewEntry);
6012
6013 return OriginalScalarPH;
6014}
6015
6016bool VPRecipeBuilder::isPredicatedInst(Instruction *I) const {
6017 return CM.isPredicatedInst(I);
6018}
6019
6020bool VPRecipeBuilder::prefersVectorizedAddressing() const {
6021 return CM.TTI.prefersVectorizedAddressing();
6022}
6023
6024VPRecipeBase *VPRecipeBuilder::tryToWidenMemory(VPInstruction *VPI,
6025 VFRange &Range) {
6026 assert((VPI->getOpcode() == Instruction::Load ||
6027 VPI->getOpcode() == Instruction::Store) &&
6028 "Must be called with either a load or store");
6029 Instruction *I = VPI->getUnderlyingInstr();
6030
6031 auto WillWiden = [&](ElementCount VF) -> bool {
6032 LoopVectorizationCostModel::InstWidening Decision =
6033 CM.getWideningDecision(I, VF);
6034 assert(Decision != LoopVectorizationCostModel::CM_Unknown &&
6035 "CM decision should be taken at this point.");
6036 if (Decision == LoopVectorizationCostModel::CM_Interleave)
6037 return true;
6038 if (CM.isScalarAfterVectorization(I, VF) ||
6039 CM.isProfitableToScalarize(I, VF))
6040 return false;
6041 return Decision != LoopVectorizationCostModel::CM_Scalarize;
6042 };
6043
6044 if (!LoopVectorizationPlanner::getDecisionAndClampRange(Predicate: WillWiden, Range))
6045 return nullptr;
6046
6047 // If a mask is not required, drop it - use unmasked version for safe loads.
6048 // TODO: Determine if mask is needed in VPlan.
6049 VPValue *Mask = CM.isMaskRequired(I) ? VPI->getMask() : nullptr;
6050
6051 // Determine if the pointer operand of the access is either consecutive or
6052 // reverse consecutive.
6053 LoopVectorizationCostModel::InstWidening Decision =
6054 CM.getWideningDecision(I, VF: Range.Start);
6055 bool Reverse = Decision == LoopVectorizationCostModel::CM_Widen_Reverse;
6056 bool Consecutive =
6057 Reverse || Decision == LoopVectorizationCostModel::CM_Widen;
6058
6059 VPValue *Ptr = VPI->getOpcode() == Instruction::Load ? VPI->getOperand(N: 0)
6060 : VPI->getOperand(N: 1);
6061 Builder.setInsertPoint(VPI);
6062 if (Consecutive) {
6063 Ptr = Builder.createConsecutiveVectorPointer(Ptr, SourceElementTy: getLoadStoreType(I),
6064 Reverse, DL: VPI->getDebugLoc());
6065 }
6066
6067 if (Reverse && Mask)
6068 Mask = Builder.createNaryOp(Opcode: VPInstruction::Reverse, Operands: Mask, DL: I->getDebugLoc());
6069
6070 if (VPI->getOpcode() == Instruction::Load) {
6071 auto *Load = cast<LoadInst>(Val: I);
6072 auto *LoadR = Builder.createWidenLoad(Load&: *Load, Addr: Ptr, Mask, Consecutive, Metadata: *VPI,
6073 DL: Load->getDebugLoc());
6074 if (Reverse)
6075 return Builder.createNaryOp(Opcode: VPInstruction::Reverse, Operands: LoadR,
6076 DL: LoadR->getDebugLoc());
6077 return LoadR;
6078 }
6079
6080 StoreInst *Store = cast<StoreInst>(Val: I);
6081 VPValue *StoredVal = VPI->getOperand(N: 0);
6082 if (Reverse)
6083 StoredVal = Builder.createNaryOp(Opcode: VPInstruction::Reverse, Operands: StoredVal,
6084 DL: Store->getDebugLoc());
6085 return Builder.createWidenStore(Store&: *Store, Addr: Ptr, StoredVal, Mask, Consecutive,
6086 Metadata: *VPI, DL: Store->getDebugLoc());
6087}
6088
6089bool VPRecipeBuilder::shouldWiden(Instruction *I, VFRange &Range) const {
6090 assert((!isa<UncondBrInst, CondBrInst, PHINode, LoadInst, StoreInst>(I)) &&
6091 "Instruction should have been handled earlier");
6092 // Instruction should be widened, unless it is scalar after vectorization,
6093 // scalarization is profitable or it is predicated.
6094 auto WillScalarize = [this, I](ElementCount VF) -> bool {
6095 return CM.isScalarAfterVectorization(I, VF) ||
6096 CM.isProfitableToScalarize(I, VF) ||
6097 CM.isScalarWithPredication(I, VF);
6098 };
6099 return !LoopVectorizationPlanner::getDecisionAndClampRange(Predicate: WillScalarize,
6100 Range);
6101}
6102
6103VPRecipeWithIRFlags *VPRecipeBuilder::tryToWiden(VPInstruction *VPI) {
6104 auto *I = VPI->getUnderlyingInstr();
6105 switch (VPI->getOpcode()) {
6106 default:
6107 return nullptr;
6108 case Instruction::SDiv:
6109 case Instruction::UDiv:
6110 case Instruction::SRem:
6111 case Instruction::URem:
6112 // If not provably safe, use a masked intrinsic.
6113 if (CM.isPredicatedInst(I))
6114 return new VPWidenIntrinsicRecipe(
6115 getMaskedDivRemIntrinsic(Opcode: VPI->getOpcode()), VPI->operands(),
6116 I->getType(), {}, {}, VPI->getDebugLoc());
6117 [[fallthrough]];
6118 case Instruction::Add:
6119 case Instruction::And:
6120 case Instruction::AShr:
6121 case Instruction::FAdd:
6122 case Instruction::FCmp:
6123 case Instruction::FDiv:
6124 case Instruction::FMul:
6125 case Instruction::FNeg:
6126 case Instruction::FRem:
6127 case Instruction::FSub:
6128 case Instruction::ICmp:
6129 case Instruction::LShr:
6130 case Instruction::Mul:
6131 case Instruction::Or:
6132 case Instruction::Select:
6133 case Instruction::Shl:
6134 case Instruction::Sub:
6135 case Instruction::Xor:
6136 case Instruction::Freeze:
6137 return new VPWidenRecipe(*I, VPI->operandsWithoutMask(), *VPI, *VPI,
6138 VPI->getDebugLoc());
6139 case Instruction::ExtractValue: {
6140 SmallVector<VPValue *> NewOps(VPI->operandsWithoutMask());
6141 auto *EVI = cast<ExtractValueInst>(Val: I);
6142 assert(EVI->getNumIndices() == 1 && "Expected one extractvalue index");
6143 unsigned Idx = EVI->getIndices()[0];
6144 NewOps.push_back(Elt: Plan.getConstantInt(BitWidth: 32, Val: Idx));
6145 return new VPWidenRecipe(*I, NewOps, *VPI, *VPI, VPI->getDebugLoc());
6146 }
6147 };
6148}
6149
6150VPHistogramRecipe *VPRecipeBuilder::widenIfHistogram(VPInstruction *VPI) {
6151 if (VPI->getOpcode() != Instruction::Store)
6152 return nullptr;
6153
6154 auto HistInfo =
6155 Legal->getHistogramInfo(I: cast<StoreInst>(Val: VPI->getUnderlyingInstr()));
6156 if (!HistInfo)
6157 return nullptr;
6158
6159 const HistogramInfo *HI = *HistInfo;
6160 // FIXME: Support other operations.
6161 unsigned Opcode = HI->Update->getOpcode();
6162 assert((Opcode == Instruction::Add || Opcode == Instruction::Sub) &&
6163 "Histogram update operation must be an Add or Sub");
6164
6165 SmallVector<VPValue *, 3> HGramOps;
6166 // Bucket address.
6167 HGramOps.push_back(Elt: VPI->getOperand(N: 1));
6168 // Increment value.
6169 HGramOps.push_back(Elt: Plan.getOrAddLiveIn(V: HI->Update->getOperand(i: 1)));
6170
6171 // In case of predicated execution (due to tail-folding, or conditional
6172 // execution, or both), pass the relevant mask.
6173 if (CM.isMaskRequired(I: HI->Store))
6174 HGramOps.push_back(Elt: VPI->getMask());
6175
6176 return new VPHistogramRecipe(Opcode, HGramOps, cast<VPIRMetadata>(Val&: *VPI),
6177 VPI->getDebugLoc());
6178}
6179
6180bool VPRecipeBuilder::replaceWithFinalIfReductionStore(
6181 VPInstruction *VPI, VPBuilder &FinalRedStoresBuilder) {
6182 StoreInst *SI;
6183 if ((SI = dyn_cast<StoreInst>(Val: VPI->getUnderlyingInstr())) &&
6184 Legal->isInvariantAddressOfReduction(V: SI->getPointerOperand())) {
6185 // Only create recipe for the final invariant store of the reduction.
6186 if (Legal->isInvariantStoreOfReduction(SI)) {
6187 VPValue *Val = VPI->getOperand(N: 0);
6188 VPValue *Addr = VPI->getOperand(N: 1);
6189 // We need to store the exiting value of the reduction, so use the blend
6190 // if tail folded.
6191 if (auto *Blend = VPlanPatternMatch::findUserOf<VPBlendRecipe>(V: Val))
6192 Val = Blend;
6193 [[maybe_unused]] auto *Rdx =
6194 VPlanPatternMatch::findUserOf<VPReductionPHIRecipe>(V: Val);
6195 assert((isa<VPIRValue>(Val) || !Rdx || Rdx->getBackedgeValue() == Val) &&
6196 "Store of reduction thats not the backedge value?");
6197 auto *Recipe = new VPReplicateRecipe(
6198 SI, {Val, Addr}, true /* IsUniform */, nullptr /*Mask*/, *VPI, *VPI,
6199 VPI->getDebugLoc());
6200 FinalRedStoresBuilder.insert(R: Recipe);
6201 }
6202 VPI->eraseFromParent();
6203 return true;
6204 }
6205
6206 return false;
6207}
6208
6209VPSingleDefRecipe *VPRecipeBuilder::handleReplication(VPInstruction *VPI,
6210 VFRange &Range) {
6211 auto *I = VPI->getUnderlyingInstr();
6212 bool IsUniform = LoopVectorizationPlanner::getDecisionAndClampRange(
6213 Predicate: [&](ElementCount VF) { return CM.isUniformAfterVectorization(I, VF); },
6214 Range);
6215
6216 bool IsPredicated = CM.isPredicatedInst(I);
6217
6218 // Even if the instruction is not marked as uniform, there are certain
6219 // intrinsic calls that can be effectively treated as such, so we check for
6220 // them here. Conservatively, we only do this for scalable vectors, since
6221 // for fixed-width VFs we can always fall back on full scalarization.
6222 if (!IsUniform && Range.Start.isScalable() && isa<IntrinsicInst>(Val: I)) {
6223 switch (cast<IntrinsicInst>(Val: I)->getIntrinsicID()) {
6224 case Intrinsic::assume:
6225 case Intrinsic::lifetime_start:
6226 case Intrinsic::lifetime_end:
6227 // For scalable vectors if one of the operands is variant then we still
6228 // want to mark as uniform, which will generate one instruction for just
6229 // the first lane of the vector. We can't scalarize the call in the same
6230 // way as for fixed-width vectors because we don't know how many lanes
6231 // there are.
6232 //
6233 // The reasons for doing it this way for scalable vectors are:
6234 // 1. For the assume intrinsic generating the instruction for the first
6235 // lane is still be better than not generating any at all. For
6236 // example, the input may be a splat across all lanes.
6237 // 2. For the lifetime start/end intrinsics the pointer operand only
6238 // does anything useful when the input comes from a stack object,
6239 // which suggests it should always be uniform. For non-stack objects
6240 // the effect is to poison the object, which still allows us to
6241 // remove the call.
6242 IsUniform = true;
6243 break;
6244 default:
6245 break;
6246 }
6247 }
6248 VPValue *BlockInMask = nullptr;
6249 if (!IsPredicated) {
6250 // Finalize the recipe for Instr, first if it is not predicated.
6251 LLVM_DEBUG(dbgs() << "LV: Scalarizing:" << *I << "\n");
6252 } else {
6253 LLVM_DEBUG(dbgs() << "LV: Scalarizing and predicating:" << *I << "\n");
6254 // Instructions marked for predication are replicated and a mask operand is
6255 // added initially. Masked replicate recipes will later be placed under an
6256 // if-then construct to prevent side-effects. Generate recipes to compute
6257 // the block mask for this region.
6258 BlockInMask = VPI->getMask();
6259 }
6260
6261 // Note that there is some custom logic to mark some intrinsics as uniform
6262 // manually above for scalable vectors, which this assert needs to account for
6263 // as well.
6264 assert((Range.Start.isScalar() || !IsUniform || !IsPredicated ||
6265 (Range.Start.isScalable() && isa<IntrinsicInst>(I))) &&
6266 "Should not predicate a uniform recipe");
6267 if (IsUniform) {
6268 return VPBuilder::createSingleScalarOp(
6269 Opcode: VPI->getOpcode(), Operands: VPI->operandsWithoutMask(), Mask: BlockInMask, Flags: *VPI, Metadata: *VPI,
6270 DL: VPI->getDebugLoc(), ResultTy: VPI->getScalarType(), UV: I);
6271 }
6272 auto *Recipe = new VPReplicateRecipe(I, VPI->operandsWithoutMask(),
6273 /*IsSingleScalar=*/false, BlockInMask,
6274 *VPI, *VPI, VPI->getDebugLoc());
6275 return Recipe;
6276}
6277
6278VPRecipeBase *
6279VPRecipeBuilder::tryToCreateWidenNonPhiRecipe(VPSingleDefRecipe *R,
6280 VFRange &Range) {
6281 assert(!R->isPhi() && "phis must be handled earlier");
6282 auto *VPI = cast<VPInstruction>(Val: R);
6283 assert(VPI->getOpcode() != Instruction::Call &&
6284 "Call should have been handled by makeCallWideningDecisions");
6285
6286 // All widen recipes below deal only with VF > 1.
6287 if (LoopVectorizationPlanner::getDecisionAndClampRange(
6288 Predicate: [&](ElementCount VF) { return VF.isScalar(); }, Range))
6289 return nullptr;
6290
6291 Instruction *Instr = R->getUnderlyingInstr();
6292 assert(!is_contained({Instruction::Load, Instruction::Store},
6293 VPI->getOpcode()) &&
6294 "Should have been handled prior to this!");
6295
6296 // We can only replicate an extractvalue if its operand generates per lane in
6297 // the same block, otherwise we would need to extract a lane from its struct
6298 // operand which is invalid.
6299 if (VPI->getOpcode() == Instruction::ExtractValue &&
6300 !vputils::isSingleScalar(VPV: VPI->getOperand(N: 0)))
6301 if (VPRecipeBase *OpR = VPI->getOperand(N: 0)->getDefiningRecipe())
6302 if (!vputils::doesGeneratePerAllLanes(R: OpR) ||
6303 OpR->getParent() != VPI->getParent())
6304 return tryToWiden(VPI);
6305
6306 if (!shouldWiden(I: Instr, Range))
6307 return nullptr;
6308
6309 if (VPI->getOpcode() == Instruction::GetElementPtr) {
6310 auto *GEP = cast<GetElementPtrInst>(Val: Instr);
6311 return new VPWidenGEPRecipe(GEP->getSourceElementType(),
6312 VPI->operandsWithoutMask(), *VPI,
6313 VPI->getDebugLoc(), GEP);
6314 }
6315
6316 if (Instruction::isCast(Opcode: VPI->getOpcode())) {
6317 auto *CI = cast<CastInst>(Val: Instr);
6318 return new VPWidenCastRecipe(CI->getOpcode(), VPI->getOperand(N: 0),
6319 VPI->getScalarType(), CI, *VPI, *VPI,
6320 VPI->getDebugLoc());
6321 }
6322
6323 return tryToWiden(VPI);
6324}
6325
6326// To allow RUN_VPLAN_PASS to print the VPlan after VF/UF independent
6327// optimizations.
6328static void printOptimizedVPlan(VPlan &) {}
6329
6330#ifndef NDEBUG
6331/// Cross-check the execution frequencies recorded in \p Plan against
6332/// BlockFrequencyInfo for the blocks of \p OrigLoop.
6333/// FIXME: Temporary verification aid, to be removed.
6334static bool verifyExecutionFrequenciesMatchBFI(VPlan &Plan, Loop *OrigLoop,
6335 LoopInfo *LI,
6336 LoopVectorizationCostModel &CM) {
6337 // Limited to loops with the latch as only exiting block
6338 if (OrigLoop->getExitingBlock() != OrigLoop->getLoopLatch())
6339 return true;
6340
6341 // Visit the loop body in the same order as recordExecutionFrequencies. Both
6342 // are reverse post-orders of the same CFG, so indices correspond.
6343 VPBasicBlock *Header = VPBlockUtils::getPlainCFGHeaderAndLatch(Plan).first;
6344 SmallVector<VPBasicBlock *> Blocks = vp_rpo_plain_cfg_loop_body(Header);
6345 assert(Blocks.size() == OrigLoop->getNumBlocks() &&
6346 "loop body and original loop must have the same blocks");
6347
6348 LoopBlocksRPO OrigRPO(OrigLoop);
6349 OrigRPO.perform(LI);
6350
6351 // Only request the expensive BFI once the cheap bail-outs are past.
6352 BlockFrequencyInfo &BFI = CM.getBFI();
6353 uint64_t HeaderFreq = BFI.getBlockFreq(OrigLoop->getHeader()).getFrequency();
6354 if (HeaderFreq == 0)
6355 return true;
6356
6357 // BFI's fixed-point mass propagation loses up to 1 ULP per edge, so bound the
6358 // error by the number of edges in the region.
6359 uint64_t Edges = 0;
6360 for (const VPBasicBlock *VPBB : Blocks)
6361 Edges += VPBB->getNumSuccessors();
6362 uint64_t Tolerance = Edges + BranchProbability::getDenominator() / HeaderFreq;
6363
6364 for (const auto &[VPBB, BB] :
6365 zip_equal(drop_begin(Blocks), drop_begin(OrigRPO))) {
6366 // Nothing to check for blocks without a recorded frequency.
6367 std::optional<VPExecutionFrequency> Freq =
6368 getRecordedExecutionFrequency(VPBB);
6369 if (!Freq)
6370 continue;
6371 BranchProbability Computed = vputils::getExecutionProbability(Freq->Freq);
6372
6373 // Clamp to the header's frequency, which BFI's rounding may exceed.
6374 uint64_t BBFreq = BFI.getBlockFreq(BB).getFrequency();
6375 BranchProbability Expected = BranchProbability::getBranchProbability(
6376 std::min(BBFreq, HeaderFreq), HeaderFreq);
6377 if (AbsoluteDifference(Computed.getNumerator(), Expected.getNumerator()) <=
6378 Tolerance)
6379 continue;
6380
6381 errs() << "Block frequency mismatch for " << VPBB->getName() << ": VPlan "
6382 << Computed << ", BlockFrequencyInfo " << Expected << "\n";
6383 return false;
6384 }
6385 return true;
6386}
6387#endif
6388
6389VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
6390 bool IsInnerLoop = OrigLoop->isInnermost();
6391
6392 // Set up loop versioning for inner loops with memory runtime checks.
6393 // Outer loops don't have LoopAccessInfo since canVectorizeMemory() is not
6394 // called for them.
6395 std::optional<LoopVersioning> LVer;
6396 if (IsInnerLoop) {
6397 const LoopAccessInfo *LAI = Legal->getLAI();
6398 LVer.emplace(args: *LAI, args: LAI->getRuntimePointerChecking()->getChecks(), args&: OrigLoop,
6399 args&: LI, args&: DT, args: PSE.getSE());
6400 if (!LAI->getRuntimePointerChecking()->getChecks().empty() &&
6401 !LAI->getRuntimePointerChecking()->getDiffChecks()) {
6402 // Only use noalias metadata when using memory checks guaranteeing no
6403 // overlap across all iterations.
6404 LVer->prepareNoAliasMetadata();
6405 }
6406 }
6407
6408 // Create initial base VPlan0, to serve as common starting point for all
6409 // candidates built later for specific VF ranges.
6410 auto VPlan0 = VPlanTransforms::buildVPlan0(
6411 TheLoop: OrigLoop, LI&: *LI, InductionTy: Legal->getWidestInductionType(), PSE,
6412 LVer: LVer ? &*LVer : nullptr, GetBPI);
6413
6414 VPDominatorTree VPDT(*VPlan0);
6415 if (const LoopAccessInfo *LAI = Legal->getLAI())
6416 RUN_VPLAN_PASS(VPlanTransforms::replaceSymbolicStrides, *VPlan0, PSE,
6417 LAI->getSymbolicStrides(), VPDT);
6418 RUN_VPLAN_PASS(VPlanTransforms::combineRecipes, *VPlan0);
6419 RUN_VPLAN_PASS(VPlanTransforms::removeDeadRecipes, *VPlan0);
6420 if (IsInnerLoop) {
6421 RUN_VPLAN_PASS(VPlanTransforms::recordExecutionFrequencies, *VPlan0);
6422 assert(verifyExecutionFrequenciesMatchBFI(*VPlan0, OrigLoop, LI, *CM) &&
6423 "execution frequencies do not match the loop's block frequencies");
6424 }
6425
6426 // Create recipes for header phis. For outer loops, reductions, recurrences
6427 // and in-loop reductions are empty since legality doesn't detect them.
6428 if (!RUN_VPLAN_PASS(
6429 VPlanTransforms::createHeaderPhiRecipes, *VPlan0, PSE, *OrigLoop, ORE,
6430 VPDT, Legal->getInductionVars(), Legal->getReductionVars(),
6431 Legal->getFixedOrderRecurrences(), Config.getInLoopReductions(),
6432 Config.getHints().allowReordering())) {
6433 return nullptr;
6434 }
6435
6436 if (const LoopAccessInfo *LAI = Legal->getLAI())
6437 RUN_VPLAN_PASS(VPlanTransforms::replaceSymbolicStrides, *VPlan0, PSE,
6438 LAI->getSymbolicStrides(), VPDT);
6439
6440 // Add surviving induction predicates to PSE and check constraints.
6441 bool ForceVectorization =
6442 Config.getHints().getForce() == LoopVectorizeHints::FK_Enabled;
6443 bool OptForSize =
6444 !ForceVectorization &&
6445 (CM->EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
6446 CM->EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
6447 unsigned SCEVCheckThreshold = ForceVectorization
6448 ? PragmaVectorizeSCEVCheckThreshold
6449 : VectorizeSCEVCheckThreshold;
6450 if (!RUN_VPLAN_PASS(VPlanTransforms::finalizeSCEVPredicates, *VPlan0, PSE,
6451 OptForSize, SCEVCheckThreshold, ORE, OrigLoop))
6452 return nullptr;
6453
6454 RUN_VPLAN_PASS(VPlanTransforms::addMiddleCheck, *VPlan0);
6455
6456 // If we're vectorizing a loop with an uncountable exit, make sure that the
6457 // recipes are safe to handle.
6458 // TODO: Remove this once we can properly check the VPlan itself for both
6459 // the presence of an uncountable exit and the presence of stores in
6460 // the loop inside handleUncountableEarlyExits itself.
6461 if (Legal->hasUncountableEarlyExit()) {
6462 if (!RUN_VPLAN_PASS(VPlanTransforms::splitCombinedExits, *VPlan0, PSE,
6463 OrigLoop))
6464 return nullptr;
6465
6466 // TODO: Check target preference for style.
6467 UncountableExitStyle EEStyle =
6468 Legal->hasUncountableExitWithSideEffects()
6469 ? UncountableExitStyle::MaskedHandleExitInScalarLoop
6470 : UncountableExitStyle::ReadOnly;
6471 if (!RUN_VPLAN_PASS(VPlanTransforms::handleUncountableEarlyExits, *VPlan0,
6472 ORE, OrigLoop, PSE, *DT, Legal->getAssumptionCache(),
6473 EEStyle)) {
6474 return nullptr;
6475 }
6476 } else {
6477 RUN_VPLAN_PASS(VPlanTransforms::handleCountableEarlyExits, *VPlan0);
6478 }
6479
6480 RUN_VPLAN_PASS(VPlanTransforms::createLoopRegions, *VPlan0,
6481 getDebugLocFromInstOrOperands(Legal->getPrimaryInduction()));
6482 if (CM->foldTailByMasking())
6483 RUN_VPLAN_PASS(VPlanTransforms::foldTailByMasking, *VPlan0);
6484
6485 RUN_VPLAN_PASS(VPlanTransforms::introduceMasksAndLinearize, *VPlan0);
6486
6487 return VPlan0;
6488}
6489
6490void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
6491 ElementCount MaxVF) {
6492 if (ElementCount::isKnownGT(LHS: MinVF, RHS: MaxVF))
6493 return;
6494
6495 auto MaxVFTimes2 = MaxVF * 2;
6496 for (ElementCount VF = MinVF; ElementCount::isKnownLT(LHS: VF, RHS: MaxVFTimes2);) {
6497 VFRange SubRange = {VF, MaxVFTimes2};
6498 auto Plan =
6499 tryToBuildVPlan(InitialPlan: std::unique_ptr<VPlan>(VPlan1.duplicate()), Range&: SubRange);
6500 VF = SubRange.End;
6501
6502 if (!Plan)
6503 continue;
6504
6505 // Now optimize the initial VPlan.
6506 RUN_VPLAN_PASS(VPlanTransforms::hoistPredicatedLoads, *Plan, PSE, OrigLoop);
6507 RUN_VPLAN_PASS(VPlanTransforms::sinkPredicatedStores, *Plan, PSE, OrigLoop);
6508 RUN_VPLAN_PASS(VPlanTransforms::truncateToMinimalBitwidths, *Plan,
6509 Config.getMinimalBitwidths());
6510 RUN_VPLAN_PASS(VPlanTransforms::optimize, *Plan);
6511 // TODO: try to put addExplicitVectorLength close to addActiveLaneMask
6512 if (CM->foldTailWithEVL()) {
6513 RUN_VPLAN_PASS(VPlanTransforms::addExplicitVectorLength, *Plan,
6514 Config.getMaxSafeElements());
6515 RUN_VPLAN_PASS(VPlanTransforms::optimizeEVLMasks, *Plan);
6516 }
6517
6518 if (auto P =
6519 RUN_VPLAN_PASS(VPlanTransforms::narrowInterleaveGroups, *Plan, TTI))
6520 VPlans.push_back(Elt: std::move(P));
6521
6522 TailFoldingStyle Style = CM->getTailFoldingStyle();
6523 RUN_VPLAN_PASS(VPlanTransforms::materializeHeaderMask, *Plan,
6524 useActiveLaneMask(Style),
6525 useActiveLaneMaskForControlFlow(Style));
6526
6527 RUN_VPLAN_PASS_NO_VERIFY(printOptimizedVPlan, *Plan);
6528 assert(verifyVPlanIsValid(*Plan) && "VPlan is invalid");
6529 VPlans.push_back(Elt: std::move(Plan));
6530 }
6531}
6532
6533VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
6534 VFRange &Range) {
6535
6536 // For outer loops, the plan only needs basic recipe conversion and induction
6537 // live-out optimization; the full inner-loop recipe building below does not
6538 // apply (no widening decisions, interleave groups, reductions, etc.).
6539 if (Plan->isOuterLoop()) {
6540 for (ElementCount VF : Range)
6541 Plan->addVF(VF);
6542 if (!RUN_VPLAN_PASS(VPlanTransforms::tryToConvertVPInstructionsToVPRecipes,
6543 *Plan, *TLI, PSE, OrigLoop))
6544 return nullptr;
6545 RUN_VPLAN_PASS(VPlanTransforms::optimizeInductionLiveOutUsers, *Plan, PSE,
6546 OrigLoop);
6547 return Plan;
6548 }
6549
6550 using namespace llvm::VPlanPatternMatch;
6551 SmallPtrSet<const InterleaveGroup<Instruction> *, 1> InterleaveGroups;
6552
6553 // ---------------------------------------------------------------------------
6554 // Build initial VPlan: Scan the body of the loop in a topological order to
6555 // visit each basic block after having visited its predecessor basic blocks.
6556 // ---------------------------------------------------------------------------
6557
6558 bool RequiresScalarEpilogueCheck =
6559 LoopVectorizationPlanner::getDecisionAndClampRange(
6560 Predicate: [this](ElementCount VF) {
6561 return !CM->requiresScalarEpilogue(IsVectorizing: VF.isVector());
6562 },
6563 Range);
6564 // Update the branch in the middle block if a scalar epilogue is required.
6565 VPBasicBlock *MiddleVPBB = Plan->getMiddleBlock();
6566 if (!RequiresScalarEpilogueCheck && MiddleVPBB->getNumSuccessors() == 2) {
6567 auto *BranchOnCond = cast<VPInstruction>(Val: MiddleVPBB->getTerminator());
6568 assert(MiddleVPBB->getSuccessors()[1] == Plan->getScalarPreheader() &&
6569 "second successor must be scalar preheader");
6570 BranchOnCond->setOperand(I: 0, New: Plan->getFalse());
6571 }
6572
6573 // Don't use getDecisionAndClampRange here, because we don't know the UF
6574 // so this function is better to be conservative, rather than to split
6575 // it up into different VPlans.
6576 // TODO: Consider using getDecisionAndClampRange here to split up VPlans.
6577 bool IVUpdateMayOverflow = false;
6578 for (ElementCount VF : Range)
6579 IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(Cost: CM.get(), VF);
6580
6581 TailFoldingStyle Style = CM->getTailFoldingStyle();
6582 // Use NUW for the induction increment if we proved that it won't overflow in
6583 // the vector loop or when not folding the tail. In the later case, we know
6584 // that the canonical induction increment will not overflow as the vector trip
6585 // count is >= increment and a multiple of the increment.
6586 VPRegionBlock *LoopRegion = Plan->getVectorLoopRegion();
6587 bool HasNUW = !IVUpdateMayOverflow || Style == TailFoldingStyle::None;
6588 if (!HasNUW) {
6589 auto *IVInc =
6590 LoopRegion->getExitingBasicBlock()->getTerminator()->getOperand(N: 0);
6591 assert(match(IVInc,
6592 m_VPInstruction<Instruction::Add>(
6593 m_Specific(LoopRegion->getCanonicalIV()), m_VPValue())) &&
6594 "Did not find the canonical IV increment");
6595 LoopRegion->clearCanonicalIVNUW(Increment: cast<VPInstruction>(Val: IVInc));
6596 }
6597
6598 // ---------------------------------------------------------------------------
6599 // Pre-construction: record ingredients whose recipes we'll need to further
6600 // process after constructing the initial VPlan.
6601 // ---------------------------------------------------------------------------
6602
6603 // For each interleave group which is relevant for this (possibly trimmed)
6604 // Range, add it to the set of groups to be later applied to the VPlan and add
6605 // placeholders for its members' Recipes which we'll be replacing with a
6606 // single VPInterleaveRecipe.
6607 for (InterleaveGroup<Instruction> *IG : IAI.getInterleaveGroups()) {
6608 auto ApplyIG = [IG, this](ElementCount VF) -> bool {
6609 bool Result = (VF.isVector() && // Query is illegal for VF == 1
6610 CM->getWideningDecision(I: IG->getInsertPos(), VF) ==
6611 LoopVectorizationCostModel::CM_Interleave);
6612 // For scalable vectors, the interleave factors must be <= 8 since we
6613 // require the (de)interleaveN intrinsics instead of shufflevectors.
6614 assert((!Result || !VF.isScalable() || IG->getFactor() <= 8) &&
6615 "Unsupported interleave factor for scalable vectors");
6616 return Result;
6617 };
6618 if (!getDecisionAndClampRange(Predicate: ApplyIG, Range))
6619 continue;
6620 InterleaveGroups.insert(Ptr: IG);
6621 }
6622
6623 // ---------------------------------------------------------------------------
6624 // Construct wide recipes and apply predication for original scalar
6625 // VPInstructions in the loop.
6626 // ---------------------------------------------------------------------------
6627 VPRecipeBuilder RecipeBuilder(*Plan, Legal, *CM, Builder);
6628
6629 RUN_VPLAN_PASS(VPlanTransforms::createInLoopReductionRecipes, *Plan,
6630 Range.Start);
6631
6632 VPCostContext CostCtx(*TLI, *Plan, *CM, Config);
6633
6634 RUN_VPLAN_PASS(VPlanTransforms::makeMemOpWideningDecisions, *Plan, Range,
6635 RecipeBuilder, CostCtx);
6636
6637 RUN_VPLAN_PASS(VPlanTransforms::makeScalarizationDecisions, *Plan, Range);
6638
6639 // Widened calls may only use the first lane of some operands.
6640 if (RUN_VPLAN_PASS(VPlanTransforms::makeCallWideningDecisions, *Plan, Range,
6641 RecipeBuilder, CostCtx))
6642 RUN_VPLAN_PASS(VPlanTransforms::makeScalarizationDecisions, *Plan, Range);
6643
6644 RUN_VPLAN_PASS(VPlanTransforms::narrowInductionTruncates, *Plan, Range, TTI,
6645 PSE);
6646
6647 // Convert remaining VPInstructions to widen or replicate recipes.
6648 // TODO: This legacy code should eventually be migrated to VPlan.
6649 VPBasicBlock *HeaderVPBB = LoopRegion->getEntryBasicBlock();
6650 for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
6651 Range: vp_depth_first_shallow(G: HeaderVPBB))) {
6652 // All types but VPInstructions are already widened and don't need extra
6653 // processing. We process VPInstructions below.
6654 assert(
6655 all_of(
6656 make_range(VPBB->getFirstNonPhi(), VPBB->end()),
6657 IsaPred<VPWidenCanonicalIVRecipe, VPBlendRecipe, VPReductionRecipe,
6658 VPReplicateRecipe, VPWidenLoadRecipe, VPWidenStoreRecipe,
6659 VPWidenCallRecipe, VPWidenIntrinsicRecipe,
6660 VPVectorPointerRecipe, VPVectorEndPointerRecipe,
6661 VPHistogramRecipe, VPInstruction>) &&
6662 "Unexpected recipe");
6663 for (VPInstruction &VPI :
6664 make_early_inc_range(Range: make_isa_range<VPInstruction>(Range&: *VPBB))) {
6665 // We represent single-scalar casts directly as VPInstructions.
6666 if (Instruction::isCast(Opcode: VPI.getOpcode()) &&
6667 vputils::onlyFirstLaneUsed(Def: &VPI))
6668 continue;
6669
6670 // Only VPInstrutions with an underlying value need to be processed.
6671 if (!VPI.getUnderlyingValue())
6672 continue;
6673
6674 Builder.setInsertPoint(&VPI);
6675
6676 VPRecipeBase *Recipe =
6677 RecipeBuilder.tryToCreateWidenNonPhiRecipe(R: &VPI, Range);
6678 if (!Recipe)
6679 Recipe = RecipeBuilder.handleReplication(VPI: &VPI, Range);
6680 Builder.insert(R: Recipe);
6681
6682 if (Recipe->getNumDefinedValues() == 1) {
6683 VPI.replaceAllUsesWith(New: Recipe->getVPSingleValue());
6684 } else {
6685 assert(Recipe->getNumDefinedValues() == 0 &&
6686 "Unexpected multidef recipe");
6687 }
6688 VPI.eraseFromParent();
6689 }
6690 }
6691
6692 assert(isa<VPRegionBlock>(LoopRegion) &&
6693 !LoopRegion->getEntryBasicBlock()->empty() &&
6694 "entry block must be set to a VPRegionBlock having a non-empty entry "
6695 "VPBasicBlock");
6696
6697 RUN_VPLAN_PASS(VPlanTransforms::adjustFirstOrderRecurrenceMiddleUsers, *Plan,
6698 Range);
6699
6700 // ---------------------------------------------------------------------------
6701 // Transform initial VPlan: Apply previously taken decisions, in order, to
6702 // bring the VPlan to its final state.
6703 // ---------------------------------------------------------------------------
6704
6705 addReductionResultComputation(Plan, MinVF: Range.Start);
6706
6707 // Optimize FindIV reductions to use sentinel-based approach when possible.
6708 RUN_VPLAN_PASS(VPlanTransforms::optimizeFindIVReductions, *Plan, PSE,
6709 *OrigLoop);
6710 RUN_VPLAN_PASS(VPlanTransforms::optimizeInductionLiveOutUsers, *Plan, PSE,
6711 OrigLoop);
6712
6713 // Apply mandatory transformation to handle reductions with multiple in-loop
6714 // uses if possible, bail out otherwise.
6715 if (!RUN_VPLAN_PASS(VPlanTransforms::handleMultiUseReductions, *Plan, ORE,
6716 OrigLoop))
6717 return nullptr;
6718 // Apply mandatory transformation to handle FP maxnum/minnum reduction with
6719 // NaNs if possible, bail out otherwise.
6720 if (!RUN_VPLAN_PASS(VPlanTransforms::handleMaxMinNumReductions, *Plan))
6721 return nullptr;
6722
6723 // Create whole-vector selects for find-last recurrences.
6724 if (!RUN_VPLAN_PASS(VPlanTransforms::handleFindLastReductions, *Plan))
6725 return nullptr;
6726
6727 RUN_VPLAN_PASS(VPlanTransforms::removeBranchOnConst, *Plan, false);
6728
6729 // Create partial reduction recipes for scaled reductions and transform
6730 // recipes to abstract recipes if it is legal and beneficial and clamp the
6731 // range for better cost estimation.
6732 RUN_VPLAN_PASS(VPlanTransforms::createPartialReductions, *Plan, CostCtx,
6733 Range);
6734 RUN_VPLAN_PASS(VPlanTransforms::convertToAbstractRecipes, *Plan, CostCtx,
6735 Range);
6736
6737 // Interleave memory: for each Interleave Group we marked earlier as relevant
6738 // for this VPlan, replace the Recipes widening its memory instructions with a
6739 // single VPInterleaveRecipe at its insertion point.
6740 RUN_VPLAN_PASS(VPlanTransforms::createInterleaveGroups, *Plan,
6741 InterleaveGroups, CM->isEpilogueAllowed());
6742
6743 // Convert memory recipes to strided access recipes if the strided access is
6744 // legal and profitable.
6745 RUN_VPLAN_PASS(VPlanTransforms::convertToStridedAccesses, *Plan, PSE,
6746 *OrigLoop, CostCtx, Range);
6747
6748 // Ensure scalar VF plans only contain VF=1, as required by hasScalarVFOnly.
6749 if (Range.Start.isScalar())
6750 Range.End = Range.Start * 2;
6751
6752 for (ElementCount VF : Range)
6753 Plan->addVF(VF);
6754 Plan->setName("Initial VPlan");
6755
6756 RUN_VPLAN_PASS(VPlanTransforms::dropPoisonGeneratingRecipes, *Plan);
6757
6758 if (CM->maskPartialAliasing())
6759 RUN_VPLAN_PASS(VPlanTransforms::attachAliasMaskToHeaderMask, *Plan);
6760
6761 assert(verifyVPlanIsValid(*Plan) && "VPlan is invalid");
6762 return Plan;
6763}
6764
6765void LoopVectorizationPlanner::addReductionResultComputation(
6766 VPlanPtr &Plan, ElementCount MinVF) {
6767 using namespace VPlanPatternMatch;
6768 VPRegionBlock *VectorLoopRegion = Plan->getVectorLoopRegion();
6769 VPBasicBlock *MiddleVPBB = Plan->getMiddleBlock();
6770 VPBasicBlock *LatchVPBB = VectorLoopRegion->getExitingBasicBlock();
6771 Builder.setInsertPoint(&*std::prev(x: std::prev(x: LatchVPBB->end())));
6772 VPBasicBlock::iterator IP = MiddleVPBB->getFirstNonPhi();
6773 VPValue *HeaderMask = Plan->getVectorLoopRegion()->getHeaderMask();
6774 for (VPRecipeBase &R : make_early_inc_range(
6775 Range: Plan->getVectorLoopRegion()->getEntryBasicBlock()->phis())) {
6776 VPReductionPHIRecipe *PhiR = dyn_cast<VPReductionPHIRecipe>(Val: &R);
6777 if (!PhiR)
6778 continue;
6779
6780 // Clean up reductions that have become invariant.
6781 if (PhiR->getBackedgeValue() == PhiR) {
6782 PhiR->replaceAllUsesWith(New: PhiR->getStartValue());
6783 PhiR->eraseFromParent();
6784 continue;
6785 }
6786
6787 RecurKind RecurrenceKind = PhiR->getRecurrenceKind();
6788 const RecurrenceDescriptor &RdxDesc = Legal->getRecurrenceDescriptor(
6789 PN: cast<PHINode>(Val: PhiR->getUnderlyingInstr()));
6790 Type *PhiTy = PhiR->getScalarType();
6791
6792 // Convert a VPBlendRecipe backedge to a select.
6793 if (auto *Blend = dyn_cast<VPBlendRecipe>(Val: PhiR->getBackedgeValue())) {
6794 if (Blend->getNumIncomingValues() == 2 &&
6795 Blend->getMask(Idx: 0) == HeaderMask) {
6796 auto *Sel = VPBuilder(Blend).createSelect(
6797 Cond: Blend->getMask(Idx: 0), TrueVal: Blend->getIncomingValue(Idx: 0),
6798 FalseVal: Blend->getIncomingValue(Idx: 1), DL: {}, Name: "", Flags: *Blend);
6799 Blend->replaceAllUsesWith(New: Sel);
6800 Blend->eraseFromParent();
6801 }
6802 }
6803
6804 auto *OrigExitingVPV = PhiR->getBackedgeValue();
6805 auto *NewExitingVPV = OrigExitingVPV;
6806
6807 // Remove the predicated select if the target doesn't want it.
6808 VPValue *V;
6809 if (!CM->usePredicatedReductionSelect(
6810 RecurrenceKind, HasUsesOutsideReductionChain: PhiR->hasUsesOutsideReductionChain()) &&
6811 match(V: PhiR->getBackedgeValue(),
6812 P: m_Select(Op0: m_Specific(VPV: HeaderMask), Op1: m_VPValue(V), Op2: m_Specific(VPV: PhiR))))
6813 PhiR->setBackedgeValue(V);
6814
6815 // We want code in the middle block to appear to execute on the location of
6816 // the scalar loop's latch terminator because: (a) it is all compiler
6817 // generated, (b) these instructions are always executed after evaluating
6818 // the latch conditional branch, and (c) other passes may add new
6819 // predecessors which terminate on this line. This is the easiest way to
6820 // ensure we don't accidentally cause an extra step back into the loop while
6821 // debugging.
6822 DebugLoc ExitDL = OrigLoop->getLoopLatch()->getTerminator()->getDebugLoc();
6823
6824 // TODO: At the moment ComputeReductionResult also drives creation of the
6825 // bc.merge.rdx phi nodes, hence it needs to be created unconditionally here
6826 // even for in-loop reductions, until the reduction resume value handling is
6827 // also modeled in VPlan.
6828 VPInstruction *FinalReductionResult;
6829 VPBuilder::InsertPointGuard Guard(Builder);
6830 Builder.setInsertPoint(TheBB: MiddleVPBB, IP);
6831 // For AnyOf reductions, find the select among PhiR's users and convert
6832 // the reduction phi to operate on bools before creating the final
6833 // reduction result.
6834 if (RecurrenceDescriptor::isAnyOfRecurrenceKind(Kind: RecurrenceKind)) {
6835 auto *AnyOfSelect = cast<VPSingleDefRecipe>(
6836 Val: findUserOf(V: PhiR, P: m_Select(Op0: m_VPValue(), Op1: m_VPValue(), Op2: m_VPValue())));
6837 VPValue *Start = PhiR->getStartValue();
6838 bool TrueValIsPhi = AnyOfSelect->getOperand(N: 1) == PhiR;
6839 // NewVal is the non-phi operand of the select.
6840 VPValue *NewVal = TrueValIsPhi ? AnyOfSelect->getOperand(N: 2)
6841 : AnyOfSelect->getOperand(N: 1);
6842
6843 // Adjust AnyOf reductions; replace the reduction phi for the selected
6844 // value with a boolean reduction phi node to check if the condition is
6845 // true in any iteration. The final value is selected by the final
6846 // ComputeReductionResult.
6847 VPValue *Cmp = AnyOfSelect->getOperand(N: 0);
6848 // If the compare is checking the reduction PHI node, adjust it to check
6849 // the start value.
6850 if (VPRecipeBase *CmpR = Cmp->getDefiningRecipe())
6851 CmpR->replaceUsesOfWith(From: PhiR, To: PhiR->getStartValue());
6852 Builder.setInsertPoint(AnyOfSelect);
6853
6854 // If the true value of the select is the reduction phi, the new value
6855 // is selected if the negated condition is true in any iteration.
6856 if (TrueValIsPhi)
6857 Cmp = Builder.createNot(Operand: Cmp);
6858
6859 // Build a fresh i1 chain (phi, or, and i1 versions of any blend/select
6860 // the exiting value flows through).
6861 auto *NewPhiR =
6862 PhiR->cloneWithOperands(Start: Plan->getFalse(), BackedgeValue: Plan->getFalse());
6863 NewPhiR->insertBefore(InsertPos: PhiR);
6864 VPValue *NewExiting = Builder.createOr(LHS: NewPhiR, RHS: Cmp);
6865
6866 // The exiting value may flow through a chain of VPBlendRecipes and
6867 // select recipes (VPInstruction, VPWidenRecipe or VPReplicateRecipe with
6868 // Select opcode) before reaching OrigExitingVPV. Clone each chain link
6869 // in topological order so each clone refers to the already-rewritten i1
6870 // operands via Substitutions.
6871 DenseMap<VPValue *, VPValue *> Substitutions = {{AnyOfSelect, NewExiting},
6872 {PhiR, NewPhiR}};
6873 std::function<void(VPSingleDefRecipe *)> CloneChain =
6874 [&](VPSingleDefRecipe *Old) {
6875 if (Substitutions.contains(Val: Old))
6876 return;
6877 SmallVector<VPValue *> NewOps;
6878 for (VPValue *Op : Old->operands()) {
6879 if (isa<VPBlendRecipe>(Val: Op) ||
6880 match(V: Op, P: m_Select(Op0: m_VPValue(), Op1: m_VPValue(), Op2: m_VPValue())))
6881 CloneChain(cast<VPSingleDefRecipe>(Val: Op));
6882 NewOps.push_back(Elt: Substitutions.lookup_or(Val: Op, Default&: Op));
6883 }
6884 VPSingleDefRecipe *New;
6885 if (auto *B = dyn_cast<VPBlendRecipe>(Val: Old))
6886 New = B->cloneWithOperands(NewOperands: NewOps);
6887 else if (auto *W = dyn_cast<VPWidenRecipe>(Val: Old))
6888 New = W->cloneWithOperands(NewOperands: NewOps);
6889 else if (auto *Rep = dyn_cast<VPReplicateRecipe>(Val: Old))
6890 New = Rep->cloneWithOperands(NewOperands: NewOps);
6891 else
6892 New = cast<VPInstruction>(Val: Old)->cloneWithOperands(NewOperands: NewOps);
6893 New->insertBefore(InsertPos: Old);
6894 Substitutions[Old] = New;
6895 };
6896
6897 if (OrigExitingVPV != AnyOfSelect) {
6898 CloneChain(cast<VPSingleDefRecipe>(Val: OrigExitingVPV));
6899 NewExiting = Substitutions.lookup(Val: OrigExitingVPV);
6900 }
6901 NewPhiR->setOperand(I: 1, New: NewExiting);
6902 PhiR->replaceAllUsesWith(New: Plan->getPoison(Ty: PhiR->getScalarType()));
6903
6904 Builder.setInsertPoint(TheBB: MiddleVPBB, IP);
6905 FinalReductionResult =
6906 Builder.createAnyOfReduction(ChainOp: NewExiting, TrueVal: NewVal, FalseVal: Start, DL: ExitDL);
6907 } else {
6908 // If the vector reduction can be performed in a smaller type, we
6909 // truncate then extend the loop exit value to enable InstCombine to
6910 // evaluate the entire expression in the smaller type.
6911 VPValue *ReductionOp = NewExitingVPV;
6912 Instruction::CastOps ExtendOpc = Instruction::CastOpsEnd;
6913 if (MinVF.isVector() && PhiTy != RdxDesc.getRecurrenceType()) {
6914 assert(!PhiR->isInLoop() && "Unexpected truncated inloop reduction!");
6915 assert(!RecurrenceDescriptor::isMinMaxRecurrenceKind(RecurrenceKind) &&
6916 "Unexpected truncated min-max recurrence!");
6917 Type *RdxTy = RdxDesc.getRecurrenceType();
6918 ExtendOpc = RdxDesc.isSigned() ? Instruction::SExt : Instruction::ZExt;
6919 {
6920 VPBuilder::InsertPointGuard Guard(Builder);
6921 Builder.setInsertPoint(
6922 TheBB: NewExitingVPV->getDefiningRecipe()->getParent(),
6923 IP: std::next(x: NewExitingVPV->getDefiningRecipe()->getIterator()));
6924 ReductionOp =
6925 Builder.createWidenCast(Opcode: Instruction::Trunc, Op: NewExitingVPV, ResultTy: RdxTy);
6926 VPWidenCastRecipe *Extnd =
6927 Builder.createWidenCast(Opcode: ExtendOpc, Op: ReductionOp, ResultTy: PhiTy);
6928 if (PhiR->getOperand(N: 1) == NewExitingVPV)
6929 PhiR->setOperand(I: 1, New: Extnd);
6930 }
6931 }
6932
6933 VPIRFlags Flags(RecurrenceKind, PhiR->isOrdered(), PhiR->isInLoop(),
6934 PhiR->getFastMathFlagsOrNone());
6935 FinalReductionResult = Builder.createNaryOp(
6936 Opcode: VPInstruction::ComputeReductionResult, Operands: {ReductionOp}, Flags, DL: ExitDL);
6937 if (ExtendOpc != Instruction::CastOpsEnd)
6938 FinalReductionResult = Builder.createScalarCast(
6939 Opcode: ExtendOpc, Op: FinalReductionResult, ResultTy: PhiTy, DL: {});
6940 }
6941
6942 // Update all users outside the vector region. Also replace redundant
6943 // extracts.
6944 for (auto *U : to_vector(Range: OrigExitingVPV->users())) {
6945 auto *Parent = cast<VPRecipeBase>(Val: U)->getParent();
6946 if (FinalReductionResult == U || Parent->getParent())
6947 continue;
6948 // Skip ComputeReductionResult and FindIV reductions when they are not the
6949 // final result.
6950 if (match(U, P: m_VPInstruction<VPInstruction::ComputeReductionResult>()) ||
6951 (RecurrenceDescriptor::isFindIVRecurrenceKind(Kind: RecurrenceKind) &&
6952 match(U, P: m_VPInstruction<Instruction::ICmp>())))
6953 continue;
6954 U->replaceUsesOfWith(From: OrigExitingVPV, To: FinalReductionResult);
6955
6956 // Look through ExtractLastPart.
6957 if (match(U, P: m_ExtractLastPart(Op0: m_VPValue())))
6958 U = cast<VPInstruction>(Val: U)->getSingleUser();
6959
6960 if (match(U, P: m_CombineOr(Ps: m_ExtractLane(Op0: m_VPValue(), Op1: m_VPValue()),
6961 Ps: m_ExtractLastLane(Op0: m_VPValue()))))
6962 cast<VPInstruction>(Val: U)->replaceAllUsesWith(New: FinalReductionResult);
6963 }
6964
6965 RecurKind RK = PhiR->getRecurrenceKind();
6966 if ((!RecurrenceDescriptor::isAnyOfRecurrenceKind(Kind: RK) &&
6967 !RecurrenceDescriptor::isFindIVRecurrenceKind(Kind: RK) &&
6968 !RecurrenceDescriptor::isMinMaxRecurrenceKind(Kind: RK) &&
6969 !RecurrenceDescriptor::isFindLastRecurrenceKind(Kind: RK))) {
6970 VPBuilder PHBuilder(Plan->getVectorPreheader());
6971 VPValue *Iden = Plan->getOrAddLiveIn(
6972 V: getRecurrenceIdentity(K: RK, Tp: PhiTy, FMF: PhiR->getFastMathFlagsOrNone()));
6973 auto *ScaleFactorVPV = Plan->getConstantInt(BitWidth: 32, Val: 1);
6974 VPValue *StartV = PHBuilder.createNaryOp(
6975 Opcode: VPInstruction::ReductionStartVector,
6976 Operands: {PhiR->getStartValue(), Iden, ScaleFactorVPV}, Flags: *PhiR);
6977 PhiR->setOperand(I: 0, New: StartV);
6978 }
6979 }
6980
6981 RUN_VPLAN_PASS(VPlanTransforms::clearReductionWrapFlags, *Plan);
6982}
6983
6984void LoopVectorizationPlanner::attachRuntimeChecks(
6985 VPlan &Plan, GeneratedRTChecks &RTChecks, bool HasBranchWeights) const {
6986 const auto &[SCEVCheckCond, SCEVCheckBlock] = RTChecks.getSCEVChecks();
6987 if (SCEVCheckBlock && SCEVCheckBlock->hasNPredecessors(N: 0)) {
6988 assert((!Config.OptForSize ||
6989 Config.getHints().getForce() == LoopVectorizeHints::FK_Enabled) &&
6990 "Cannot SCEV check stride or overflow when optimizing for size");
6991 RUN_VPLAN_PASS(VPlanTransforms::attachCheckBlock, Plan, SCEVCheckCond,
6992 SCEVCheckBlock, HasBranchWeights);
6993 }
6994 const auto &[MemCheckCond, MemCheckBlock] = RTChecks.getMemRuntimeChecks();
6995 if (MemCheckBlock && MemCheckBlock->hasNPredecessors(N: 0)) {
6996 // VPlan-native path does not do any analysis for runtime checks
6997 // currently.
6998 assert((!EnableVPlanNativePath || !Plan.isOuterLoop()) &&
6999 "Runtime checks are not supported for outer loops yet");
7000
7001 if (Config.OptForSize) {
7002 assert(
7003 Config.getHints().getForce() == LoopVectorizeHints::FK_Enabled &&
7004 "Cannot emit memory checks when optimizing for size, unless forced "
7005 "to vectorize.");
7006 ORE->emit(RemarkBuilder: [&]() {
7007 return OptimizationRemarkAnalysis(DEBUG_TYPE, "VectorizationCodeSize",
7008 OrigLoop->getStartLoc(),
7009 OrigLoop->getHeader())
7010 << "Code-size may be reduced by not forcing "
7011 "vectorization, or by source-code modifications "
7012 "eliminating the need for runtime checks "
7013 "(e.g., adding 'restrict').";
7014 });
7015 }
7016 // VPSCEVExpander expands AddRecs in the plan's entry, not the check block.
7017 auto IsUnsupported = IsaPred<SCEVAddRecExpr>;
7018 // Diff checks are not modelled in VPlan yet, and the VPlan expander cannot
7019 // hoist bounds out of an enclosing loop.
7020 const auto &RtPtrChecking = *Legal->getRuntimePointerChecking();
7021 if (RtPtrChecking.getDiffChecks() || OrigLoop->getParentLoop() ||
7022 any_of(Range: RtPtrChecking.CheckingGroups,
7023 P: [&](const RuntimeCheckingPtrGroup &CG) {
7024 return SCEVExprContains(Root: CG.Low, Pred: IsUnsupported) ||
7025 SCEVExprContains(Root: CG.High, Pred: IsUnsupported);
7026 }))
7027 return RUN_VPLAN_PASS(VPlanTransforms::attachCheckBlock, Plan,
7028 MemCheckCond, MemCheckBlock, HasBranchWeights);
7029
7030 // Erase the temporary IR before recipe expansion can reuse its values.
7031 RTChecks.eraseMemCheckBlock();
7032 RUN_VPLAN_PASS(VPlanTransforms::attachMemoryChecks, Plan,
7033 RtPtrChecking.getChecks(), *PSE.getSE(),
7034 OrigLoop->getStartLoc(), HasBranchWeights);
7035 }
7036}
7037
7038void LoopVectorizationPlanner::addMinimumIterationCheck(
7039 VPlan &Plan, ElementCount VF, unsigned UF,
7040 ElementCount MinProfitableTripCount) const {
7041 const uint32_t *BranchWeights =
7042 hasBranchWeightMD(I: *OrigLoop->getLoopLatch()->getTerminator())
7043 ? &MinItersBypassWeights[0]
7044 : nullptr;
7045 RUN_VPLAN_PASS(VPlanTransforms::addMinimumIterationCheck, Plan, VF, UF,
7046 MinProfitableTripCount, Plan.requiresScalarEpilogue(),
7047 Plan.hasTailFolded(), OrigLoop, BranchWeights,
7048 OrigLoop->getLoopPredecessor()->getTerminator()->getDebugLoc(),
7049 PSE, Plan.getEntry());
7050}
7051
7052// Determine how to lower the epilogue, which depends on 1) optimising
7053// for minimum code-size, 2) tail-folding compiler options, 3) loop
7054// hints forcing tail-folding, and 4) a TTI hook that analyses whether the loop
7055// is suitable for tail-folding.
7056// This function determines epilogue lowering for the main vector loop while
7057// epilogue lowering for the tail-folded epilogue path will be handled
7058// separately in getEpilogueTailLowering.
7059static EpilogueLowering
7060getEpilogueLowering(Function *F, Loop *L, LoopVectorizeHints &Hints,
7061 bool OptForSize, TargetTransformInfo *TTI,
7062 TargetLibraryInfo *TLI, LoopVectorizationLegality &LVL,
7063 InterleavedAccessInfo *IAI) {
7064 // 1) OptSize takes precedence over all other options, i.e. if this is set,
7065 // don't look at hints or options, and don't request an epilogue.
7066 if (F->hasOptSize() ||
7067 (OptForSize && Hints.getForce() != LoopVectorizeHints::FK_Enabled))
7068 return CM_EpilogueNotAllowedOptSize;
7069
7070 // 2) If set, obey the directives
7071 if (TailFoldingPolicy.getNumOccurrences()) {
7072 switch (TailFoldingPolicy) {
7073 case TailFoldingPolicyTy::None:
7074 return CM_EpilogueAllowed;
7075 case TailFoldingPolicyTy::PreferFoldTail:
7076 return CM_EpilogueNotNeededFoldTail;
7077 case TailFoldingPolicyTy::MustFoldTail:
7078 return CM_EpilogueNotAllowedFoldTail;
7079 };
7080 }
7081
7082 // 3) If set, obey the hints
7083 switch (Hints.getPredicate()) {
7084 case LoopVectorizeHints::FK_Enabled:
7085 return CM_EpilogueNotNeededFoldTail;
7086 case LoopVectorizeHints::FK_Disabled:
7087 return CM_EpilogueAllowed;
7088 };
7089
7090 // 4) if the TTI hook indicates this is profitable, request tail-folding.
7091 TailFoldingInfo TFI(TLI, &LVL, IAI);
7092 if (TTI->preferTailFoldingOverEpilogue(TFI: &TFI))
7093 return CM_EpilogueNotNeededFoldTail;
7094
7095 return CM_EpilogueAllowed;
7096}
7097
7098// Emit a remark if there are stores to floats that required a floating point
7099// extension. If the vectorized loop was generated with floating point there
7100// will be a performance penalty from the conversion overhead and the change in
7101// the vector width.
7102static void checkMixedPrecision(Loop *L, OptimizationRemarkEmitter *ORE) {
7103 SmallVector<Instruction *, 4> Worklist;
7104 for (BasicBlock *BB : L->getBlocks()) {
7105 for (Instruction &Inst : *BB) {
7106 if (auto *S = dyn_cast<StoreInst>(Val: &Inst)) {
7107 if (S->getValueOperand()->getType()->isFloatTy())
7108 Worklist.push_back(Elt: S);
7109 }
7110 }
7111 }
7112
7113 // Traverse the floating point stores upwards searching, for floating point
7114 // conversions.
7115 SmallPtrSet<const Instruction *, 4> Visited;
7116 SmallPtrSet<const Instruction *, 4> EmittedRemark;
7117 while (!Worklist.empty()) {
7118 auto *I = Worklist.pop_back_val();
7119 if (!L->contains(Inst: I))
7120 continue;
7121 if (!Visited.insert(Ptr: I).second)
7122 continue;
7123
7124 // Emit a remark if the floating point store required a floating
7125 // point conversion.
7126 // TODO: More work could be done to identify the root cause such as a
7127 // constant or a function return type and point the user to it.
7128 if (isa<FPExtInst>(Val: I) && EmittedRemark.insert(Ptr: I).second)
7129 ORE->emit(RemarkBuilder: [&]() {
7130 return OptimizationRemarkAnalysis(LV_NAME, "VectorMixedPrecision",
7131 I->getDebugLoc(), L->getHeader())
7132 << "floating point conversion changes vector width. "
7133 << "Mixed floating point precision requires an up/down "
7134 << "cast that will negatively impact performance.";
7135 });
7136
7137 for (Use &Op : I->operands())
7138 if (auto *OpI = dyn_cast<Instruction>(Val&: Op))
7139 Worklist.push_back(Elt: OpI);
7140 }
7141}
7142
7143/// For loops with uncountable early exits, find the cost of doing work when
7144/// exiting the loop early, such as calculating the final exit values of
7145/// variables used outside the loop.
7146/// TODO: This is currently overly pessimistic because the loop may not take
7147/// the early exit, but better to keep this conservative for now. In future,
7148/// it might be possible to relax this by using branch probabilities.
7149static InstructionCost calculateEarlyExitCost(VPCostContext &CostCtx,
7150 VPlan &Plan, ElementCount VF) {
7151 InstructionCost Cost = 0;
7152 for (auto *ExitVPBB : Plan.getExitBlocks()) {
7153 for (auto *PredVPBB : ExitVPBB->getPredecessors()) {
7154 // If the predecessor is not the middle.block, then it must be the
7155 // vector.early.exit block, which may contain work to calculate the exit
7156 // values of variables used outside the loop.
7157 if (PredVPBB != Plan.getMiddleBlock()) {
7158 LLVM_DEBUG(dbgs() << "Calculating cost of work in exit block "
7159 << PredVPBB->getName() << ":\n");
7160 Cost += PredVPBB->cost(VF, Ctx&: CostCtx);
7161 }
7162 }
7163 }
7164 return Cost;
7165}
7166
7167/// This function determines whether or not it's still profitable to vectorize
7168/// the loop given the extra work we have to do outside of the loop:
7169/// 1. Perform the runtime checks before entering the loop to ensure it's safe
7170/// to vectorize.
7171/// 2. In the case of loops with uncountable early exits, we may have to do
7172/// extra work when exiting the loop early, such as calculating the final
7173/// exit values of variables used outside the loop.
7174/// 3. The middle block.
7175static bool isOutsideLoopWorkProfitable(GeneratedRTChecks &Checks,
7176 VectorizationFactor &VF, Loop *L,
7177 PredicatedScalarEvolution &PSE,
7178 VPCostContext &CostCtx, VPlan &Plan,
7179 EpilogueLowering SEL,
7180 std::optional<unsigned> VScale) {
7181 InstructionCost RtC = Checks.getCost();
7182 if (!RtC.isValid())
7183 return false;
7184
7185 // When interleaving only scalar and vector cost will be equal, which in turn
7186 // would lead to a divide by 0. Fall back to hard threshold.
7187 if (VF.Width.isScalar()) {
7188 // TODO: Should we rename VectorizeMemoryCheckThreshold?
7189 if (RtC > VectorizerParams::VectorizeMemoryCheckThreshold) {
7190 LLVM_DEBUG(
7191 dbgs()
7192 << "LV: Interleaving only is not profitable due to runtime checks\n");
7193 return false;
7194 }
7195 return true;
7196 }
7197
7198 // The scalar cost should only be 0 when vectorizing with a user specified
7199 // VF/IC. In those cases, runtime checks should always be generated.
7200 uint64_t ScalarC = VF.ScalarCost.getValue();
7201 if (ScalarC == 0)
7202 return true;
7203
7204 InstructionCost TotalCost = RtC;
7205 // Add on the cost of any work required in the vector early exit block, if
7206 // one exists.
7207 TotalCost += calculateEarlyExitCost(CostCtx, Plan, VF: VF.Width);
7208 TotalCost += Plan.getMiddleBlock()->cost(VF: VF.Width, Ctx&: CostCtx);
7209
7210 // First, compute the minimum iteration count required so that the vector
7211 // loop outperforms the scalar loop.
7212 // The total cost of the scalar loop is
7213 // ScalarC * TC
7214 // where
7215 // * TC is the actual trip count of the loop.
7216 // * ScalarC is the cost of a single scalar iteration.
7217 //
7218 // The total cost of the vector loop is
7219 // TotalCost + VecC * (TC / VF) + EpiC
7220 // where
7221 // * TotalCost is the sum of the costs cost of
7222 // - the generated runtime checks, i.e. RtC
7223 // - performing any additional work in the vector.early.exit block for
7224 // loops with uncountable early exits.
7225 // - the middle block, if ExpectedTC <= VF.Width.
7226 // * VecC is the cost of a single vector iteration.
7227 // * TC is the actual trip count of the loop
7228 // * VF is the vectorization factor
7229 // * EpiCost is the cost of the generated epilogue, including the cost
7230 // of the remaining scalar operations.
7231 //
7232 // Vectorization is profitable once the total vector cost is less than the
7233 // total scalar cost:
7234 // TotalCost + VecC * (TC / VF) + EpiC < ScalarC * TC
7235 //
7236 // Now we can compute the minimum required trip count TC as
7237 // VF * (TotalCost + EpiC) / (ScalarC * VF - VecC) < TC
7238 //
7239 // For now we assume the epilogue cost EpiC = 0 for simplicity. Note that
7240 // the computations are performed on doubles, not integers and the result
7241 // is rounded up, hence we get an upper estimate of the TC.
7242 unsigned IntVF = estimateElementCount(VF: VF.Width, VScale);
7243 uint64_t Div = ScalarC * IntVF - VF.Cost.getValue();
7244 uint64_t MinTC1 =
7245 Div == 0 ? 0 : divideCeil(Numerator: TotalCost.getValue() * IntVF, Denominator: Div);
7246
7247 // Second, compute a minimum iteration count so that the cost of the
7248 // runtime checks is only a fraction of the total scalar loop cost. This
7249 // adds a loop-dependent bound on the overhead incurred if the runtime
7250 // checks fail. In case the runtime checks fail, the cost is RtC + ScalarC
7251 // * TC. To bound the runtime check to be a fraction 1/X of the scalar
7252 // cost, compute
7253 // RtC < ScalarC * TC * (1 / X) ==> RtC * X / ScalarC < TC
7254 uint64_t MinTC2 = divideCeil(Numerator: RtC.getValue() * 10, Denominator: ScalarC);
7255
7256 // Now pick the larger minimum. If it is not a multiple of VF and an epilogue
7257 // is allowed, choose the next closest multiple of VF. This should partly
7258 // compensate for ignoring the epilogue cost.
7259 uint64_t MinTC = std::max(a: MinTC1, b: MinTC2);
7260 if (SEL == CM_EpilogueAllowed)
7261 MinTC = alignTo(Value: MinTC, Align: IntVF);
7262 VF.MinProfitableTripCount = ElementCount::getFixed(MinVal: MinTC);
7263
7264 LLVM_DEBUG(
7265 dbgs() << "LV: Minimum required TC for runtime checks to be profitable:"
7266 << VF.MinProfitableTripCount << "\n");
7267
7268 // Skip vectorization if the expected trip count is less than the minimum
7269 // required trip count.
7270 if (auto ExpectedTC = getSmallBestKnownTC(PSE, L)) {
7271 if (ElementCount::isKnownLT(LHS: *ExpectedTC, RHS: VF.MinProfitableTripCount)) {
7272 LLVM_DEBUG(dbgs() << "LV: Vectorization is not beneficial: expected "
7273 "trip count < minimum profitable VF ("
7274 << *ExpectedTC << " < " << VF.MinProfitableTripCount
7275 << ")\n");
7276
7277 return false;
7278 }
7279 }
7280 return true;
7281}
7282
7283LoopVectorizePass::LoopVectorizePass(LoopVectorizeOptions Opts)
7284 : InterleaveOnlyWhenForced(Opts.InterleaveOnlyWhenForced ||
7285 !EnableLoopInterleaving),
7286 VectorizeOnlyWhenForced(Opts.VectorizeOnlyWhenForced ||
7287 !EnableLoopVectorization) {}
7288
7289/// ResumeForEpilogue markers in the main plan, used by the epilogue plan.
7290struct MainPlanResumeMarkers {
7291 VPInstruction *CanIVResume;
7292 VPInstruction *VectorTC;
7293 SmallVector<VPInstruction *> ResumeValues;
7294};
7295
7296/// Prepare \p MainPlan for vectorizing the main vector loop during epilogue
7297/// vectorization.
7298static MainPlanResumeMarkers preparePlanForMainVectorLoop(VPlan &MainPlan) {
7299 using namespace VPlanPatternMatch;
7300 // When vectorizing the epilogue, FindFirstIV & FindLastIV reductions can
7301 // introduce multiple uses of undef/poison. If the reduction start value may
7302 // be undef or poison it needs to be frozen and the frozen start has to be
7303 // used when computing the reduction result. We also need to use the frozen
7304 // value in the resume phi generated by the main vector loop, as this is also
7305 // used to compute the reduction result after the epilogue vector loop. The
7306 // epilogue plan re-uses the frozen values.
7307 VPBuilder Builder(MainPlan.getEntry());
7308 for (VPInstruction &VPI :
7309 make_isa_range<VPInstruction>(Range&: *MainPlan.getMiddleBlock())) {
7310 VPValue *OrigStart;
7311 if (!matchFindIVResult(VPI: &VPI, ReducedIV: m_VPValue(), Start: m_VPValue(V&: OrigStart)))
7312 continue;
7313 if (isGuaranteedNotToBeUndefOrPoison(V: OrigStart->getLiveInIRValue()))
7314 continue;
7315 VPInstruction *Freeze = Builder.createFreeze(Op: OrigStart, DL: {}, Name: "fr");
7316 VPI.setOperand(I: 2, New: Freeze);
7317 OrigStart->replaceUsesWithIf(New: Freeze, ShouldReplace: IsaPred<VPPhi>);
7318 }
7319
7320 VPValue *VectorTC = nullptr;
7321 auto *Term =
7322 MainPlan.getVectorLoopRegion()->getExitingBasicBlock()->getTerminator();
7323 [[maybe_unused]] bool MatchedTC =
7324 match(V: Term, P: m_BranchOnCount(Op0: m_VPValue(), Op1: m_VPValue(V&: VectorTC)));
7325 assert(MatchedTC && "must match vector trip count");
7326
7327 VPBasicBlock *MiddleVPBB = MainPlan.getMiddleBlock();
7328 VPBuilder MiddleBuilder(MiddleVPBB, MiddleVPBB->getFirstNonPhi());
7329 VPInstruction *VectorTCMarker = MiddleBuilder.createNaryOp(
7330 Opcode: VPInstruction::ResumeForEpilogue,
7331 Operands: {VectorTC, MainPlan.getZero(Ty: VectorTC->getScalarType())});
7332
7333 // If there is a suitable resume value for the canonical induction in the
7334 // scalar (which will become vector) epilogue loop, use it and move it to the
7335 // beginning of the scalar preheader. Otherwise create it below.
7336 VPBasicBlock *MainScalarPH = MainPlan.getScalarPreheader();
7337 auto ResumePhiIter =
7338 find_if(Range: MainScalarPH->phis(), P: [VectorTC](VPRecipeBase &R) {
7339 return match(V: &R, P: m_VPInstruction<Instruction::PHI>(Ops: m_Specific(VPV: VectorTC),
7340 Ops: m_ZeroInt()));
7341 });
7342 VPPhi *ResumePhi = nullptr;
7343 if (ResumePhiIter == MainScalarPH->phis().end()) {
7344 assert(MainPlan.getVectorLoopRegion()->getCanonicalIV() &&
7345 "canonical IV must exist");
7346 Type *Ty = VectorTC->getScalarType();
7347 VPBuilder ScalarPHBuilder(MainScalarPH, MainScalarPH->begin());
7348 ResumePhi = ScalarPHBuilder.createScalarPhi(
7349 IncomingValues: {VectorTC, MainPlan.getZero(Ty)}, DL: {}, Name: "vec.epilog.resume.val");
7350 } else {
7351 ResumePhi = cast<VPPhi>(Val: &*ResumePhiIter);
7352 ResumePhi->setName("vec.epilog.resume.val");
7353 if (&MainScalarPH->front() != ResumePhi)
7354 ResumePhi->moveBefore(BB&: *MainScalarPH, I: MainScalarPH->begin());
7355 }
7356
7357 // Create a ResumeForEpilogue for the canonical IV resume and its bypass value
7358 // as the first non-phi, to keep them alive for the epilogue.
7359 VPBuilder ResumeBuilder(MainScalarPH);
7360 VPInstruction *CanIVResume = ResumeBuilder.createNaryOp(
7361 Opcode: VPInstruction::ResumeForEpilogue, Operands: {ResumePhi, ResumePhi->getOperand(N: 1)});
7362
7363 // Create ResumeForEpilogue instructions for the resume phis of the
7364 // VPIRPhis and their bypass values in the scalar header of the main plan and
7365 // return them so they can be used as resume values when vectorizing the
7366 // epilogue.
7367 auto ResumeValues = to_vector(
7368 Range: map_range(C: MainPlan.getScalarHeader()->phis(), F: [&](VPRecipeBase &R) {
7369 assert(isa<VPIRPhi>(R) &&
7370 "only VPIRPhis expected in the scalar header");
7371 VPValue *MainResumePhi = R.getOperand(N: 0);
7372 VPValue *Bypass = MainResumePhi->getDefiningRecipe()->getOperand(N: 1);
7373 return ResumeBuilder.createNaryOp(Opcode: VPInstruction::ResumeForEpilogue,
7374 Operands: {MainResumePhi, Bypass});
7375 }));
7376 return {.CanIVResume: CanIVResume, .VectorTC: VectorTCMarker, .ResumeValues: std::move(ResumeValues)};
7377}
7378
7379/// Prepare \p Plan for vectorizing the epilogue loop. That is, re-use expanded
7380/// SCEVs from \p ExpandedSCEVs and set resume values for header recipes.
7381static void preparePlanForEpilogueVectorLoop(
7382 VPlan &MainPlan, VPlan &Plan, Loop *L, const SCEV2ValueTy &ExpandedSCEVs,
7383 ElementCount MainLoopVF, unsigned MainLoopUF, ElementCount EpilogueVF,
7384 LoopVectorizationPlanner &LVP, VFSelectionContext &Config,
7385 ScalarEvolution &SE, const MainPlanResumeMarkers &Markers) {
7386 // Build a map from the scalar-header PHI to the ResumeForEpilogue markers
7387 // from the main plan.
7388 // TODO: Replace the IR PHI key.
7389 DenseMap<PHINode *, VPInstruction *> IRPhiToResumeForEpi;
7390 for (auto [HeaderPhi, ResumeForEpi] :
7391 zip_equal(t: MainPlan.getScalarHeader()->phis(), u: Markers.ResumeValues))
7392 IRPhiToResumeForEpi[&cast<VPIRPhi>(Val&: HeaderPhi).getIRPhi()] = ResumeForEpi;
7393 VPRegionBlock *VectorLoop = Plan.getVectorLoopRegion();
7394 VPBasicBlock *Header = VectorLoop->getEntryBasicBlock();
7395 Header->setName("vec.epilog.vector.body");
7396
7397 VPValue *IV = VectorLoop->getCanonicalIV();
7398 // When vectorizing the epilogue loop, the canonical induction needs to start
7399 // at the resume value from the main vector loop. Find the resume value
7400 // created during execution of the main VPlan. Add this resume value as an
7401 // offset to the canonical IV of the epilogue loop.
7402 VPValue *VPV = Plan.getOrAddLiveIn(V: Markers.CanIVResume->getUnderlyingValue());
7403 assert(all_of(IV->users(),
7404 [](const VPUser *U) {
7405 if (isa<VPScalarIVStepsRecipe, VPDerivedIVRecipe>(U))
7406 return true;
7407 unsigned Opc = cast<VPInstruction>(U)->getOpcode();
7408 return Instruction::isCast(Opc) || Opc == Instruction::Add;
7409 }) &&
7410 "the canonical IV should only be used by its increment or "
7411 "ScalarIVSteps when resetting the start value");
7412 VPBuilder Builder(Header, Header->getFirstNonPhi());
7413 VPInstruction *Add = Builder.createAdd(LHS: IV, RHS: VPV);
7414 // Replace all users of the canonical IV and its increment with the offset
7415 // version, except for the Add itself and the canonical IV increment.
7416 auto *Increment = vputils::findCanonicalIVIncrement(Plan);
7417 assert(Increment && "Must have a canonical IV increment at this point");
7418 IV->replaceUsesWithIf(New: Add, ShouldReplace: [Add, Increment](VPUser &U) {
7419 return &U != Add && &U != Increment;
7420 });
7421 VPInstruction *OffsetIVInc =
7422 VPBuilder::getToInsertAfter(R: Increment).createAdd(LHS: Increment, RHS: VPV);
7423 Increment->replaceAllUsesWith(New: OffsetIVInc);
7424 OffsetIVInc->setOperand(I: 0, New: Increment);
7425
7426 // Resume values must be created in the vector preheader.
7427 VPBasicBlock *VectorPH = Plan.getVectorPreheader();
7428 VPBuilder PHBuilder(VectorPH, VectorPH->getFirstNonPhi());
7429
7430 // Ensure that the start values for all header phi recipes are updated before
7431 // vectorizing the epilogue loop.
7432 for (VPRecipeBase &R : Header->phis()) {
7433 VPValue *ResumeVPV = nullptr;
7434 // TODO: Move setting of resume values to prepareToExecute.
7435 if (auto *ReductionPhi = dyn_cast<VPReductionPHIRecipe>(Val: &R)) {
7436 // Find the reduction result by searching users of the phi or its backedge
7437 // value.
7438 auto IsReductionResult = [](VPRecipeBase *R) {
7439 auto *VPI = dyn_cast<VPInstruction>(Val: R);
7440 return VPI && VPI->getOpcode() == VPInstruction::ComputeReductionResult;
7441 };
7442 auto *RdxResult = cast<VPInstruction>(
7443 Val: vputils::findRecipe(Start: ReductionPhi->getBackedgeValue(), Pred: IsReductionResult));
7444 assert(RdxResult && "expected to find reduction result");
7445
7446 VPInstruction *ResumeForEpi = IRPhiToResumeForEpi.at(
7447 Val: cast<PHINode>(Val: ReductionPhi->getUnderlyingInstr()));
7448
7449 // Check for FindIV pattern by looking for icmp user of RdxResult.
7450 // The pattern is: select(icmp ne RdxResult, Sentinel), RdxResult, Start
7451 using namespace VPlanPatternMatch;
7452 VPValue *SentinelVPV = nullptr;
7453 bool IsFindIV = any_of(Range: RdxResult->users(), P: [&](VPUser *U) {
7454 return match(U, P: VPlanPatternMatch::m_SpecificICmp(
7455 MatchPred: ICmpInst::ICMP_NE, Op0: m_Specific(VPV: RdxResult),
7456 Op1: m_VPValue(V&: SentinelVPV)));
7457 });
7458
7459 RecurKind RK = ReductionPhi->getRecurrenceKind();
7460 ResumeVPV = Plan.getOrAddLiveIn(V: ResumeForEpi->getUnderlyingValue());
7461 if (RecurrenceDescriptor::isAnyOfRecurrenceKind(Kind: RK) || IsFindIV) {
7462 VPValue *BypassOp = ResumeForEpi->getOperand(N: 1);
7463 assert((isa<VPIRValue>(BypassOp) ||
7464 VPlanPatternMatch::match(
7465 BypassOp,
7466 m_VPInstruction<Instruction::Freeze>(m_VPValue()))) &&
7467 "expected live-in or Freeze");
7468 VPValue *StartV = Plan.getOrAddLiveIn(V: BypassOp->getUnderlyingValue());
7469 if (RecurrenceDescriptor::isAnyOfRecurrenceKind(Kind: RK)) {
7470 // VPReductionPHIRecipes for AnyOf reductions expect a boolean as
7471 // start value; compare the final value from the main vector loop
7472 // to the start value.
7473 ResumeVPV = PHBuilder.createICmp(Pred: CmpInst::ICMP_NE, A: ResumeVPV, B: StartV);
7474 } else {
7475 assert(isa<VPIRValue>(SentinelVPV) &&
7476 "sentinel must be a live-in to be used in the preheader");
7477 // Use the start value frozen for the main plan, if any, when
7478 // computing the reduction result.
7479 if (match(V: BypassOp, P: m_Freeze(Op0: m_VPValue())))
7480 for (VPUser *U : RdxResult->users()) {
7481 auto *VPI = dyn_cast<VPInstruction>(Val: U);
7482 if (VPI && matchFindIVResult(VPI, ReducedIV: m_VPValue(), Start: m_VPValue()))
7483 VPI->setOperand(I: 2, New: StartV);
7484 }
7485
7486 // Adjust resume: select(icmp eq ResumeVPV, StartV), Sentinel,
7487 // ResumeVPV
7488 VPValue *Cmp =
7489 PHBuilder.createICmp(Pred: CmpInst::ICMP_EQ, A: ResumeVPV, B: StartV);
7490 ResumeVPV = PHBuilder.createSelect(Cond: Cmp, TrueVal: SentinelVPV, FalseVal: ResumeVPV);
7491 }
7492 // TODO: materializeBroadcasts does not cover values in the vector
7493 // preheader.
7494 ReductionPhi->setStartValue(
7495 PHBuilder.createNaryOp(Opcode: VPInstruction::Broadcast, Operands: ResumeVPV));
7496 continue;
7497 }
7498 if (auto *VPI = dyn_cast<VPInstruction>(Val: ReductionPhi->getStartValue())) {
7499 assert(VPI->getOpcode() == VPInstruction::ReductionStartVector &&
7500 "unexpected start value");
7501 // Partial sub-reductions always start at 0 and account for the
7502 // reduction start value in a final subtraction. Update it to use the
7503 // resume value from the main vector loop.
7504 if (ReductionPhi->getVFScaleFactor() > 1 &&
7505 RecurrenceDescriptor::isSubRecurrenceKind(Kind: RK)) {
7506 auto *Sub = cast<VPInstruction>(Val: RdxResult->getSingleUser());
7507 assert((Sub->getOpcode() == Instruction::Sub ||
7508 Sub->getOpcode() == Instruction::FSub) &&
7509 "Unexpected opcode");
7510 assert(isa<VPIRValue>(Sub->getOperand(0)) &&
7511 "Expected operand to match the original start value of the "
7512 "reduction");
7513 // For integer sub-reductions, verify start value is zero.
7514 // For FP sub-reductions, verify start value is negative zero.
7515 [[maybe_unused]] auto StartValueIsIdentity = [&] {
7516 Value *IdentityValue =
7517 getRecurrenceIdentity(K: RK, Tp: ResumeVPV->getScalarType(),
7518 FMF: ReductionPhi->getFastMathFlagsOrNone());
7519 auto *StartValue = dyn_cast<VPIRValue>(Val: VPI->getOperand(N: 0));
7520 return StartValue && StartValue->getValue() == IdentityValue;
7521 };
7522 assert(StartValueIsIdentity() &&
7523 "Expected start value for partial sub-reduction to be zero "
7524 "(or negative zero)");
7525
7526 Sub->setOperand(I: 0, New: ResumeVPV);
7527 } else
7528 VPI->setOperand(I: 0, New: ResumeVPV);
7529 continue;
7530 }
7531 } else {
7532 // Retrieve the induction resume value via ResumeForEpilogue.
7533 PHINode *IndPhi = cast<VPWidenInductionRecipe>(Val: &R)->getPHINode();
7534 ResumeVPV = Plan.getOrAddLiveIn(
7535 V: IRPhiToResumeForEpi.at(Val: IndPhi)->getUnderlyingValue());
7536 }
7537 assert(ResumeVPV && "Must have a resume value");
7538 cast<VPHeaderPHIRecipe>(Val: &R)->setStartValue(ResumeVPV);
7539 }
7540
7541 // Re-use the trip count and steps expanded for the main loop, as skeleton
7542 // creation needs it as a value that dominates both the scalar and vector
7543 // epilogue loops.
7544 // TODO: This is a workaround needed for epilogue vectorization and it
7545 // should be removed once induction resume value creation is done
7546 // directly in VPlan.
7547 for (VPExpandSCEVRecipe &ExpandR : make_early_inc_range(
7548 Range: make_isa_range<VPExpandSCEVRecipe>(Range&: *Plan.getEntry()))) {
7549 assert(ExpandedSCEVs.contains(ExpandR.getSCEV()) &&
7550 "Epilogue plan needs a SCEV not expanded for the main loop");
7551 VPValue *ExpandedVal =
7552 Plan.getOrAddLiveIn(V: ExpandedSCEVs.lookup(Val: ExpandR.getSCEV()));
7553 ExpandR.replaceAllUsesWith(New: ExpandedVal);
7554 if (Plan.getTripCount() == &ExpandR)
7555 Plan.resetTripCount(NewTripCount: ExpandedVal);
7556 ExpandR.eraseFromParent();
7557 }
7558
7559 auto VScale = Config.getVScaleForTuning();
7560 unsigned MainLoopStep = estimateElementCount(VF: MainLoopVF * MainLoopUF, VScale);
7561 unsigned EpilogueLoopStep = estimateElementCount(VF: EpilogueVF, VScale);
7562 RUN_VPLAN_PASS(VPlanTransforms::addMinimumVectorEpilogueIterationCheck, Plan,
7563 Plan.getOrAddLiveIn(Markers.VectorTC->getUnderlyingValue()),
7564 Plan.requiresScalarEpilogue(), EpilogueVF, MainLoopStep,
7565 EpilogueLoopStep, SE);
7566}
7567
7568static void
7569fixScalarResumeValuesFromBypass(BasicBlock *BypassBlock, VPlan &BestEpiPlan,
7570 ArrayRef<VPInstruction *> ResumeValues) {
7571 auto *ScalarPH = cast<VPIRBasicBlock>(Val: BestEpiPlan.getScalarPreheader());
7572 BasicBlock *PH = ScalarPH->getIRBasicBlock();
7573 if (ScalarPH->hasPredecessors()) {
7574 // Fix resume values for inductions and reductions from the additional
7575 // bypass block using the incoming values from the main loop's resume phis.
7576 // ResumeValues correspond 1:1 with the scalar loop header phis.
7577 for (auto [ResumeV, HeaderPhi] :
7578 zip(t&: ResumeValues, u: BestEpiPlan.getScalarHeader()->phis())) {
7579 auto *HeaderPhiR = cast<VPIRPhi>(Val: &HeaderPhi);
7580 auto *EpiResumePhi =
7581 cast<PHINode>(Val: HeaderPhiR->getIRPhi().getIncomingValueForBlock(BB: PH));
7582 if (EpiResumePhi->getBasicBlockIndex(BB: BypassBlock) == -1)
7583 continue;
7584 auto *MainResumePhi = cast<PHINode>(Val: ResumeV->getUnderlyingValue());
7585 EpiResumePhi->setIncomingValueForBlock(
7586 BB: BypassBlock, V: MainResumePhi->getIncomingValueForBlock(BB: BypassBlock));
7587 }
7588 }
7589}
7590
7591/// Connect the epilogue vector loop generated for \p EpiPlan to the main vector
7592/// loop, after both plans have executed, updating the branch from the iteration
7593/// count check of the main loop, as well as updating various phis.
7594static void connectEpilogueVectorLoop(VPlan &EpiPlan, DominatorTree *DT,
7595 VPIRBasicBlock *VecEpilogueIterCheckVPBB,
7596 ArrayRef<VPInstruction *> ResumeValues) {
7597 ArrayRef<VPBlockBase *> Preds = VecEpilogueIterCheckVPBB->getPredecessors();
7598 BasicBlock *MainLoopIterationCountCheck =
7599 cast<VPIRBasicBlock>(Val: Preds.front())->getIRBasicBlock();
7600 BasicBlock *VecEpilogueIterationCountCheck =
7601 VecEpilogueIterCheckVPBB->getIRBasicBlock();
7602 BasicBlock *VecEpiloguePreHeader =
7603 cast<CondBrInst>(Val: VecEpilogueIterationCountCheck->getTerminator())
7604 ->getSuccessor(i: 1);
7605 DomTreeUpdater DTU(DT, DomTreeUpdater::UpdateStrategy::Eager);
7606
7607 MainLoopIterationCountCheck->getTerminator()->replaceSuccessorWith(
7608 OldBB: VecEpilogueIterationCountCheck, NewBB: VecEpiloguePreHeader);
7609 DTU.applyUpdates(Updates: {{DominatorTree::Delete, MainLoopIterationCountCheck,
7610 VecEpilogueIterationCountCheck},
7611 {DominatorTree::Insert, MainLoopIterationCountCheck,
7612 VecEpiloguePreHeader}});
7613
7614 // The vec.epilog.iter.check block may contain Phi nodes from inductions
7615 // or reductions which merge control-flow from the latch block and the
7616 // middle block. Update the incoming values here and move the Phi into the
7617 // preheader.
7618 SmallVector<PHINode *, 4> PhisInBlock(
7619 llvm::make_pointer_range(Range: VecEpilogueIterationCountCheck->phis()));
7620
7621 for (PHINode *Phi : PhisInBlock) {
7622 Phi->moveBefore(InsertPos: VecEpiloguePreHeader->getFirstNonPHIIt());
7623 Phi->replaceIncomingBlockWith(
7624 Old: VecEpilogueIterationCountCheck->getSinglePredecessor(),
7625 New: VecEpilogueIterationCountCheck);
7626 }
7627
7628 // VecEpilogueIterationCountCheck conditionally skips over the epilogue loop
7629 // after executing the main loop. We need to update the resume values of
7630 // inductions and reductions during epilogue vectorization.
7631 fixScalarResumeValuesFromBypass(BypassBlock: VecEpilogueIterationCountCheck, BestEpiPlan&: EpiPlan,
7632 ResumeValues);
7633
7634 // Remove dead phis that were moved to the epilogue preheader but are unused
7635 // (e.g., resume phis for inductions not widened in the epilogue vector loop).
7636 for (PHINode &Phi : make_early_inc_range(Range: VecEpiloguePreHeader->phis()))
7637 if (Phi.use_empty())
7638 Phi.eraseFromParent();
7639}
7640
7641bool LoopVectorizePass::processLoop(Loop *L) {
7642 assert((EnableVPlanNativePath || L->isInnermost()) &&
7643 "VPlan-native path is not enabled. Only process inner loops.");
7644
7645 LLVM_DEBUG(dbgs() << "\nLV: Checking a loop in '"
7646 << L->getHeader()->getParent()->getName() << "' from "
7647 << L->getLocStr() << "\n");
7648
7649 LoopVectorizeHints Hints(L, InterleaveOnlyWhenForced, *ORE, TTI);
7650
7651 LLVM_DEBUG(
7652 dbgs() << "LV: Loop hints:"
7653 << " force="
7654 << (Hints.getForce() == LoopVectorizeHints::FK_Disabled
7655 ? "disabled"
7656 : (Hints.getForce() == LoopVectorizeHints::FK_Enabled
7657 ? "enabled"
7658 : "?"))
7659 << " width=" << Hints.getWidth()
7660 << " interleave=" << Hints.getInterleave() << "\n");
7661
7662 // Function containing loop
7663 Function *F = L->getHeader()->getParent();
7664
7665 // Looking at the diagnostic output is the only way to determine if a loop
7666 // was vectorized (other than looking at the IR or machine code), so it
7667 // is important to generate an optimization remark for each loop. Most of
7668 // these messages are generated as OptimizationRemarkAnalysis. Remarks
7669 // generated as OptimizationRemark and OptimizationRemarkMissed are
7670 // less verbose reporting vectorized loops and unvectorized loops that may
7671 // benefit from vectorization, respectively.
7672
7673 if (!Hints.allowVectorization(F, L, VectorizeOnlyWhenForced)) {
7674 LLVM_DEBUG(dbgs() << "LV: Loop hints prevent vectorization.\n");
7675 return false;
7676 }
7677
7678 PredicatedScalarEvolution PSE(*SE, *L);
7679
7680 // Query this against the original loop and save it here because the profile
7681 // of the original loop header may change as the transformation happens.
7682 bool OptForSize = llvm::shouldOptimizeForSize(
7683 BB: L->getHeader(), PSI,
7684 BFI: PSI && PSI->hasProfileSummary() ? &GetBFI() : nullptr,
7685 QueryType: PGSOQueryType::IRPass);
7686
7687 // Check if it is legal to vectorize the loop.
7688 LoopVectorizationRequirements Requirements;
7689 LoopVectorizationLegality LVL(L, PSE, DT, TTI, TLI, F, *LAIs, LI, ORE,
7690 &Requirements, &Hints, DB, AC,
7691 /*AllowRuntimeSCEVChecks=*/!OptForSize, AA);
7692 if (!LVL.canVectorize(UseVPlanNativePath: EnableVPlanNativePath)) {
7693 LLVM_DEBUG(dbgs() << "LV: Not vectorizing: Cannot prove legality.\n");
7694 Hints.emitRemarkWithHints();
7695 return false;
7696 }
7697
7698 bool IsInnerLoop = L->isInnermost();
7699
7700 // Outer loops require a computable trip count.
7701 if (!IsInnerLoop && isa<SCEVCouldNotCompute>(Val: PSE.getBackedgeTakenCount())) {
7702 LLVM_DEBUG(dbgs() << "LV: cannot compute the outer-loop trip count\n");
7703 return false;
7704 }
7705
7706 if (LVL.hasUncountableEarlyExit()) {
7707 if (!EnableEarlyExitVectorization) {
7708 reportVectorizationFailure(DebugMsg: "Auto-vectorization of loops with uncountable "
7709 "early exit is not enabled",
7710 ORETag: "UncountableEarlyExitLoopsDisabled", ORE, TheLoop: L);
7711 return false;
7712 }
7713 if (LVL.hasUncountableExitWithSideEffects() &&
7714 !EnableEarlyExitVectorizationWithSideEffects) {
7715 reportVectorizationFailure(DebugMsg: "Auto-vectorization of loops with uncountable "
7716 "early exit and side effects is not enabled",
7717 ORETag: "UncountableEarlyExitSideEffectLoopsDisabled",
7718 ORE, TheLoop: L);
7719 return false;
7720 }
7721 }
7722
7723 InterleavedAccessInfo IAI(PSE, L, DT, LI, LVL.getLAI(), OptForSize);
7724 bool UseInterleaved =
7725 IsInnerLoop && TTI->enableInterleavedAccessVectorization();
7726
7727 // If an override option has been passed in for interleaved accesses, use it.
7728 if (EnableInterleavedMemAccesses.getNumOccurrences() > 0)
7729 UseInterleaved = IsInnerLoop && EnableInterleavedMemAccesses;
7730
7731 // Analyze interleaved memory accesses.
7732 if (UseInterleaved)
7733 IAI.analyzeInterleaving(EnableMaskedInterleavedGroup: useMaskedInterleavedAccesses(TTI: *TTI));
7734
7735 if (LVL.hasUncountableEarlyExit()) {
7736 BasicBlock *LoopLatch = L->getLoopLatch();
7737 if (IAI.requiresScalarEpilogue() ||
7738 any_of(Range: LVL.getCountableExitingBlocks(), P: not_equal_to(Arg&: LoopLatch))) {
7739 reportVectorizationFailure(DebugMsg: "Auto-vectorization of early exit loops "
7740 "requiring a scalar epilogue is unsupported",
7741 ORETag: "UncountableEarlyExitUnsupported", ORE, TheLoop: L);
7742 return false;
7743 }
7744 }
7745
7746 // Check the function attributes and profiles to find out if this function
7747 // should be optimized for size.
7748 EpilogueLowering SEL =
7749 getEpilogueLowering(F, L, Hints, OptForSize, TTI, TLI, LVL, IAI: &IAI);
7750
7751 // Check the loop for a trip count threshold: vectorize loops with a tiny trip
7752 // count by optimizing for size, to minimize overheads.
7753 auto ExpectedTC = getSmallBestKnownTC(PSE, L);
7754 if (ExpectedTC && ExpectedTC->isFixed() &&
7755 ExpectedTC->getFixedValue() < TinyTripCountVectorThreshold) {
7756 LLVM_DEBUG(dbgs() << "LV: Found a loop with a very small trip count. "
7757 << "This loop is worth vectorizing only if no scalar "
7758 << "iteration overheads are incurred.");
7759 if (Hints.getForce() == LoopVectorizeHints::FK_Enabled)
7760 LLVM_DEBUG(dbgs() << " But vectorizing was explicitly forced.\n");
7761 else {
7762 LLVM_DEBUG(dbgs() << "\n");
7763 // Tail-folded loops are efficient even when the loop
7764 // iteration count is low. However, setting the epilogue policy to
7765 // `CM_EpilogueNotAllowedLowTripLoop` prevents vectorizing loops
7766 // with runtime checks. It's more effective to let
7767 // `isOutsideLoopWorkProfitable` determine if vectorization is
7768 // beneficial for the loop. If the trip count is below the target's
7769 // minimum for tail-folding, the tail cannot be folded, so treat it like
7770 // any other low trip count loop.
7771 if (SEL != CM_EpilogueNotNeededFoldTail ||
7772 ExpectedTC->getFixedValue() <=
7773 TTI->getMinTripCountTailFoldingThreshold())
7774 SEL = CM_EpilogueNotAllowedLowTripLoop;
7775 }
7776 }
7777
7778 // Check the function attributes to see if implicit floats or vectors are
7779 // allowed.
7780 if (F->hasFnAttribute(Kind: Attribute::NoImplicitFloat)) {
7781 reportVectorizationFailure(
7782 DebugMsg: "Can't vectorize when the NoImplicitFloat attribute is used",
7783 OREMsg: "loop not vectorized due to NoImplicitFloat attribute",
7784 ORETag: "NoImplicitFloat", ORE, TheLoop: L);
7785 Hints.emitRemarkWithHints();
7786 return false;
7787 }
7788
7789 // Check if the target supports potentially unsafe FP vectorization.
7790 // FIXME: Add a check for the type of safety issue (denormal, signaling)
7791 // for the target we're vectorizing for, to make sure none of the
7792 // additional fp-math flags can help.
7793 if (Hints.isPotentiallyUnsafe() &&
7794 TTI->isFPVectorizationPotentiallyUnsafe()) {
7795 reportVectorizationFailure(
7796 DebugMsg: "Potentially unsafe FP op prevents vectorization",
7797 OREMsg: "loop not vectorized due to unsafe FP support.", ORETag: "UnsafeFP", ORE, TheLoop: L);
7798 Hints.emitRemarkWithHints();
7799 return false;
7800 }
7801
7802 bool AllowOrderedReductions;
7803 // If the flag is set, use that instead and override the TTI behaviour.
7804 if (ForceOrderedReductions.getNumOccurrences() > 0)
7805 AllowOrderedReductions = ForceOrderedReductions;
7806 else
7807 AllowOrderedReductions = TTI->enableOrderedReductions();
7808 if (!LVL.canVectorizeFPMath(EnableStrictReductions: AllowOrderedReductions)) {
7809 ORE->emit(RemarkBuilder: [&]() {
7810 auto *ExactFPMathInst = Requirements.getExactFPInst();
7811 return OptimizationRemarkAnalysisFPCommute(DEBUG_TYPE, "CantReorderFPOps",
7812 ExactFPMathInst->getDebugLoc(),
7813 ExactFPMathInst->getParent())
7814 << "loop not vectorized: cannot prove it is safe to reorder "
7815 "floating-point operations";
7816 });
7817 LLVM_DEBUG(dbgs() << "LV: loop not vectorized: cannot prove it is safe to "
7818 "reorder floating-point operations\n");
7819 Hints.emitRemarkWithHints();
7820 return false;
7821 }
7822
7823 // Use the cost model.
7824 VFSelectionContext Config(*TTI, &LVL, L, *F, PSE, DB, ORE, &Hints,
7825 OptForSize);
7826 // Use the planner for vectorization.
7827 LoopVectorizationPlanner LVP(
7828 L, LI, DT, TLI, *TTI, &LVL,
7829 std::make_unique<LoopVectorizationCostModel>(
7830 args&: SEL, args&: L, args&: PSE, args&: LI, args: &LVL, args&: *TTI, args&: TLI, args&: AC, args&: ORE, args&: GetBFI, args&: F, args&: IAI, args&: Config),
7831 Config, IAI, PSE, ORE, GetBPI);
7832
7833 EpilogueLowering EpilogueTailLoweringStatus =
7834 getEpilogueTailLowering(MainCM: LVP.getCostModel(), L, ORE, LVL, Hints, TTI);
7835 if (EpilogueTailLoweringStatus ==
7836 EpilogueLowering::CM_EpilogueNotNeededFoldTail) {
7837 // TODO: Apply tail-folding on the vectorized epilogue loop.
7838 LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is not supported yet\n");
7839 reportVectorizationInfo(
7840 Msg: "The epilogue-tail-folding policy prefer-fold-tail is not supported "
7841 "yet, fall back to a normal epilogue",
7842 ORETag: "UnsupportedEpilogueTailFoldingPolicy", ORE, TheLoop: L);
7843 }
7844
7845 // Get user vectorization factor and interleave count.
7846 ElementCount UserVF = Hints.getWidth();
7847 unsigned UserIC = Hints.getInterleave();
7848 // Outer loops don't have LoopAccessInfo, so skip the safety check and reset
7849 // UserIC (interleaving is not supported for outer loops).
7850 if (!IsInnerLoop)
7851 UserIC = 0;
7852 else if (UserIC > 1 && !LVL.isSafeForAnyVectorWidth())
7853 UserIC = 1;
7854
7855 // Plan how to best vectorize.
7856 LVP.plan(UserVF, UserIC);
7857 auto [VF, BestPlanPtr] = LVP.computeBestVF();
7858 unsigned IC = 1;
7859
7860 // For VPlan build stress testing of outer loops, bail after plan
7861 // construction.
7862 if (!IsInnerLoop && VPlanBuildOuterloopStressTest)
7863 return false;
7864
7865 if (IsInnerLoop && ORE->allowExtraAnalysis(LV_NAME))
7866 LVP.emitInvalidCostRemarks(ORE);
7867
7868 assert((IsInnerLoop || !LVP.getCostModel().maskPartialAliasing()) &&
7869 "Did not expect to alias-mask outer loop");
7870
7871 GeneratedRTChecks Checks(PSE, DT, LI, TTI, Config.CostKind,
7872 LVP.getCostModel().maskPartialAliasing());
7873 if (IsInnerLoop && LVP.hasPlanWithVF(VF: VF.Width)) {
7874 // Select the interleave count.
7875 IC = LVP.selectInterleaveCount(Plan&: *BestPlanPtr, VF: VF.Width, LoopCost: VF.Cost);
7876
7877 unsigned SelectedIC = UserIC > 0 ? UserIC : IC;
7878 // Optimistically generate runtime checks if they are needed. Drop them if
7879 // they turn out to not be profitable.
7880 if (VF.Width.isVector() || SelectedIC > 1) {
7881 Checks.create(L, LAI: *LVL.getLAI(), UnionPred: PSE.getPredicate(), VF: VF.Width, IC: SelectedIC,
7882 ORE&: *ORE);
7883
7884 // Bail out early if either the SCEV or memory runtime checks are known to
7885 // fail. In that case, the vector loop would never execute.
7886 using namespace llvm::PatternMatch;
7887 if ((Checks.getSCEVChecks().first &&
7888 match(V: Checks.getSCEVChecks().first, P: m_One())) ||
7889 (Checks.getMemRuntimeChecks().first &&
7890 match(V: Checks.getMemRuntimeChecks().first, P: m_One()))) {
7891 reportVectorizationFailure(
7892 DebugMsg: "runtime checks are known to fail, so we will never enter the "
7893 "vector loop",
7894 ORETag: "RuntimeChecksNeverEnterVectorLoop", ORE, TheLoop: L);
7895 return false;
7896 }
7897 }
7898
7899 // Check if it is profitable to vectorize with runtime checks.
7900 bool ForceVectorization =
7901 Hints.getForce() == LoopVectorizeHints::FK_Enabled;
7902 VPCostContext CostCtx(*TLI, *BestPlanPtr, LVP.getCostModel(), Config,
7903 /*ReusePrintingSlotTracker=*/true);
7904 if (!ForceVectorization &&
7905 !isOutsideLoopWorkProfitable(Checks, VF, L, PSE, CostCtx, Plan&: *BestPlanPtr,
7906 SEL, VScale: Config.getVScaleForTuning())) {
7907 ORE->emit(RemarkBuilder: [&]() {
7908 return OptimizationRemarkAnalysisAliasing(
7909 DEBUG_TYPE, "CantReorderMemOps", L->getStartLoc(),
7910 L->getHeader())
7911 << "loop not vectorized: cannot prove it is safe to reorder "
7912 "memory operations";
7913 });
7914 LLVM_DEBUG(dbgs() << "LV: Too many memory checks needed.\n");
7915 Hints.emitRemarkWithHints();
7916 return false;
7917 }
7918 }
7919
7920 // Identify the diagnostic messages that should be produced.
7921 std::pair<StringRef, std::string> VecDiagMsg, IntDiagMsg;
7922 bool VectorizeLoop = true, InterleaveLoop = true;
7923 if (VF.Width.isScalar()) {
7924 if (LVP.hasVectorPlan()) {
7925 LLVM_DEBUG(
7926 dbgs() << "LV: Vectorization is possible but not beneficial.\n");
7927 VecDiagMsg = {
7928 "VectorizationNotBeneficial",
7929 "the cost-model indicates that vectorization is not beneficial"};
7930 } else {
7931 LLVM_DEBUG(dbgs() << "LV: Vectorization is not possible. Failed to "
7932 "create any vector VPlans.\n");
7933 VecDiagMsg = {"VectorizationNotPossible",
7934 "vectorization is not possible"};
7935 }
7936 VectorizeLoop = false;
7937 }
7938
7939 if (UserIC == 1 && Hints.getInterleave() > 1) {
7940 assert(!LVL.isSafeForAnyVectorWidth() &&
7941 "UserIC should only be ignored due to unsafe dependencies");
7942 LLVM_DEBUG(dbgs() << "LV: Ignoring user-specified interleave count.\n");
7943 IntDiagMsg = {"InterleavingUnsafe",
7944 "Ignoring user-specified interleave count due to possibly "
7945 "unsafe dependencies in the loop."};
7946 InterleaveLoop = false;
7947 } else if (!LVP.hasPlanWithVF(VF: VF.Width) && UserIC > 1) {
7948 // Tell the user interleaving was avoided up-front, despite being explicitly
7949 // requested.
7950 LLVM_DEBUG(dbgs() << "LV: Ignoring UserIC, because vectorization and "
7951 "interleaving should be avoided up front\n");
7952 IntDiagMsg = {"InterleavingAvoided",
7953 "Ignoring UserIC, because interleaving was avoided up front"};
7954 InterleaveLoop = false;
7955 } else if (IC == 1 && UserIC <= 1) {
7956 // Tell the user interleaving is not beneficial.
7957 if (LVP.hasAPlan()) {
7958 LLVM_DEBUG(dbgs() << "LV: Interleaving is not beneficial.\n");
7959 IntDiagMsg = {
7960 "InterleavingNotBeneficial",
7961 "the cost-model indicates that interleaving is not beneficial"};
7962 if (UserIC == 1) {
7963 IntDiagMsg.first = "InterleavingNotBeneficialAndDisabled";
7964 IntDiagMsg.second +=
7965 " and is explicitly disabled or interleave count is set to 1";
7966 }
7967 } else {
7968 LLVM_DEBUG(dbgs() << "LV: Interleaving is not possible. Failed to create"
7969 << " any vplans\n");
7970 IntDiagMsg = {"InterleavingNotPossible", "interleaving is not possible"};
7971 }
7972 InterleaveLoop = false;
7973 } else if (IC > 1 && UserIC == 1) {
7974 // Tell the user interleaving is beneficial, but it explicitly disabled.
7975 LLVM_DEBUG(dbgs() << "LV: Interleaving is beneficial but is explicitly "
7976 "disabled.\n");
7977 IntDiagMsg = {"InterleavingBeneficialButDisabled",
7978 "the cost-model indicates that interleaving is beneficial "
7979 "but is explicitly disabled or interleave count is set to 1"};
7980 InterleaveLoop = false;
7981 }
7982
7983 // If there is a histogram in the loop, do not just interleave without
7984 // vectorizing. The order of operations will be incorrect without the
7985 // histogram intrinsics, which are only used for recipes with VF > 1.
7986 if (!VectorizeLoop && InterleaveLoop && LVL.hasHistograms()) {
7987 LLVM_DEBUG(dbgs() << "LV: Not interleaving without vectorization due "
7988 << "to histogram operations.\n");
7989 IntDiagMsg = {
7990 "HistogramPreventsScalarInterleaving",
7991 "Unable to interleave without vectorization due to constraints on "
7992 "the order of histogram operations"};
7993 InterleaveLoop = false;
7994 }
7995
7996 // Override IC if user provided an interleave count.
7997 IC = UserIC > 0 ? UserIC : IC;
7998
7999 if (LVP.getCostModel().maskPartialAliasing()) {
8000 LLVM_DEBUG(
8001 dbgs()
8002 << "LV: Not interleaving due to partial aliasing vectorization.\n");
8003 IntDiagMsg = {
8004 "PartialAliasingVectorization",
8005 "Unable to interleave due to partial aliasing vectorization."};
8006 InterleaveLoop = false;
8007 IC = 1;
8008 }
8009
8010 // FIXME: Enable interleaving for EE-with-side-effects.
8011 if (InterleaveLoop && LVL.hasUncountableExitWithSideEffects()) {
8012 LLVM_DEBUG(dbgs() << "LV: Not interleaving due to EE with side effects.\n");
8013 IntDiagMsg = {"EEWithSideEffectsPreventsInterleaving",
8014 "Unable to interleave due to early exit with side effects."};
8015 InterleaveLoop = false;
8016 IC = 1;
8017 }
8018
8019 // Emit diagnostic messages, if any.
8020 if (!VectorizeLoop && !InterleaveLoop) {
8021 // Do not vectorize or interleaving the loop.
8022 ORE->emit(RemarkBuilder: [&]() {
8023 return OptimizationRemarkMissed(LV_NAME, VecDiagMsg.first,
8024 L->getStartLoc(), L->getHeader())
8025 << VecDiagMsg.second;
8026 });
8027 ORE->emit(RemarkBuilder: [&]() {
8028 return OptimizationRemarkMissed(LV_NAME, IntDiagMsg.first,
8029 L->getStartLoc(), L->getHeader())
8030 << IntDiagMsg.second;
8031 });
8032 return false;
8033 }
8034
8035 if (!VectorizeLoop && InterleaveLoop) {
8036 LLVM_DEBUG(dbgs() << "LV: Interleave Count is " << IC << '\n');
8037 ORE->emit(RemarkBuilder: [&]() {
8038 return OptimizationRemarkAnalysis(LV_NAME, VecDiagMsg.first,
8039 L->getStartLoc(), L->getHeader())
8040 << VecDiagMsg.second;
8041 });
8042 } else if (VectorizeLoop && !InterleaveLoop) {
8043 LLVM_DEBUG(dbgs() << "LV: Found a vectorizable loop (" << VF.Width
8044 << ") in " << L->getLocStr() << '\n');
8045 ORE->emit(RemarkBuilder: [&]() {
8046 return OptimizationRemarkAnalysis(LV_NAME, IntDiagMsg.first,
8047 L->getStartLoc(), L->getHeader())
8048 << IntDiagMsg.second;
8049 });
8050 } else if (VectorizeLoop && InterleaveLoop) {
8051 LLVM_DEBUG(dbgs() << "LV: Found a vectorizable loop (" << VF.Width
8052 << ") in " << L->getLocStr() << '\n');
8053 LLVM_DEBUG(dbgs() << "LV: Interleave Count is " << IC << '\n');
8054 }
8055
8056 // Report the vectorization decision.
8057 if (VF.Width.isScalar()) {
8058 using namespace ore;
8059 assert(IC > 1);
8060 ORE->emit(RemarkBuilder: [&]() {
8061 return OptimizationRemark(LV_NAME, "Interleaved", L->getStartLoc(),
8062 L->getHeader())
8063 << "interleaved loop (interleaved count: "
8064 << NV("InterleaveCount", IC) << ")";
8065 });
8066 } else {
8067 // Report the vectorization decision.
8068 reportVectorization(ORE, TheLoop: L, VFWidth: VF.Width, IC);
8069 }
8070 if (ORE->allowExtraAnalysis(LV_NAME))
8071 checkMixedPrecision(L, ORE);
8072
8073 // If we decided that it is *legal* to interleave or vectorize the loop, then
8074 // do it.
8075
8076 // Whether a scalar epilogue may be created is decided by the epilogue
8077 // lowering policy.
8078 // TODO: Also move check to be based on VPlan.
8079 bool ScalarEpilogueAllowed = LVP.getCostModel().isEpilogueAllowed();
8080
8081 // Destroy the cost model before executing any plan, so that code generation
8082 // cannot rely on cost-modeling decisions.
8083 LVP.clearCostModel();
8084
8085 VPlan &BestPlan = *BestPlanPtr;
8086 // Consider vectorizing the epilogue too if it's profitable.
8087 std::unique_ptr<VPlan> EpiPlan =
8088 LVP.selectBestEpiloguePlan(MainPlan&: BestPlan, MainLoopVF: VF.Width, IC, ScalarEpilogueAllowed);
8089 bool HasBranchWeights =
8090 hasBranchWeightMD(I: *L->getLoopLatch()->getTerminator());
8091 if (EpiPlan) {
8092 VPlan &BestEpiPlan = *EpiPlan;
8093 VPlan &BestMainPlan = BestPlan;
8094 ElementCount EpilogueVF = BestEpiPlan.getSingleVF();
8095
8096 // The first pass vectorizes the main loop and creates a scalar epilogue
8097 // to be vectorized by executing the plan (potentially with a different
8098 // factor) again shortly afterwards.
8099 BestEpiPlan.getMiddleBlock()->setName("vec.epilog.middle.block");
8100 BestEpiPlan.getVectorPreheader()->setName("vec.epilog.ph");
8101 MainPlanResumeMarkers Markers = preparePlanForMainVectorLoop(MainPlan&: BestMainPlan);
8102
8103 // Add minimum iteration check for the epilogue plan, followed by runtime
8104 // checks for the main plan.
8105 LVP.addMinimumIterationCheck(Plan&: BestMainPlan, VF: EpilogueVF, /*UF=*/1,
8106 MinProfitableTripCount: ElementCount::getFixed(MinVal: 0));
8107 LVP.attachRuntimeChecks(Plan&: BestMainPlan, RTChecks&: Checks, HasBranchWeights);
8108 RUN_VPLAN_PASS(VPlanTransforms::addIterationCountCheckBlock, BestMainPlan,
8109 VF.Width, IC, BestMainPlan.requiresScalarEpilogue(), L,
8110 HasBranchWeights ? MinItersBypassWeights : nullptr,
8111 L->getLoopPredecessor()->getTerminator()->getDebugLoc(),
8112 PSE);
8113
8114 LLVM_DEBUG({
8115 dbgs() << "Create Skeleton for epilogue vectorized loop (first pass)\n"
8116 << "Main Loop VF:" << VF.Width << ", Main Loop UF:" << IC
8117 << ", Epilogue Loop VF:" << EpilogueVF << ", Epilogue Loop UF:1\n";
8118 });
8119 InnerLoopVectorizer MainILV(L, PSE, LI, DT, TTI, AC, VF.Width, IC, Checks,
8120 BestMainPlan);
8121 auto ExpandedSCEVs = LVP.executePlan(
8122 BestVF: VF.Width, BestUF: IC, BestVPlan&: BestMainPlan, ILV&: MainILV, DT,
8123 EpilogueVecKind: LoopVectorizationPlanner::EpilogueVectorizationKind::MainLoop);
8124 ++LoopsVectorized;
8125 DEBUG_WITH_TYPE(VerboseDebug, {
8126 dbgs() << "intermediate fn:\n" << *L->getHeader()->getParent() << "\n";
8127 });
8128
8129 BasicBlock *EntryBB =
8130 cast<VPIRBasicBlock>(Val: BestMainPlan.getEntry())->getIRBasicBlock();
8131 EntryBB->setName("iter.check");
8132
8133 // Second pass vectorizes the epilogue and adjusts the control flow
8134 // edges from the first pass.
8135 EpilogueVectorizerEpilogueLoop EpilogILV(L, PSE, LI, DT, TTI, AC,
8136 EpilogueVF, /*UnrollFactor=*/1,
8137 Checks, BestEpiPlan, BestMainPlan);
8138 preparePlanForEpilogueVectorLoop(MainPlan&: BestMainPlan, Plan&: BestEpiPlan, L,
8139 ExpandedSCEVs, MainLoopVF: VF.Width, MainLoopUF: IC, EpilogueVF,
8140 LVP, Config, SE&: *PSE.getSE(), Markers);
8141 RUN_VPLAN_PASS(VPlanTransforms::simplifyLiveInsWithSCEV, BestEpiPlan, PSE);
8142 LLVM_DEBUG({
8143 dbgs() << "Create Skeleton for epilogue vectorized loop (second pass)\n"
8144 << "Epilogue Loop VF:" << EpilogueVF << ", Epilogue Loop UF:1\n";
8145 });
8146 LVP.executePlan(
8147 BestVF: EpilogueVF, /*BestUF=*/1, BestVPlan&: BestEpiPlan, ILV&: EpilogILV, DT,
8148 EpilogueVecKind: LoopVectorizationPlanner::EpilogueVectorizationKind::Epilogue);
8149 DEBUG_WITH_TYPE(VerboseDebug, {
8150 dbgs() << "final fn:\n" << *L->getHeader()->getParent() << "\n";
8151 });
8152 connectEpilogueVectorLoop(EpiPlan&: BestEpiPlan, DT,
8153 VecEpilogueIterCheckVPBB: EpilogILV.VecEpilogueIterationCountCheck,
8154 ResumeValues: Markers.ResumeValues);
8155 ++LoopsEpilogueVectorized;
8156 } else {
8157 InnerLoopVectorizer LB(L, PSE, LI, DT, TTI, AC, VF.Width, IC, Checks,
8158 BestPlan);
8159 LVP.addMinimumIterationCheck(Plan&: BestPlan, VF: VF.Width, UF: IC,
8160 MinProfitableTripCount: VF.MinProfitableTripCount);
8161 LVP.attachRuntimeChecks(Plan&: BestPlan, RTChecks&: Checks, HasBranchWeights);
8162
8163 if (!IsInnerLoop)
8164 LLVM_DEBUG(dbgs() << "Vectorizing outer loop in \"" << F->getName()
8165 << "\"\n");
8166 LVP.executePlan(BestVF: VF.Width, BestUF: IC, BestVPlan&: BestPlan, ILV&: LB, DT);
8167 ++LoopsVectorized;
8168 }
8169
8170 assert(DT->verify(DominatorTree::VerificationLevel::Fast) &&
8171 "DT not preserved correctly");
8172
8173 return true;
8174}
8175
8176LoopVectorizeResult LoopVectorizePass::runImpl(Function &F) {
8177 CFGChanged = false;
8178
8179 // Don't attempt if
8180 // 1. the target claims to have no vector registers, and
8181 // 2. interleaving won't help ILP.
8182 //
8183 // The second condition is necessary because, even if the target has no
8184 // vector registers, loop vectorization may still enable scalar
8185 // interleaving.
8186 if (!TTI->getNumberOfRegisters(ClassID: TTI->getRegisterClassForType(Vector: true)) &&
8187 (TTI->getMaxInterleaveFactor(VF: ElementCount::getFixed(MinVal: 1), HasUnorderedReductions: false) < 2 ||
8188 TTI->getMaxInterleaveFactor(VF: ElementCount::getFixed(MinVal: 1), HasUnorderedReductions: true) < 2))
8189 return LoopVectorizeResult(false, false);
8190
8191 bool Changed = false;
8192
8193 // The vectorizer requires loops to be in simplified form.
8194 // Since simplification may add new inner loops, it has to run before the
8195 // legality and profitability checks. This means running the loop vectorizer
8196 // will simplify all loops, regardless of whether anything end up being
8197 // vectorized.
8198 for (const auto &L : *LI)
8199 Changed |= CFGChanged |=
8200 simplifyLoop(L, DT, LI, SE, AC, MSSAU: nullptr, PreserveLCSSA: false /* PreserveLCSSA */);
8201
8202 // Build up a worklist of inner-loops to vectorize. This is necessary as
8203 // the act of vectorizing or partially unrolling a loop creates new loops
8204 // and can invalidate iterators across the loops.
8205 SmallVector<Loop *, 8> Worklist;
8206
8207 for (Loop *L : *LI)
8208 collectSupportedLoops(L&: *L, LI, ORE, V&: Worklist);
8209
8210 LoopsAnalyzed += Worklist.size();
8211
8212 // Now walk the identified inner loops.
8213 while (!Worklist.empty()) {
8214 Loop *L = Worklist.pop_back_val();
8215
8216 // For the inner loops we actually process, form LCSSA to simplify the
8217 // transform.
8218 Changed |= formLCSSARecursively(L&: *L, DT: *DT, LI, SE);
8219
8220 Changed |= CFGChanged |= processLoop(L);
8221
8222 if (Changed) {
8223 LAIs->clear();
8224
8225#ifndef NDEBUG
8226 if (VerifySCEV)
8227 SE->verify();
8228#endif
8229 }
8230 }
8231
8232 // Verify once per function rather than once per processed loop, which would
8233 // make the pass quadratic in the number of loops.
8234 assert((!Changed || !verifyFunction(F, &dbgs())) &&
8235 "Invalid IR produced by LoopVectorize");
8236
8237 // Process each loop nest in the function.
8238 return LoopVectorizeResult(Changed, CFGChanged);
8239}
8240
8241PreservedAnalyses LoopVectorizePass::run(Function &F,
8242 FunctionAnalysisManager &AM) {
8243 LI = &AM.getResult<LoopAnalysis>(IR&: F);
8244 // There are no loops in the function. Return before computing other
8245 // expensive analyses.
8246 if (LI->empty())
8247 return PreservedAnalyses::all();
8248 SE = &AM.getResult<ScalarEvolutionAnalysis>(IR&: F);
8249 TTI = &AM.getResult<TargetIRAnalysis>(IR&: F);
8250 DT = &AM.getResult<DominatorTreeAnalysis>(IR&: F);
8251 TLI = &AM.getResult<TargetLibraryAnalysis>(IR&: F);
8252 AC = &AM.getResult<AssumptionAnalysis>(IR&: F);
8253 DB = &AM.getResult<DemandedBitsAnalysis>(IR&: F);
8254 ORE = &AM.getResult<OptimizationRemarkEmitterAnalysis>(IR&: F);
8255 LAIs = &AM.getResult<LoopAccessAnalysis>(IR&: F);
8256 AA = &AM.getResult<AAManager>(IR&: F);
8257
8258 auto &MAMProxy = AM.getResult<ModuleAnalysisManagerFunctionProxy>(IR&: F);
8259 PSI = MAMProxy.getCachedResult<ProfileSummaryAnalysis>(IR&: *F.getParent());
8260 // CycleInfo cached by an earlier pass is invalidated when the CFG changes.
8261 // Both BlockFrequencyAnalysis and BranchProbabilityAnalysis depend on it, so
8262 // drop the stale result before either is (re-)computed.
8263 auto ClearStaleCycleInfo = [this, &AM, &F] {
8264 if (CFGChanged && AM.getCachedResult<CycleAnalysis>(IR&: F))
8265 AM.clearAnalysis<CycleAnalysis>(IR&: F);
8266 };
8267 GetBFI = [&AM, &F, ClearStaleCycleInfo]() -> BlockFrequencyInfo & {
8268 ClearStaleCycleInfo();
8269 return AM.getResult<BlockFrequencyAnalysis>(IR&: F);
8270 };
8271 GetBPI = [&AM, &F, ClearStaleCycleInfo]() -> const BranchProbabilityInfo & {
8272 ClearStaleCycleInfo();
8273 return AM.getResult<BranchProbabilityAnalysis>(IR&: F);
8274 };
8275 LoopVectorizeResult Result = runImpl(F);
8276 if (!Result.MadeAnyChange)
8277 return PreservedAnalyses::all();
8278 PreservedAnalyses PA;
8279
8280 if (isAssignmentTrackingEnabled(M: *F.getParent())) {
8281 for (auto &BB : F)
8282 RemoveRedundantDbgInstrs(BB: &BB);
8283 }
8284
8285 PA.preserve<LoopAnalysis>();
8286 PA.preserve<DominatorTreeAnalysis>();
8287 PA.preserve<ScalarEvolutionAnalysis>();
8288 PA.preserve<LoopAccessAnalysis>();
8289
8290 if (Result.MadeCFGChange) {
8291 // Making CFG changes likely means a loop got vectorized. Indicate that
8292 // extra simplification passes should be run.
8293 // TODO: MadeCFGChanges is not a prefect proxy. Extra passes should only
8294 // be run if runtime checks have been added.
8295 AM.getResult<ShouldRunExtraVectorPasses>(IR&: F);
8296 PA.preserve<ShouldRunExtraVectorPasses>();
8297 } else {
8298 PA.preserveSet<CFGAnalyses>();
8299 }
8300 return PA;
8301}
8302
8303void LoopVectorizePass::printPipeline(
8304 raw_ostream &OS, function_ref<StringRef(StringRef)> MapClassName2PassName) {
8305 static_cast<PassInfoMixin<LoopVectorizePass> *>(this)->printPipeline(
8306 OS, MapClassName2PassName);
8307
8308 OS << '<';
8309 OS << (InterleaveOnlyWhenForced ? "" : "no-") << "interleave-forced-only;";
8310 OS << (VectorizeOnlyWhenForced ? "" : "no-") << "vectorize-forced-only;";
8311 OS << '>';
8312}
8313