LLVM 24.0.0git
LoopVectorize.cpp
Go to the documentation of this file.
1//===- LoopVectorize.cpp - A Loop Vectorizer ------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This is the LLVM loop vectorizer. This pass modifies 'vectorizable' loops
10// and generates target-independent LLVM-IR.
11// The vectorizer uses the TargetTransformInfo analysis to estimate the costs
12// of instructions in order to estimate the profitability of vectorization.
13//
14// The loop vectorizer combines consecutive loop iterations into a single
15// 'wide' iteration. After this transformation the index is incremented
16// by the SIMD vector width, and not by one.
17//
18// This pass has three parts:
19// 1. The main loop pass that drives the different parts.
20// 2. LoopVectorizationLegality - A unit that checks for the legality
21// of the vectorization.
22// 3. InnerLoopVectorizer - A unit that performs the actual
23// widening of instructions.
24// 4. LoopVectorizationCostModel - A unit that checks for the profitability
25// of vectorization. It decides on the optimal vector width, which
26// can be one, if vectorization is not profitable.
27//
28// There is a development effort going on to migrate loop vectorizer to the
29// VPlan infrastructure and to introduce outer loop vectorization support (see
30// docs/VectorizationPlan.rst and
31// http://lists.llvm.org/pipermail/llvm-dev/2017-December/119523.html). For this
32// purpose, we temporarily introduced the VPlan-native vectorization path: an
33// alternative vectorization path that is natively implemented on top of the
34// VPlan infrastructure. See EnableVPlanNativePath for enabling.
35//
36//===----------------------------------------------------------------------===//
37//
38// The reduction-variable vectorization is based on the paper:
39// D. Nuzman and R. Henderson. Multi-platform Auto-vectorization.
40//
41// Variable uniformity checks are inspired by:
42// Karrenberg, R. and Hack, S. Whole Function Vectorization.
43//
44// The interleaved access vectorization is based on the paper:
45// Dorit Nuzman, Ira Rosen and Ayal Zaks. Auto-Vectorization of Interleaved
46// Data for SIMD
47//
48// Other ideas/concepts are from:
49// A. Zaks and D. Nuzman. Autovectorization in GCC-two years later.
50//
51// S. Maleki, Y. Gao, M. Garzaran, T. Wong and D. Padua. An Evaluation of
52// Vectorizing Compilers.
53//
54//===----------------------------------------------------------------------===//
55
58#include "VPRecipeBuilder.h"
59#include "VPlan.h"
60#include "VPlanAnalysis.h"
61#include "VPlanCFG.h"
62#include "VPlanHelpers.h"
63#include "VPlanPatternMatch.h"
64#include "VPlanTransforms.h"
65#include "VPlanUtils.h"
66#include "VPlanVerifier.h"
67#include "llvm/ADT/APInt.h"
68#include "llvm/ADT/ArrayRef.h"
69#include "llvm/ADT/DenseMap.h"
70#include "llvm/ADT/Hashing.h"
71#include "llvm/ADT/MapVector.h"
72#include "llvm/ADT/STLExtras.h"
75#include "llvm/ADT/Statistic.h"
76#include "llvm/ADT/StringRef.h"
77#include "llvm/ADT/Twine.h"
78#include "llvm/ADT/TypeSwitch.h"
84#include "llvm/Analysis/CFG.h"
102#include "llvm/IR/Attributes.h"
103#include "llvm/IR/BasicBlock.h"
104#include "llvm/IR/CFG.h"
105#include "llvm/IR/Constant.h"
106#include "llvm/IR/Constants.h"
107#include "llvm/IR/DataLayout.h"
108#include "llvm/IR/DebugInfo.h"
109#include "llvm/IR/DebugLoc.h"
110#include "llvm/IR/DerivedTypes.h"
112#include "llvm/IR/Dominators.h"
113#include "llvm/IR/Function.h"
114#include "llvm/IR/IRBuilder.h"
115#include "llvm/IR/InstrTypes.h"
116#include "llvm/IR/Instruction.h"
117#include "llvm/IR/Instructions.h"
119#include "llvm/IR/Intrinsics.h"
120#include "llvm/IR/MDBuilder.h"
121#include "llvm/IR/Metadata.h"
122#include "llvm/IR/Module.h"
123#include "llvm/IR/Operator.h"
124#include "llvm/IR/PatternMatch.h"
126#include "llvm/IR/Type.h"
127#include "llvm/IR/Use.h"
128#include "llvm/IR/User.h"
129#include "llvm/IR/Value.h"
130#include "llvm/IR/Verifier.h"
131#include "llvm/Support/Casting.h"
133#include "llvm/Support/Debug.h"
148#include <algorithm>
149#include <cassert>
150#include <cmath>
151#include <cstdint>
152#include <functional>
153#include <iterator>
154#include <memory>
155#include <string>
156#include <tuple>
157#include <utility>
158
159using namespace llvm;
160using namespace SCEVPatternMatch;
161using namespace LoopVectorizationUtils;
162
163#define LV_NAME "loop-vectorize"
164#define DEBUG_TYPE LV_NAME
165
166#ifndef NDEBUG
167const char VerboseDebug[] = DEBUG_TYPE "-verbose";
168#endif
169
170STATISTIC(LoopsVectorized, "Number of loops vectorized");
171STATISTIC(LoopsAnalyzed, "Number of loops analyzed for vectorization");
172STATISTIC(LoopsEpilogueVectorized, "Number of epilogues vectorized");
173STATISTIC(LoopsEarlyExitVectorized, "Number of early exit loops vectorized");
174STATISTIC(LoopsPartialAliasVectorized,
175 "Number of partial aliasing loops vectorized");
176
178 "enable-epilogue-vectorization", cl::init(true), cl::Hidden,
179 cl::desc("Enable vectorization of epilogue loops."));
180
182 "epilogue-vectorization-force-VF", cl::init(ElementCount::getFixed(1)),
184 cl::desc("When epilogue vectorization is enabled, and a value greater than "
185 "1 is specified, forces the given VF for all applicable epilogue "
186 "loops. Note: This allows all scalable VFs >= vscale x 1."));
187
189 "epilogue-vectorization-minimum-VF", cl::Hidden,
190 cl::desc("Only loops with vectorization factor equal to or larger than "
191 "the specified value are considered for epilogue vectorization."));
192
193/// Loops with a known constant trip count below this number are vectorized only
194/// if no scalar iteration overheads are incurred.
196 "vectorizer-min-trip-count", cl::init(16), cl::Hidden,
197 cl::desc("Loops with a constant trip count that is smaller than this "
198 "value are vectorized only if no scalar iteration overheads "
199 "are incurred."));
200
202 "force-partial-aliasing-vectorization", cl::init(false), cl::Hidden,
203 cl::desc("Replace pointer diff checks with alias masks."));
204
205/// Option tail-folding-policy controls the tail-folding strategy and lists all
206/// available options. The vectorizer will attempt to fold the tail-loop into
207/// the vector loop (main/epilogue loops) and predicate the instructions
208/// accordingly. If tail-folding fails, there are different fallback strategies
209/// depending on these values:
211
213 "tail-folding-policy", cl::init(TailFoldingPolicyTy::None), cl::Hidden,
214 cl::desc("Tail-folding preferences over creating an epilogue loop."),
216 clEnumValN(TailFoldingPolicyTy::None, "dont-fold-tail",
217 "Don't tail-fold loops."),
219 "prefer tail-folding, otherwise create an epilogue when "
220 "appropriate."),
222 "always tail-fold, don't attempt vectorization if "
223 "tail-folding fails.")));
224
226 "epilogue-tail-folding-policy", cl::Hidden,
227 cl::desc(
228 "Epilogue-tail-folding preferences over creating an epilogue loop."),
230 clEnumValN(TailFoldingPolicyTy::None, "dont-fold-tail",
231 "Don't tail-fold loops."),
233 "prefer tail-folding, otherwise create an epilogue when "
234 "appropriate.")));
235
237 "force-tail-folding-style", cl::desc("Force the tail folding style"),
240 clEnumValN(TailFoldingStyle::None, "none", "Disable tail folding"),
243 "Create lane mask for data only, using active.lane.mask intrinsic"),
245 "data-without-lane-mask",
246 "Create lane mask with compare/stepvector"),
248 "Create lane mask using active.lane.mask intrinsic, and use "
249 "it for both data and control flow"),
251 "Use predicated EVL instructions for tail folding. If EVL "
252 "is unsupported, fallback to data-without-lane-mask.")));
253
255 "enable-interleaved-mem-accesses", cl::init(false), cl::Hidden,
256 cl::desc("Enable vectorization on interleaved memory accesses in a loop"));
257
258/// An interleave-group may need masking if it resides in a block that needs
259/// predication, or in order to mask away gaps.
261 "enable-masked-interleaved-mem-accesses", cl::init(false), cl::Hidden,
262 cl::desc("Enable vectorization on masked interleaved memory accesses in a loop"));
263
265 "force-target-num-scalar-regs", cl::init(0), cl::Hidden,
266 cl::desc("A flag that overrides the target's number of scalar registers."));
267
269 "force-target-num-vector-regs", cl::init(0), cl::Hidden,
270 cl::desc("A flag that overrides the target's number of vector registers."));
271
273 "force-target-max-scalar-interleave", cl::init(0), cl::Hidden,
274 cl::desc("A flag that overrides the target's max interleave factor for "
275 "scalar loops."));
276
278 "force-target-max-vector-interleave", cl::init(0), cl::Hidden,
279 cl::desc("A flag that overrides the target's max interleave factor for "
280 "vectorized loops."));
281
283 "small-loop-cost", cl::init(20), cl::Hidden,
284 cl::desc(
285 "The cost of a loop that is considered 'small' by the interleaver."));
286
288 "loop-vectorize-with-block-frequency", cl::init(true), cl::Hidden,
289 cl::desc("Enable the use of the block frequency analysis to access PGO "
290 "heuristics minimizing code growth in cold regions and being more "
291 "aggressive in hot regions."));
292
293// Runtime interleave loops for load/store throughput.
295 "enable-loadstore-runtime-interleave", cl::init(true), cl::Hidden,
296 cl::desc(
297 "Enable runtime interleaving until load/store ports are saturated"));
298
299// TODO: Move size-based thresholds out of legality checking, make cost based
300// decisions instead of hard thresholds.
302 "vectorize-scev-check-threshold", cl::init(16), cl::Hidden,
303 cl::desc("The maximum number of SCEV checks allowed."));
304
306 "pragma-vectorize-scev-check-threshold", cl::init(128), cl::Hidden,
307 cl::desc("The maximum number of SCEV checks allowed with a "
308 "vectorize(enable) pragma"));
309
311 "enable-ind-var-reg-heur", cl::init(true), cl::Hidden,
312 cl::desc("Count the induction variable only once when interleaving"));
313
315 "max-nested-scalar-reduction-interleave", cl::init(2), cl::Hidden,
316 cl::desc("The maximum interleave count to use when interleaving a scalar "
317 "reduction in a nested loop."));
318
320 "force-ordered-reductions", cl::init(false), cl::Hidden,
321 cl::desc("Enable the vectorisation of loops with in-order (strict) "
322 "FP reductions"));
323
325 "prefer-predicated-reduction-select", cl::init(false), cl::Hidden,
326 cl::desc(
327 "Prefer predicating a reduction operation over an after loop select."));
328
330 "enable-vplan-native-path", cl::Hidden,
331 cl::desc("Enable VPlan-native vectorization path with "
332 "support for outer loop vectorization."));
333
335 llvm::VerifyEachVPlan("vplan-verify-each",
336#ifdef EXPENSIVE_CHECKS
337 cl::init(true),
338#else
339 cl::init(false),
340#endif
342 cl::desc("Verify VPlans after VPlan transforms."));
343
344#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
346 "vplan-print-before-all", cl::init(false), cl::Hidden,
347 cl::desc("Print VPlans before all VPlan transformations."));
348
350 "vplan-print-after-all", cl::init(false), cl::Hidden,
351 cl::desc("Print VPlans after all VPlan transformations."));
352
354 "vplan-print-before", cl::Hidden,
355 cl::desc("Print VPlans before specified VPlan transformations (regexp)."));
356
358 "vplan-print-after", cl::Hidden,
359 cl::desc("Print VPlans after specified VPlan transformations (regexp)."));
360
362 "vplan-print-vector-region-scope", cl::init(false), cl::Hidden,
363 cl::desc("Limit VPlan printing to vector loop region in "
364 "`-vplan-print-after*` if the plan has one."));
365#endif
366
368 "interleave-loops", cl::init(true), cl::Hidden,
369 cl::desc("Enable loop interleaving in Loop vectorization passes"));
371 "vectorize-loops", cl::init(true), cl::Hidden,
372 cl::desc("Run the Loop vectorization passes"));
373
374namespace llvm {
376 "force-target-instruction-cost", cl::init(0), cl::Hidden,
377 cl::desc("A flag that overrides the target's expected cost for "
378 "an instruction to a single constant value. Mostly "
379 "useful for getting consistent testing."));
380
381/// The number of stores in a loop that are allowed to need predication.
383 "vectorize-num-stores-pred", cl::init(1), cl::Hidden,
384 cl::desc("Max number of stores to be predicated behind an if."));
385
386// This flag enables the stress testing of the VPlan H-CFG construction in the
387// VPlan-native vectorization path. It must be used in conjuction with
388// -enable-vplan-native-path. -vplan-verify-hcfg can also be used to enable the
389// verification of the H-CFGs built.
391 "vplan-build-outerloop-stress-test", cl::init(false), cl::Hidden,
392 cl::desc(
393 "Build VPlan for every supported loop nest in the function and bail "
394 "out right after the build (stress test the VPlan H-CFG construction "
395 "in the VPlan-native vectorization path)."));
396} // namespace llvm
397
399 ForceMaskedDivRem("force-widen-divrem-via-masked-intrinsic", cl::Hidden,
400 cl::desc("Override cost based masked intrinsic widening "
401 "for div/rem instructions"));
402
404 "enable-early-exit-vectorization", cl::init(true), cl::Hidden,
405 cl::desc(
406 "Enable vectorization of early exit loops with uncountable exits."));
407
409 "enable-early-exit-vectorization-with-side-effects", cl::init(false),
411 cl::desc("Enable vectorization of early exit loops with uncountable exits "
412 "and side effects"));
413
415 "low-trip-count-loop-body-size-limit", cl::init(20), cl::Hidden,
416 cl::desc("Minimum number of instructions to vectorize loops with trip "
417 "counts below tail folding threshold"));
418
419// Returns true if the epilogue VF has been set to a non-zero value other than
420// VF=1 (scalar).
425
426// Likelyhood of bypassing the vectorized loop because there are zero trips left
427// after prolog. See `emitIterationCountCheck`.
428static constexpr uint32_t MinItersBypassWeights[] = {1, 127};
429
430/// A version of ScalarEvolution::getSmallConstantTripCount that returns an
431/// ElementCount to include loops whose trip count is a function of vscale.
433 const Loop *L) {
434 if (unsigned ExpectedTC = SE->getSmallConstantTripCount(L))
435 return ElementCount::getFixed(ExpectedTC);
436
437 const SCEV *BTC = SE->getBackedgeTakenCount(L);
439 return ElementCount::getFixed(0);
440
441 const SCEV *ExitCount = SE->getTripCountFromExitCount(BTC, BTC->getType(), L);
442 if (isa<SCEVVScale>(ExitCount))
444
445 const APInt *Scale;
446 if (match(ExitCount, m_scev_Mul(m_scev_APInt(Scale), m_SCEVVScale())))
447 if (cast<SCEVMulExpr>(ExitCount)->hasNoUnsignedWrap())
448 if (Scale->getActiveBits() <= 32)
450
451 return ElementCount::getFixed(0);
452}
453
454/// Get the maximum trip count for \p L from the SCEV unsigned range, excluding
455/// zero from the range. Only valid when not folding the tail, as the minimum
456/// iteration count check guards against a zero trip count. Returns 0 if
457/// unknown.
459 Loop *L) {
460 const SCEV *BTC = PSE.getBackedgeTakenCount();
462 return 0;
463 ScalarEvolution *SE = PSE.getSE();
464 const SCEV *TripCount = SE->getTripCountFromExitCount(BTC, BTC->getType(), L);
465 ConstantRange TCRange = SE->getUnsignedRange(TripCount);
466 APInt MaxTCFromRange = TCRange.getUnsignedMax();
467 if (!MaxTCFromRange.isZero() && MaxTCFromRange.getActiveBits() <= 32)
468 return MaxTCFromRange.getZExtValue();
469 return 0;
470}
471
472/// Returns "best known" trip count, which is either a valid positive trip count
473/// or std::nullopt when an estimate cannot be made (including when the trip
474/// count would overflow), for the specified loop \p L as defined by the
475/// following procedure:
476/// 1) Returns exact trip count if it is known.
477/// 2) Returns expected trip count according to profile data if any.
478/// 3) Returns upper bound estimate if known, if \p CanUseConstantMax, and
479/// if \p ComputeUpperBoundOnly is false.
480/// 4) Returns the maximum trip count from the SCEV range excluding zero,
481/// if \p CanUseConstantMax and \p CanExcludeZeroTrips.
482/// 5) Returns std::nullopt if all of the above failed.
483static std::optional<ElementCount> getSmallBestKnownTC(
484 PredicatedScalarEvolution &PSE, Loop *L, bool CanUseConstantMax = true,
485 bool CanExcludeZeroTrips = false, bool ComputeUpperBoundOnly = false) {
486 // Check if exact trip count is known.
487 if (auto ExpectedTC = getSmallConstantTripCount(PSE.getSE(), L))
488 return ExpectedTC;
489
490 // Check if there is an expected trip count available from profile data.
491 // An estimate of zero means the loop is estimated not to be entered; it is
492 // not a usable trip count for the profitability decisions below (and would
493 // e.g. divide by zero when scaling runtime check cost), so treat it as
494 // unknown.
495 if (LoopVectorizeWithBlockFrequency && !ComputeUpperBoundOnly)
496 if (unsigned EstimatedTC = getLoopEstimatedTripCount(L).value_or(0))
497 return ElementCount::getFixed(EstimatedTC);
498
499 if (!CanUseConstantMax)
500 return std::nullopt;
501
502 // Check if upper bound estimate is known.
503 if (unsigned ExpectedTC = PSE.getSmallConstantMaxTripCount())
504 return ElementCount::getFixed(ExpectedTC);
505
506 // Get the maximum trip count from the SCEV range excluding zero. This is
507 // only safe when not folding the tail, as the minimum iteration count check
508 // prevents entering the vector loop with a zero trip count.
509 if (CanUseConstantMax && CanExcludeZeroTrips)
510 if (unsigned RefinedTC = getMaxTCFromNonZeroRange(PSE, L))
511 return ElementCount::getFixed(RefinedTC);
512
513 return std::nullopt;
514}
515
516namespace {
517// Forward declare GeneratedRTChecks.
518class GeneratedRTChecks;
519
520using SCEV2ValueTy = DenseMap<const SCEV *, Value *>;
521} // namespace
522
523namespace llvm {
524
526
527/// InnerLoopVectorizer vectorizes loops which contain only one basic
528/// block to a specified vectorization factor (VF).
529/// This class performs the widening of scalars into vectors, or multiple
530/// scalars. This class also implements the following features:
531/// * It inserts an epilogue loop for handling loops that don't have iteration
532/// counts that are known to be a multiple of the vectorization factor.
533/// * It handles the code generation for reduction variables.
534/// * Scalarization (implementation using scalars) of un-vectorizable
535/// instructions.
536/// InnerLoopVectorizer does not perform any vectorization-legality
537/// checks, and relies on the caller to check for the different legality
538/// aspects. The InnerLoopVectorizer relies on the
539/// LoopVectorizationLegality class to provide information about the induction
540/// and reduction variables that were found to a given vectorization factor.
542public:
546 ElementCount VecWidth, unsigned UnrollFactor,
547 GeneratedRTChecks &RTChecks, VPlan &Plan)
548 : OrigLoop(OrigLoop), PSE(PSE), LI(LI), DT(DT), TTI(TTI), AC(AC),
549 VF(VecWidth), UF(UnrollFactor), Builder(PSE.getSE()->getContext()),
552 Plan.getVectorLoopRegion()->getSinglePredecessor())) {}
553
554 virtual ~InnerLoopVectorizer() = default;
555
556 /// Creates a basic block for the scalar preheader.
557 /// EpilogueVectorizerEpilogueLoop overrides the method to create additional
558 /// blocks and checks needed for epilogue vectorization.
560
561 /// Fix the vectorized code, taking care of header phi's, and more.
563
564protected:
566
567 /// Create and return a new IR basic block for the scalar preheader whose name
568 /// is prefixed with \p Prefix.
570
571 /// The original loop.
573
574 /// A wrapper around ScalarEvolution used to add runtime SCEV checks. Applies
575 /// dynamic knowledge to simplify SCEV expressions and converts them to a
576 /// more usable form.
578
579 /// Loop Info.
581
582 /// Dominator Tree.
584
585 /// Target Transform Info.
587
588 /// Assumption Cache.
590
591 /// The vectorization SIMD factor to use. Each vector will have this many
592 /// vector elements.
594
595 /// The vectorization unroll factor to use. Each scalar is vectorized to this
596 /// many different vector instructions.
597 unsigned UF;
598
599 /// The builder that we use
601
602 // --- Vectorization state ---
603
604 /// Structure to hold information about generated runtime checks, responsible
605 /// for cleaning the checks, if vectorization turns out unprofitable.
606 GeneratedRTChecks &RTChecks;
607
609
610 /// The vector preheader block of \p Plan, used as target for check blocks
611 /// introduced during skeleton creation.
613};
614
615/// Encapsulate information regarding vectorization of a loop and its epilogue.
616/// This information is meant to be updated and used across two stages of
617/// epilogue vectorization.
628
629/// A specialized derived class of inner loop vectorizer that performs
630/// vectorization of *epilogue* loops in the process of vectorizing loops and
631/// their epilogues. The idea is to run the vplan on a given loop twice, firstly
632/// to vectorize the main loop, and secondly to complete the skeleton from the
633/// first step and vectorize the epilogue. This helps us avoid regenerating and
634/// recomputing runtime safety checks, and shortens the iteration-count-check
635/// path length for loops whose iteration count is so small that the main vector
636/// loop is completely skipped.
638 VPlan &MainPlan;
639
640public:
642
647 unsigned UnrollFactor,
648 GeneratedRTChecks &Checks, VPlan &Plan,
649 VPlan &MainPlan)
650 : InnerLoopVectorizer(OrigLoop, PSE, LI, DT, TTI, AC, VecWidth,
651 UnrollFactor, Checks, Plan),
652 MainPlan(MainPlan) {}
653 /// Implements the interface for creating a vectorized skeleton using the
654 /// *epilogue loop* strategy (i.e., the second pass of VPlan execution).
656};
657} // end namespace llvm
658
659/// Look for a meaningful debug location on the instruction or its operands.
661 if (!I)
662 return DebugLoc::getUnknown();
663
665 if (I->getDebugLoc() != Empty)
666 return I->getDebugLoc();
667
668 for (Use &Op : I->operands()) {
669 if (Instruction *OpInst = dyn_cast<Instruction>(Op))
670 if (OpInst->getDebugLoc() != Empty)
671 return OpInst->getDebugLoc();
672 }
673
674 return I->getDebugLoc();
675}
676
677namespace llvm {
678
679/// Return the runtime value for VF.
681 return B.CreateElementCount(Ty, VF);
682}
683
684} // end namespace llvm
685
686namespace llvm {
687
688// Loop vectorization cost-model hints how the epilogue/tail loop should be
689// lowered.
691
692 // The default: allowing epilogues.
694
695 // Vectorization with OptForSize: don't allow epilogues.
697
698 // A special case of vectorisation with OptForSize: loops with a very small
699 // trip count are considered for vectorization under OptForSize, thereby
700 // making sure the cost of their loop body is dominant, free of runtime
701 // guards and scalar iteration overheads.
703
704 // Loop hint indicating an epilogue is undesired, apply tail folding.
706
707 // Directive indicating we must either fold the epilogue/tail or not vectorize
709};
710
712
713/// LoopVectorizationCostModel - estimates the expected speedups due to
714/// vectorization.
715/// In many cases vectorization is not profitable. This can happen because of
716/// a number of reasons. In this class we mainly attempt to predict the
717/// expected speedup/slowdowns due to the supported instruction set. We use the
718/// TargetTransformInfo to query the different backends for the cost of
719/// different operations.
722
723public:
730 std::function<BlockFrequencyInfo &()> GetBFI,
731 const Function *F, InterleavedAccessInfo &IAI,
732 VFSelectionContext &Config)
733 : Config(Config), EpilogueLoweringStatus(SEL), TheLoop(L), PSE(PSE),
734 LI(LI), Legal(Legal), TTI(TTI), TLI(TLI), AC(AC), ORE(ORE),
736
737 /// \return An upper bound for the vectorization factors (both fixed and
738 /// scalable). If the factors are 0, vectorization and interleaving should be
739 /// avoided up front.
740 FixedScalableVFPair computeMaxVF(ElementCount UserVF, unsigned UserIC);
741
742 /// Memory access instruction may be vectorized in more than one way.
743 /// Form of instruction after vectorization depends on cost.
744 /// This function takes cost-based decisions for Load/Store instructions
745 /// and collects them in a map. This decisions map is used for building
746 /// the lists of loop-uniform and loop-scalar instructions.
747 /// The calculated cost is saved with widening decision in order to
748 /// avoid redundant calculations.
749 void setCostBasedWideningDecision(ElementCount VF);
750
751 /// Collect values we want to ignore in the cost model.
752 void collectValuesToIgnore();
753
754 /// \returns True if it is more profitable to scalarize instruction \p I for
755 /// vectorization factor \p VF.
757 assert(VF.isVector() &&
758 "Profitable to scalarize relevant only for VF > 1.");
759 assert(
760 TheLoop->isInnermost() &&
761 "cost-model should not be used for outer loops (in VPlan-native path)");
762
763 auto Scalars = InstsToScalarize.find(VF);
764 assert(Scalars != InstsToScalarize.end() &&
765 "VF not yet analyzed for scalarization profitability");
766 return Scalars->second.contains(I);
767 }
768
769 /// Returns true if \p I is known to be uniform after vectorization.
771 assert(
772 TheLoop->isInnermost() &&
773 "cost-model should not be used for outer loops (in VPlan-native path)");
774
775 // If VF is scalar, then all instructions are trivially uniform.
776 if (VF.isScalar())
777 return true;
778
779 // Pseudo probes must be duplicated per vector lane so that the
780 // profiled loop trip count is not undercounted.
782 return false;
783
784 auto UniformsPerVF = Uniforms.find(VF);
785 assert(UniformsPerVF != Uniforms.end() &&
786 "VF not yet analyzed for uniformity");
787 return UniformsPerVF->second.count(I);
788 }
789
790 /// Returns true if \p I is known to be scalar after vectorization.
792 assert(
793 TheLoop->isInnermost() &&
794 "cost-model should not be used for outer loops (in VPlan-native path)");
795 if (VF.isScalar())
796 return true;
797
798 auto ScalarsPerVF = Scalars.find(VF);
799 assert(ScalarsPerVF != Scalars.end() &&
800 "Scalar values are not calculated for VF");
801 return ScalarsPerVF->second.count(I);
802 }
803
804 /// \returns True if instruction \p I can be truncated to a smaller bitwidth
805 /// for vectorization factor \p VF.
807 const auto &MinBWs = Config.getMinimalBitwidths();
808 // Truncs must truncate at most to their destination type.
809 if (isa_and_nonnull<TruncInst>(I) && MinBWs.contains(I) &&
810 I->getType()->getScalarSizeInBits() < MinBWs.lookup(I))
811 return false;
812 return VF.isVector() && MinBWs.contains(I) &&
815 }
816
817 /// Decision that was taken during cost calculation for memory instruction.
820 CM_Widen, // For consecutive accesses with stride +1.
821 CM_Widen_Reverse, // For consecutive accesses with stride -1.
825 /// A widening decision that has been invalidated after replacing the
826 /// corresponding recipe during VPlan transforms.
827 /// TODO: Remove once the legacy exit cost computation is retired.
829 };
830
831#ifndef NDEBUG
833 constexpr StringLiteral WideningStr[] = {
834 "Unknown", "Widen", "Widen_Reverse", "Interleave",
835 "GatherScatter", "Scalarize", "InvalidatedDecision"};
836 return WideningStr[W];
837 }
838#endif
839
840 /// Save vectorization decision \p W and \p Cost taken by the cost model for
841 /// instruction \p I and vector width \p VF.
844 assert(VF.isVector() && "Expected VF >=2");
845 LLVM_DEBUG(dbgs() << "LV: Setting widening decision to "
846 << getInstWideningStr(W) << " for VF " << VF
847 << " and instruction: " << *I << '\n');
848 WideningDecisions[{I, VF}] = {W, Cost};
849 }
850
851 /// Save vectorization decision \p W and \p Cost taken by the cost model for
852 /// interleaving group \p Grp and vector width \p VF.
856 assert(VF.isVector() && "Expected VF >=2");
857 /// Broadcast this decicion to all instructions inside the group.
858 /// When interleaving, the cost will only be assigned one instruction, the
859 /// insert position. For other cases, add the appropriate fraction of the
860 /// total cost to each instruction. This ensures accurate costs are used,
861 /// even if the insert position instruction is not used.
862 InstructionCost InsertPosCost = Cost;
863 InstructionCost OtherMemberCost = 0;
864 if (W != CM_Interleave)
865 OtherMemberCost = InsertPosCost = Cost / Grp->getNumMembers();
866 ;
867 for (auto *I : Grp->members()) {
868 LLVM_DEBUG(dbgs() << "LV: Setting widening decision to "
869 << getInstWideningStr(W) << " for VF " << VF
870 << " and instruction: " << *I << '\n');
871 if (Grp->getInsertPos() == I)
872 WideningDecisions[{I, VF}] = {W, InsertPosCost};
873 else
874 WideningDecisions[{I, VF}] = {W, OtherMemberCost};
875 }
876 }
877
878 /// Return the cost model decision for the given instruction \p I and vector
879 /// width \p VF. Return CM_Unknown if this instruction did not pass
880 /// through the cost modeling.
882 assert(VF.isVector() && "Expected VF to be a vector VF");
883 assert(
884 TheLoop->isInnermost() &&
885 "cost-model should not be used for outer loops (in VPlan-native path)");
886
887 std::pair<Instruction *, ElementCount> InstOnVF(I, VF);
888 auto Itr = WideningDecisions.find(InstOnVF);
889 if (Itr == WideningDecisions.end())
890 return CM_Unknown;
891 return Itr->second.first;
892 }
893
894 /// Return the vectorization cost for the given instruction \p I and vector
895 /// width \p VF.
897 assert(VF.isVector() && "Expected VF >=2");
898 std::pair<Instruction *, ElementCount> InstOnVF(I, VF);
899 assert(WideningDecisions.contains(InstOnVF) &&
900 "The cost is not calculated");
901 return WideningDecisions[InstOnVF].second;
902 }
903
904 /// Return True if instruction \p I is an optimizable truncate whose operand
905 /// is an induction variable. Such a truncate will be removed by adding a new
906 /// induction variable with the destination type.
908 // If the instruction is not a truncate, return false.
909 auto *Trunc = dyn_cast<TruncInst>(I);
910 if (!Trunc)
911 return false;
912
913 // Get the source and destination types of the truncate.
914 Type *SrcTy = toVectorTy(Trunc->getSrcTy(), VF);
915 Type *DestTy = toVectorTy(Trunc->getDestTy(), VF);
916
917 // If the truncate is free for the given types, return false. Replacing a
918 // free truncate with an induction variable would add an induction variable
919 // update instruction to each iteration of the loop. We exclude from this
920 // check the primary induction variable since it will need an update
921 // instruction regardless.
922 Value *Op = Trunc->getOperand(0);
923 if (Op != Legal->getPrimaryInduction() && TTI.isTruncateFree(SrcTy, DestTy))
924 return false;
925
926 // If the truncated value is not an induction variable, return false.
927 return Legal->isInductionPhi(Op);
928 }
929
930 /// Collects the instructions to scalarize for each predicated instruction in
931 /// the loop.
932 void collectInstsToScalarize(ElementCount VF);
933
934 /// Collect values that will not be widened, including Uniforms, Scalars, and
935 /// Instructions to Scalarize for the given \p VF.
936 /// The sets depend on CM decision for Load/Store instructions
937 /// that may be vectorized as interleave, gather-scatter or scalarized.
938 /// Also make a decision on what to do about call instructions in the loop
939 /// at that VF -- scalarize, call a known vector routine, or call a
940 /// vector intrinsic.
942 // Do the analysis once.
943 if (VF.isScalar() || Uniforms.contains(VF))
944 return;
946 collectLoopUniforms(VF);
947 collectLoopScalars(VF);
949 }
950
951 /// Given costs for both strategies, return true if the scalar predication
952 /// lowering should be used for div/rem. This incorporates an override
953 /// option so it is not simply a cost comparison.
955 InstructionCost MaskedCost) const {
956 switch (ForceMaskedDivRem) {
958 return ScalarCost < MaskedCost;
960 return false;
962 return true;
963 }
964 llvm_unreachable("impossible case value");
965 }
966
967 /// Returns true if \p I is an instruction which requires predication and
968 /// for which our chosen predication strategy is scalarization (i.e. we
969 /// don't have an alternate strategy such as masking available).
970 /// \p VF is the vectorization factor that will be used to vectorize \p I.
971 bool isScalarWithPredication(Instruction *I, ElementCount VF);
972
973 /// Wrapper function for LoopVectorizationLegality::isMaskRequired,
974 /// that passes the Instruction \p I and if we fold tail.
975 bool isMaskRequired(Instruction *I) const;
976
977 /// Returns true if \p I is an instruction that needs to be predicated
978 /// at runtime. The result is independent of the predication mechanism.
979 /// Superset of instructions that return true for isScalarWithPredication.
980 bool isPredicatedInst(Instruction *I) const;
981
982 /// A helper function that returns how much we should divide the cost of a
983 /// predicated block by. Typically this is the reciprocal of the block
984 /// probability, i.e. if we return X we are assuming the predicated block will
985 /// execute once for every X iterations of the loop header so the block should
986 /// only contribute 1/X of its cost to the total cost calculation, but when
987 /// optimizing for code size it will just be 1 as code size costs don't depend
988 /// on execution probabilities.
989 ///
990 /// Note that if a block wasn't originally predicated but was predicated due
991 /// to tail folding, the divisor will still be 1 because it will execute for
992 /// every iteration of the loop header.
993 inline uint64_t
994 getPredBlockCostDivisor(TargetTransformInfo::TargetCostKind CostKind,
995 const BasicBlock *BB);
996
997 /// Returns true if an artificially high cost for emulated masked memrefs
998 /// should be used.
999 bool useEmulatedMaskMemRefHack(Instruction *I, ElementCount VF) const;
1000
1001 /// Return the costs for our two available strategies for lowering a
1002 /// div/rem operation which requires speculating at least one lane.
1003 /// First result is for scalarization (will be invalid for scalable
1004 /// vectors); second is for the masked intrinsic strategy.
1005 std::pair<InstructionCost, InstructionCost>
1006 getDivRemSpeculationCost(Instruction *I, ElementCount VF);
1007
1008 /// If \p I is a memory instruction with a consecutive pointer that can be
1009 /// widened, returns the widening kind (CM_Widen or CM_Widen_Reverse) and
1010 /// std::nullopt otherwise.
1011 std::optional<InstWidening> memoryInstructionCanBeWidened(Instruction *I,
1012 ElementCount VF);
1013
1014 /// Returns true if \p I is a memory instruction in an interleaved-group
1015 /// of memory accesses that can be vectorized with wide vector loads/stores
1016 /// and shuffles.
1017 bool interleavedAccessCanBeWidened(Instruction *I, ElementCount VF) const;
1018
1019 /// Returns true if the target machine supports masked loads or stores
1020 /// for \p I's data type and alignment. The caller must ensure the access is
1021 /// consecutive or part of an interleave group.
1022 bool isLegalMaskedLoadOrStore(Instruction *I, ElementCount VF) const;
1023
1024 /// Returns true if the target machine supports gather or scatter for \p I's
1025 /// data type and alignment.
1026 bool isLegalGatherOrScatter(Instruction *I, ElementCount VF) const;
1027
1028 /// Check if \p Instr belongs to any interleaved access group.
1030 return InterleaveInfo.isInterleaved(Instr);
1031 }
1032
1033 /// Get the interleaved access group that \p Instr belongs to.
1036 return InterleaveInfo.getInterleaveGroup(Instr);
1037 }
1038
1039 /// Returns true if we're required to use a scalar epilogue for at least
1040 /// the final iteration of the original loop.
1041 bool requiresScalarEpilogue(bool IsVectorizing) const {
1042 if (!isEpilogueAllowed()) {
1043 LLVM_DEBUG(dbgs() << "LV: Loop does not require scalar epilogue\n");
1044 return false;
1045 }
1046 // If we might exit from anywhere but the latch and early exit vectorization
1047 // is disabled, we must run the exiting iteration in scalar form.
1048 if (TheLoop->getExitingBlock() != TheLoop->getLoopLatch() &&
1049 !(EnableEarlyExitVectorization && Legal->hasUncountableEarlyExit())) {
1050 LLVM_DEBUG(dbgs() << "LV: Loop requires scalar epilogue: not exiting "
1051 "from latch block\n");
1052 return true;
1053 }
1054 if (IsVectorizing && InterleaveInfo.requiresScalarEpilogue()) {
1055 LLVM_DEBUG(dbgs() << "LV: Loop requires scalar epilogue: "
1056 "interleaved group requires scalar epilogue\n");
1057 return true;
1058 }
1059 LLVM_DEBUG(dbgs() << "LV: Loop does not require scalar epilogue\n");
1060 return false;
1061 }
1062
1063 /// Returns true if an epilogue is allowed (e.g., not prevented by
1064 /// optsize or a loop hint annotation).
1065 bool isEpilogueAllowed() const {
1066 return EpilogueLoweringStatus == CM_EpilogueAllowed;
1067 }
1068
1069 /// Returns the TailFoldingStyle that is best for the current loop.
1071 return ChosenTailFoldingStyle;
1072 }
1073
1074 /// Selects and saves TailFoldingStyle.
1075 /// \param IsScalableVF true if scalable vector factors enabled.
1076 /// \param UserIC User specific interleave count.
1077 void setTailFoldingStyle(bool IsScalableVF, unsigned UserIC) {
1078 assert(ChosenTailFoldingStyle == TailFoldingStyle::None &&
1079 "Tail folding must not be selected yet.");
1080 if (!Legal->canFoldTailByMasking()) {
1081 ChosenTailFoldingStyle = TailFoldingStyle::None;
1082 return;
1083 }
1084
1085 // Default to TTI preference, but allow command line override.
1086 ChosenTailFoldingStyle = TTI.getPreferredTailFoldingStyle();
1087 if (ForceTailFoldingStyle.getNumOccurrences())
1088 ChosenTailFoldingStyle = ForceTailFoldingStyle.getValue();
1089
1090 if (ChosenTailFoldingStyle != TailFoldingStyle::DataWithEVL)
1091 return;
1092 // Override EVL styles if needed.
1093 // FIXME: Investigate opportunity for fixed vector factor.
1094 bool EVLIsLegal = UserIC <= 1 && IsScalableVF &&
1095 TTI.hasActiveVectorLength() && !EnableVPlanNativePath;
1096 if (EVLIsLegal)
1097 return;
1098 // If for some reason EVL mode is unsupported, fallback to an epilogue
1099 // if it's allowed, or DataWithoutLaneMask otherwise.
1100 if (EpilogueLoweringStatus == CM_EpilogueAllowed ||
1101 EpilogueLoweringStatus == CM_EpilogueNotNeededFoldTail)
1102 ChosenTailFoldingStyle = TailFoldingStyle::None;
1103 else
1104 ChosenTailFoldingStyle = TailFoldingStyle::DataWithoutLaneMask;
1105
1106 LLVM_DEBUG(
1107 dbgs() << "LV: Preference for VP intrinsics indicated. Will "
1108 "not try to generate VP Intrinsics "
1109 << (UserIC > 1
1110 ? "since interleave count specified is greater than 1.\n"
1111 : "due to non-interleaving reasons.\n"));
1112 }
1113
1114 /// Returns true if all loop blocks should be masked to fold tail loop.
1115 bool foldTailByMasking() const {
1117 }
1118
1120 assert(foldTailByMasking() && "Expected tail folding to be enabled!");
1122 "Did not expect to enable alias masking with EVL!");
1123 assert(PartialAliasMaskingStatus == AliasMaskingStatus::NotDecided);
1124
1125 // Assume we fail to enable alias masking (in case we early exit).
1126 PartialAliasMaskingStatus = AliasMaskingStatus::Disabled;
1127
1128 // Note: FixedOrderRecurrences are not supported yet as we cannot handle
1129 // the required `splice.right` with the alias-mask.
1131 !Legal->getFixedOrderRecurrences().empty())
1132 return;
1133
1134 const RuntimePointerChecking *Checks = Legal->getRuntimePointerChecking();
1135 if (!Checks)
1136 return;
1137
1138 auto DiffChecks = Checks->getDiffChecks();
1139 if (!DiffChecks || DiffChecks->empty())
1140 return;
1141
1142 [[maybe_unused]] auto HasPointerArgs = [](CallBase *CB) {
1143 return any_of(CB->args(), [](Value const *Arg) {
1144 return Arg->getType()->isPointerTy();
1145 });
1146 };
1147
1148 for (BasicBlock *BB : TheLoop->blocks()) {
1149 for (Instruction &I : *BB) {
1151 [[maybe_unused]] auto *Call = dyn_cast<CallInst>(&I);
1152 assert(
1153 (!I.mayReadOrWriteMemory() || (Call && !HasPointerArgs(Call))) &&
1154 "Skipped unexpected memory access");
1155 continue;
1156 }
1157
1158 Type *ScalarTy = getLoadStoreType(&I);
1160
1161 // Currently, we can't handle alias masking in reverse. Reversing the
1162 // alias mask is not correct (or necessary). When combined with
1163 // tail-folding the active lane mask should only be reversed where the
1164 // alias-mask is true.
1165 if (Legal->isConsecutivePtr(ScalarTy, Ptr) == -1)
1166 return;
1167 }
1168 }
1169
1170 PartialAliasMaskingStatus = AliasMaskingStatus::Enabled;
1171 }
1172
1173 /// Returns true if all loop blocks should have partial aliases masked.
1174 bool maskPartialAliasing() const {
1175 return PartialAliasMaskingStatus == AliasMaskingStatus::Enabled;
1176 }
1177
1178 /// Returns true if the instructions in this block requires predication
1179 /// for any reason, e.g. because tail folding now requires a predicate
1180 /// or because the block in the original loop was predicated.
1182 return foldTailByMasking() || Legal->blockNeedsPredication(BB);
1183 }
1184
1185 /// Returns true if VP intrinsics with explicit vector length support should
1186 /// be generated in the tail folded loop.
1190
1191 /// Returns true if the predicated reduction select should be used to set the
1192 /// incoming value for the reduction phi.
1193 bool usePredicatedReductionSelect(RecurKind RecurrenceKind) const {
1194 // Force to use predicated reduction select since the EVL of the
1195 // second-to-last iteration might not be VF*UF.
1196 if (foldTailWithEVL())
1197 return true;
1198
1199 // Force a predicated select with alias-masking to avoid propagating poison
1200 // values to the header phi for lanes outside the alias-mask.
1201 if (maskPartialAliasing())
1202 return true;
1203
1204 // Note: For FindLast recurrences we prefer a predicated select to simplify
1205 // matching in handleFindLastReductions(), rather than handle multiple
1206 // cases.
1208 return true;
1209
1211 TTI.preferPredicatedReductionSelect();
1212 }
1213
1214 /// Estimate cost of an intrinsic call instruction CI if it were vectorized
1215 /// with factor VF. Return the cost of the instruction, including
1216 /// scalarization overhead if it's needed.
1217 InstructionCost getVectorIntrinsicCost(CallInst *CI, ElementCount VF) const;
1218
1219 /// Estimate cost of a call instruction CI if it were vectorized with factor
1220 /// VF. Return the cost of the instruction, including scalarization overhead
1221 /// if it's needed.
1222 InstructionCost getVectorCallCost(CallInst *CI, ElementCount VF) const;
1223
1224 /// Invalidates decisions already taken by the cost model.
1226 WideningDecisions.clear();
1227 Uniforms.clear();
1228 Scalars.clear();
1229 }
1230
1231 /// Returns the expected execution cost. The unit of the cost does
1232 /// not matter because we use the 'cost' units to compare different
1233 /// vector widths. The cost that is returned is *not* normalized by
1234 /// the factor width.
1235 InstructionCost expectedCost(ElementCount VF);
1236
1237 /// Returns the execution time cost of an instruction for a given vector
1238 /// width. Vector width of one means scalar.
1239 InstructionCost getInstructionCost(Instruction *I, ElementCount VF);
1240
1241 /// Returns true if \p Op should be considered invariant and if it is
1242 /// trivially hoistable.
1243 bool shouldConsiderInvariant(Value *Op);
1244
1245 /// Returns true if \p I has been forced to be scalarized at \p VF.
1247 auto FS = ForcedScalars.find(VF);
1248 return FS != ForcedScalars.end() && FS->second.contains(I);
1249 }
1250
1251private:
1252 unsigned NumPredStores = 0;
1253
1254 /// VF selection state independent of cost-modeling decisions.
1255 VFSelectionContext &Config;
1256
1257 /// Wrapper around LoopVectorizationLegality::isUniform() that takes into
1258 /// account if alias-masking is enabled. We consider the VF to be unknown when
1259 /// alias masking.
1260 bool isUniform(Value *V, ElementCount VF) const {
1261 // With alias-masking our runtime VF is [2, VF] (and not necessarily a
1262 // power-of-two). Something that is uniform for VF may not be for the full
1263 // range.
1264 assert(PartialAliasMaskingStatus != AliasMaskingStatus::NotDecided &&
1265 "alias-mask status must be decided already");
1266 return Legal->isUniform(V, PartialAliasMaskingStatus ==
1268 ? std::optional(VF)
1269 : std::nullopt);
1270 }
1271
1272 /// Wrapper around LoopVectorizationLegality::isUniformMemOp() that takes into
1273 /// account if alias-masking is enabled. We consider the VF to be unknown when
1274 /// alias masking.
1275 bool isUniformMemOp(Instruction &I, ElementCount VF) const {
1276 assert(PartialAliasMaskingStatus != AliasMaskingStatus::NotDecided &&
1277 "alias-mask status must be decided already");
1278 return Legal->isUniformMemOp(I, PartialAliasMaskingStatus ==
1280 ? std::optional(VF)
1281 : std::nullopt);
1282 }
1283
1284 /// Calculate vectorization cost of memory instruction \p I.
1285 InstructionCost getMemoryInstructionCost(Instruction *I, ElementCount VF);
1286
1287 /// The cost computation for scalarized memory instruction.
1288 InstructionCost getMemInstScalarizationCost(Instruction *I, ElementCount VF);
1289
1290 /// The cost computation for interleaving group of memory instructions.
1291 InstructionCost getInterleaveGroupCost(Instruction *I, ElementCount VF) const;
1292
1293 /// The cost computation for Gather/Scatter instruction.
1294 InstructionCost getGatherScatterCost(Instruction *I, ElementCount VF) const;
1295
1296 /// The cost computation for widening instruction \p I with consecutive
1297 /// memory access.
1298 InstructionCost getConsecutiveMemOpCost(Instruction *I, ElementCount VF,
1299 InstWidening Kind);
1300
1301 /// The cost calculation for Load/Store instruction \p I with uniform pointer -
1302 /// Load: scalar load + broadcast.
1303 /// Store: scalar store + (loop invariant value stored? 0 : extract of last
1304 /// element)
1305 InstructionCost getUniformMemOpCost(Instruction *I, ElementCount VF) const;
1306
1307 /// Estimate the overhead of scalarizing an instruction. This is a
1308 /// convenience wrapper for the type-based getScalarizationOverhead API.
1310 ElementCount VF) const;
1311
1312 /// A type representing the costs for instructions if they were to be
1313 /// scalarized rather than vectorized. The entries are Instruction-Cost
1314 /// pairs.
1315 using ScalarCostsTy = MapVector<Instruction *, InstructionCost>;
1316
1317 /// A set containing all BasicBlocks that are known to present after
1318 /// vectorization as a predicated block.
1319 DenseMap<ElementCount, SmallPtrSet<BasicBlock *, 4>>
1320 PredicatedBBsAfterVectorization;
1321
1322 /// Records whether it is allowed to have the original scalar loop execute at
1323 /// least once. This may be needed as a fallback loop in case runtime
1324 /// aliasing/dependence checks fail, or to handle the tail/remainder
1325 /// iterations when the trip count is unknown or doesn't divide by the VF,
1326 /// or as a peel-loop to handle gaps in interleave-groups.
1327 /// Under optsize and when the trip count is very small we don't allow any
1328 /// iterations to execute in the scalar loop.
1329 EpilogueLowering EpilogueLoweringStatus = CM_EpilogueAllowed;
1330
1331 /// Control finally chosen tail folding style.
1332 TailFoldingStyle ChosenTailFoldingStyle = TailFoldingStyle::None;
1333
1334 /// If partial alias masking is enabled/disabled or not decided.
1335 AliasMaskingStatus PartialAliasMaskingStatus = AliasMaskingStatus::NotDecided;
1336
1337 /// A map holding scalar costs for different vectorization factors. The
1338 /// presence of a cost for an instruction in the mapping indicates that the
1339 /// instruction will be scalarized when vectorizing with the associated
1340 /// vectorization factor. The entries are VF-ScalarCostTy pairs.
1341 MapVector<ElementCount, ScalarCostsTy> InstsToScalarize;
1342
1343 /// Holds the instructions known to be uniform after vectorization.
1344 /// The data is collected per VF.
1345 DenseMap<ElementCount, SmallPtrSet<Instruction *, 4>> Uniforms;
1346
1347 /// Holds the instructions known to be scalar after vectorization.
1348 /// The data is collected per VF.
1349 DenseMap<ElementCount, SmallPtrSet<Instruction *, 4>> Scalars;
1350
1351 /// Holds the instructions (address computations) that are forced to be
1352 /// scalarized.
1353 DenseMap<ElementCount, SmallSetVector<Instruction *, 4>> ForcedScalars;
1354
1355 /// Returns the expected difference in cost from scalarizing the expression
1356 /// feeding a predicated instruction \p PredInst. The instructions to
1357 /// scalarize and their scalar costs are collected in \p ScalarCosts. A
1358 /// non-negative return value implies the expression will be scalarized.
1359 /// Currently, only single-use chains are considered for scalarization.
1360 InstructionCost computePredInstDiscount(Instruction *PredInst,
1361 ScalarCostsTy &ScalarCosts,
1362 ElementCount VF);
1363
1364 /// Collect the instructions that are uniform after vectorization. An
1365 /// instruction is uniform if we represent it with a single scalar value in
1366 /// the vectorized loop corresponding to each vector iteration. Examples of
1367 /// uniform instructions include pointer operands of consecutive or
1368 /// interleaved memory accesses. Note that although uniformity implies an
1369 /// instruction will be scalar, the reverse is not true. In general, a
1370 /// scalarized instruction will be represented by VF scalar values in the
1371 /// vectorized loop, each corresponding to an iteration of the original
1372 /// scalar loop.
1373 void collectLoopUniforms(ElementCount VF);
1374
1375 /// Collect the instructions that are scalar after vectorization. An
1376 /// instruction is scalar if it is known to be uniform or will be scalarized
1377 /// during vectorization. collectLoopScalars should only add non-uniform nodes
1378 /// to the list if they are used by a load/store instruction that is marked as
1379 /// CM_Scalarize. Non-uniform scalarized instructions will be represented by
1380 /// VF values in the vectorized loop, each corresponding to an iteration of
1381 /// the original scalar loop.
1382 void collectLoopScalars(ElementCount VF);
1383
1384 /// Keeps cost model vectorization decision and cost for instructions.
1385 /// Right now it is used for memory instructions only.
1386 using DecisionList = DenseMap<std::pair<Instruction *, ElementCount>,
1387 std::pair<InstWidening, InstructionCost>>;
1388
1389 DecisionList WideningDecisions;
1390
1391 /// Returns true if \p V is expected to be vectorized and it needs to be
1392 /// extracted.
1393 bool needsExtract(Value *V, ElementCount VF) const {
1395 if (VF.isScalar() || !I || !TheLoop->contains(I) ||
1396 TheLoop->isLoopInvariant(I) ||
1397 getWideningDecision(I, VF) == CM_Scalarize)
1398 return false;
1399
1400 // Assume we can vectorize V (and hence we need extraction) if the
1401 // scalars are not computed yet. This can happen, because it is called
1402 // via getScalarizationOverhead from setCostBasedWideningDecision, before
1403 // the scalars are collected. That should be a safe assumption in most
1404 // cases, because we check if the operands have vectorizable types
1405 // beforehand in LoopVectorizationLegality.
1406 return !Scalars.contains(VF) || !isScalarAfterVectorization(I, VF);
1407 };
1408
1409 /// Returns a range containing only operands needing to be extracted.
1410 SmallVector<Value *, 4> filterExtractingOperands(Instruction::op_range Ops,
1411 ElementCount VF) const {
1412
1413 SmallPtrSet<const Value *, 4> UniqueOperands;
1414 SmallVector<Value *, 4> Res;
1415 for (Value *Op : Ops) {
1416 if (isa<Constant>(Op) || !UniqueOperands.insert(Op).second ||
1417 !needsExtract(Op, VF))
1418 continue;
1419 Res.push_back(Op);
1420 }
1421 return Res;
1422 }
1423
1424public:
1425 /// The loop that we evaluate.
1427
1428 /// Predicated scalar evolution analysis.
1430
1431 /// Loop Info analysis.
1433
1434 /// Vectorization legality.
1436
1437 /// Vector target information.
1439
1440 /// Target Library Info.
1442
1443 /// Assumption cache.
1445
1446 /// Interface to emit optimization remarks.
1448
1449 /// A function to lazily fetch BlockFrequencyInfo. This avoids computing it
1450 /// unless necessary, e.g. when the loop isn't legal to vectorize or when
1451 /// there is no predication.
1452 std::function<BlockFrequencyInfo &()> GetBFI;
1453 /// The BlockFrequencyInfo returned from GetBFI.
1455 /// Returns the BlockFrequencyInfo for the function if cached, otherwise
1456 /// fetches it via GetBFI. Avoids an indirect call to the std::function.
1458 if (!BFI)
1459 BFI = &GetBFI();
1460 return *BFI;
1461 }
1462
1464
1465 /// The interleave access information contains groups of interleaved accesses
1466 /// with the same stride and close to each other.
1468
1469 /// Values to ignore in the cost model.
1471
1472 /// Values to ignore in the cost model when VF > 1.
1474};
1475} // end namespace llvm
1476
1477namespace {
1478/// Helper struct to manage generating runtime checks for vectorization.
1479///
1480/// The runtime checks are created up-front in temporary blocks to allow better
1481/// estimating the cost and un-linked from the existing IR. After deciding to
1482/// vectorize, the checks are attached to VPlan as IR or recipes. If deciding
1483/// not to vectorize, the temporary blocks are completely removed.
1484class GeneratedRTChecks {
1485 /// Basic block which contains the generated SCEV checks, if any.
1486 BasicBlock *SCEVCheckBlock = nullptr;
1487
1488 /// The value representing the result of the generated SCEV checks. If it is
1489 /// nullptr no SCEV checks have been generated.
1490 Value *SCEVCheckCond = nullptr;
1491
1492 /// Basic block which contains the generated memory runtime checks, if any.
1493 BasicBlock *MemCheckBlock = nullptr;
1494
1495 /// The value representing the result of the generated memory runtime checks.
1496 /// If it is nullptr no memory runtime checks have been generated.
1497 Value *MemRuntimeCheckCond = nullptr;
1498
1499 /// Whether checks were generated, retained after their IR is replaced or
1500 /// removed during VPlan execution.
1501 bool HasChecks = false;
1502
1503 DominatorTree *DT;
1504 LoopInfo *LI;
1506
1507 SCEVExpander SCEVExp;
1508 SCEVExpander MemCheckExp;
1509
1510 bool CostTooHigh = false;
1511
1512 Loop *OuterLoop = nullptr;
1513
1515
1516 /// The kind of cost that we are calculating
1518
1519 /// True if the loop is alias-masked (which allows us to omit diff checks).
1520 bool LoopUsesPartialAliasMasking = false;
1521
1522public:
1523 GeneratedRTChecks(PredicatedScalarEvolution &PSE, DominatorTree *DT,
1526 bool LoopUsesPartialAliasMasking)
1527 : DT(DT), LI(LI), TTI(TTI),
1528 SCEVExp(*PSE.getSE(), "scev.check", /*PreserveLCSSA=*/false),
1529 MemCheckExp(*PSE.getSE(), "scev.check", /*PreserveLCSSA=*/false),
1530 PSE(PSE), CostKind(CostKind),
1531 LoopUsesPartialAliasMasking(LoopUsesPartialAliasMasking) {}
1532
1533 /// Generate runtime checks in SCEVCheckBlock and MemCheckBlock, so we can
1534 /// accurately estimate the cost of the runtime checks. The blocks are
1535 /// un-linked from the IR and attached to VPlan as IR or recipes if
1536 /// profitable. Otherwise, the check blocks are removed completely.
1537 void create(Loop *L, const LoopAccessInfo &LAI,
1538 const SCEVPredicate &UnionPred, ElementCount VF, unsigned IC,
1539 OptimizationRemarkEmitter &ORE) {
1540
1541 // Hard cutoff to limit compile-time increase in case a very large number of
1542 // runtime checks needs to be generated.
1543 // TODO: Skip cutoff if the loop is guaranteed to execute, e.g. due to
1544 // profile info.
1545 CostTooHigh = LAI.getNumRuntimePointerChecks() >
1547 if (CostTooHigh) {
1548 // Mark runtime checks as never succeeding when they exceed the threshold.
1549 MemRuntimeCheckCond = ConstantInt::getTrue(L->getHeader()->getContext());
1550 SCEVCheckCond = ConstantInt::getTrue(L->getHeader()->getContext());
1551 ORE.emit([&]() {
1552 return OptimizationRemarkAnalysisAliasing(
1553 DEBUG_TYPE, "TooManyMemoryRuntimeChecks", L->getStartLoc(),
1554 L->getHeader())
1555 << "loop not vectorized: too many memory checks needed";
1556 });
1557 LLVM_DEBUG(dbgs() << "LV: Too many memory checks needed.\n");
1558 return;
1559 }
1560
1561 BasicBlock *LoopHeader = L->getHeader();
1562 BasicBlock *Preheader = L->getLoopPreheader();
1563
1564 // Use SplitBlock to create blocks for SCEV & memory runtime checks to
1565 // ensure the blocks are properly added to LoopInfo & DominatorTree. Those
1566 // may be used by SCEVExpander. The blocks will be un-linked from their
1567 // predecessors and removed from LI & DT at the end of the function.
1568 if (!UnionPred.isAlwaysTrue()) {
1569 SCEVCheckBlock = SplitBlock(Preheader, Preheader->getTerminator(), DT, LI,
1570 nullptr, "vector.scevcheck");
1571
1572 SCEVCheckCond = SCEVExp.expandCodeForPredicate(
1573 &UnionPred, SCEVCheckBlock->getTerminator());
1574 if (isa<Constant>(SCEVCheckCond)) {
1575 // Clean up directly after expanding the predicate to a constant, to
1576 // avoid further expansions re-using anything left over from SCEVExp.
1577 SCEVExpanderCleaner SCEVCleaner(SCEVExp);
1578 SCEVCleaner.cleanup();
1579 }
1580 }
1581
1582 const auto &RtPtrChecking = *LAI.getRuntimePointerChecking();
1583 // TODO: We need to estimate the cost of alias-masking in
1584 // GeneratedRTChecks::getCost(). We can't check the MemCheckBlock as the
1585 // alias-mask is generated later in VPlan.
1586 if (RtPtrChecking.Need && !LoopUsesPartialAliasMasking) {
1587 auto *Pred = SCEVCheckBlock ? SCEVCheckBlock : Preheader;
1588 MemCheckBlock = SplitBlock(Pred, Pred->getTerminator(), DT, LI, nullptr,
1589 "vector.memcheck");
1590
1591 auto DiffChecks = RtPtrChecking.getDiffChecks();
1592 if (DiffChecks) {
1593 MemRuntimeCheckCond = addDiffRuntimeChecks(
1594 MemCheckBlock->getTerminator(), *DiffChecks, MemCheckExp, VF, IC);
1595 } else {
1596 MemRuntimeCheckCond = addRuntimeChecks(
1597 MemCheckBlock->getTerminator(), L, RtPtrChecking.getChecks(),
1599 }
1600 assert(MemRuntimeCheckCond &&
1601 "no RT checks generated although RtPtrChecking "
1602 "claimed checks are required");
1603 }
1604
1605 SCEVExp.eraseDeadInstructions(SCEVCheckCond);
1606 HasChecks = getSCEVChecks().first || getMemRuntimeChecks().first;
1607
1608 if (!MemCheckBlock && !SCEVCheckBlock)
1609 return;
1610
1611 // Unhook the temporary block with the checks, update various places
1612 // accordingly.
1613 if (SCEVCheckBlock)
1614 SCEVCheckBlock->replaceAllUsesWith(Preheader);
1615 if (MemCheckBlock)
1616 MemCheckBlock->replaceAllUsesWith(Preheader);
1617
1618 if (SCEVCheckBlock) {
1619 SCEVCheckBlock->getTerminator()->moveBefore(
1620 Preheader->getTerminator()->getIterator());
1621 auto *UI = new UnreachableInst(Preheader->getContext(), SCEVCheckBlock);
1622 UI->setDebugLoc(DebugLoc::getTemporary());
1623 Preheader->getTerminator()->eraseFromParent();
1624 }
1625 if (MemCheckBlock) {
1626 MemCheckBlock->getTerminator()->moveBefore(
1627 Preheader->getTerminator()->getIterator());
1628 auto *UI = new UnreachableInst(Preheader->getContext(), MemCheckBlock);
1629 UI->setDebugLoc(DebugLoc::getTemporary());
1630 Preheader->getTerminator()->eraseFromParent();
1631 }
1632
1633 DT->changeImmediateDominator(LoopHeader, Preheader);
1634 if (MemCheckBlock) {
1635 DT->eraseNode(MemCheckBlock);
1636 LI->removeBlock(MemCheckBlock);
1637 }
1638 if (SCEVCheckBlock) {
1639 DT->eraseNode(SCEVCheckBlock);
1640 LI->removeBlock(SCEVCheckBlock);
1641 }
1642
1643 // Outer loop is used as part of the later cost calculations.
1644 OuterLoop = L->getParentLoop();
1645 }
1646
1648 if (SCEVCheckBlock || MemCheckBlock)
1649 LLVM_DEBUG(dbgs() << "Calculating cost of runtime checks:\n");
1650
1651 if (CostTooHigh) {
1653 Cost.setInvalid();
1654 LLVM_DEBUG(dbgs() << " number of checks exceeded threshold\n");
1655 return Cost;
1656 }
1657
1658 InstructionCost RTCheckCost = 0;
1659 if (SCEVCheckBlock)
1660 for (Instruction &I : *SCEVCheckBlock) {
1661 if (SCEVCheckBlock->getTerminator() == &I)
1662 continue;
1664 LLVM_DEBUG(dbgs() << " " << C << " for " << I << "\n");
1665 RTCheckCost += C;
1666 }
1667 if (MemCheckBlock) {
1668 InstructionCost MemCheckCost = 0;
1669 for (Instruction &I : *MemCheckBlock) {
1670 if (MemCheckBlock->getTerminator() == &I)
1671 continue;
1673 LLVM_DEBUG(dbgs() << " " << C << " for " << I << "\n");
1674 MemCheckCost += C;
1675 }
1676
1677 // If the runtime memory checks are being created inside an outer loop
1678 // we should find out if these checks are outer loop invariant. If so,
1679 // the checks will likely be hoisted out and so the effective cost will
1680 // reduce according to the outer loop trip count.
1681 if (OuterLoop) {
1682 ScalarEvolution *SE = MemCheckExp.getSE();
1683 // TODO: If profitable, we could refine this further by analysing every
1684 // individual memory check, since there could be a mixture of loop
1685 // variant and invariant checks that mean the final condition is
1686 // variant.
1687 const SCEV *Cond = SE->getSCEV(MemRuntimeCheckCond);
1688 if (SE->isLoopInvariant(Cond, OuterLoop)) {
1689 // It seems reasonable to assume that we can reduce the effective
1690 // cost of the checks even when we know nothing about the trip
1691 // count. Assume that the outer loop executes at least twice.
1692 unsigned BestTripCount = 2;
1693
1694 // Get the best known TC estimate.
1695 if (auto EstimatedTC = getSmallBestKnownTC(
1696 PSE, OuterLoop, /* CanUseConstantMax = */ false))
1697 if (EstimatedTC->isFixed())
1698 BestTripCount = EstimatedTC->getFixedValue();
1699
1700 InstructionCost NewMemCheckCost = MemCheckCost / BestTripCount;
1701
1702 // Let's ensure the cost is always at least 1.
1703 NewMemCheckCost = std::max(NewMemCheckCost.getValue(),
1704 (InstructionCost::CostType)1);
1705
1706 if (BestTripCount > 1)
1708 << "We expect runtime memory checks to be hoisted "
1709 << "out of the outer loop. Cost reduced from "
1710 << MemCheckCost << " to " << NewMemCheckCost << '\n');
1711
1712 MemCheckCost = NewMemCheckCost;
1713 }
1714 }
1715
1716 RTCheckCost += MemCheckCost;
1717 }
1718
1719 if (SCEVCheckBlock || MemCheckBlock)
1720 LLVM_DEBUG(dbgs() << "Total cost of runtime checks: " << RTCheckCost
1721 << "\n");
1722
1723 return RTCheckCost;
1724 }
1725
1726 /// Remove the created SCEV & memory runtime check blocks & instructions, if
1727 /// unused.
1728 ~GeneratedRTChecks() {
1729 SCEVExpanderCleaner SCEVCleaner(SCEVExp);
1730 bool SCEVChecksUsed = !SCEVCheckBlock || !pred_empty(SCEVCheckBlock);
1731 if (SCEVChecksUsed)
1732 SCEVCleaner.markResultUsed();
1733
1734 if (MemCheckBlock && pred_empty(MemCheckBlock))
1735 eraseMemCheckBlock();
1736
1737 SCEVCleaner.cleanup();
1738
1739 if (!SCEVChecksUsed)
1740 SCEVCheckBlock->eraseFromParent();
1741 }
1742
1743 /// Retrieves the SCEVCheckCond and SCEVCheckBlock that were generated as IR
1744 /// outside VPlan.
1745 std::pair<Value *, BasicBlock *> getSCEVChecks() const {
1746 using namespace llvm::PatternMatch;
1747 if (!SCEVCheckCond || match(SCEVCheckCond, m_ZeroInt()))
1748 return {nullptr, nullptr};
1749
1750 return {SCEVCheckCond, SCEVCheckBlock};
1751 }
1752
1753 /// Retrieves the MemCheckCond and MemCheckBlock that were generated as IR
1754 /// outside VPlan.
1755 std::pair<Value *, BasicBlock *> getMemRuntimeChecks() const {
1756 using namespace llvm::PatternMatch;
1757 if (MemRuntimeCheckCond && match(MemRuntimeCheckCond, m_ZeroInt()))
1758 return {nullptr, nullptr};
1759 return {MemRuntimeCheckCond, MemCheckBlock};
1760 }
1761
1762 /// Return true if any runtime checks have been added
1763 bool hasChecks() const { return HasChecks; }
1764
1765 /// Erase the memory check block, its instructions and their SCEV expansions.
1766 void eraseMemCheckBlock() {
1767 SCEVExpanderCleaner MemCheckCleaner(MemCheckExp);
1768 auto &SE = *MemCheckExp.getSE();
1769 // Memory runtime check generation creates compares that use expanded
1770 // values. Remove them before running the SCEVExpanderCleaner.
1771 for (auto &I : make_early_inc_range(reverse(*MemCheckBlock))) {
1772 if (MemCheckExp.isInsertedInstruction(&I))
1773 continue;
1774 SE.forgetValue(&I);
1775 I.eraseFromParent();
1776 }
1777 MemCheckCleaner.cleanup();
1778 MemCheckBlock->eraseFromParent();
1779 MemCheckBlock = nullptr;
1780 MemRuntimeCheckCond = nullptr;
1781 }
1782};
1783} // namespace
1784
1786 return Style == TailFoldingStyle::Data ||
1788}
1789
1793
1794// Return true if \p OuterLp is an outer loop annotated with hints for explicit
1795// vectorization. The loop needs to be annotated with #pragma omp simd
1796// simdlen(#) or #pragma clang vectorize(enable) vectorize_width(#). If the
1797// vector length information is not provided, vectorization is not considered
1798// explicit. Interleave hints are not allowed either. These limitations will be
1799// relaxed in the future.
1800// Please, note that we are currently forced to abuse the pragma 'clang
1801// vectorize' semantics. This pragma provides *auto-vectorization hints*
1802// (i.e., LV must check that vectorization is legal) whereas pragma 'omp simd'
1803// provides *explicit vectorization hints* (LV can bypass legal checks and
1804// assume that vectorization is legal). However, both hints are implemented
1805// using the same metadata (llvm.loop.vectorize, processed by
1806// LoopVectorizeHints). This will be fixed in the future when the native IR
1807// representation for pragma 'omp simd' is introduced.
1808static bool isExplicitVecOuterLoop(Loop *OuterLp,
1810 assert(!OuterLp->isInnermost() && "This is not an outer loop");
1811 LoopVectorizeHints Hints(OuterLp, true /*DisableInterleaving*/, *ORE);
1812
1813 // Only outer loops with an explicit vectorization hint are supported.
1814 // Unannotated outer loops are ignored.
1816 return false;
1817
1818 Function *Fn = OuterLp->getHeader()->getParent();
1819 if (!Hints.allowVectorization(Fn, OuterLp,
1820 true /*VectorizeOnlyWhenForced*/)) {
1821 LLVM_DEBUG(dbgs() << "LV: Loop hints prevent outer loop vectorization.\n");
1822 return false;
1823 }
1824
1825 if (Hints.getInterleave() > 1) {
1826 // TODO: Interleave support is future work.
1827 LLVM_DEBUG(dbgs() << "LV: Not vectorizing: Interleave is not supported for "
1828 "outer loops.\n");
1829 Hints.emitRemarkWithHints();
1830 return false;
1831 }
1832
1833 return true;
1834}
1835
1839 // Collect inner loops and outer loops without irreducible control flow. For
1840 // now, only collect outer loops that have explicit vectorization hints. If we
1841 // are stress testing the VPlan H-CFG construction, we collect the outermost
1842 // loop of every loop nest.
1843 if (L.isInnermost() || VPlanBuildOuterloopStressTest ||
1845 LoopBlocksRPO RPOT(&L);
1846 RPOT.perform(LI);
1848 V.push_back(&L);
1849 // TODO: Collect inner loops inside marked outer loops in case
1850 // vectorization fails for the outer loop. Do not invoke
1851 // 'containsIrreducibleCFG' again for inner loops when the outer loop is
1852 // already known to be reducible. We can use an inherited attribute for
1853 // that.
1854 return;
1855 }
1856 }
1857 for (Loop *InnerL : L)
1858 collectSupportedLoops(*InnerL, LI, ORE, V);
1859}
1860
1861//===----------------------------------------------------------------------===//
1862// Implementation of LoopVectorizationLegality, InnerLoopVectorizer and
1863// LoopVectorizationCostModel and LoopVectorizationPlanner.
1864//===----------------------------------------------------------------------===//
1865
1866/// For the given VF and UF and maximum trip count computed for the loop, return
1867/// whether the induction variable might overflow in the vectorized loop. If not,
1868/// then we know a runtime overflow check always evaluates to false and can be
1869/// removed.
1871 const LoopVectorizationCostModel *Cost,
1872 ElementCount VF, std::optional<unsigned> UF = std::nullopt) {
1873 // Always be conservative if we don't know the exact unroll factor.
1874 uint64_t MaxUF = UF ? *UF
1875 : std::max(Cost->TTI.getMaxInterleaveFactor(VF, false),
1876 Cost->TTI.getMaxInterleaveFactor(VF, true));
1877
1878 IntegerType *IdxTy = Cost->Legal->getWidestInductionType();
1879 APInt MaxUIntTripCount = IdxTy->getMask();
1880
1881 // We know the runtime overflow check is known false iff the (max) trip-count
1882 // is known and (max) trip-count + (VF * UF) does not overflow in the type of
1883 // the vector loop induction variable.
1884 if (std::optional<ElementCount> TC = getSmallBestKnownTC(
1885 Cost->PSE, Cost->TheLoop,
1886 /*CanUseConstantMax=*/true, /*CanExcludeZeroTrips=*/false,
1887 /*ComputeUpperBoundOnly=*/true)) {
1888 // Compute the maximum runtime values of VF and the trip count.
1889 std::optional<uint64_t> MaxStep =
1890 getMaxRuntimeElementCount(VF * MaxUF, *Cost->TheFunction);
1891 std::optional<uint64_t> MaxTC =
1892 getMaxRuntimeElementCount(*TC, *Cost->TheFunction);
1893 if (!MaxStep || !MaxTC)
1894 return false;
1895
1896 // Bail out if the maximum trip count is not representable in the induction
1897 // variable's type.
1898 if (MaxUIntTripCount.ult(*MaxTC))
1899 return false;
1900
1901 return (MaxUIntTripCount - *MaxTC).ugt(*MaxStep);
1902 }
1903
1904 return false;
1905}
1906
1907// Return whether we allow using masked interleave-groups (for dealing with
1908// strided loads/stores that reside in predicated blocks, or for dealing
1909// with gaps).
1911 // If an override option has been passed in for interleaved accesses, use it.
1912 if (EnableMaskedInterleavedMemAccesses.getNumOccurrences() > 0)
1914
1915 return TTI.enableMaskedInterleavedAccessVectorization();
1916}
1917
1918/// Replace \p VPBB with a VPIRBasicBlock wrapping \p IRBB. All recipes from \p
1919/// VPBB are moved to the end of the newly created VPIRBasicBlock. All
1920/// predecessors and successors of VPBB, if any, are rewired to the new
1921/// VPIRBasicBlock. If \p VPBB may be unreachable, \p Plan must be passed.
1923 BasicBlock *IRBB,
1924 VPlan *Plan = nullptr) {
1925 if (!Plan)
1926 Plan = VPBB->getPlan();
1927 VPIRBasicBlock *IRVPBB = Plan->createEmptyVPIRBasicBlock(IRBB);
1928 auto IP = IRVPBB->begin();
1929 for (auto &R : make_early_inc_range(VPBB->phis()))
1930 R.moveBefore(*IRVPBB, IP);
1931
1932 for (auto &R :
1934 R.moveBefore(*IRVPBB, IRVPBB->end());
1935
1936 VPBlockUtils::reassociateBlocks(VPBB, IRVPBB);
1937 // VPBB is now dead and will be cleaned up when the plan gets destroyed.
1938 return IRVPBB;
1939}
1940
1942 BasicBlock *VectorPH = OrigLoop->getLoopPreheader();
1943 assert(VectorPH && "Invalid loop structure");
1944
1945 // NOTE: The Plan's scalar preheader VPBB isn't replaced with a VPIRBasicBlock
1946 // wrapping the newly created scalar preheader here at the moment, because the
1947 // Plan's scalar preheader may be unreachable at this point. Instead it is
1948 // replaced in executePlan.
1949 return SplitBlock(VectorPH, VectorPH->getTerminator(), DT, LI, nullptr,
1950 Twine(Prefix) + "scalar.ph");
1951}
1952
1953/// Knowing that loop \p L executes a single vector iteration, add instructions
1954/// that will get simplified and thus should not have any cost to \p
1955/// InstsToIgnore.
1958 SmallPtrSetImpl<Instruction *> &InstsToIgnore) {
1959 auto *Cmp = L->getLatchCmpInst();
1960 if (Cmp)
1961 InstsToIgnore.insert(Cmp);
1962 for (const auto &KV : IL) {
1963 // Extract the key by hand so that it can be used in the lambda below. Note
1964 // that captured structured bindings are a C++20 extension.
1965 PHINode *IV = KV.first;
1966
1967 // The induction is free: a widened induction generates a vector phi with
1968 // its start value and an increment that is dead without a backedge.
1969 InstsToIgnore.insert(IV);
1970
1971 // Get next iteration value of the induction variable.
1972 Instruction *IVInst =
1973 cast<Instruction>(IV->getIncomingValueForBlock(L->getLoopLatch()));
1974 if (all_of(IVInst->users(),
1975 [&](const User *U) { return U == IV || U == Cmp; }))
1976 InstsToIgnore.insert(IVInst);
1977 }
1978}
1979
1981 // Create a new IR basic block for the scalar preheader.
1982 BasicBlock *ScalarPH = createScalarPreheader("");
1983 return ScalarPH->getSinglePredecessor();
1984}
1985
1986namespace {
1987
1988struct CSEDenseMapInfo {
1989 static bool canHandle(const Instruction *I) {
1992 }
1993
1994 static unsigned getHashValue(const Instruction *I) {
1995 assert(canHandle(I) && "Unknown instruction!");
1996 return hash_combine(I->getOpcode(),
1997 hash_combine_range(I->operand_values()));
1998 }
1999
2000 static bool isEqual(const Instruction *LHS, const Instruction *RHS) {
2001 return LHS->isIdenticalTo(RHS);
2002 }
2003};
2004
2005} // end anonymous namespace
2006
2007/// FIXME: This legacy common-subexpression-elimination routine is scheduled for
2008/// removal, in favor of the VPlan-based one.
2009static void legacyCSE(BasicBlock *BB) {
2010 // Perform simple cse.
2012 for (Instruction &In : llvm::make_early_inc_range(*BB)) {
2013 if (!CSEDenseMapInfo::canHandle(&In))
2014 continue;
2015
2016 // Check if we can replace this instruction with any of the
2017 // visited instructions.
2018 if (Instruction *V = CSEMap.lookup(&In)) {
2019 In.replaceAllUsesWith(V);
2020 In.eraseFromParent();
2021 continue;
2022 }
2023
2024 CSEMap[&In] = &In;
2025 }
2026}
2027
2028/// This function attempts to return a value that represents the ElementCount
2029/// at runtime. For fixed-width VFs we know this precisely at compile
2030/// time, but for scalable VFs we calculate it based on an estimate of the
2031/// vscale value.
2033 std::optional<unsigned> VScale) {
2034 unsigned EstimatedVF = VF.getKnownMinValue();
2035 if (VF.isScalable())
2036 if (VScale)
2037 EstimatedVF *= *VScale;
2038 assert(EstimatedVF >= 1 && "Estimated VF shouldn't be less than 1");
2039 return EstimatedVF;
2040}
2041
2042/// Returns the vector library variant function of \p CI usable at \p VF,
2043/// respecting \p MaskRequired, or nullptr if none is found: a mapping with
2044/// matching VF, masked if required, whose vector function is declared in the
2045/// module.
2047 bool MaskRequired,
2048 const TargetLibraryInfo *TLI) {
2049 if (!TLI || CI.isNoBuiltin())
2050 return nullptr;
2051 for (const VFInfo &Info : VFDatabase::getMappings(CI))
2052 if (Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()))
2053 if (Function *F = CI.getModule()->getFunction(Info.VectorName))
2054 return F;
2055 return nullptr;
2056}
2057
2058/// Returns true iff \p CI has a library vector variant usable at \p VF.
2060 bool MaskRequired,
2061 const TargetLibraryInfo *TLI) {
2062 return getVectorLibraryVariantFor(CI, VF, MaskRequired, TLI) != nullptr;
2063}
2064
2067 ElementCount VF) const {
2068 Type *RetTy = CI->getType();
2070 for (auto &ArgOp : CI->args())
2071 Tys.push_back(ArgOp->getType());
2072
2073 InstructionCost ScalarCallCost = TTI.getCallInstrCost(
2074 CI->getCalledFunction(), RetTy, Tys, Config.CostKind);
2075
2076 // Cost of the scalar call (scalar VF) or its scalarization (vector VF). The
2077 // scalarization cost is only meaningful for fixed VFs.
2080 : ScalarCallCost * VF.getKnownMinValue() +
2081 getScalarizationOverhead(CI, VF);
2082
2083 // The call may be vectorized at this VF, via a vector intrinsic or a vector
2084 // library variant.
2086 Cost = std::min(Cost, getVectorIntrinsicCost(CI, VF));
2087
2088 if (Function *Variant =
2090 Cost = std::min(Cost,
2091 TTI.getCallInstrCost(
2092 /*F=*/nullptr, Variant->getReturnType(),
2093 Variant->getFunctionType()->params(), Config.CostKind));
2094
2095 return Cost;
2096}
2097
2099 if (VF.isScalar() || !canVectorizeTy(Ty))
2100 return Ty;
2101 return toVectorizedTy(Ty, VF);
2102}
2103
2106 ElementCount VF) const {
2108 assert(ID && "Expected intrinsic call!");
2109 Type *RetTy = maybeVectorizeType(CI->getType(), VF);
2110 FastMathFlags FMF;
2111 if (auto *FPMO = dyn_cast<FPMathOperator>(CI))
2112 FMF = FPMO->getFastMathFlags();
2113
2116 SmallVector<Type *> ParamTys;
2117 std::transform(FTy->param_begin(), FTy->param_end(),
2118 std::back_inserter(ParamTys),
2119 [&](Type *Ty) { return maybeVectorizeType(Ty, VF); });
2120
2121 IntrinsicCostAttributes CostAttrs(ID, RetTy, Arguments, ParamTys, FMF,
2124 return TTI.getIntrinsicInstrCost(CostAttrs, Config.CostKind);
2125}
2126
2128 // Don't apply optimizations below when no (vector) loop remains, as they all
2129 // require one at the moment.
2130 VPBasicBlock *HeaderVPBB =
2131 vputils::getFirstLoopHeader(*State.Plan, State.VPDT);
2132 if (!HeaderVPBB)
2133 return;
2134
2135 BasicBlock *HeaderBB = State.CFG.VPBB2IRBB[HeaderVPBB];
2136
2137 // Remove redundant induction instructions.
2138 legacyCSE(HeaderBB);
2139}
2140
2141void LoopVectorizationCostModel::collectLoopScalars(ElementCount VF) {
2142 // We should not collect Scalars more than once per VF. Right now, this
2143 // function is called from collectUniformsAndScalars(), which already does
2144 // this check. Collecting Scalars for VF=1 does not make any sense.
2145 assert(VF.isVector() && !Scalars.contains(VF) &&
2146 "This function should not be visited twice for the same VF");
2147
2148 // This avoids any chances of creating a REPLICATE recipe during planning
2149 // since that would result in generation of scalarized code during execution,
2150 // which is not supported for scalable vectors.
2151 if (VF.isScalable()) {
2152 Scalars[VF].insert_range(Uniforms[VF]);
2153 return;
2154 }
2155
2157
2158 // These sets are used to seed the analysis with pointers used by memory
2159 // accesses that will remain scalar.
2161 SmallPtrSet<Instruction *, 8> PossibleNonScalarPtrs;
2162 auto *Latch = TheLoop->getLoopLatch();
2163
2164 // A helper that returns true if the use of Ptr by MemAccess will be scalar.
2165 // The pointer operands of loads and stores will be scalar as long as the
2166 // memory access is not a gather/scatter or histogram operation. The value
2167 // operand of a store will remain scalar if the store is scalarized.
2168 auto IsScalarUse = [&](Instruction *MemAccess, Value *Ptr) {
2169 InstWidening WideningDecision = getWideningDecision(MemAccess, VF);
2170 assert(WideningDecision != CM_Unknown &&
2171 "Widening decision should be ready at this moment");
2172 auto *Store = dyn_cast<StoreInst>(MemAccess);
2173 if (Store && Ptr == Store->getValueOperand())
2174 return WideningDecision == CM_Scalarize;
2175 assert(Ptr == getLoadStorePointerOperand(MemAccess) &&
2176 "Ptr is neither a value or pointer operand");
2177 return WideningDecision != CM_GatherScatter &&
2178 !(Store && Legal->getHistogramInfo(Store));
2179 };
2180
2181 // A helper that returns true if the given value is a getelementptr
2182 // instruction contained in the loop.
2183 auto IsLoopVaryingGEP = [&](Value *V) {
2184 return isa<GetElementPtrInst>(V) && !TheLoop->isLoopInvariant(V);
2185 };
2186
2187 // A helper that evaluates a memory access's use of a pointer. If the use will
2188 // be a scalar use and the pointer is only used by memory accesses, we place
2189 // the pointer in ScalarPtrs. Otherwise, the pointer is placed in
2190 // PossibleNonScalarPtrs.
2191 auto EvaluatePtrUse = [&](Instruction *MemAccess, Value *Ptr) {
2192 // We only care about bitcast and getelementptr instructions contained in
2193 // the loop.
2194 if (!IsLoopVaryingGEP(Ptr))
2195 return;
2196
2197 // If the pointer has already been identified as scalar (e.g., if it was
2198 // also identified as uniform), there's nothing to do.
2199 auto *I = cast<Instruction>(Ptr);
2200 if (Worklist.count(I))
2201 return;
2202
2203 // If the use of the pointer will be a scalar use, and all users of the
2204 // pointer are memory accesses, place the pointer in ScalarPtrs. Otherwise,
2205 // place the pointer in PossibleNonScalarPtrs.
2206 if (IsScalarUse(MemAccess, Ptr) &&
2208 ScalarPtrs.insert(I);
2209 else
2210 PossibleNonScalarPtrs.insert(I);
2211 };
2212
2213 // We seed the scalars analysis with three classes of instructions: (1)
2214 // instructions marked uniform-after-vectorization and (2) bitcast,
2215 // getelementptr and (pointer) phi instructions used by memory accesses
2216 // requiring a scalar use.
2217 //
2218 // (1) Add to the worklist all instructions that have been identified as
2219 // uniform-after-vectorization.
2220 Worklist.insert_range(Uniforms[VF]);
2221
2222 // (2) Add to the worklist all bitcast and getelementptr instructions used by
2223 // memory accesses requiring a scalar use. The pointer operands of loads and
2224 // stores will be scalar unless the operation is a gather or scatter.
2225 // The value operand of a store will remain scalar if the store is scalarized.
2226 for (auto *BB : TheLoop->blocks())
2227 for (auto &I : *BB) {
2228 if (auto *Load = dyn_cast<LoadInst>(&I)) {
2229 EvaluatePtrUse(Load, Load->getPointerOperand());
2230 } else if (auto *Store = dyn_cast<StoreInst>(&I)) {
2231 EvaluatePtrUse(Store, Store->getPointerOperand());
2232 EvaluatePtrUse(Store, Store->getValueOperand());
2233 }
2234 }
2235 for (auto *I : ScalarPtrs)
2236 if (!PossibleNonScalarPtrs.count(I)) {
2237 LLVM_DEBUG(dbgs() << "LV: Found scalar instruction: " << *I << "\n");
2238 Worklist.insert(I);
2239 }
2240
2241 // Insert the forced scalars.
2242 // FIXME: Currently VPWidenPHIRecipe() often creates a dead vector
2243 // induction variable when the PHI user is scalarized.
2244 auto ForcedScalar = ForcedScalars.find(VF);
2245 if (ForcedScalar != ForcedScalars.end())
2246 for (auto *I : ForcedScalar->second) {
2247 LLVM_DEBUG(dbgs() << "LV: Found (forced) scalar instruction: " << *I << "\n");
2248 Worklist.insert(I);
2249 }
2250
2251 // Expand the worklist by looking through any bitcasts and getelementptr
2252 // instructions we've already identified as scalar. This is similar to the
2253 // expansion step in collectLoopUniforms(); however, here we're only
2254 // expanding to include additional bitcasts and getelementptr instructions.
2255 unsigned Idx = 0;
2256 while (Idx != Worklist.size()) {
2257 Instruction *Dst = Worklist[Idx++];
2258 if (!IsLoopVaryingGEP(Dst->getOperand(0)))
2259 continue;
2260 auto *Src = cast<Instruction>(Dst->getOperand(0));
2261 if (llvm::all_of(Src->users(), [&](User *U) -> bool {
2262 auto *J = cast<Instruction>(U);
2263 return !TheLoop->contains(J) || Worklist.count(J) ||
2264 ((isa<LoadInst>(J) || isa<StoreInst>(J)) &&
2265 IsScalarUse(J, Src));
2266 })) {
2267 Worklist.insert(Src);
2268 LLVM_DEBUG(dbgs() << "LV: Found scalar instruction: " << *Src << "\n");
2269 }
2270 }
2271
2272 // An induction variable will remain scalar if all users of the induction
2273 // variable and induction variable update remain scalar.
2274 for (const auto &Induction : Legal->getInductionVars()) {
2275 auto *Ind = Induction.first;
2276 auto *IndUpdate = cast<Instruction>(Ind->getIncomingValueForBlock(Latch));
2277
2278 // If tail-folding is applied, the primary induction variable will be used
2279 // to feed a vector compare.
2280 if (Ind == Legal->getPrimaryInduction() && foldTailByMasking())
2281 continue;
2282
2283 // Returns true if \p Indvar is a pointer induction that is used directly by
2284 // load/store instruction \p I.
2285 auto IsDirectLoadStoreFromPtrIndvar = [&](Instruction *Indvar,
2286 Instruction *I) {
2287 return Induction.second.getKind() ==
2290 Indvar == getLoadStorePointerOperand(I) && IsScalarUse(I, Indvar);
2291 };
2292
2293 // Determine if all users of the induction variable are scalar after
2294 // vectorization.
2295 bool ScalarInd = all_of(Ind->users(), [&](User *U) -> bool {
2296 auto *I = cast<Instruction>(U);
2297 return I == IndUpdate || !TheLoop->contains(I) || Worklist.count(I) ||
2298 IsDirectLoadStoreFromPtrIndvar(Ind, I);
2299 });
2300 if (!ScalarInd)
2301 continue;
2302
2303 // If the induction variable update is a fixed-order recurrence, neither the
2304 // induction variable or its update should be marked scalar after
2305 // vectorization.
2306 auto *IndUpdatePhi = dyn_cast<PHINode>(IndUpdate);
2307 if (IndUpdatePhi && Legal->isFixedOrderRecurrence(IndUpdatePhi))
2308 continue;
2309
2310 // Determine if all users of the induction variable update instruction are
2311 // scalar after vectorization.
2312 bool ScalarIndUpdate = all_of(IndUpdate->users(), [&](User *U) -> bool {
2313 auto *I = cast<Instruction>(U);
2314 return I == Ind || !TheLoop->contains(I) || Worklist.count(I) ||
2315 IsDirectLoadStoreFromPtrIndvar(IndUpdate, I);
2316 });
2317 if (!ScalarIndUpdate)
2318 continue;
2319
2320 // The induction variable and its update instruction will remain scalar.
2321 Worklist.insert(Ind);
2322 Worklist.insert(IndUpdate);
2323 LLVM_DEBUG(dbgs() << "LV: Found scalar instruction: " << *Ind << "\n");
2324 LLVM_DEBUG(dbgs() << "LV: Found scalar instruction: " << *IndUpdate
2325 << "\n");
2326 }
2327
2328 Scalars[VF].insert_range(Worklist);
2329}
2330
2338
2340 ElementCount VF) const {
2342 return Config.isLegalGatherOrScatter(isa<LoadInst>(I), getLoadStoreType(I),
2344}
2345
2347 ElementCount VF) {
2348 if (!isPredicatedInst(I))
2349 return false;
2350
2351 // Do we have a non-scalar lowering for this predicated
2352 // instruction? No - it is scalar with predication.
2353 switch(I->getOpcode()) {
2354 default:
2355 return true;
2356 case Instruction::Call: {
2357 if (VF.isScalar())
2358 return true;
2359 auto *CI = cast<CallInst>(I);
2360 // A vector intrinsic or library variant lowering avoids scalarization.
2361 return !getVectorIntrinsicIDForCall(CI, TLI) &&
2363 }
2364 case Instruction::Load:
2365 case Instruction::Store: {
2366 bool IsConsecutive = Legal->isConsecutivePtr(getLoadStoreType(I),
2368 return !(IsConsecutive && isLegalMaskedLoadOrStore(I, VF)) &&
2370 }
2371 case Instruction::UDiv:
2372 case Instruction::SDiv:
2373 case Instruction::SRem:
2374 case Instruction::URem: {
2375 // We have the option to use the llvm.masked.udiv intrinsics to avoid
2376 // predication. The cost based decision here will always select the masked
2377 // intrinsics for scalable vectors as scalarization isn't legal.
2378 const auto [ScalarCost, MaskedCost] = getDivRemSpeculationCost(I, VF);
2379 return isDivRemScalarWithPredication(ScalarCost, MaskedCost);
2380 }
2381 }
2382}
2383
2385 return Legal->isMaskRequired(I, foldTailByMasking());
2386}
2387
2388// TODO: Fold into LoopVectorizationLegality::isMaskRequired.
2390 // TODO: We can use the loop-preheader as context point here and get
2391 // context sensitive reasoning for isSafeToSpeculativelyExecute.
2395 return false;
2396
2397 // If the instruction was executed conditionally in the original scalar loop,
2398 // predication is needed with a mask whose lanes are all possibly inactive.
2399 if (Legal->blockNeedsPredication(I->getParent()))
2400 return true;
2401
2402 // If we're not folding the tail by masking and not vectorizing a loop with
2403 // uncountable exits and side effects, predication is unnecessary.
2404 if (!foldTailByMasking() && !Legal->hasUncountableExitWithSideEffects())
2405 return false;
2406
2407 // All that remain are instructions with side-effects originally executed in
2408 // the loop unconditionally, but now execute under a tail-fold mask (only)
2409 // having at least one active lane (the first). If the side-effects of the
2410 // instruction are invariant, executing it w/o (the tail-folding) mask is safe
2411 // - it will cause the same side-effects as when masked.
2412 switch(I->getOpcode()) {
2413 default:
2415 "instruction should have been considered by earlier checks");
2416 case Instruction::Call:
2417 // Side-effects of a Call are assumed to be non-invariant, needing a
2418 // (fold-tail) mask.
2420 "should have returned earlier for calls not needing a mask");
2421 return true;
2422 case Instruction::Load:
2423 // If the address is loop invariant no predication is needed.
2424 return !Legal->isInvariant(getLoadStorePointerOperand(I));
2425 case Instruction::Store: {
2426 // For stores, we need to prove both speculation safety (which follows from
2427 // the same argument as loads), but also must prove the value being stored
2428 // is correct. The easiest form of the later is to require that all values
2429 // stored are the same.
2430 return !(Legal->isInvariant(getLoadStorePointerOperand(I)) &&
2431 TheLoop->isLoopInvariant(cast<StoreInst>(I)->getValueOperand()));
2432 }
2433 case Instruction::UDiv:
2434 case Instruction::URem:
2435 // If the divisor is loop-invariant no predication is needed.
2436 return !Legal->isInvariant(I->getOperand(1));
2437 case Instruction::SDiv:
2438 case Instruction::SRem:
2439 // Conservative for now, since masked-off lanes may be poison and could
2440 // trigger signed overflow.
2441 return true;
2442 }
2443}
2444
2448 return 1;
2449 // If the block wasn't originally predicated then return early to avoid
2450 // computing BlockFrequencyInfo unnecessarily.
2451 if (!Legal->blockNeedsPredication(BB))
2452 return 1;
2453
2454 uint64_t HeaderFreq =
2455 getBFI().getBlockFreq(TheLoop->getHeader()).getFrequency();
2456 uint64_t BBFreq = getBFI().getBlockFreq(BB).getFrequency();
2457 assert(HeaderFreq >= BBFreq &&
2458 "Header has smaller block freq than dominated BB?");
2459 return std::round((double)HeaderFreq / BBFreq);
2460}
2461
2463 switch (Opcode) {
2464 case Instruction::UDiv:
2465 return Intrinsic::masked_udiv;
2466 case Instruction::SDiv:
2467 return Intrinsic::masked_sdiv;
2468 case Instruction::URem:
2469 return Intrinsic::masked_urem;
2470 case Instruction::SRem:
2471 return Intrinsic::masked_srem;
2472 default:
2473 llvm_unreachable("Unexpected opcode");
2474 }
2475}
2476
2477std::pair<InstructionCost, InstructionCost>
2479 ElementCount VF) {
2480 assert(I->getOpcode() == Instruction::UDiv ||
2481 I->getOpcode() == Instruction::SDiv ||
2482 I->getOpcode() == Instruction::SRem ||
2483 I->getOpcode() == Instruction::URem);
2485
2486 // Scalarization isn't legal for scalable vector types
2487 InstructionCost ScalarizationCost = InstructionCost::getInvalid();
2488 if (!VF.isScalable()) {
2489 // Get the scalarization cost and scale this amount by the probability of
2490 // executing the predicated block. If the instruction is not predicated,
2491 // we fall through to the next case.
2492 ScalarizationCost = 0;
2493
2494 // These instructions have a non-void type, so account for the phi nodes
2495 // that we will create. This cost is likely to be zero. The phi node
2496 // cost, if any, should be scaled by the block probability because it
2497 // models a copy at the end of each predicated block.
2498 ScalarizationCost += VF.getFixedValue() *
2499 TTI.getCFInstrCost(Instruction::PHI, Config.CostKind);
2500
2501 // The cost of the non-predicated instruction.
2502 ScalarizationCost +=
2503 VF.getFixedValue() * TTI.getArithmeticInstrCost(
2504 I->getOpcode(), I->getType(), Config.CostKind);
2505
2506 // The cost of insertelement and extractelement instructions needed for
2507 // scalarization.
2508 ScalarizationCost += getScalarizationOverhead(I, VF);
2509
2510 // Scale the cost by the probability of executing the predicated blocks.
2511 // This assumes the predicated block for each vector lane is equally
2512 // likely.
2513 ScalarizationCost =
2514 ScalarizationCost /
2515 getPredBlockCostDivisor(Config.CostKind, I->getParent());
2516 }
2517
2518 auto *VecTy = toVectorTy(I->getType(), VF);
2519 auto *MaskTy = toVectorTy(Type::getInt1Ty(I->getContext()), VF);
2520 IntrinsicCostAttributes ICA(getMaskedDivRemIntrinsic(I->getOpcode()), VecTy,
2521 {VecTy, VecTy, MaskTy});
2522 InstructionCost MaskedCost = TTI.getIntrinsicInstrCost(ICA, Config.CostKind);
2523 return {ScalarizationCost, MaskedCost};
2524}
2525
2527 Instruction *I, ElementCount VF) const {
2528 assert(isAccessInterleaved(I) && "Expecting interleaved access.");
2530 "Decision should not be set yet.");
2531 auto *Group = getInterleavedAccessGroup(I);
2532 assert(Group && "Must have a group.");
2533 unsigned InterleaveFactor = Group->getFactor();
2534
2535 // If the instruction's allocated size doesn't equal its type size, it
2536 // requires padding and will be scalarized.
2537 auto &DL = I->getDataLayout();
2538 auto *ScalarTy = getLoadStoreType(I);
2539 if (hasIrregularType(ScalarTy, DL))
2540 return false;
2541
2542 // For scalable vectors, the interleave factors must be <= 8 since we require
2543 // the (de)interleaveN intrinsics instead of shufflevectors.
2544 if (VF.isScalable() && InterleaveFactor > 8)
2545 return false;
2546
2547 // If the group involves a non-integral pointer, we may not be able to
2548 // losslessly cast all values to a common type.
2549 bool ScalarNI = DL.isNonIntegralPointerType(ScalarTy);
2550 for (Instruction *Member : Group->members()) {
2551 auto *MemberTy = getLoadStoreType(Member);
2552 bool MemberNI = DL.isNonIntegralPointerType(MemberTy);
2553 // Don't coerce non-integral pointers to integers or vice versa.
2554 if (MemberNI != ScalarNI)
2555 // TODO: Consider adding special nullptr value case here
2556 return false;
2557 if (MemberNI && ScalarNI &&
2558 ScalarTy->getPointerAddressSpace() !=
2559 MemberTy->getPointerAddressSpace())
2560 return false;
2561 }
2562
2563 // Check if masking is required.
2564 // A Group may need masking for one of two reasons: it resides in a block that
2565 // needs predication, or it was decided to use masking to deal with gaps
2566 // (either a gap at the end of a load-access that may result in a speculative
2567 // load, or any gaps in a store-access).
2568 bool PredicatedAccessRequiresMasking =
2570 bool LoadAccessWithGapsRequiresEpilogMasking =
2571 isa<LoadInst>(I) && Group->requiresScalarEpilogue() &&
2573 bool StoreAccessWithGapsRequiresMasking =
2574 isa<StoreInst>(I) && !Group->isFull();
2575 if (!PredicatedAccessRequiresMasking &&
2576 !LoadAccessWithGapsRequiresEpilogMasking &&
2577 !StoreAccessWithGapsRequiresMasking)
2578 return true;
2579
2580 // If masked interleaving is required, we expect that the user/target had
2581 // enabled it, because otherwise it either wouldn't have been created or
2582 // it should have been invalidated by the CostModel.
2584 "Masked interleave-groups for predicated accesses are not enabled.");
2585
2586 if (Group->isReverse())
2587 return false;
2588
2589 // TODO: Support interleaved access that requires a gap mask for scalable VFs.
2590 bool NeedsMaskForGaps = LoadAccessWithGapsRequiresEpilogMasking ||
2591 StoreAccessWithGapsRequiresMasking;
2592 if (VF.isScalable() && NeedsMaskForGaps)
2593 return false;
2594
2595 return isLegalMaskedLoadOrStore(I, VF);
2596}
2597
2598std::optional<LoopVectorizationCostModel::InstWidening>
2600 ElementCount VF) {
2601 // Get and ensure we have a valid memory instruction.
2602 assert((isa<LoadInst, StoreInst>(I)) && "Invalid memory instruction");
2603
2604 auto *Ptr = getLoadStorePointerOperand(I);
2605 auto *ScalarTy = getLoadStoreType(I);
2606
2607 // In order to be widened, the pointer should be consecutive, first of all.
2608 int Stride = Legal->isConsecutivePtr(ScalarTy, Ptr);
2609 if (!Stride)
2610 return std::nullopt;
2611
2612 // If the instruction is a store located in a predicated block, it will be
2613 // scalarized.
2614 if (isScalarWithPredication(I, VF))
2615 return std::nullopt;
2616
2617 // If the instruction's allocated size doesn't equal it's type size, it
2618 // requires padding and will be scalarized.
2619 auto &DL = I->getDataLayout();
2620 if (hasIrregularType(ScalarTy, DL))
2621 return std::nullopt;
2622
2623 return Stride == 1 ? CM_Widen : CM_Widen_Reverse;
2624}
2625
2626void LoopVectorizationCostModel::collectLoopUniforms(ElementCount VF) {
2627 // We should not collect Uniforms more than once per VF. Right now,
2628 // this function is called from collectUniformsAndScalars(), which
2629 // already does this check. Collecting Uniforms for VF=1 does not make any
2630 // sense.
2631
2632 assert(VF.isVector() && !Uniforms.contains(VF) &&
2633 "This function should not be visited twice for the same VF");
2634
2635 // Visit the list of Uniforms. If we find no uniform value, we won't
2636 // analyze again. Uniforms.count(VF) will return 1.
2637 Uniforms[VF].clear();
2638
2639 // Now we know that the loop is vectorizable!
2640 // Collect instructions inside the loop that will remain uniform after
2641 // vectorization.
2642
2643 // Global values, params and instructions outside of current loop are out of
2644 // scope.
2645 auto IsOutOfScope = [&](Value *V) -> bool {
2647 return (!I || !TheLoop->contains(I));
2648 };
2649
2650 // Worklist containing uniform instructions demanding lane 0.
2651 SetVector<Instruction *> Worklist;
2652
2653 // Add uniform instructions demanding lane 0 to the worklist. Instructions
2654 // that require predication must not be considered uniform after
2655 // vectorization, because that would create an erroneous replicating region
2656 // where only a single instance out of VF should be formed.
2657 auto AddToWorklistIfAllowed = [&](Instruction *I) -> void {
2658 if (IsOutOfScope(I)) {
2659 LLVM_DEBUG(dbgs() << "LV: Found not uniform due to scope: "
2660 << *I << "\n");
2661 return;
2662 }
2663 if (isPredicatedInst(I)) {
2664 LLVM_DEBUG(
2665 dbgs() << "LV: Found not uniform due to requiring predication: " << *I
2666 << "\n");
2667 return;
2668 }
2669 LLVM_DEBUG(dbgs() << "LV: Found uniform instruction: " << *I << "\n");
2670 Worklist.insert(I);
2671 };
2672
2673 // Start with the conditional branches exiting the loop. If the branch
2674 // condition is an instruction contained in the loop that is only used by the
2675 // branch, it is uniform. Note conditions from uncountable early exits are not
2676 // uniform.
2678 TheLoop->getExitingBlocks(Exiting);
2679 for (BasicBlock *E : Exiting) {
2680 if (Legal->hasUncountableEarlyExit() && TheLoop->getLoopLatch() != E)
2681 continue;
2682 auto *Cmp = dyn_cast<Instruction>(E->getTerminator()->getOperand(0));
2683 if (!Cmp || !TheLoop->contains(Cmp) || !Cmp->hasOneUse())
2684 continue;
2685
2686 // If we have an exit condition that is actually two conditions (one
2687 // countable and the other uncountable) combined via an or, only add the
2688 // countable comparison as a uniform value.
2689 if (Legal->hasUncountableExitWithSideEffects() &&
2690 TheLoop->getLoopLatch() == E) {
2691 if (Instruction *Countable =
2692 Legal->findCountableComparisonInCombinedCondition(Cmp)) {
2693 if (Countable->hasOneUse())
2694 AddToWorklistIfAllowed(Countable);
2695 continue;
2696 }
2697 }
2698
2699 // Normal exit comparisons are uniform.
2700 AddToWorklistIfAllowed(Cmp);
2701 }
2702
2703 auto PrevVF = VF.divideCoefficientBy(2);
2704 // Return true if all lanes perform the same memory operation, and we can
2705 // thus choose to execute only one.
2706 auto IsUniformMemOpUse = [&](Instruction *I) {
2707 // If the value was already known to not be uniform for the previous
2708 // (smaller VF), it cannot be uniform for the larger VF.
2709 if (PrevVF.isVector()) {
2710 auto Iter = Uniforms.find(PrevVF);
2711 if (Iter != Uniforms.end() && !Iter->second.contains(I))
2712 return false;
2713 }
2714 if (!isUniformMemOp(*I, VF))
2715 return false;
2716 if (isa<LoadInst>(I))
2717 // Loading the same address always produces the same result - at least
2718 // assuming aliasing and ordering which have already been checked.
2719 return true;
2720 // Storing the same value on every iteration.
2721 return TheLoop->isLoopInvariant(cast<StoreInst>(I)->getValueOperand());
2722 };
2723
2724 auto IsUniformDecision = [&](Instruction *I, ElementCount VF) {
2725 InstWidening WideningDecision = getWideningDecision(I, VF);
2726 assert(WideningDecision != CM_Unknown &&
2727 "Widening decision should be ready at this moment");
2728
2729 if (IsUniformMemOpUse(I))
2730 return true;
2731
2732 return (WideningDecision == CM_Widen ||
2733 WideningDecision == CM_Widen_Reverse ||
2734 WideningDecision == CM_Interleave);
2735 };
2736
2737 // Returns true if Ptr is the pointer operand of a memory access instruction
2738 // I, I is known to not require scalarization, and the pointer is not also
2739 // stored.
2740 auto IsVectorizedMemAccessUse = [&](Instruction *I, Value *Ptr) -> bool {
2741 if (isa<StoreInst>(I) && I->getOperand(0) == Ptr)
2742 return false;
2743 return getLoadStorePointerOperand(I) == Ptr &&
2744 (IsUniformDecision(I, VF) || Legal->isInvariant(Ptr));
2745 };
2746
2747 // Holds a list of values which are known to have at least one uniform use.
2748 // Note that there may be other uses which aren't uniform. A "uniform use"
2749 // here is something which only demands lane 0 of the unrolled iterations;
2750 // it does not imply that all lanes produce the same value (e.g. this is not
2751 // the usual meaning of uniform)
2752 SetVector<Value *> HasUniformUse;
2753
2754 // Scan the loop for instructions which are either a) known to have only
2755 // lane 0 demanded or b) are uses which demand only lane 0 of their operand.
2756 for (auto *BB : TheLoop->blocks())
2757 for (auto &I : *BB) {
2758 if (IntrinsicInst *II = dyn_cast<IntrinsicInst>(&I)) {
2759 switch (II->getIntrinsicID()) {
2760 case Intrinsic::sideeffect:
2761 case Intrinsic::experimental_noalias_scope_decl:
2762 case Intrinsic::assume:
2763 case Intrinsic::lifetime_start:
2764 case Intrinsic::lifetime_end:
2765 if (TheLoop->hasLoopInvariantOperands(&I))
2766 AddToWorklistIfAllowed(&I);
2767 break;
2768 default:
2769 break;
2770 }
2771 }
2772
2773 if (auto *EVI = dyn_cast<ExtractValueInst>(&I)) {
2774 if (IsOutOfScope(EVI->getAggregateOperand())) {
2775 AddToWorklistIfAllowed(EVI);
2776 continue;
2777 }
2778 // Only ExtractValue instructions where the aggregate value comes from a
2779 // call are allowed to be non-uniform.
2780 assert(isa<CallInst>(EVI->getAggregateOperand()) &&
2781 "Expected aggregate value to be call return value");
2782 }
2783
2784 // If there's no pointer operand, there's nothing to do.
2785 auto *Ptr = getLoadStorePointerOperand(&I);
2786 if (!Ptr)
2787 continue;
2788
2789 // If the pointer can be proven to be uniform, always add it to the
2790 // worklist.
2791 if (isa<Instruction>(Ptr) && isUniform(Ptr, VF))
2792 AddToWorklistIfAllowed(cast<Instruction>(Ptr));
2793
2794 if (IsUniformMemOpUse(&I))
2795 AddToWorklistIfAllowed(&I);
2796
2797 if (IsVectorizedMemAccessUse(&I, Ptr))
2798 HasUniformUse.insert(Ptr);
2799 }
2800
2801 // Add to the worklist any operands which have *only* uniform (e.g. lane 0
2802 // demanding) users. Since loops are assumed to be in LCSSA form, this
2803 // disallows uses outside the loop as well.
2804 for (auto *V : HasUniformUse) {
2805 if (IsOutOfScope(V))
2806 continue;
2807 auto *I = cast<Instruction>(V);
2808 bool UsersAreMemAccesses = all_of(I->users(), [&](User *U) -> bool {
2809 auto *UI = cast<Instruction>(U);
2810 return TheLoop->contains(UI) && IsVectorizedMemAccessUse(UI, V);
2811 });
2812 if (UsersAreMemAccesses)
2813 AddToWorklistIfAllowed(I);
2814 }
2815
2816 // Expand Worklist in topological order: whenever a new instruction
2817 // is added , its users should be already inside Worklist. It ensures
2818 // a uniform instruction will only be used by uniform instructions.
2819 unsigned Idx = 0;
2820 while (Idx != Worklist.size()) {
2821 Instruction *I = Worklist[Idx++];
2822
2823 for (auto *OV : I->operand_values()) {
2824 // isOutOfScope operands cannot be uniform instructions.
2825 if (IsOutOfScope(OV))
2826 continue;
2827 // First order recurrence Phi's should typically be considered
2828 // non-uniform.
2829 auto *OP = dyn_cast<PHINode>(OV);
2830 if (OP && Legal->isFixedOrderRecurrence(OP))
2831 continue;
2832 // If all the users of the operand are uniform, then add the
2833 // operand into the uniform worklist.
2834 auto *OI = cast<Instruction>(OV);
2835 if (llvm::all_of(OI->users(), [&](User *U) -> bool {
2836 auto *J = cast<Instruction>(U);
2837 return Worklist.count(J) || IsVectorizedMemAccessUse(J, OI);
2838 }))
2839 AddToWorklistIfAllowed(OI);
2840 }
2841 }
2842
2843 // For an instruction to be added into Worklist above, all its users inside
2844 // the loop should also be in Worklist. However, this condition cannot be
2845 // true for phi nodes that form a cyclic dependence. We must process phi
2846 // nodes separately. An induction variable will remain uniform if all users
2847 // of the induction variable and induction variable update remain uniform.
2848 // The code below handles both pointer and non-pointer induction variables.
2849 BasicBlock *Latch = TheLoop->getLoopLatch();
2850 for (const auto &Induction : Legal->getInductionVars()) {
2851 auto *Ind = Induction.first;
2852 auto *IndUpdate = cast<Instruction>(Ind->getIncomingValueForBlock(Latch));
2853
2854 // Determine if all users of the induction variable are uniform after
2855 // vectorization.
2856 bool UniformInd = all_of(Ind->users(), [&](User *U) -> bool {
2857 auto *I = cast<Instruction>(U);
2858 return I == IndUpdate || !TheLoop->contains(I) || Worklist.count(I) ||
2859 IsVectorizedMemAccessUse(I, Ind);
2860 });
2861 if (!UniformInd)
2862 continue;
2863
2864 // Determine if all users of the induction variable update instruction are
2865 // uniform after vectorization.
2866 bool UniformIndUpdate = all_of(IndUpdate->users(), [&](User *U) -> bool {
2867 auto *I = cast<Instruction>(U);
2868 return I == Ind || Worklist.count(I) ||
2869 IsVectorizedMemAccessUse(I, IndUpdate);
2870 });
2871 if (!UniformIndUpdate)
2872 continue;
2873
2874 // The induction variable and its update instruction will remain uniform.
2875 AddToWorklistIfAllowed(Ind);
2876 AddToWorklistIfAllowed(IndUpdate);
2877 }
2878
2879 Uniforms[VF].insert_range(Worklist);
2880}
2881
2882FixedScalableVFPair
2884 // Make sure once we return PartialAliasMaskingStatus is not "NotDecided".
2885 scope_exit EnsureAliasMaskingStatusIsDecidedOnReturn([this] {
2886 if (PartialAliasMaskingStatus == AliasMaskingStatus::NotDecided)
2887 PartialAliasMaskingStatus = AliasMaskingStatus::Disabled;
2888 });
2889
2890 // For outer loops, use simple type-based heuristic VF. No cost model or
2891 // memory dependence analysis is available.
2892 if (!TheLoop->isInnermost()) {
2893 return Config.computeVPlanOuterloopVF(UserVF);
2894 }
2895
2896 if (Legal->getRuntimePointerChecking()->Need && TTI.hasBranchDivergence()) {
2897 // TODO: It may be useful to do since it's still likely to be dynamically
2898 // uniform if the target can skip.
2900 "Not inserting runtime ptr check for divergent target",
2901 "runtime pointer checks needed. Not enabled for divergent target",
2902 "CantVersionLoopWithDivergentTarget", ORE, TheLoop);
2904 }
2905
2906 ScalarEvolution *SE = PSE.getSE();
2908 unsigned MaxTC = PSE.getSmallConstantMaxTripCount();
2909 if (!MaxTC && EpilogueLoweringStatus == CM_EpilogueAllowed)
2911 LLVM_DEBUG(dbgs() << "LV: Found trip count: " << TC << '\n');
2912 if (TC != ElementCount::getFixed(MaxTC))
2913 LLVM_DEBUG(dbgs() << "LV: Found maximum trip count: " << MaxTC << '\n');
2914 if (TC.isScalar()) {
2916 "Single iteration (non) loop",
2917 "loop trip count is one, irrelevant for vectorization",
2918 "SingleIterationLoop", ORE, TheLoop);
2920 }
2921
2922 // If BTC matches the widest induction type and is -1 then the trip count
2923 // computation will wrap to 0 and the vector trip count will be 0. Do not try
2924 // to vectorize.
2925 const SCEV *BTC = SE->getBackedgeTakenCount(TheLoop);
2926 if (!isa<SCEVCouldNotCompute>(BTC) &&
2927 BTC->getType()->getScalarSizeInBits() >=
2928 Legal->getWidestInductionType()->getScalarSizeInBits() &&
2930 SE->getMinusOne(BTC->getType()))) {
2932 "Trip count computation wrapped",
2933 "backedge-taken count is -1, loop trip count wrapped to 0",
2934 "TripCountWrapped", ORE, TheLoop);
2936 }
2937
2938 assert(WideningDecisions.empty() && Uniforms.empty() && Scalars.empty() &&
2939 "No cost-modeling decisions should have been taken at this point");
2940
2941 switch (EpilogueLoweringStatus) {
2942 case CM_EpilogueAllowed:
2943 return Config.computeFeasibleMaxVF(MaxTC, UserVF, UserIC, false,
2946 [[fallthrough]];
2948 LLVM_DEBUG(dbgs() << "LV: tail-folding hint/switch found.\n"
2949 << "LV: Not allowing epilogue, creating tail-folded "
2950 << "vector loop.\n");
2951 break;
2953 // fallthrough as a special case of OptForSize
2955 if (EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize)
2956 LLVM_DEBUG(dbgs() << "LV: Not allowing epilogue due to -Os/-Oz.\n");
2957 else
2958 LLVM_DEBUG(dbgs() << "LV: Not allowing epilogue due to low trip "
2959 << "count.\n");
2960
2961 // Bail if runtime checks are required, which are not good when optimising
2962 // for size.
2963 if (Config.runtimeChecksRequired())
2965
2966 break;
2967 }
2968
2969 // Now try the tail folding
2970
2971 // Invalidate interleave groups that require an epilogue if we can't mask
2972 // the interleave-group.
2974 // Note: There is no need to invalidate any cost modeling decisions here, as
2975 // none were taken so far (see assertion above).
2976 InterleaveInfo.invalidateGroupsRequiringScalarEpilogue();
2977 }
2978
2979 FixedScalableVFPair MaxFactors = Config.computeFeasibleMaxVF(
2980 MaxTC, UserVF, UserIC, true, requiresScalarEpilogue(true));
2981
2982 // Avoid tail folding if the trip count is known to be a multiple of any VF
2983 // we choose.
2984 std::optional<uint64_t> MaxPowerOf2RuntimeVF =
2985 MaxFactors.FixedVF.getFixedValue();
2986 if (MaxFactors.ScalableVF) {
2987 if (std::optional<uint64_t> MaxRuntimeScalableVF =
2989 MaxPowerOf2RuntimeVF =
2990 std::max(*MaxPowerOf2RuntimeVF, *MaxRuntimeScalableVF);
2991 else
2992 MaxPowerOf2RuntimeVF = std::nullopt; // Stick with tail-folding for now.
2993 }
2994
2995 auto NoScalarEpilogueNeeded = [this, &UserIC](uint64_t MaxRuntimeVF) {
2996 // Return false if the loop is neither a single-latch-exit loop nor an
2997 // early-exit loop as tail-folding is not supported in that case.
2998 if (TheLoop->getExitingBlock() != TheLoop->getLoopLatch() &&
2999 !Legal->hasUncountableEarlyExit())
3000 return false;
3001 uint64_t MaxVFtimesIC = MaxRuntimeVF * std::max<uint64_t>(UserIC, 1);
3002 ScalarEvolution *SE = PSE.getSE();
3003 // Calling getSymbolicMaxBackedgeTakenCount enables support for loops
3004 // with uncountable exits. For countable loops, the symbolic maximum must
3005 // remain identical to the known back-edge taken count.
3006 const SCEV *BackedgeTakenCount = PSE.getSymbolicMaxBackedgeTakenCount();
3007 assert((Legal->hasUncountableEarlyExit() ||
3008 BackedgeTakenCount == PSE.getBackedgeTakenCount()) &&
3009 "Invalid loop count");
3010 const SCEV *ExitCount = SE->getAddExpr(
3011 BackedgeTakenCount, SE->getOne(BackedgeTakenCount->getType()));
3012 const SCEV *Rem = SE->getURemExpr(
3013 SE->applyLoopGuards(ExitCount, TheLoop),
3014 SE->getConstant(BackedgeTakenCount->getType(), MaxVFtimesIC));
3015 return Rem->isZero();
3016 };
3017
3018 if (MaxPowerOf2RuntimeVF > 0u) {
3019 assert((UserVF.isNonZero() || isPowerOf2_64(*MaxPowerOf2RuntimeVF)) &&
3020 "MaxFixedVF must be a power of 2");
3021 if (NoScalarEpilogueNeeded(*MaxPowerOf2RuntimeVF)) {
3022 // Accept MaxFixedVF if we do not have a tail.
3023 LLVM_DEBUG(dbgs() << "LV: No tail will remain for any chosen VF.\n");
3024 return MaxFactors;
3025 }
3026 }
3027
3028 auto ExpectedTC = getSmallBestKnownTC(PSE, TheLoop);
3029 if (ExpectedTC && ExpectedTC->isFixed() &&
3030 ExpectedTC->getFixedValue() <=
3031 TTI.getMinTripCountTailFoldingThreshold()) {
3032 // If we have a low-trip-count, and the fixed-width VF is known to divide
3033 // the trip count the fixed-width factor in preference to allow the
3034 // generation of a non-predicated loop.
3035 if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
3036 NoScalarEpilogueNeeded(MaxFactors.FixedVF.getFixedValue())) {
3037 LLVM_DEBUG(dbgs() << "LV: Picking a fixed-width so that no tail will "
3038 "remain for any chosen VF.\n");
3039 MaxFactors.ScalableVF = ElementCount::getScalable(0);
3040 return MaxFactors;
3041 }
3042
3043 // Allow cases where the ExactTC == (VF * IC) + 1.
3044 //
3045 // This produces 1 vector iteration, and 1 scalar iteration with no
3046 // remainder. Later passes will eliminate the loop and leave straight-line
3047 // code as the both iteration counts are statically known.
3048 //
3049 // If a function is marked as minsize/optsize or OptForSize is set, do not
3050 // allow this form of transformation as this will increase CodeSize.
3051 //
3052 // For loops with small bodies, the cost model is not currently reliable
3053 // enough to accurately determine if vectorization is beneficial.
3054 unsigned EffectiveIC = UserIC > 0 ? UserIC : 1;
3055 unsigned MaxVFForTC = llvm::bit_floor(TC.getFixedValue());
3056 if (TC.getFixedValue() - MaxVFForTC == 1 && MaxVFForTC / EffectiveIC > 1 &&
3057 MaxVFForTC <= (MaxFactors.FixedVF.getFixedValue() * EffectiveIC) &&
3058 !Config.OptForSize) {
3059 unsigned NumOfInstructions = llvm::sum_of(
3060 llvm::map_range(TheLoop->blocks(),
3061 [](BasicBlock *BB) { return BB->size(); }),
3062 unsigned(0));
3063 if (NumOfInstructions > LowTripCountLoopBodySizeLimit) {
3064 unsigned VF = MaxVFForTC / EffectiveIC;
3065 LLVM_DEBUG(dbgs() << "LV: Picking MaxVF=" << VF
3066 << " with 1 scalar iteration remaining.\n");
3067 MaxFactors.FixedVF = ElementCount::getFixed(VF);
3068 MaxFactors.ScalableVF = ElementCount::getScalable(0);
3069 return MaxFactors;
3070 }
3071 }
3072
3074 "The trip count is below the minial threshold value.",
3075 "loop trip count is too low, avoiding vectorization", "LowTripCount",
3076 ORE, TheLoop);
3078 }
3079
3080 // If we don't know the precise trip count, or if the trip count that we
3081 // found modulo the vectorization factor is not zero, try to fold the tail
3082 // by masking.
3083 // FIXME: look for a smaller MaxVF that does divide TC rather than masking.
3084 bool ContainsScalableVF = MaxFactors.ScalableVF.isNonZero();
3085 setTailFoldingStyle(ContainsScalableVF, UserIC);
3086 if (foldTailByMasking()) {
3087 if (foldTailWithEVL()) {
3088 LLVM_DEBUG(
3089 dbgs()
3090 << "LV: tail is folded with EVL, forcing unroll factor to be 1. Will "
3091 "try to generate VP Intrinsics with scalable vector "
3092 "factors only.\n");
3093 // Tail folded loop using VP intrinsics restricts the VF to be scalable
3094 // for now.
3095 // TODO: extend it for fixed vectors, if required.
3096 assert(ContainsScalableVF && "Expected scalable vector factor.");
3097
3098 MaxFactors.FixedVF = ElementCount::getFixed(1);
3099 } else {
3101 }
3102 return MaxFactors;
3103 }
3104
3105 // If there was a tail-folding hint/switch, but we can't fold the tail by
3106 // masking, fallback to a vectorization with an epilogue.
3107 if (EpilogueLoweringStatus == CM_EpilogueNotNeededFoldTail) {
3108 LLVM_DEBUG(dbgs() << "LV: Cannot fold tail by masking: vectorize with an "
3109 "epilogue instead.\n");
3110 EpilogueLoweringStatus = CM_EpilogueAllowed;
3111 return MaxFactors;
3112 }
3113
3114 if (EpilogueLoweringStatus == CM_EpilogueNotAllowedFoldTail) {
3115 LLVM_DEBUG(dbgs() << "LV: Can't fold tail by masking: don't vectorize\n");
3117 }
3118
3119 if (TC.isZero()) {
3121 "unable to calculate the loop count due to complex control flow",
3122 "UnknownLoopCountComplexCFG", ORE, TheLoop);
3124 }
3125
3127 "Cannot optimize for size and vectorize at the same time.",
3128 "cannot optimize for size and vectorize at the same time. "
3129 "Enable vectorization of this loop with '#pragma clang loop "
3130 "vectorize(enable)' when compiling with -Os/-Oz",
3131 "NoTailLoopWithOptForSize", ORE, TheLoop);
3133}
3134
3137 using RecipeVFPair = std::pair<VPRecipeBase *, ElementCount>;
3138 SmallVector<RecipeVFPair> InvalidCosts;
3139 for (const auto &Plan : VPlans) {
3140 for (ElementCount VF : Plan->vectorFactors()) {
3141 // The VPlan-based cost model is designed for computing vector cost.
3142 // Querying VPlan-based cost model with a scarlar VF will cause some
3143 // errors because we expect the VF is vector for most of the widen
3144 // recipes.
3145 if (VF.isScalar())
3146 continue;
3147
3148 VPCostContext CostCtx(*TLI, *Plan, *CM, Config,
3149 /*ReusePrintingSlotTracker=*/true);
3150 precomputeCosts(*Plan, VF, CostCtx);
3151 auto Iter = vp_depth_first_deep(Plan->getVectorLoopRegion()->getEntry());
3153 for (auto &R : *VPBB) {
3154 if (!R.cost(VF, CostCtx).isValid())
3155 InvalidCosts.emplace_back(&R, VF);
3156 }
3157 }
3158 }
3159 }
3160 if (InvalidCosts.empty())
3161 return;
3162
3163 // Emit a report of VFs with invalid costs in the loop.
3164
3165 // Group the remarks per recipe, keeping the recipe order from InvalidCosts.
3167 unsigned I = 0;
3168 for (auto &Pair : InvalidCosts)
3169 if (Numbering.try_emplace(Pair.first, I).second)
3170 ++I;
3171
3172 // Sort the list, first on recipe(number) then on VF.
3173 sort(InvalidCosts, [&Numbering](RecipeVFPair &A, RecipeVFPair &B) {
3174 unsigned NA = Numbering[A.first];
3175 unsigned NB = Numbering[B.first];
3176 if (NA != NB)
3177 return NA < NB;
3178 return ElementCount::isKnownLT(A.second, B.second);
3179 });
3180
3181 // For a list of ordered recipe-VF pairs:
3182 // [(load, VF1), (load, VF2), (store, VF1)]
3183 // group the recipes together to emit separate remarks for:
3184 // load (VF1, VF2)
3185 // store (VF1)
3186 auto Tail = ArrayRef<RecipeVFPair>(InvalidCosts);
3187 auto Subset = ArrayRef<RecipeVFPair>();
3188 do {
3189 if (Subset.empty())
3190 Subset = Tail.take_front(1);
3191
3192 VPRecipeBase *R = Subset.front().first;
3193
3194 unsigned Opcode =
3196 .Case([](const VPHeaderPHIRecipe *R) { return Instruction::PHI; })
3197 .Case(
3198 [](const VPWidenStoreRecipe *R) { return Instruction::Store; })
3199 .Case([](const VPWidenLoadRecipe *R) { return Instruction::Load; })
3200 .Case<VPWidenCallRecipe, VPWidenIntrinsicRecipe>(
3201 [](const auto *R) { return Instruction::Call; })
3204 [](const auto *R) { return R->getOpcode(); })
3205 .Case([](const VPInterleaveRecipe *R) {
3206 return R->getStoredValues().empty() ? Instruction::Load
3207 : Instruction::Store;
3208 })
3209 .Case([](const VPReductionRecipe *R) {
3210 return RecurrenceDescriptor::getOpcode(R->getRecurrenceKind());
3211 });
3212
3213 // If the next recipe is different, or if there are no other pairs,
3214 // emit a remark for the collated subset. e.g.
3215 // [(load, VF1), (load, VF2))]
3216 // to emit:
3217 // remark: invalid costs for 'load' at VF=(VF1, VF2)
3218 if (Subset == Tail || Tail[Subset.size()].first != R) {
3219 std::string OutString;
3220 raw_string_ostream OS(OutString);
3221 assert(!Subset.empty() && "Unexpected empty range");
3222 OS << "Recipe with invalid costs prevented vectorization at VF=(";
3223 for (const auto &Pair : Subset)
3224 OS << (Pair.second == Subset.front().second ? "" : ", ") << Pair.second;
3225 OS << "):";
3226 if (Opcode == Instruction::Call) {
3227 StringRef Name = "";
3228 if (auto *Int = dyn_cast<VPWidenIntrinsicRecipe>(R)) {
3229 Name = Int->getIntrinsicName();
3230 } else {
3231 auto *WidenCall = dyn_cast<VPWidenCallRecipe>(R);
3232 Function *CalledFn =
3233 WidenCall ? WidenCall->getCalledScalarFunction()
3234 : cast<Function>(R->getOperand(R->getNumOperands() - 1)
3235 ->getLiveInIRValue());
3236 Name = CalledFn->getName();
3237 }
3238 OS << " call to " << Name;
3239 } else
3240 OS << " " << Instruction::getOpcodeName(Opcode);
3241 reportVectorizationInfo(OutString, "InvalidCost", ORE, OrigLoop, nullptr,
3242 R->getDebugLoc());
3243 Tail = Tail.drop_front(Subset.size());
3244 Subset = {};
3245 } else
3246 // Grow the subset by one element
3247 Subset = Tail.take_front(Subset.size() + 1);
3248 } while (!Tail.empty());
3249}
3250
3251/// Check if any recipe of \p Plan will generate a vector value, which will be
3252/// assigned a vector register.
3254 const TargetTransformInfo &TTI) {
3255 assert(VF.isVector() && "Checking a scalar VF?");
3256 DenseSet<VPRecipeBase *> EphemeralRecipes;
3257 collectEphemeralRecipesForVPlan(Plan, EphemeralRecipes);
3258 // Set of already visited types.
3259 DenseSet<Type *> Visited;
3262 for (VPRecipeBase &R : *VPBB) {
3263 if (EphemeralRecipes.contains(&R))
3264 continue;
3265 // Continue early if the recipe is considered to not produce a vector
3266 // result. Note that this includes VPInstruction where some opcodes may
3267 // produce a vector, to preserve existing behavior as VPInstructions model
3268 // aspects not directly mapped to existing IR instructions.
3269 switch (R.getVPRecipeID()) {
3270 case VPRecipeBase::VPDerivedIVSC:
3271 case VPRecipeBase::VPScalarIVStepsSC:
3272 case VPRecipeBase::VPReplicateSC:
3273 case VPRecipeBase::VPInstructionSC:
3274 case VPRecipeBase::VPCurrentIterationPHISC:
3275 case VPRecipeBase::VPVectorPointerSC:
3276 case VPRecipeBase::VPVectorEndPointerSC:
3277 case VPRecipeBase::VPExpandSCEVSC:
3278 case VPRecipeBase::VPPredInstPHISC:
3279 case VPRecipeBase::VPBranchOnMaskSC:
3280 continue;
3281 case VPRecipeBase::VPReductionSC:
3282 case VPRecipeBase::VPActiveLaneMaskPHISC:
3283 case VPRecipeBase::VPWidenCallSC:
3284 case VPRecipeBase::VPWidenCanonicalIVSC:
3285 case VPRecipeBase::VPWidenCastSC:
3286 case VPRecipeBase::VPWidenGEPSC:
3287 case VPRecipeBase::VPWidenIntrinsicSC:
3288 case VPRecipeBase::VPWidenMemIntrinsicSC:
3289 case VPRecipeBase::VPWidenSC:
3290 case VPRecipeBase::VPBlendSC:
3291 case VPRecipeBase::VPFirstOrderRecurrencePHISC:
3292 case VPRecipeBase::VPHistogramSC:
3293 case VPRecipeBase::VPWidenPHISC:
3294 case VPRecipeBase::VPWidenIntOrFpInductionSC:
3295 case VPRecipeBase::VPWidenPointerInductionSC:
3296 case VPRecipeBase::VPReductionPHISC:
3297 case VPRecipeBase::VPInterleaveEVLSC:
3298 case VPRecipeBase::VPInterleaveSC:
3299 case VPRecipeBase::VPWidenLoadEVLSC:
3300 case VPRecipeBase::VPWidenLoadSC:
3301 case VPRecipeBase::VPWidenStoreEVLSC:
3302 case VPRecipeBase::VPWidenStoreSC:
3303 break;
3304 default:
3305 llvm_unreachable("unhandled recipe");
3306 }
3307
3308 auto WillGenerateTargetVectors = [&TTI, VF](Type *VectorTy) {
3309 unsigned NumLegalParts = TTI.getNumberOfParts(VectorTy);
3310 if (!NumLegalParts)
3311 return false;
3312 if (VF.isScalable()) {
3313 // <vscale x 1 x iN> is assumed to be profitable over iN because
3314 // scalable registers are a distinct register class from scalar
3315 // ones. If we ever find a target which wants to lower scalable
3316 // vectors back to scalars, we'll need to update this code to
3317 // explicitly ask TTI about the register class uses for each part.
3318 return NumLegalParts <= VF.getKnownMinValue();
3319 }
3320 // Two or more elements that share a register - are vectorized.
3321 return NumLegalParts < VF.getFixedValue();
3322 };
3323
3324 // If no def nor is a store, e.g., branches, continue - no value to check.
3325 if (R.getNumDefinedValues() == 0 &&
3327 continue;
3328 // For multi-def recipes, currently only interleaved loads, suffice to
3329 // check first def only.
3330 // For stores check their stored value; for interleaved stores suffice
3331 // the check first stored value only. In all cases this is the second
3332 // operand.
3333 VPValue *ToCheck =
3334 R.getNumDefinedValues() >= 1 ? R.getVPValue(0) : R.getOperand(1);
3335 Type *ScalarTy = ToCheck->getScalarType();
3336 if (!Visited.insert({ScalarTy}).second)
3337 continue;
3338 Type *WideTy = toVectorizedTy(ScalarTy, VF);
3339 if (any_of(getContainedTypes(WideTy), WillGenerateTargetVectors))
3340 return true;
3341 }
3342 }
3343
3344 return false;
3345}
3346
3347static bool hasReplicatorRegion(VPlan &Plan) {
3349 Plan.getVectorLoopRegion()->getEntry())),
3350 [](auto *VPRB) { return VPRB->isReplicator(); });
3351}
3352
3353/// Returns true if the VPlan contains a VPReductionPHIRecipe with
3354/// FindLast recurrence kind.
3355static bool hasFindLastReductionPhi(VPlan &Plan) {
3358 [](VPReductionPHIRecipe &RedPhi) {
3359 return RecurrenceDescriptor::isFindLastRecurrenceKind(
3360 RedPhi.getRecurrenceKind());
3361 });
3362}
3363
3364/// Determine how to lower the epilogue for the vector epilogue loop.
3365/// Check if there are any conflicts that prevent tail-folding the epilogue.
3366/// \return CM_EpilogueNotNeededFoldTail if epilogue tail-folding is possible,
3367/// otherwise CM_EpilogueAllowed.
3369 const LoopVectorizationCostModel &MainCM, const Loop *L,
3372 // Epilogue TF is only enabled when explicitly requested via command line.
3373 if (!EpilogueTailFoldingPolicy.getNumOccurrences() ||
3375 return CM_EpilogueAllowed;
3376
3379 "Options conflict, epilogue vectorization is disallowed while "
3380 "epilogue tail-folding allowed!",
3381 "UnsupportedEpilogueTailFoldingPolicy", ORE, L);
3382 return CM_EpilogueAllowed;
3383 }
3384
3385 if (!Hints.getWidth() || !hasForcedEpilogueVF()) {
3386 reportVectorizationInfo("For now, epilogue tail-folding can't be "
3387 "applied without forced main/epilogue loop VF",
3388 "UnsupportedEpilogueTailFoldingPolicy", ORE, L);
3389 return CM_EpilogueAllowed;
3390 }
3391
3393 reportVectorizationInfo("For now, epilogue tail-folding can't be applied "
3394 "when VF of the main loop <= VF of the epilogue",
3395 "UnsupportedEpilogueTailFoldingPolicy", ORE, L);
3396 return CM_EpilogueAllowed;
3397 }
3398
3399 if (!L->isInnermost()) {
3401 "Epilogue tail-folding is not supported for outer loop",
3402 "InvalidTailFoldedEpilogue", ORE, L);
3403 return CM_EpilogueAllowed;
3404 }
3405
3406 // If scalar epilogue is explicitly required, we can't apply TF.
3407 if (MainCM.requiresScalarEpilogue(/*IsVectorizing*/ true)) {
3409 "Epilogue tail-folding can't be applied because scalar epilogue is "
3410 "required. Fall back to a normal epilogue",
3411 "InvalidTailFoldedEpilogue", ORE, L);
3412 return CM_EpilogueAllowed;
3413 }
3414
3415 // If having epilogue is NOT allowed, then no epilogue to apply TF for.
3416 if (!MainCM.isEpilogueAllowed()) {
3417 reportVectorizationInfo("Not applying tail-folding to the epilogue, since "
3418 "no epilogue is allowed.",
3419 "InvalidTailFoldedEpilogue", ORE, L);
3420 return CM_EpilogueAllowed;
3421 }
3422
3423 if (L->getExitingBlock() != L->getLoopLatch() ||
3426 "Epilogue tail-folding is not supported yet for early-exit loops",
3427 "InvalidTailFoldedEpilogue", ORE, L);
3428 return CM_EpilogueAllowed;
3429 }
3430
3431 // The epilogue reuses the main loop's interleave groups, so it can't be
3432 // tail-folded if the target can't mask interleaved accesses.
3433 // TODO: Add support once the epilogue has its own IAI, separate from the main
3434 // loop's.
3435 if (MainCM.InterleaveInfo.hasGroups() &&
3438 "Epilogue tail-folding is not supported with interleaved accesses "
3439 "when masking them isn't supported",
3440 "InvalidTailFoldedEpilogue", ORE, L);
3441 return CM_EpilogueAllowed;
3442 }
3443
3446 "Epilogue tail-folding is not supported with alias masking",
3447 "InvalidTailFoldedEpilogue", ORE, L);
3448 return CM_EpilogueAllowed;
3449 }
3450
3451 if (!LVL.getReductionVars().empty()) {
3453 "Epilogue tail-folding is not supported with reductions",
3454 "InvalidTailFoldedEpilogue", ORE, L);
3455 return CM_EpilogueAllowed;
3456 }
3457
3458 if (!LVL.getFixedOrderRecurrences().empty()) {
3460 "Epilogue tail-folding is not supported with fixed-order recurrence",
3461 "InvalidTailFoldedEpilogue", ORE, L);
3462 return CM_EpilogueAllowed;
3463 }
3464
3465 // We can apply tail-folding on the vectorized epilogue loop.
3467}
3468
3470 const ElementCount VF, const unsigned IC) const {
3471 // FIXME: We need a much better cost-model to take different parameters such
3472 // as register pressure, code size increase and cost of extra branches into
3473 // account. For now we apply a very crude heuristic and only consider loops
3474 // with vectorization factors larger than a certain value.
3475
3476 // Allow the target to opt out.
3477 if (!TTI.preferEpilogueVectorization(VF * IC))
3478 return false;
3479
3480 unsigned MinVFThreshold = EpilogueVectorizationMinVF.getNumOccurrences() > 0
3482 : TTI.getEpilogueVectorizationMinVF();
3483 return estimateElementCount(VF * IC, getVScaleForTuning()) >= MinVFThreshold;
3484}
3485
3487 VPlan &MainPlan, ElementCount MainLoopVF, unsigned IC,
3488 bool ScalarEpilogueAllowed) {
3490 LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization is disabled.\n");
3491 return nullptr;
3492 }
3493
3494 if (!ScalarEpilogueAllowed) {
3495 LLVM_DEBUG(dbgs() << "LEV: Unable to vectorize epilogue because no "
3496 "epilogue is allowed.\n");
3497 return nullptr;
3498 }
3499
3500 if (vputils::findIncomingAliasMask(MainPlan)) {
3501 LLVM_DEBUG(
3502 dbgs()
3503 << "LEV: Epilogue vectorization not supported with alias masking.\n");
3504 return nullptr;
3505 }
3506
3507 // Not really a cost consideration, but check for unsupported cases here to
3508 // simplify the logic.
3509 if (!isCandidateForEpilogueVectorization(MainPlan)) {
3510 LLVM_DEBUG(dbgs() << "LEV: Unable to vectorize epilogue because the loop "
3511 "is not a supported candidate.\n");
3512 return nullptr;
3513 }
3514
3515 if (hasForcedEpilogueVF()) {
3517 Config.getVScaleForTuning()) >=
3518 IC * estimateElementCount(MainLoopVF, Config.getVScaleForTuning())) {
3519 // Note that the main loop leaves IC * MainLoopVF iterations iff a scalar
3520 // epilogue is required, but then the epilogue loop also requires a scalar
3521 // epilogue.
3522 LLVM_DEBUG(dbgs() << "LEV: Forced epilogue VF results in dead epilogue "
3523 "vector loop, skipping vectorizing epilogue.\n");
3524 return nullptr;
3525 }
3526
3527 LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization factor is forced.\n");
3529 std::unique_ptr<VPlan> Clone(
3531 Clone->setVF(EpilogueVectorizationForceVF);
3532 return Clone;
3533 }
3534
3535 LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization forced factor is not "
3536 "viable.\n");
3537 return nullptr;
3538 }
3539
3540 if (OrigLoop->getHeader()->getParent()->hasOptSize()) {
3541 LLVM_DEBUG(
3542 dbgs() << "LEV: Epilogue vectorization skipped due to opt for size.\n");
3543 return nullptr;
3544 }
3545
3546 if (!Config.isEpilogueVectorizationProfitable(MainLoopVF, IC)) {
3547 LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization is not profitable for "
3548 "this loop\n");
3549 return nullptr;
3550 }
3551
3552 // Check if a plan's vector loop processes fewer iterations than VF (e.g. when
3553 // interleave groups have been narrowed) narrowInterleaveGroups) and return
3554 // the adjusted, effective VF.
3555 using namespace VPlanPatternMatch;
3556 auto GetEffectiveVF = [](VPlan &Plan, ElementCount VF) -> ElementCount {
3557 auto *Exiting = Plan.getVectorLoopRegion()->getExitingBasicBlock();
3558 if (match(&Exiting->back(),
3559 m_BranchOnCount(m_Add(m_CanonicalIV(), m_Specific(&Plan.getUF())),
3560 m_VPValue())))
3561 return ElementCount::get(1, VF.isScalable());
3562 return VF;
3563 };
3564
3565 // Check if the main loop processes fewer than MainLoopVF elements per
3566 // iteration (e.g. due to narrowing interleave groups). Adjust MainLoopVF
3567 // as needed.
3568 MainLoopVF = GetEffectiveVF(MainPlan, MainLoopVF);
3569
3570 // If MainLoopVF = vscale x 2, and vscale is expected to be 4, then we know
3571 // the main loop handles 8 lanes per iteration. We could still benefit from
3572 // vectorizing the epilogue loop with VF=4.
3573 ElementCount EstimatedRuntimeVF = ElementCount::getFixed(
3574 estimateElementCount(MainLoopVF, Config.getVScaleForTuning()));
3575
3576 Type *TCType = Legal->getWidestInductionType();
3577 const SCEV *RemainingIterations = nullptr;
3578 unsigned MaxTripCount = 0;
3579 const SCEV *TC = vputils::getSCEVExprForVPValue(MainPlan.getTripCount(), PSE);
3580 assert(!isa<SCEVCouldNotCompute>(TC) && "Trip count SCEV must be computable");
3581 const SCEV *KnownMinTC;
3582 bool ScalableTC = match(TC, m_scev_c_Mul(m_SCEV(KnownMinTC), m_SCEVVScale()));
3583 bool ScalableRemIter = false;
3584 ScalarEvolution &SE = *PSE.getSE();
3585 // Use versions of TC and VF in which both are either scalable or fixed.
3586 if (ScalableTC == MainLoopVF.isScalable()) {
3587 ScalableRemIter = ScalableTC;
3588 RemainingIterations =
3589 SE.getURemExpr(TC, SE.getElementCount(TCType, MainLoopVF * IC));
3590 } else if (ScalableTC) {
3591 const SCEV *EstimatedTC = SE.getMulExpr(
3592 KnownMinTC,
3593 SE.getConstant(TCType, Config.getVScaleForTuning().value_or(1)));
3594 RemainingIterations = SE.getURemExpr(
3595 EstimatedTC, SE.getElementCount(TCType, MainLoopVF * IC));
3596 } else
3597 RemainingIterations =
3598 SE.getURemExpr(TC, SE.getElementCount(TCType, EstimatedRuntimeVF * IC));
3599
3600 // No iterations left to process in the epilogue.
3601 if (RemainingIterations->isZero())
3602 return nullptr;
3603
3604 if (MainLoopVF.isFixed()) {
3605 MaxTripCount = MainLoopVF.getFixedValue() * IC - 1;
3606 if (SE.isKnownPredicate(CmpInst::ICMP_ULT, RemainingIterations,
3607 SE.getConstant(TCType, MaxTripCount))) {
3608 MaxTripCount = SE.getUnsignedRangeMax(RemainingIterations).getZExtValue();
3609 }
3610 LLVM_DEBUG(dbgs() << "LEV: Maximum Trip Count for Epilogue: "
3611 << MaxTripCount << "\n");
3612 }
3613
3614 auto SkipVF = [&](const SCEV *VF, const SCEV *RemIter) -> bool {
3615 return SE.isKnownPredicate(CmpInst::ICMP_UGT, VF, RemIter);
3616 };
3618 VPlan *BestPlan = nullptr;
3619 for (auto &NextVF : ProfitableVFs) {
3620 // Skip candidate VFs without a corresponding VPlan.
3621 if (!hasPlanWithVF(NextVF.Width))
3622 continue;
3623
3624 VPlan &CurrentPlan = getPlanFor(NextVF.Width);
3625 ElementCount EffectiveVF = GetEffectiveVF(CurrentPlan, NextVF.Width);
3626 // Skip fixed vector VFs > than the estimated runtime VF, or any VF > than
3627 // the VF of the main loop.
3628 if ((!EffectiveVF.isScalable() && MainLoopVF.isScalable() &&
3629 ElementCount::isKnownGT(EffectiveVF, EstimatedRuntimeVF)) ||
3630 ElementCount::isKnownGT(EffectiveVF, MainLoopVF))
3631 continue;
3632
3633 // If EffectiveVF is greater than the number of remaining iterations, the
3634 // epilogue loop would be dead. Skip such factors. If the epilogue plan
3635 // also has narrowed interleave groups, use the effective VF since
3636 // the epilogue step will be reduced to its IC.
3637 // TODO: We should also consider comparing against a scalable
3638 // RemainingIterations when SCEV be able to evaluate non-canonical
3639 // vscale-based expressions.
3640 if (!ScalableRemIter) {
3641 // Handle the case where EffectiveVF and RemainingIterations are in
3642 // different numerical spaces.
3643 if (EffectiveVF.isScalable())
3644 EffectiveVF = ElementCount::getFixed(
3645 estimateElementCount(EffectiveVF, Config.getVScaleForTuning()));
3646 if (SkipVF(SE.getElementCount(TCType, EffectiveVF), RemainingIterations))
3647 continue;
3648 }
3649
3650 if (Result.Width.isScalar() ||
3651 isMoreProfitable(NextVF, Result, MaxTripCount,
3652 !MainPlan.hasTailFolded(),
3653 /*IsEpilogue*/ true)) {
3654 Result = NextVF;
3655 BestPlan = &CurrentPlan;
3656 }
3657 }
3658
3659 if (!BestPlan)
3660 return nullptr;
3661
3662 LLVM_DEBUG(dbgs() << "LEV: Vectorizing epilogue loop with VF = "
3663 << Result.Width << "\n");
3664 std::unique_ptr<VPlan> Clone(BestPlan->duplicate());
3665 Clone->setVF(Result.Width);
3666 return Clone;
3667}
3668
3669unsigned
3671 InstructionCost LoopCost) {
3672 // -- The interleave heuristics --
3673 // We interleave the loop in order to expose ILP and reduce the loop overhead.
3674 // There are many micro-architectural considerations that we can't predict
3675 // at this level. For example, frontend pressure (on decode or fetch) due to
3676 // code size, or the number and capabilities of the execution ports.
3677 //
3678 // We use the following heuristics to select the interleave count:
3679 // 1. If the code has reductions, then we interleave to break the cross
3680 // iteration dependency.
3681 // 2. If the loop is really small, then we interleave to reduce the loop
3682 // overhead.
3683 // 3. We don't interleave if we think that we will spill registers to memory
3684 // due to the increased register pressure.
3685
3686 // Do not interleave tail-folded loops, as the overhead of multiple
3687 // instructions to calculate the predicate is likely not beneficial.
3688 // If an epilogue is not allowed for any other reason, do not interleave.
3689 if (!CM->isEpilogueAllowed())
3690 return 1;
3691
3694 LLVM_DEBUG(dbgs() << "LV: Loop requires variable-length step. "
3695 "Unroll factor forced to be 1.\n");
3696 return 1;
3697 }
3698
3699 // We used the distance for the interleave count.
3700 if (!Legal->isSafeForAnyVectorWidth())
3701 return 1;
3702
3703 // We don't attempt to perform interleaving for loops with uncountable early
3704 // exits because the VPInstruction::AnyOf code cannot currently handle
3705 // multiple parts.
3706 if (Plan.hasEarlyExit())
3707 return 1;
3708
3709 const bool HasReductions =
3712
3713 // FIXME: implement interleaving for FindLast transform correctly.
3714 if (hasFindLastReductionPhi(Plan))
3715 return 1;
3716
3717 VPRegisterUsage R = calculateRegisterUsageForPlan(Plan, {VF}, TTI)[0];
3718
3719 // If we did not calculate the cost for VF (because the user selected the VF)
3720 // then we calculate the cost of VF here.
3721 if (LoopCost == 0) {
3722 if (VF.isScalar())
3723 LoopCost = CM->expectedCost(VF);
3724 else
3725 LoopCost = cost(Plan, VF, &R);
3726 assert(LoopCost.isValid() && "Expected to have chosen a VF with valid cost");
3727
3728 // Loop body is free and there is no need for interleaving.
3729 if (LoopCost == 0)
3730 return 1;
3731 }
3732
3733 // We divide by these constants so assume that we have at least one
3734 // instruction that uses at least one register.
3735 for (auto &Pair : R.MaxLocalUsers) {
3736 Pair.second = std::max(Pair.second, 1U);
3737 }
3738
3739 // We calculate the interleave count using the following formula.
3740 // Subtract the number of loop invariants from the number of available
3741 // registers. These registers are used by all of the interleaved instances.
3742 // Next, divide the remaining registers by the number of registers that is
3743 // required by the loop, in order to estimate how many parallel instances
3744 // fit without causing spills. All of this is rounded down if necessary to be
3745 // a power of two. We want power of two interleave count to simplify any
3746 // addressing operations or alignment considerations.
3747 // We also want power of two interleave counts to ensure that the induction
3748 // variable of the vector loop wraps to zero, when tail is folded by masking;
3749 // this currently happens when OptForSize, in which case IC is set to 1 above.
3750 unsigned IC = UINT_MAX;
3751
3752 for (const auto &Pair : R.MaxLocalUsers) {
3753 unsigned TargetNumRegisters = TTI.getNumberOfRegisters(Pair.first);
3754 LLVM_DEBUG(dbgs() << "LV: The target has " << TargetNumRegisters
3755 << " registers of "
3756 << TTI.getRegisterClassName(Pair.first)
3757 << " register class\n");
3758 if (VF.isScalar()) {
3759 if (ForceTargetNumScalarRegs.getNumOccurrences() > 0)
3760 TargetNumRegisters = ForceTargetNumScalarRegs;
3761 } else {
3762 if (ForceTargetNumVectorRegs.getNumOccurrences() > 0)
3763 TargetNumRegisters = ForceTargetNumVectorRegs;
3764 }
3765 unsigned MaxLocalUsers = Pair.second;
3766 unsigned LoopInvariantRegs = 0;
3767 if (R.LoopInvariantRegs.contains(Pair.first))
3768 LoopInvariantRegs = R.LoopInvariantRegs[Pair.first];
3769
3770 unsigned TmpIC = llvm::bit_floor((TargetNumRegisters - LoopInvariantRegs) /
3771 MaxLocalUsers);
3772 // Don't count the induction variable as interleaved.
3774 TmpIC = llvm::bit_floor((TargetNumRegisters - LoopInvariantRegs - 1) /
3775 std::max(1U, (MaxLocalUsers - 1)));
3776 }
3777
3778 IC = std::min(IC, TmpIC);
3779 }
3780
3781 // Clamp the interleave ranges to reasonable counts.
3782 bool HasUnorderedReductions =
3783 HasReductions &&
3786 [](VPReductionPHIRecipe &RedR) { return RedR.isOrdered(); });
3787 unsigned MaxInterleaveCount =
3788 TTI.getMaxInterleaveFactor(VF, HasUnorderedReductions);
3789 LLVM_DEBUG(dbgs() << "LV: MaxInterleaveFactor for the target is "
3790 << MaxInterleaveCount << "\n");
3791
3792 // Check if the user has overridden the max.
3793 if (VF.isScalar()) {
3794 if (ForceTargetMaxScalarInterleaveFactor.getNumOccurrences() > 0)
3795 MaxInterleaveCount = ForceTargetMaxScalarInterleaveFactor;
3796 } else {
3797 if (ForceTargetMaxVectorInterleaveFactor.getNumOccurrences() > 0)
3798 MaxInterleaveCount = ForceTargetMaxVectorInterleaveFactor;
3799 }
3800
3801 // Try to get the exact trip count, or an estimate based on profiling data or
3802 // ConstantMax from PSE, failing that.
3803 auto BestKnownTC =
3804 getSmallBestKnownTC(PSE, OrigLoop,
3805 /*CanUseConstantMax=*/true,
3806 /*CanExcludeZeroTrips=*/CM->isEpilogueAllowed());
3807
3808 // For fixed length VFs treat a scalable trip count as unknown.
3809 if (BestKnownTC && (BestKnownTC->isFixed() || VF.isScalable())) {
3810 // Re-evaluate trip counts and VFs to be in the same numerical space.
3811 unsigned AvailableTC =
3812 estimateElementCount(*BestKnownTC, Config.getVScaleForTuning());
3813 unsigned EstimatedVF =
3814 estimateElementCount(VF, Config.getVScaleForTuning());
3815
3816 // At least one iteration must be scalar when this constraint holds. So the
3817 // maximum available iterations for interleaving is one less.
3818 if (Plan.requiresScalarEpilogue())
3819 --AvailableTC;
3820
3821 unsigned InterleaveCountLB = bit_floor(std::max(
3822 1u, std::min(AvailableTC / (EstimatedVF * 2), MaxInterleaveCount)));
3823
3824 if (getSmallConstantTripCount(PSE.getSE(), OrigLoop).isNonZero()) {
3825 // If the best known trip count is exact, we select between two
3826 // prospective ICs, where
3827 //
3828 // 1) the aggressive IC is capped by the trip count divided by VF
3829 // 2) the conservative IC is capped by the trip count divided by (VF * 2)
3830 //
3831 // The final IC is selected in a way that the epilogue loop trip count is
3832 // minimized while maximizing the IC itself, so that we either run the
3833 // vector loop at least once if it generates a small epilogue loop, or
3834 // else we run the vector loop at least twice.
3835
3836 unsigned InterleaveCountUB = bit_floor(std::max(
3837 1u, std::min(AvailableTC / EstimatedVF, MaxInterleaveCount)));
3838 MaxInterleaveCount = InterleaveCountLB;
3839
3840 if (InterleaveCountUB != InterleaveCountLB) {
3841 unsigned TailTripCountUB =
3842 (AvailableTC % (EstimatedVF * InterleaveCountUB));
3843 unsigned TailTripCountLB =
3844 (AvailableTC % (EstimatedVF * InterleaveCountLB));
3845 // If both produce same scalar tail, maximize the IC to do the same work
3846 // in fewer vector loop iterations
3847 if (TailTripCountUB == TailTripCountLB)
3848 MaxInterleaveCount = InterleaveCountUB;
3849 }
3850 } else {
3851 // If trip count is an estimated compile time constant, limit the
3852 // IC to be capped by the trip count divided by VF * 2, such that the
3853 // vector loop runs at least twice to make interleaving seem profitable
3854 // when there is an epilogue loop present. Since exact Trip count is not
3855 // known we choose to be conservative in our IC estimate.
3856 MaxInterleaveCount = InterleaveCountLB;
3857 }
3858 }
3859
3860 assert(MaxInterleaveCount > 0 &&
3861 "Maximum interleave count must be greater than 0");
3862
3863 // Clamp the calculated IC to be between the 1 and the max interleave count
3864 // that the target and trip count allows.
3865 if (IC > MaxInterleaveCount)
3866 IC = MaxInterleaveCount;
3867 else
3868 // Make sure IC is greater than 0.
3869 IC = std::max(1u, IC);
3870
3871 assert(IC > 0 && "Interleave count must be greater than 0.");
3872
3873 // Interleave if we vectorized this loop and there is a reduction that could
3874 // benefit from interleaving.
3875 if (VF.isVector() && HasReductions) {
3876 LLVM_DEBUG(dbgs() << "LV: Interleaving because of reductions.\n");
3877 return IC;
3878 }
3879
3880 // For any scalar loop that either requires runtime checks or tail-folding we
3881 // are better off leaving this to the unroller. Note that if we've already
3882 // vectorized the loop we will have done the runtime check and so interleaving
3883 // won't require further checks.
3884 bool ScalarInterleavingRequiresPredication =
3885 (VF.isScalar() && any_of(OrigLoop->blocks(), [this](BasicBlock *BB) {
3886 return Legal->blockNeedsPredication(BB);
3887 }));
3888 bool ScalarInterleavingRequiresRuntimePointerCheck =
3889 (VF.isScalar() && Legal->getRuntimePointerChecking()->Need);
3890
3891 // We want to interleave small loops in order to reduce the loop overhead and
3892 // potentially expose ILP opportunities.
3893 LLVM_DEBUG(dbgs() << "LV: Loop cost is " << LoopCost << '\n'
3894 << "LV: IC is " << IC << '\n'
3895 << "LV: VF is " << VF << '\n');
3896 const bool AggressivelyInterleave =
3897 TTI.enableAggressiveInterleaving(HasReductions);
3898 if (!ScalarInterleavingRequiresRuntimePointerCheck &&
3899 !ScalarInterleavingRequiresPredication && LoopCost < SmallLoopCost) {
3900 // We assume that the cost overhead is 1 and we use the cost model
3901 // to estimate the cost of the loop and interleave until the cost of the
3902 // loop overhead is about 5% of the cost of the loop.
3903 unsigned SmallIC = std::min(IC, (unsigned)llvm::bit_floor<uint64_t>(
3904 SmallLoopCost / LoopCost.getValue()));
3905
3906 // Interleave until store/load ports (estimated by max interleave count) are
3907 // saturated.
3908 unsigned NumStores = 0;
3909 unsigned NumLoads = 0;
3912 for (VPRecipeBase &R : *VPBB) {
3914 NumLoads++;
3915 continue;
3916 }
3918 NumStores++;
3919 continue;
3920 }
3921
3922 if (auto *InterleaveR = dyn_cast<VPInterleaveRecipe>(&R)) {
3923 if (unsigned StoreOps = InterleaveR->getNumStoreOperands())
3924 NumStores += StoreOps;
3925 else
3926 NumLoads += InterleaveR->getNumDefinedValues();
3927 continue;
3928 }
3929 if (auto *RepR = dyn_cast<VPReplicateRecipe>(&R)) {
3930 NumLoads += isa<LoadInst>(RepR->getUnderlyingInstr());
3931 NumStores += isa<StoreInst>(RepR->getUnderlyingInstr());
3932 continue;
3933 }
3934 if (isa<VPHistogramRecipe>(&R)) {
3935 NumLoads++;
3936 NumStores++;
3937 continue;
3938 }
3939 }
3940 }
3941 unsigned StoresIC = IC / (NumStores ? NumStores : 1);
3942 unsigned LoadsIC = IC / (NumLoads ? NumLoads : 1);
3943
3944 // There is little point in interleaving for reductions containing selects
3945 // and compares when VF=1 since it may just create more overhead than it's
3946 // worth for loops with small trip counts. This is because we still have to
3947 // do the final reduction after the loop.
3948 bool HasSelectCmpReductions =
3949 HasReductions &&
3952 [](VPReductionPHIRecipe &RedR) {
3953 return RecurrenceDescriptor::isAnyOfRecurrenceKind(
3954 RedR.getRecurrenceKind()) ||
3955 RecurrenceDescriptor::isFindIVRecurrenceKind(
3956 RedR.getRecurrenceKind());
3957 });
3958 if (HasSelectCmpReductions) {
3959 LLVM_DEBUG(dbgs() << "LV: Not interleaving select-cmp reductions.\n");
3960 return 1;
3961 }
3962
3963 // If we have a scalar reduction (vector reductions are already dealt with
3964 // by this point), we can increase the critical path length if the loop
3965 // we're interleaving is inside another loop. For tree-wise reductions
3966 // set the limit to 2, and for ordered reductions it's best to disable
3967 // interleaving entirely.
3968 if (HasReductions && OrigLoop->getLoopDepth() > 1) {
3969 bool HasOrderedReductions =
3972 [](VPReductionPHIRecipe &RedR) { return RedR.isOrdered(); });
3973 if (HasOrderedReductions) {
3974 LLVM_DEBUG(
3975 dbgs() << "LV: Not interleaving scalar ordered reductions.\n");
3976 return 1;
3977 }
3978
3979 unsigned F = MaxNestedScalarReductionIC;
3980 SmallIC = std::min(SmallIC, F);
3981 StoresIC = std::min(StoresIC, F);
3982 LoadsIC = std::min(LoadsIC, F);
3983 }
3984
3986 std::max(StoresIC, LoadsIC) > SmallIC) {
3987 LLVM_DEBUG(
3988 dbgs() << "LV: Interleaving to saturate store or load ports.\n");
3989 return std::max(StoresIC, LoadsIC);
3990 }
3991
3992 // If there are scalar reductions and TTI has enabled aggressive
3993 // interleaving for reductions, we will interleave to expose ILP.
3994 if (VF.isScalar() && AggressivelyInterleave) {
3995 LLVM_DEBUG(dbgs() << "LV: Interleaving to expose ILP.\n");
3996 // Interleave no less than SmallIC but not as aggressive as the normal IC
3997 // to satisfy the rare situation when resources are too limited.
3998 return std::max(IC / 2, SmallIC);
3999 }
4000
4001 LLVM_DEBUG(dbgs() << "LV: Interleaving to reduce branch cost.\n");
4002 return SmallIC;
4003 }
4004
4005 // Interleave if this is a large loop (small loops are already dealt with by
4006 // this point) that could benefit from interleaving.
4007 if (AggressivelyInterleave) {
4008 LLVM_DEBUG(dbgs() << "LV: Interleaving to expose ILP.\n");
4009 return IC;
4010 }
4011
4012 LLVM_DEBUG(dbgs() << "LV: Not Interleaving.\n");
4013 return 1;
4014}
4015
4017 Instruction *I, ElementCount VF) const {
4018 // TODO: Cost model for emulated masked load/store is completely
4019 // broken. This hack guides the cost model to use an artificially
4020 // high enough value to practically disable vectorization with such
4021 // operations, except where previously deployed legality hack allowed
4022 // using very low cost values. This is to avoid regressions coming simply
4023 // from moving "masked load/store" check from legality to cost model.
4024 // Masked Load/Gather emulation was previously never allowed.
4025 // Limited number of Masked Store/Scatter emulation was allowed.
4027 "Expecting a scalar emulated instruction");
4028 return isa<LoadInst>(I) ||
4029 (isa<StoreInst>(I) &&
4030 NumPredStores > NumberOfStoresToPredicate);
4031}
4032
4034 assert(VF.isVector() && "Expected VF >= 2");
4035
4036 // If we've already collected the instructions to scalarize or the predicated
4037 // BBs after vectorization, there's nothing to do. Collection may already have
4038 // occurred if we have a user-selected VF and are now computing the expected
4039 // cost for interleaving.
4040 if (InstsToScalarize.contains(VF) ||
4041 PredicatedBBsAfterVectorization.contains(VF))
4042 return;
4043
4044 // Initialize a mapping for VF in InstsToScalalarize. If we find that it's
4045 // not profitable to scalarize any instructions, the presence of VF in the
4046 // map will indicate that we've analyzed it already.
4047 ScalarCostsTy &ScalarCostsVF = InstsToScalarize[VF];
4048
4049 // Find all the instructions that are scalar with predication in the loop and
4050 // determine if it would be better to not if-convert the blocks they are in.
4051 // If so, we also record the instructions to scalarize.
4052 for (BasicBlock *BB : TheLoop->blocks()) {
4054 continue;
4055 for (Instruction &I : *BB)
4056 if (isScalarWithPredication(&I, VF)) {
4057 ScalarCostsTy ScalarCosts;
4058 // Do not apply discount logic for:
4059 // 1. Scalars after vectorization, as there will only be a single copy
4060 // of the instruction.
4061 // 2. Scalable VF, as that would lead to invalid scalarization costs.
4062 // 3. Emulated masked memrefs, if a hacked cost is needed.
4063 if (!isScalarAfterVectorization(&I, VF) && !VF.isScalable() &&
4065 computePredInstDiscount(&I, ScalarCosts, VF) >= 0) {
4066 for (const auto &[I, IC] : ScalarCosts)
4067 ScalarCostsVF.insert({I, IC});
4068 }
4069 // Remember that BB will remain after vectorization.
4070 PredicatedBBsAfterVectorization[VF].insert(BB);
4071 for (auto *Pred : predecessors(BB)) {
4072 if (Pred->getSingleSuccessor() == BB)
4073 PredicatedBBsAfterVectorization[VF].insert(Pred);
4074 }
4075 }
4076 }
4077}
4078
4079InstructionCost LoopVectorizationCostModel::computePredInstDiscount(
4080 Instruction *PredInst, ScalarCostsTy &ScalarCosts, ElementCount VF) {
4081 assert(!isUniformAfterVectorization(PredInst, VF) &&
4082 "Instruction marked uniform-after-vectorization will be predicated");
4083
4084 // Initialize the discount to zero, meaning that the scalar version and the
4085 // vector version cost the same.
4086 InstructionCost Discount = 0;
4087
4088 // Holds instructions to analyze. The instructions we visit are mapped in
4089 // ScalarCosts. Those instructions are the ones that would be scalarized if
4090 // we find that the scalar version costs less.
4092
4093 // Returns true if the given instruction can be scalarized.
4094 auto CanBeScalarized = [&](Instruction *I) -> bool {
4095 // We only attempt to scalarize instructions forming a single-use chain
4096 // from the original predicated block that would otherwise be vectorized.
4097 // Although not strictly necessary, we give up on instructions we know will
4098 // already be scalar to avoid traversing chains that are unlikely to be
4099 // beneficial.
4100 if (!I->hasOneUse() || PredInst->getParent() != I->getParent() ||
4101 isScalarAfterVectorization(I, VF))
4102 return false;
4103
4104 // If the instruction is scalar with predication, it will be analyzed
4105 // separately. We ignore it within the context of PredInst.
4106 if (isScalarWithPredication(I, VF))
4107 return false;
4108
4109 // If any of the instruction's operands are uniform after vectorization,
4110 // the instruction cannot be scalarized. This prevents, for example, a
4111 // masked load from being scalarized.
4112 //
4113 // We assume we will only emit a value for lane zero of an instruction
4114 // marked uniform after vectorization, rather than VF identical values.
4115 // Thus, if we scalarize an instruction that uses a uniform, we would
4116 // create uses of values corresponding to the lanes we aren't emitting code
4117 // for. This behavior can be changed by allowing getScalarValue to clone
4118 // the lane zero values for uniforms rather than asserting.
4119 for (Use &U : I->operands())
4120 if (auto *J = dyn_cast<Instruction>(U.get()))
4121 if (isUniformAfterVectorization(J, VF))
4122 return false;
4123
4124 // Otherwise, we can scalarize the instruction.
4125 return true;
4126 };
4127
4128 // Compute the expected cost discount from scalarizing the entire expression
4129 // feeding the predicated instruction. We currently only consider expressions
4130 // that are single-use instruction chains.
4131 Worklist.push_back(PredInst);
4132 while (!Worklist.empty()) {
4133 Instruction *I = Worklist.pop_back_val();
4134
4135 // If we've already analyzed the instruction, there's nothing to do.
4136 if (ScalarCosts.contains(I))
4137 continue;
4138
4139 // Cannot scalarize fixed-order recurrence phis at the moment.
4140 if (isa<PHINode>(I) && Legal->isFixedOrderRecurrence(cast<PHINode>(I)))
4141 continue;
4142
4143 // Compute the cost of the vector instruction. Note that this cost already
4144 // includes the scalarization overhead of the predicated instruction.
4145 InstructionCost VectorCost = getInstructionCost(I, VF);
4146
4147 // Compute the cost of the scalarized instruction. This cost is the cost of
4148 // the instruction as if it wasn't if-converted and instead remained in the
4149 // predicated block. We will scale this cost by block probability after
4150 // computing the scalarization overhead.
4151 InstructionCost ScalarCost =
4152 VF.getFixedValue() * getInstructionCost(I, ElementCount::getFixed(1));
4153
4154 // Compute the scalarization overhead of needed insertelement instructions
4155 // and phi nodes.
4156 if (isScalarWithPredication(I, VF) && !I->getType()->isVoidTy()) {
4157 Type *WideTy = toVectorizedTy(I->getType(), VF);
4158 for (Type *VectorTy : getContainedTypes(WideTy)) {
4159 ScalarCost += TTI.getScalarizationOverhead(
4161 /*Insert=*/true,
4162 /*Extract=*/false, Config.CostKind);
4163 }
4164 ScalarCost += VF.getFixedValue() *
4165 TTI.getCFInstrCost(Instruction::PHI, Config.CostKind);
4166 }
4167
4168 // Compute the scalarization overhead of needed extractelement
4169 // instructions. For each of the instruction's operands, if the operand can
4170 // be scalarized, add it to the worklist; otherwise, account for the
4171 // overhead.
4172 for (Use &U : I->operands())
4173 if (auto *J = dyn_cast<Instruction>(U.get())) {
4174 assert(canVectorizeTy(J->getType()) &&
4175 "Instruction has non-scalar type");
4176 if (CanBeScalarized(J))
4177 Worklist.push_back(J);
4178 else if (needsExtract(J, VF)) {
4179 Type *WideTy = toVectorizedTy(J->getType(), VF);
4180 for (Type *VectorTy : getContainedTypes(WideTy)) {
4181 ScalarCost += TTI.getScalarizationOverhead(
4182 cast<VectorType>(VectorTy),
4183 APInt::getAllOnes(VF.getFixedValue()), /*Insert*/ false,
4184 /*Extract*/ true, Config.CostKind);
4185 }
4186 }
4187 }
4188
4189 // Scale the total scalar cost by block probability.
4190 ScalarCost /= getPredBlockCostDivisor(Config.CostKind, I->getParent());
4191
4192 // Compute the discount. A non-negative discount means the vector version
4193 // of the instruction costs more, and scalarizing would be beneficial.
4194 Discount += VectorCost - ScalarCost;
4195 ScalarCosts[I] = ScalarCost;
4196 }
4197
4198 return Discount;
4199}
4200
4203 assert(VF.isScalar() && "must only be called for scalar VFs");
4204
4205 // For each block.
4206 for (BasicBlock *BB : TheLoop->blocks()) {
4207 InstructionCost BlockCost;
4208
4209 // For each instruction in the old loop.
4210 for (Instruction &I : *BB) {
4211 // Skip ignored values.
4212 if (ValuesToIgnore.count(&I) ||
4213 (VF.isVector() && VecValuesToIgnore.count(&I)))
4214 continue;
4215
4217
4218 // Check if we should override the cost.
4219 if (C.isValid() && ForceTargetInstructionCost.getNumOccurrences() > 0)
4221
4222 BlockCost += C;
4223 LLVM_DEBUG(dbgs() << "LV: Found an estimated cost of " << C << " for VF "
4224 << VF << " For instruction: " << I << '\n');
4225 }
4226
4227 // In the scalar loop, we may not always execute the predicated block, if it
4228 // is an if-else block. Thus, scale the block's cost by the probability of
4229 // executing it. getPredBlockCostDivisor will return 1 for blocks that are
4230 // only predicated by the header mask when folding the tail.
4231 Cost += BlockCost / getPredBlockCostDivisor(Config.CostKind, BB);
4232 }
4233
4234 return Cost;
4235}
4236
4237/// Gets the address access SCEV for Ptr, if it should be used for cost modeling
4238/// according to isAddressSCEVForCost.
4239///
4240/// This SCEV can be sent to the Target in order to estimate the address
4241/// calculation cost.
4243 Value *Ptr,
4245 const Loop *TheLoop) {
4246 const SCEV *Addr = PSE.getSCEV(Ptr);
4247 return vputils::isAddressSCEVForCost(Addr, *PSE.getSE(), TheLoop) ? Addr
4248 : nullptr;
4249}
4250
4252LoopVectorizationCostModel::getMemInstScalarizationCost(Instruction *I,
4253 ElementCount VF) {
4254 assert(VF.isVector() &&
4255 "Scalarization cost of instruction implies vectorization.");
4256 if (VF.isScalable())
4257 return InstructionCost::getInvalid();
4258
4259 Type *ValTy = getLoadStoreType(I);
4260 auto *SE = PSE.getSE();
4261
4262 unsigned AS = getLoadStoreAddressSpace(I);
4264 Type *PtrTy = toVectorTy(Ptr->getType(), VF);
4265 // NOTE: PtrTy is a vector to signal `TTI::getAddressComputationCost`
4266 // that it is being called from this specific place.
4267
4268 // Figure out whether the access is strided and get the stride value
4269 // if it's known in compile time
4270 const SCEV *PtrSCEV = getAddressAccessSCEV(Ptr, PSE, TheLoop);
4271
4272 // Get the cost of the scalar memory instruction and address computation.
4274 VF.getFixedValue() *
4275 TTI.getAddressComputationCost(PtrTy, SE, PtrSCEV, Config.CostKind);
4276
4277 // Don't pass *I here, since it is scalar but will actually be part of a
4278 // vectorized loop where the user of it is a vectorized instruction.
4280 TTI::OperandValueInfo OpInfo = TTI::getOperandInfo(I->getOperand(0));
4281 Cost += VF.getFixedValue() *
4282 TTI.getMemoryOpCost(I->getOpcode(), ValTy->getScalarType(), Alignment,
4283 AS, Config.CostKind, OpInfo);
4284
4285 // Get the overhead of the extractelement and insertelement instructions
4286 // we might create due to scalarization.
4288
4289 // If we have a predicated load/store, it will need extra i1 extracts and
4290 // conditional branches, but may not be executed for each vector lane. Scale
4291 // the cost by the probability of executing the predicated block.
4292 if (isPredicatedInst(I)) {
4293 Cost /= getPredBlockCostDivisor(Config.CostKind, I->getParent());
4294
4295 // Add the cost of an i1 extract and a branch
4296 auto *VecI1Ty =
4297 VectorType::get(IntegerType::getInt1Ty(ValTy->getContext()), VF);
4299 VecI1Ty, APInt::getAllOnes(VF.getFixedValue()),
4300 /*Insert=*/false, /*Extract=*/true, Config.CostKind);
4301 Cost += TTI.getCFInstrCost(Instruction::CondBr, Config.CostKind);
4302
4303 if (useEmulatedMaskMemRefHack(I, VF))
4304 // Artificially setting to a high enough value to practically disable
4305 // vectorization with such operations.
4306 Cost = 3000000;
4307 }
4308
4309 return Cost;
4310}
4311
4312InstructionCost LoopVectorizationCostModel::getConsecutiveMemOpCost(
4313 Instruction *I, ElementCount VF, InstWidening Kind) {
4314 assert((Kind == CM_Widen || Kind == CM_Widen_Reverse) &&
4315 "Expected a consecutive widening decision");
4316 Type *ValTy = getLoadStoreType(I);
4317 auto *VectorTy = cast<VectorType>(toVectorTy(ValTy, VF));
4318 unsigned AS = getLoadStoreAddressSpace(I);
4319
4322 if (isMaskRequired(I)) {
4323 unsigned IID = I->getOpcode() == Instruction::Load
4324 ? Intrinsic::masked_load
4325 : Intrinsic::masked_store;
4327 MemIntrinsicCostAttributes(IID, VectorTy, Alignment, AS),
4328 Config.CostKind);
4329 } else {
4330 TTI::OperandValueInfo OpInfo = TTI::getOperandInfo(I->getOperand(0));
4331 Cost += TTI.getMemoryOpCost(I->getOpcode(), VectorTy, Alignment, AS,
4332 Config.CostKind, OpInfo, I);
4333 }
4334
4335 if (Kind == CM_Widen_Reverse)
4337 VectorTy, Config.CostKind, {}, 0);
4338 return Cost;
4339}
4340
4342LoopVectorizationCostModel::getUniformMemOpCost(Instruction *I,
4343 ElementCount VF) const {
4344 assert(isUniformMemOp(*I, VF));
4345
4346 Type *ValTy = getLoadStoreType(I);
4348 auto *VectorTy = cast<VectorType>(toVectorTy(ValTy, VF));
4350 unsigned AS = getLoadStoreAddressSpace(I);
4351 if (isa<LoadInst>(I)) {
4352 return TTI.getAddressComputationCost(PtrTy, nullptr, nullptr,
4353 Config.CostKind) +
4354 TTI.getMemoryOpCost(Instruction::Load, ValTy, Alignment, AS,
4355 Config.CostKind) +
4357 VectorTy, Config.CostKind);
4358 }
4359 StoreInst *SI = cast<StoreInst>(I);
4360
4361 bool IsLoopInvariantStoreValue = Legal->isInvariant(SI->getValueOperand());
4362 // TODO: We have existing tests that request the cost of extracting element
4363 // VF.getKnownMinValue() - 1 from a scalable vector. This does not represent
4364 // the actual generated code, which involves extracting the last element of
4365 // a scalable vector where the lane to extract is unknown at compile time.
4367 TTI.getAddressComputationCost(PtrTy, nullptr, nullptr, Config.CostKind) +
4368 TTI.getMemoryOpCost(Instruction::Store, ValTy, Alignment, AS,
4369 Config.CostKind);
4370 if (!IsLoopInvariantStoreValue)
4371 Cost += TTI.getIndexedVectorInstrCostFromEnd(Instruction::ExtractElement,
4372 VectorTy, Config.CostKind, 0);
4373 return Cost;
4374}
4375
4377LoopVectorizationCostModel::getGatherScatterCost(Instruction *I,
4378 ElementCount VF) const {
4379 Type *ValTy = getLoadStoreType(I);
4380 auto *VectorTy = cast<VectorType>(toVectorTy(ValTy, VF));
4383 Type *PtrTy = Ptr->getType();
4384
4385 if (!isUniform(Ptr, VF))
4386 PtrTy = toVectorTy(PtrTy, VF);
4387
4388 unsigned IID = I->getOpcode() == Instruction::Load
4389 ? Intrinsic::masked_gather
4390 : Intrinsic::masked_scatter;
4391 return TTI.getAddressComputationCost(PtrTy, nullptr, nullptr,
4392 Config.CostKind) +
4394 MemIntrinsicCostAttributes(IID, VectorTy, Ptr, isMaskRequired(I),
4395 Alignment, I),
4396 Config.CostKind);
4397}
4398
4400LoopVectorizationCostModel::getInterleaveGroupCost(Instruction *I,
4401 ElementCount VF) const {
4402 const auto *Group = getInterleavedAccessGroup(I);
4403 assert(Group && "Fail to get an interleaved access group.");
4404
4405 Instruction *InsertPos = Group->getInsertPos();
4406 Type *ValTy = getLoadStoreType(InsertPos);
4407 auto *VectorTy = cast<VectorType>(toVectorTy(ValTy, VF));
4408 unsigned AS = getLoadStoreAddressSpace(InsertPos);
4409
4410 unsigned InterleaveFactor = Group->getFactor();
4411 auto *WideVecTy = VectorType::get(ValTy, VF * InterleaveFactor);
4412
4413 // Holds the indices of existing members in the interleaved group.
4414 SmallVector<unsigned, 4> Indices;
4415 for (unsigned IF = 0; IF < InterleaveFactor; IF++)
4416 if (Group->getMember(IF))
4417 Indices.push_back(IF);
4418
4419 // Calculate the cost of the whole interleaved group.
4420 bool UseMaskForGaps =
4421 (Group->requiresScalarEpilogue() && !isEpilogueAllowed()) ||
4422 (isa<StoreInst>(I) && !Group->isFull());
4424 InsertPos->getOpcode(), WideVecTy, Group->getFactor(), Indices,
4425 Group->getAlign(), AS, Config.CostKind, isMaskRequired(I),
4426 UseMaskForGaps);
4427
4428 if (Group->isReverse()) {
4429 // TODO: Add support for reversed masked interleaved access.
4430 assert(!isMaskRequired(I) &&
4431 "Reverse masked interleaved access not supported.");
4432 Cost += Group->getNumMembers() *
4434 VectorTy, Config.CostKind, {}, 0);
4435 }
4436 return Cost;
4437}
4438
4440LoopVectorizationCostModel::getMemoryInstructionCost(Instruction *I,
4441 ElementCount VF) {
4442 // Calculate scalar cost only. Vectorization cost should be ready at this
4443 // moment.
4444 if (VF.isScalar()) {
4445 Type *ValTy = getLoadStoreType(I);
4448 unsigned AS = getLoadStoreAddressSpace(I);
4449
4450 TTI::OperandValueInfo OpInfo = TTI::getOperandInfo(I->getOperand(0));
4451 return TTI.getAddressComputationCost(PtrTy, nullptr, nullptr,
4452 Config.CostKind) +
4453 TTI.getMemoryOpCost(I->getOpcode(), ValTy, Alignment, AS,
4454 Config.CostKind, OpInfo, I);
4455 }
4456 return getWideningCost(I, VF);
4457}
4458
4460LoopVectorizationCostModel::getScalarizationOverhead(Instruction *I,
4461 ElementCount VF) const {
4462
4463 // There is no mechanism yet to create a scalable scalarization loop,
4464 // so this is currently Invalid.
4465 if (VF.isScalable())
4466 return InstructionCost::getInvalid();
4467
4468 if (VF.isScalar())
4469 return 0;
4470
4472 Type *RetTy = toVectorizedTy(I->getType(), VF);
4473 if (!RetTy->isVoidTy() &&
4475
4477 if (isa<LoadInst>(I))
4478 VIC = TTI::VectorInstrContext::Load;
4479 else if (isa<StoreInst>(I))
4480 VIC = TTI::VectorInstrContext::Store;
4481
4482 for (Type *VectorTy : getContainedTypes(RetTy)) {
4485 /*Insert=*/true, /*Extract=*/false, Config.CostKind,
4486 /*ForPoisonSrc=*/true, {}, VIC);
4487 }
4488 }
4489
4490 // Some targets keep addresses scalar.
4492 return Cost;
4493
4494 // Some targets support efficient element stores.
4496 return Cost;
4497
4498 // Collect operands to consider.
4499 CallInst *CI = dyn_cast<CallInst>(I);
4500 Instruction::op_range Ops = CI ? CI->args() : I->operands();
4501
4502 // Skip operands that do not require extraction/scalarization and do not incur
4503 // any overhead.
4505 for (auto *V : filterExtractingOperands(Ops, VF))
4506 Tys.push_back(maybeVectorizeType(V->getType(), VF));
4507
4509 ? TTI::VectorInstrContext::Store
4511 return Cost +
4512 TTI.getOperandsScalarizationOverhead(Tys, Config.CostKind, OperandVIC);
4513}
4514
4516 if (VF.isScalar())
4517 return;
4518
4519 // TODO: We should generate better code and update the cost model for
4520 // predicated uniform stores. Today they are treated as any other
4521 // predicated store (see added test cases in
4522 // invariant-store-vectorization.ll).
4523 NumPredStores = 0;
4524 for (BasicBlock *BB : TheLoop->blocks())
4525 for (Instruction &I : *BB)
4527 ++NumPredStores;
4528
4529 for (BasicBlock *BB : TheLoop->blocks()) {
4530 // For each instruction in the old loop.
4531 for (Instruction &I : *BB) {
4533 if (!Ptr)
4534 continue;
4535
4536 LLVM_DEBUG(dbgs() << "LV: Memory widening: calculating best strategy for "
4537 << I << '\n');
4538 if (isUniformMemOp(I, VF)) {
4539 auto IsLegalToScalarize = [&]() {
4540 if (!VF.isScalable())
4541 // Scalarization of fixed length vectors "just works".
4542 return true;
4543
4544 // We have dedicated lowering for unpredicated uniform loads and
4545 // stores. Note that even with tail folding we know that at least
4546 // one lane is active (i.e. generalized predication is not possible
4547 // here), and the logic below depends on this fact.
4548 if (!foldTailByMasking())
4549 return true;
4550
4551 // For scalable vectors, a uniform memop load is always
4552 // uniform-by-parts and we know how to scalarize that.
4553 if (isa<LoadInst>(I))
4554 return true;
4555
4556 // A uniform store isn't neccessarily uniform-by-part
4557 // and we can't assume scalarization.
4558 auto &SI = cast<StoreInst>(I);
4559 return TheLoop->isLoopInvariant(SI.getValueOperand());
4560 };
4561
4562 const InstructionCost GatherScatterCost =
4563 isLegalGatherOrScatter(&I, VF) ? getGatherScatterCost(&I, VF)
4565
4566 // Load: Scalar load + broadcast
4567 // Store: Scalar store + isLoopInvariantStoreValue ? 0 : extract
4568 // FIXME: This cost is a significant under-estimate for tail folded
4569 // memory ops.
4570 const InstructionCost ScalarizationCost =
4571 IsLegalToScalarize() ? getUniformMemOpCost(&I, VF)
4573
4574 // Choose better solution for the current VF, Note that Invalid
4575 // costs compare as maximumal large. If both are invalid, we get
4576 // scalable invalid which signals a failure and a vectorization abort.
4577 LLVM_DEBUG(dbgs() << "LV: Memory widening: uniform memory op has "
4578 "GatherScatterCost = "
4579 << GatherScatterCost << ", ScalarizationCost = "
4580 << ScalarizationCost << '\n');
4581 if (GatherScatterCost < ScalarizationCost)
4582 setWideningDecision(&I, VF, CM_GatherScatter, GatherScatterCost);
4583 else
4584 setWideningDecision(&I, VF, CM_Scalarize, ScalarizationCost);
4585 continue;
4586 }
4587
4588 // We assume that widening is the best solution when possible.
4589 if (std::optional<InstWidening> Decision =
4591 InstructionCost WidenCost = getConsecutiveMemOpCost(&I, VF, *Decision);
4592 LLVM_DEBUG(
4593 dbgs() << "LV: Memory widening: can be widened normally with cost "
4594 << WidenCost << '\n');
4595 setWideningDecision(&I, VF, *Decision, WidenCost);
4596 continue;
4597 }
4598
4599 // Choose between Interleaving, Gather/Scatter or Scalarization.
4601 unsigned NumAccesses = 1;
4602 if (isAccessInterleaved(&I)) {
4603 const auto *Group = getInterleavedAccessGroup(&I);
4604 assert(Group && "Fail to get an interleaved access group.");
4605
4606 // Make one decision for the whole group.
4607 if (getWideningDecision(&I, VF) != CM_Unknown)
4608 continue;
4609
4610 NumAccesses = Group->getNumMembers();
4612 InterleaveCost = getInterleaveGroupCost(&I, VF);
4613 }
4614
4615 InstructionCost GatherScatterCost =
4617 ? getGatherScatterCost(&I, VF) * NumAccesses
4619
4620 InstructionCost ScalarizationCost =
4621 getMemInstScalarizationCost(&I, VF) * NumAccesses;
4622
4623 // Choose better solution for the current VF,
4624 // write down this decision and use it during vectorization.
4626 InstWidening Decision;
4627 if (InterleaveCost <= GatherScatterCost &&
4628 InterleaveCost < ScalarizationCost) {
4629 Decision = CM_Interleave;
4630 Cost = InterleaveCost;
4631 } else if (GatherScatterCost < ScalarizationCost) {
4632 Decision = CM_GatherScatter;
4633 Cost = GatherScatterCost;
4634 } else {
4635 Decision = CM_Scalarize;
4636 Cost = ScalarizationCost;
4637 }
4638 LLVM_DEBUG(
4639 dbgs() << "LV: Memory widening: InterleaveCost = " << InterleaveCost
4640 << ", GatherScatterCost = " << GatherScatterCost
4641 << ", ScalarizationCost = " << ScalarizationCost << '\n');
4642
4643 // If the instructions belongs to an interleave group, the whole group
4644 // receives the same decision. The whole group receives the cost, but
4645 // the cost will actually be assigned to one instruction.
4646 if (const auto *Group = getInterleavedAccessGroup(&I)) {
4647 if (Decision == CM_Scalarize) {
4648 for (Instruction *I : Group->members())
4649 setWideningDecision(I, VF, Decision,
4650 getMemInstScalarizationCost(I, VF));
4651 } else {
4652 setWideningDecision(Group, VF, Decision, Cost);
4653 }
4654 } else
4655 setWideningDecision(&I, VF, Decision, Cost);
4656 }
4657 }
4658
4659 // Make sure that any load of address and any other address computation
4660 // remains scalar unless there is gather/scatter support. This avoids
4661 // inevitable extracts into address registers, and also has the benefit of
4662 // activating LSR more, since that pass can't optimize vectorized
4663 // addresses.
4664 if (TTI.prefersVectorizedAddressing())
4665 return;
4666
4667 // Start with all scalar pointer uses.
4669 for (BasicBlock *BB : TheLoop->blocks())
4670 for (Instruction &I : *BB) {
4671 Instruction *PtrDef =
4673 if (PtrDef && TheLoop->contains(PtrDef) &&
4675 AddrDefs.insert(PtrDef);
4676 }
4677
4678 // Add all instructions used to generate the addresses.
4680 append_range(Worklist, AddrDefs);
4681 while (!Worklist.empty()) {
4682 Instruction *I = Worklist.pop_back_val();
4683 for (auto &Op : I->operands())
4684 if (auto *InstOp = dyn_cast<Instruction>(Op))
4685 if (TheLoop->contains(InstOp) && !isa<PHINode>(InstOp) &&
4686 AddrDefs.insert(InstOp))
4687 Worklist.push_back(InstOp);
4688 }
4689
4690 auto UpdateMemOpUserCost = [this, VF](LoadInst *LI) {
4691 // If there are direct memory op users of the newly scalarized load,
4692 // their cost may have changed because there's no scalarization
4693 // overhead for the operand. Update it.
4694 for (User *U : LI->users()) {
4696 continue;
4698 continue;
4699 auto UI = cast<Instruction>(U);
4700 LLVM_DEBUG(
4701 dbgs() << "LV: Memory widening: updating decision for load user "
4702 << *UI << '\n');
4704 UI, VF, CM_Scalarize,
4705 getMemInstScalarizationCost(cast<Instruction>(U), VF));
4706 }
4707 };
4708 for (auto *I : AddrDefs) {
4709 if (isa<LoadInst>(I)) {
4710 // Setting the desired widening decision should ideally be handled in
4711 // by cost functions, but since this involves the task of finding out
4712 // if the loaded register is involved in an address computation, it is
4713 // instead changed here when we know this is the case.
4714 InstWidening Decision = getWideningDecision(I, VF);
4715 if (!isPredicatedInst(I) &&
4716 (Decision == CM_Widen || Decision == CM_Widen_Reverse ||
4717 (!isUniformMemOp(*I, VF) && Decision == CM_Scalarize))) {
4718 // Scalarize a widened load of address or update the cost of a scalar
4719 // load of an address.
4720 LLVM_DEBUG(dbgs() << "LV: Memory widening: updating decision for load "
4721 << *I << '\n');
4723 I, VF, CM_Scalarize,
4724 (VF.getKnownMinValue() *
4725 getMemoryInstructionCost(I, ElementCount::getFixed(1))));
4726 UpdateMemOpUserCost(cast<LoadInst>(I));
4727 } else if (const auto *Group = getInterleavedAccessGroup(I)) {
4728 // Scalarize all members of this interleaved group when any member
4729 // is used as an address. The address-used load skips scalarization
4730 // overhead, other members include it.
4731 for (Instruction *Member : Group->members()) {
4732 InstructionCost Cost = AddrDefs.contains(Member)
4733 ? (VF.getKnownMinValue() *
4734 getMemoryInstructionCost(
4735 Member, ElementCount::getFixed(1)))
4736 : getMemInstScalarizationCost(Member, VF);
4737 LLVM_DEBUG(
4738 dbgs()
4739 << "LV: Memory widening: updating decision for interleave member "
4740 << *Member << '\n');
4742 UpdateMemOpUserCost(cast<LoadInst>(Member));
4743 }
4744 }
4745 } else {
4746 // Cannot scalarize fixed-order recurrence phis at the moment.
4747 if (isa<PHINode>(I) && Legal->isFixedOrderRecurrence(cast<PHINode>(I)))
4748 continue;
4749
4750 // Make sure I gets scalarized and a cost estimate without
4751 // scalarization overhead.
4752 ForcedScalars[VF].insert(I);
4753 }
4754 }
4755}
4756
4758 if (!Legal->isInvariant(Op))
4759 return false;
4760 // Consider Op invariant, if it or its operands aren't predicated
4761 // instruction in the loop. In that case, it is not trivially hoistable.
4762 auto *OpI = dyn_cast<Instruction>(Op);
4763 return !OpI || !TheLoop->contains(OpI) ||
4764 (!isPredicatedInst(OpI) &&
4765 (!isa<PHINode>(OpI) || OpI->getParent() != TheLoop->getHeader()) &&
4766 all_of(OpI->operands(),
4767 [this](Value *Op) { return shouldConsiderInvariant(Op); }));
4768}
4769
4772 ElementCount VF) {
4773 // If we know that this instruction will remain uniform, check the cost of
4774 // the scalar version.
4776 VF = ElementCount::getFixed(1);
4777
4778 if (VF.isVector() && isProfitableToScalarize(I, VF))
4779 return InstsToScalarize[VF][I];
4780
4781 // Forced scalars do not have any scalarization overhead.
4782 auto ForcedScalar = ForcedScalars.find(VF);
4783 if (VF.isVector() && ForcedScalar != ForcedScalars.end()) {
4784 auto InstSet = ForcedScalar->second;
4785 if (InstSet.count(I))
4787 VF.getKnownMinValue();
4788 }
4789
4790 const auto &MinBWs = Config.getMinimalBitwidths();
4791 uint64_t InstrMinBWs = MinBWs.lookup(I);
4792 Type *RetTy = I->getType();
4794 RetTy = IntegerType::get(RetTy->getContext(), InstrMinBWs);
4795 auto *SE = PSE.getSE();
4796
4797 Type *VectorTy;
4798 if (isScalarAfterVectorization(I, VF)) {
4799 [[maybe_unused]] auto HasSingleCopyAfterVectorization =
4800 [this](Instruction *I, ElementCount VF) -> bool {
4801 if (VF.isScalar())
4802 return true;
4803
4804 auto Scalarized = InstsToScalarize.find(VF);
4805 assert(Scalarized != InstsToScalarize.end() &&
4806 "VF not yet analyzed for scalarization profitability");
4807 return !Scalarized->second.count(I) &&
4808 llvm::all_of(I->users(), [&](User *U) {
4809 auto *UI = cast<Instruction>(U);
4810 return !Scalarized->second.count(UI);
4811 });
4812 };
4813
4814 // With the exception of GEPs and PHIs, after scalarization there should
4815 // only be one copy of the instruction generated in the loop. This is
4816 // because the VF is either 1, or any instructions that need scalarizing
4817 // have already been dealt with by the time we get here. As a result,
4818 // it means we don't have to multiply the instruction cost by VF.
4819 assert(I->getOpcode() == Instruction::GetElementPtr ||
4820 I->getOpcode() == Instruction::PHI ||
4821 (I->getOpcode() == Instruction::BitCast &&
4822 I->getType()->isPointerTy()) ||
4823 HasSingleCopyAfterVectorization(I, VF));
4824 VectorTy = RetTy;
4825 } else
4826 VectorTy = toVectorizedTy(RetTy, VF);
4827
4828 if (VF.isVector() && VectorTy->isVectorTy() &&
4829 !TTI.getNumberOfParts(VectorTy))
4831
4832 // TODO: We need to estimate the cost of intrinsic calls.
4833 switch (I->getOpcode()) {
4834 case Instruction::GetElementPtr:
4835 // We mark this instruction as zero-cost because the cost of GEPs in
4836 // vectorized code depends on whether the corresponding memory instruction
4837 // is scalarized or not. Therefore, we handle GEPs with the memory
4838 // instruction cost.
4839 return 0;
4840 case Instruction::UncondBr:
4841 case Instruction::CondBr: {
4842 // In cases of scalarized and predicated instructions, there will be VF
4843 // predicated blocks in the vectorized loop. Each branch around these
4844 // blocks requires also an extract of its vector compare i1 element.
4845 // Note that the conditional branch from the loop latch will be replaced by
4846 // a single branch controlling the loop, so there is no extra overhead from
4847 // scalarization.
4848 bool ScalarPredicatedBB = false;
4850 if (VF.isVector() && BI &&
4851 (PredicatedBBsAfterVectorization[VF].count(BI->getSuccessor(0)) ||
4852 PredicatedBBsAfterVectorization[VF].count(BI->getSuccessor(1))) &&
4853 BI->getParent() != TheLoop->getLoopLatch())
4854 ScalarPredicatedBB = true;
4855
4856 if (ScalarPredicatedBB) {
4857 // Not possible to scalarize scalable vector with predicated instructions.
4858 if (VF.isScalable())
4860 // Return cost for branches around scalarized and predicated blocks.
4861 auto *VecI1Ty =
4863 return (TTI.getScalarizationOverhead(
4864 VecI1Ty, APInt::getAllOnes(VF.getFixedValue()),
4865 /*Insert*/ false, /*Extract*/ true, Config.CostKind) +
4866 (TTI.getCFInstrCost(Instruction::CondBr, Config.CostKind) *
4867 VF.getFixedValue()));
4868 }
4869
4870 if (I->getParent() == TheLoop->getLoopLatch() || VF.isScalar())
4871 // The back-edge branch will remain, as will all scalar branches.
4872 return TTI.getCFInstrCost(Instruction::UncondBr, Config.CostKind);
4873
4874 // This branch will be eliminated by if-conversion.
4875 return 0;
4876 // Note: We currently assume zero cost for an unconditional branch inside
4877 // a predicated block since it will become a fall-through, although we
4878 // may decide in the future to call TTI for all branches.
4879 }
4880 case Instruction::Switch: {
4881 if (VF.isScalar())
4882 return TTI.getCFInstrCost(Instruction::Switch, Config.CostKind);
4883 auto *Switch = cast<SwitchInst>(I);
4884 return Switch->getNumCases() *
4885 TTI.getCmpSelInstrCost(
4886 Instruction::ICmp,
4887 toVectorTy(Switch->getCondition()->getType(), VF),
4888 toVectorTy(Type::getInt1Ty(I->getContext()), VF),
4889 CmpInst::ICMP_EQ, Config.CostKind);
4890 }
4891 case Instruction::PHI: {
4892 auto *Phi = cast<PHINode>(I);
4893
4894 // First-order recurrences are replaced by vector shuffles inside the loop.
4895 if (VF.isVector() && Legal->isFixedOrderRecurrence(Phi)) {
4896 return TTI.getShuffleCost(
4898 cast<VectorType>(VectorTy), Config.CostKind, {}, -1);
4899 }
4900
4901 // Phi nodes in non-header blocks (not inductions, reductions, etc.) are
4902 // converted into select instructions. We require N - 1 selects per phi
4903 // node, where N is the number of incoming values.
4904 if (VF.isVector() && Phi->getParent() != TheLoop->getHeader()) {
4905 Type *ResultTy = Phi->getType();
4906
4907 // All instructions in an Any-of reduction chain are narrowed to bool.
4908 // Check if that is the case for this phi node.
4909 auto *HeaderUser = cast_if_present<PHINode>(
4910 find_singleton<User>(Phi->users(), [this](User *U, bool) -> User * {
4911 auto *Phi = dyn_cast<PHINode>(U);
4912 if (Phi && Phi->getParent() == TheLoop->getHeader())
4913 return Phi;
4914 return nullptr;
4915 }));
4916 if (HeaderUser) {
4917 auto &ReductionVars = Legal->getReductionVars();
4918 auto Iter = ReductionVars.find(HeaderUser);
4919 if (Iter != ReductionVars.end() &&
4921 Iter->second.getRecurrenceKind()))
4922 ResultTy = Type::getInt1Ty(Phi->getContext());
4923 }
4924 return (Phi->getNumIncomingValues() - 1) *
4925 TTI.getCmpSelInstrCost(
4926 Instruction::Select, toVectorTy(ResultTy, VF),
4927 toVectorTy(Type::getInt1Ty(Phi->getContext()), VF),
4928 CmpInst::BAD_ICMP_PREDICATE, Config.CostKind);
4929 }
4930
4931 // When tail folding with EVL, if the phi is part of an out of loop
4932 // reduction then it will be transformed into a wide vp_merge.
4933 if (VF.isVector() && foldTailWithEVL() &&
4934 Legal->getReductionVars().contains(Phi) &&
4935 !Config.isInLoopReduction(Phi)) {
4937 Intrinsic::vp_merge, toVectorTy(Phi->getType(), VF),
4938 {toVectorTy(Type::getInt1Ty(Phi->getContext()), VF)});
4939 return TTI.getIntrinsicInstrCost(ICA, Config.CostKind);
4940 }
4941
4942 return TTI.getCFInstrCost(Instruction::PHI, Config.CostKind);
4943 }
4944 case Instruction::UDiv:
4945 case Instruction::SDiv:
4946 case Instruction::URem:
4947 case Instruction::SRem:
4948 if (VF.isVector() && isPredicatedInst(I)) {
4949 const auto [ScalarCost, MaskedCost] = getDivRemSpeculationCost(I, VF);
4950 return isDivRemScalarWithPredication(ScalarCost, MaskedCost) ? ScalarCost
4951 : MaskedCost;
4952 }
4953 // We've proven all lanes safe to speculate, fall through.
4954 [[fallthrough]];
4955 case Instruction::Add:
4956 case Instruction::Sub: {
4957 auto Info = Legal->getHistogramInfo(I);
4958 if (Info && VF.isVector()) {
4959 const HistogramInfo *HGram = Info.value();
4960 // Assume that a non-constant update value (or a constant != 1) requires
4961 // a multiply, and add that into the cost.
4963 ConstantInt *RHS = dyn_cast<ConstantInt>(I->getOperand(1));
4964 if (!RHS || RHS->getZExtValue() != 1)
4965 MulCost = TTI.getArithmeticInstrCost(Instruction::Mul, VectorTy,
4966 Config.CostKind);
4967
4968 // Find the cost of the histogram operation itself.
4969 Type *PtrTy = VectorType::get(HGram->Load->getPointerOperandType(), VF);
4970 Type *ScalarTy = I->getType();
4971 Type *MaskTy = VectorType::get(Type::getInt1Ty(I->getContext()), VF);
4972 IntrinsicCostAttributes ICA(Intrinsic::experimental_vector_histogram_add,
4973 Type::getVoidTy(I->getContext()),
4974 {PtrTy, ScalarTy, MaskTy});
4975
4976 // Add the costs together with the add/sub operation.
4977 return TTI.getIntrinsicInstrCost(ICA, Config.CostKind) + MulCost +
4978 TTI.getArithmeticInstrCost(I->getOpcode(), VectorTy,
4979 Config.CostKind);
4980 }
4981 [[fallthrough]];
4982 }
4983 case Instruction::FAdd:
4984 case Instruction::FSub:
4985 case Instruction::Mul:
4986 case Instruction::FMul:
4987 case Instruction::FDiv:
4988 case Instruction::FRem:
4989 case Instruction::Shl:
4990 case Instruction::LShr:
4991 case Instruction::AShr:
4992 case Instruction::And:
4993 case Instruction::Or:
4994 case Instruction::Xor: {
4995 // If we're speculating on the stride being 1, the multiplication may
4996 // fold away. We can generalize this for all operations using the notion
4997 // of neutral elements. (TODO)
4998 if (I->getOpcode() == Instruction::Mul &&
4999 ((TheLoop->isLoopInvariant(I->getOperand(0)) &&
5000 PSE.getSCEV(I->getOperand(0))->isOne()) ||
5001 (TheLoop->isLoopInvariant(I->getOperand(1)) &&
5002 PSE.getSCEV(I->getOperand(1))->isOne())))
5003 return 0;
5004
5005 // Certain instructions can be cheaper to vectorize if they have a constant
5006 // second vector operand. One example of this are shifts on x86.
5007 Value *Op2 = I->getOperand(1);
5008 if (!isa<Constant>(Op2) && TheLoop->isLoopInvariant(Op2) &&
5009 PSE.getSE()->isSCEVable(Op2->getType()) &&
5010 isa<SCEVConstant>(PSE.getSCEV(Op2))) {
5011 Op2 = cast<SCEVConstant>(PSE.getSCEV(Op2))->getValue();
5012 }
5013 auto Op2Info = TTI.getOperandInfo(Op2);
5014 if (Op2Info.Kind == TargetTransformInfo::OK_AnyValue &&
5017
5018 SmallVector<const Value *, 4> Operands(I->operand_values());
5019 return TTI.getArithmeticInstrCost(
5020 I->getOpcode(), VectorTy, Config.CostKind,
5021 {TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None},
5022 Op2Info, Operands, I, TLI);
5023 }
5024 case Instruction::FNeg: {
5025 return TTI.getArithmeticInstrCost(
5026 I->getOpcode(), VectorTy, Config.CostKind,
5027 {TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None},
5028 {TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None},
5029 I->getOperand(0), I);
5030 }
5031 case Instruction::Select: {
5033 const SCEV *CondSCEV = SE->getSCEV(SI->getCondition());
5034 bool ScalarCond = (SE->isLoopInvariant(CondSCEV, TheLoop));
5035
5036 const Value *Op0, *Op1;
5037 using namespace llvm::PatternMatch;
5038 if (!ScalarCond && (match(I, m_LogicalAnd(m_Value(Op0), m_Value(Op1))) ||
5039 match(I, m_LogicalOr(m_Value(Op0), m_Value(Op1))))) {
5040 // select x, y, false --> x & y
5041 // select x, true, y --> x | y
5042 const auto [Op1VK, Op1VP] = TTI::getOperandInfo(Op0);
5043 const auto [Op2VK, Op2VP] = TTI::getOperandInfo(Op1);
5044 assert(Op0->getType()->getScalarSizeInBits() == 1 &&
5045 Op1->getType()->getScalarSizeInBits() == 1);
5046
5047 return TTI.getArithmeticInstrCost(
5048 match(I, m_LogicalOr()) ? Instruction::Or : Instruction::And,
5049 VectorTy, Config.CostKind, {Op1VK, Op1VP}, {Op2VK, Op2VP}, {Op0, Op1},
5050 I);
5051 }
5052
5053 Type *CondTy = SI->getCondition()->getType();
5054 if (!ScalarCond)
5055 CondTy = VectorType::get(CondTy, VF);
5056
5058 if (auto *Cmp = dyn_cast<CmpInst>(SI->getCondition()))
5059 Pred = Cmp->getPredicate();
5060 return TTI.getCmpSelInstrCost(
5061 I->getOpcode(), VectorTy, CondTy, Pred, Config.CostKind,
5062 {TTI::OK_AnyValue, TTI::OP_None}, {TTI::OK_AnyValue, TTI::OP_None}, I);
5063 }
5064 case Instruction::ICmp:
5065 case Instruction::FCmp: {
5066 Type *ValTy = I->getOperand(0)->getType();
5067
5069 [[maybe_unused]] Instruction *Op0AsInstruction =
5070 dyn_cast<Instruction>(I->getOperand(0));
5071 assert((!canTruncateToMinimalBitwidth(Op0AsInstruction, VF) ||
5072 InstrMinBWs == MinBWs.lookup(Op0AsInstruction)) &&
5073 "if both the operand and the compare are marked for "
5074 "truncation, they must have the same bitwidth");
5075 ValTy = IntegerType::get(ValTy->getContext(), InstrMinBWs);
5076 }
5077
5078 VectorTy = toVectorTy(ValTy, VF);
5079 return TTI.getCmpSelInstrCost(
5080 I->getOpcode(), VectorTy, CmpInst::makeCmpResultType(VectorTy),
5081 cast<CmpInst>(I)->getPredicate(), Config.CostKind,
5082 {TTI::OK_AnyValue, TTI::OP_None}, {TTI::OK_AnyValue, TTI::OP_None}, I);
5083 }
5084 case Instruction::Store:
5085 case Instruction::Load: {
5086 ElementCount Width = VF;
5087 if (Width.isVector()) {
5088 InstWidening Decision = getWideningDecision(I, Width);
5089 assert(Decision != CM_Unknown &&
5090 "CM decision should be taken at this point");
5093 if (Decision == CM_Scalarize)
5094 Width = ElementCount::getFixed(1);
5095 }
5096 VectorTy = toVectorTy(getLoadStoreType(I), Width);
5097 return getMemoryInstructionCost(I, VF);
5098 }
5099 case Instruction::BitCast:
5100 if (I->getType()->isPointerTy())
5101 return 0;
5102 [[fallthrough]];
5103 case Instruction::ZExt:
5104 case Instruction::SExt:
5105 case Instruction::FPToUI:
5106 case Instruction::FPToSI:
5107 case Instruction::FPExt:
5108 case Instruction::PtrToInt:
5109 case Instruction::IntToPtr:
5110 case Instruction::SIToFP:
5111 case Instruction::UIToFP:
5112 case Instruction::Trunc:
5113 case Instruction::FPTrunc: {
5114 // Computes the CastContextHint from a Load/Store instruction.
5115 auto ComputeCCH = [&](Instruction *I) -> TTI::CastContextHint {
5117 "Expected a load or a store!");
5118
5119 if (VF.isScalar() || !TheLoop->contains(I))
5121
5122 switch (getWideningDecision(I, VF)) {
5134 llvm_unreachable("Instr did not go through cost modelling?");
5137 }
5138
5139 llvm_unreachable("Unhandled case!");
5140 };
5141
5142 unsigned Opcode = I->getOpcode();
5144 // For Trunc, the context is the only user, which must be a StoreInst.
5145 if (Opcode == Instruction::Trunc || Opcode == Instruction::FPTrunc) {
5146 if (I->hasOneUse())
5147 if (StoreInst *Store = dyn_cast<StoreInst>(*I->user_begin()))
5148 CCH = ComputeCCH(Store);
5149 }
5150 // For Z/Sext, the context is the operand, which must be a LoadInst.
5151 else if (Opcode == Instruction::ZExt || Opcode == Instruction::SExt ||
5152 Opcode == Instruction::FPExt) {
5153 if (LoadInst *Load = dyn_cast<LoadInst>(I->getOperand(0)))
5154 CCH = ComputeCCH(Load);
5155 }
5156
5157 // We optimize the truncation of induction variables having constant
5158 // integer steps. The cost of these truncations is the same as the scalar
5159 // operation.
5160 if (isOptimizableIVTruncate(I, VF)) {
5161 auto *Trunc = cast<TruncInst>(I);
5162 return TTI.getCastInstrCost(Instruction::Trunc, Trunc->getDestTy(),
5163 Trunc->getSrcTy(), CCH, Config.CostKind,
5164 Trunc);
5165 }
5166
5167 Type *SrcScalarTy = I->getOperand(0)->getType();
5168 Instruction *Op0AsInstruction = dyn_cast<Instruction>(I->getOperand(0));
5169 if (canTruncateToMinimalBitwidth(Op0AsInstruction, VF))
5170 SrcScalarTy = IntegerType::get(SrcScalarTy->getContext(),
5171 MinBWs.lookup(Op0AsInstruction));
5172 Type *SrcVecTy =
5173 VectorTy->isVectorTy() ? toVectorTy(SrcScalarTy, VF) : SrcScalarTy;
5174
5176 // If the result type is <= the source type, there will be no extend
5177 // after truncating the users to the minimal required bitwidth.
5178 if (VectorTy->getScalarSizeInBits() <= SrcVecTy->getScalarSizeInBits() &&
5179 (I->getOpcode() == Instruction::ZExt ||
5180 I->getOpcode() == Instruction::SExt))
5181 return 0;
5182 }
5183
5184 return TTI.getCastInstrCost(Opcode, VectorTy, SrcVecTy, CCH,
5185 Config.CostKind, I);
5186 }
5187 case Instruction::Call:
5188 return getVectorCallCost(cast<CallInst>(I), VF);
5189 case Instruction::ExtractValue:
5190 return TTI.getInstructionCost(I, Config.CostKind);
5191 case Instruction::Alloca:
5192 // We cannot easily widen alloca to a scalable alloca, as
5193 // the result would need to be a vector of pointers.
5194 if (VF.isScalable())
5196 return TTI.getArithmeticInstrCost(Instruction::Mul, RetTy, Config.CostKind);
5197 case Instruction::Freeze:
5198 return TTI::TCC_Free;
5199 default:
5200 // This opcode is unknown. Assume that it is the same as 'mul'.
5201 return TTI.getArithmeticInstrCost(Instruction::Mul, VectorTy,
5202 Config.CostKind);
5203 } // end of switch.
5204}
5205
5207 // Ignore ephemeral values.
5209
5210 SmallVector<Value *, 4> DeadInterleavePointerOps;
5212
5213 // If a scalar epilogue is required, users outside the loop won't use
5214 // live-outs from the vector loop but from the scalar epilogue. Ignore them if
5215 // that is the case.
5216 bool RequiresScalarEpilogue = requiresScalarEpilogue(true);
5217 auto IsLiveOutDead = [this, RequiresScalarEpilogue](User *U) {
5218 return RequiresScalarEpilogue &&
5219 !TheLoop->contains(cast<Instruction>(U)->getParent());
5220 };
5221
5223 DFS.perform(LI);
5224 for (BasicBlock *BB : reverse(make_range(DFS.beginRPO(), DFS.endRPO())))
5225 for (Instruction &I : reverse(*BB)) {
5226 if (VecValuesToIgnore.contains(&I) || ValuesToIgnore.contains(&I))
5227 continue;
5228
5229 // Add instructions that would be trivially dead and are only used by
5230 // values already ignored to DeadOps to seed worklist.
5232 all_of(I.users(), [this, IsLiveOutDead](User *U) {
5233 return VecValuesToIgnore.contains(U) ||
5234 ValuesToIgnore.contains(U) || IsLiveOutDead(U);
5235 }))
5236 DeadOps.push_back(&I);
5237
5238 // For interleave groups, we only create a pointer for the start of the
5239 // interleave group. Queue up addresses of group members except the insert
5240 // position for further processing.
5241 if (isAccessInterleaved(&I)) {
5242 auto *Group = getInterleavedAccessGroup(&I);
5243 if (Group->getInsertPos() == &I)
5244 continue;
5245 Value *PointerOp = getLoadStorePointerOperand(&I);
5246 DeadInterleavePointerOps.push_back(PointerOp);
5247 }
5248
5249 // Queue branches for analysis. They are dead, if their successors only
5250 // contain dead instructions.
5251 if (isa<CondBrInst>(&I))
5252 DeadOps.push_back(&I);
5253 }
5254
5255 // Mark ops feeding interleave group members as free, if they are only used
5256 // by other dead computations.
5257 for (unsigned I = 0; I != DeadInterleavePointerOps.size(); ++I) {
5258 auto *Op = dyn_cast<Instruction>(DeadInterleavePointerOps[I]);
5259 if (!Op || !TheLoop->contains(Op) || any_of(Op->users(), [this](User *U) {
5260 Instruction *UI = cast<Instruction>(U);
5261 return !VecValuesToIgnore.contains(U) &&
5262 (!isAccessInterleaved(UI) ||
5263 getInterleavedAccessGroup(UI)->getInsertPos() == UI);
5264 }))
5265 continue;
5266 VecValuesToIgnore.insert(Op);
5267 append_range(DeadInterleavePointerOps, Op->operands());
5268 }
5269
5270 // Mark ops that would be trivially dead and are only used by ignored
5271 // instructions as free.
5272 BasicBlock *Header = TheLoop->getHeader();
5273
5274 // Returns true if the block contains only dead instructions. Such blocks will
5275 // be removed by VPlan-to-VPlan transforms and won't be considered by the
5276 // VPlan-based cost model, so skip them in the legacy cost-model as well.
5277 auto IsEmptyBlock = [this](BasicBlock *BB) {
5278 return all_of(*BB, [this](Instruction &I) {
5279 return ValuesToIgnore.contains(&I) || VecValuesToIgnore.contains(&I) ||
5281 });
5282 };
5283 for (unsigned I = 0; I != DeadOps.size(); ++I) {
5284 auto *Op = dyn_cast<Instruction>(DeadOps[I]);
5285
5286 // Check if the branch should be considered dead.
5287 if (auto *Br = dyn_cast_or_null<CondBrInst>(Op)) {
5288 BasicBlock *ThenBB = Br->getSuccessor(0);
5289 BasicBlock *ElseBB = Br->getSuccessor(1);
5290 // Don't considers branches leaving the loop for simplification.
5291 if (!TheLoop->contains(ThenBB) || !TheLoop->contains(ElseBB))
5292 continue;
5293 bool ThenEmpty = IsEmptyBlock(ThenBB);
5294 bool ElseEmpty = IsEmptyBlock(ElseBB);
5295 if ((ThenEmpty && ElseEmpty) ||
5296 (ThenEmpty && ThenBB->getSingleSuccessor() == ElseBB &&
5297 ElseBB->phis().empty()) ||
5298 (ElseEmpty && ElseBB->getSingleSuccessor() == ThenBB &&
5299 ThenBB->phis().empty())) {
5300 VecValuesToIgnore.insert(Br);
5301 DeadOps.push_back(Br->getCondition());
5302 }
5303 continue;
5304 }
5305
5306 // Skip any op that shouldn't be considered dead.
5307 if (!Op || !TheLoop->contains(Op) ||
5308 (isa<PHINode>(Op) && Op->getParent() == Header) ||
5310 any_of(Op->users(), [this, IsLiveOutDead](User *U) {
5311 return !VecValuesToIgnore.contains(U) &&
5312 !ValuesToIgnore.contains(U) && !IsLiveOutDead(U);
5313 }))
5314 continue;
5315
5316 // If all of Op's users are in ValuesToIgnore, add it to ValuesToIgnore
5317 // which applies for both scalar and vector versions. Otherwise it is only
5318 // dead in vector versions, so only add it to VecValuesToIgnore.
5319 if (all_of(Op->users(),
5320 [this](User *U) { return ValuesToIgnore.contains(U); }))
5321 ValuesToIgnore.insert(Op);
5322
5323 VecValuesToIgnore.insert(Op);
5324 append_range(DeadOps, Op->operands());
5325 }
5326
5327 // Ignore type-promoting instructions we identified during reduction
5328 // detection.
5329 for (const auto &Reduction : Legal->getReductionVars()) {
5330 const RecurrenceDescriptor &RedDes = Reduction.second;
5331 const SmallPtrSetImpl<Instruction *> &Casts = RedDes.getCastInsts();
5332 VecValuesToIgnore.insert_range(Casts);
5333 }
5334 // Ignore type-casting instructions we identified during induction
5335 // detection.
5336 for (const auto &Induction : Legal->getInductionVars()) {
5337 const InductionDescriptor &IndDes = Induction.second;
5338 VecValuesToIgnore.insert_range(IndDes.getCastInsts());
5339 }
5340}
5341
5342void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
5343 CM->collectValuesToIgnore();
5344 Config.collectElementTypesForWidening(&CM->ValuesToIgnore);
5345
5346 FixedScalableVFPair MaxFactors = CM->computeMaxVF(UserVF, UserIC);
5347 if (!MaxFactors) // Cases that should not to be vectorized nor interleaved.
5348 return;
5349
5350 Config.collectInLoopReductions();
5351 // Cases that may be vectorized may be optimized by unit stride predicates.
5352 // TODO: Currently unit stride predicates are added unconditionally, even if
5353 // they are not used for the selected VF (e.g. when only interleaving).
5354 if (MaxFactors.FixedVF.isVector() || MaxFactors.ScalableVF.isVector())
5355 Legal->collectUnitStridePredicates();
5356
5357 auto VPlan1 = tryToBuildVPlan1();
5358 if (!VPlan1)
5359 return;
5360
5361 LLVM_DEBUG(dbgs() << "LV: VPlan created successfully. Loop can be "
5362 "vectorized.\n");
5363
5364 if (!OrigLoop->isInnermost()) {
5365 // For outer loops, computeMaxVF returns a single non-scalar VF; build a
5366 // plan for that VF only.
5367 ElementCount VF =
5368 MaxFactors.FixedVF ? MaxFactors.FixedVF : MaxFactors.ScalableVF;
5369 buildVPlans(*VPlan1, VF, VF);
5371 return;
5372 }
5373
5374 // Compute the minimal bitwidths required for integer operations in the loop
5375 // for later use by the cost model.
5376 Config.computeMinimalBitwidths();
5377
5378 // Invalidate interleave groups if all blocks of loop will be predicated.
5379 if (CM->blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
5381 LLVM_DEBUG(
5382 dbgs()
5383 << "LV: Invalidate all interleaved groups due to fold-tail by masking "
5384 "which requires masked-interleaved support.\n");
5385 if (CM->InterleaveInfo.invalidateGroups())
5386 // Invalidating interleave groups also requires invalidating all decisions
5387 // based on them, which includes widening decisions and uniform and scalar
5388 // values.
5389 CM->invalidateCostModelingDecisions();
5390 }
5391
5392 if (CM->foldTailByMasking())
5393 Legal->prepareToFoldTailByMasking();
5394
5395 ElementCount MaxUserVF =
5396 UserVF.isScalable() ? MaxFactors.ScalableVF : MaxFactors.FixedVF;
5397 if (UserVF) {
5398 if (!ElementCount::isKnownLE(UserVF, MaxUserVF)) {
5400 "UserVF ignored because it may be larger than the maximal safe VF",
5401 "InvalidUserVF", ORE, OrigLoop);
5402 } else {
5404 "VF needs to be a power of two");
5405 // Collect the instructions (and their associated costs) that will be more
5406 // profitable to scalarize.
5407 CM->collectNonVectorizedAndSetWideningDecisions(UserVF);
5408 buildVPlans(*VPlan1, UserVF, UserVF);
5410 if (EpilogueUserVF.isVector() &&
5411 ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
5412 CM->collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
5413 buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF);
5414 }
5415 if (!VPlans.empty() && VPlans.front()->getSingleVF() == UserVF) {
5416 // For scalar VF, skip VPlan cost check as VPlan cost is designed for
5417 // vector VFs only.
5418 if (UserVF.isScalar() ||
5419 cost(*VPlans.front(), UserVF, /*RU=*/nullptr).isValid()) {
5420 LLVM_DEBUG(dbgs() << "LV: Using user VF " << UserVF << ".\n");
5422 return;
5423 }
5424 }
5425 VPlans.clear();
5426 reportVectorizationInfo("UserVF ignored because of invalid costs.",
5427 "InvalidCost", ORE, OrigLoop);
5428 }
5429 }
5430
5431 // Collect the Vectorization Factor Candidates.
5432 SmallVector<ElementCount> VFCandidates;
5433 for (auto VF = ElementCount::getFixed(1);
5434 ElementCount::isKnownLE(VF, MaxFactors.FixedVF); VF *= 2)
5435 VFCandidates.push_back(VF);
5436 for (auto VF = ElementCount::getScalable(1);
5437 ElementCount::isKnownLE(VF, MaxFactors.ScalableVF); VF *= 2)
5438 VFCandidates.push_back(VF);
5439
5440 for (const auto &VF : VFCandidates) {
5441 // Collect Uniform and Scalar instructions after vectorization with VF.
5442 CM->collectNonVectorizedAndSetWideningDecisions(VF);
5443 }
5444
5445 buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF);
5446 buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF);
5447
5449}
5450
5454 bool ReusePrintingSlotTracker)
5455 : TTI(Config.getTTI()), TLI(TLI), LLVMCtx(Plan.getContext()), CM(CM),
5457 L(Config.getLoop()) {
5458#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5459 if (ReusePrintingSlotTracker)
5460 PlanForSlotTracker = &Plan;
5461#endif
5462}
5463
5465 ElementCount VF) const {
5466 InstructionCost Cost = CM.getInstructionCost(UI, VF);
5467 if (Cost.isValid() && ForceTargetInstructionCost.getNumOccurrences())
5469 return Cost;
5470}
5471
5472bool VPCostContext::skipCostComputation(Instruction *UI, bool IsVector) const {
5473 return CM.ValuesToIgnore.contains(UI) ||
5474 (IsVector && CM.VecValuesToIgnore.contains(UI)) ||
5475 SkipCostComputation.contains(UI);
5476}
5477
5483
5485 return CM.isScalarWithPredication(I, VF) ||
5486 CM.isUniformAfterVectorization(I, VF) || CM.isForcedScalar(I, VF) ||
5487 (VF.isVector() && CM.isProfitableToScalarize(I, VF));
5488}
5489
5491 return CM.isMaskRequired(I);
5492}
5493
5497 return TC && TC->getValue().ule(VF.getKnownMinValue());
5498}
5499
5501LoopVectorizationPlanner::precomputeCosts(VPlan &Plan, ElementCount VF,
5502 VPCostContext &CostCtx) const {
5504
5505 // If the vector loop gets executed exactly once with the given VF, ignore the
5506 // costs of comparison and induction instructions, as they'll get simplified
5507 // away.
5508 // TODO: Remove this code after stepping away from the legacy cost model and
5509 // adding code to simplify VPlans before calculating their costs.
5510 auto TC = getSmallConstantTripCount(PSE.getSE(), OrigLoop);
5511 if (TC == VF && !Plan.hasTailFolded())
5512 addFullyUnrolledInstructionsToIgnore(OrigLoop, Legal->getInductionVars(),
5513 CostCtx.SkipCostComputation);
5514
5515 // Pre-compute the costs for branches except for the backedge, as the number
5516 // of replicate regions in a VPlan may not directly match the number of
5517 // branches, which would lead to different decisions.
5518 // TODO: Compute cost of branches for each replicate region in the VPlan,
5519 // which is more accurate than the legacy cost model.
5520 for (BasicBlock *BB : OrigLoop->blocks()) {
5521 if (CostCtx.skipCostComputation(BB->getTerminator(), VF.isVector()))
5522 continue;
5523 CostCtx.SkipCostComputation.insert(BB->getTerminator());
5524 if (BB == OrigLoop->getLoopLatch())
5525 continue;
5526 auto BranchCost = CostCtx.getLegacyCost(BB->getTerminator(), VF);
5527 Cost += BranchCost;
5528 }
5529
5530 // Don't apply special costs when instruction cost is forced to make sure the
5531 // forced cost is used for each recipe.
5532 if (ForceTargetInstructionCost.getNumOccurrences())
5533 return Cost;
5534
5535 // Pre-compute costs for instructions that are forced-scalar or profitable to
5536 // scalarize. For most such instructions, their scalarization costs are
5537 // accounted for here using the legacy cost model. However, some opcodes
5538 // are excluded from these precomputed scalarization costs and are instead
5539 // modeled later by the VPlan cost model (see UseVPlanCostModel below).
5540 for (Instruction *ForcedScalar : CostCtx.CM.ForcedScalars[VF]) {
5541 if (CostCtx.skipCostComputation(ForcedScalar, VF.isVector()))
5542 continue;
5543 CostCtx.SkipCostComputation.insert(ForcedScalar);
5544 InstructionCost ForcedCost = CostCtx.getLegacyCost(ForcedScalar, VF);
5545 LLVM_DEBUG({
5546 dbgs() << "Cost of " << ForcedCost << " for VF " << VF
5547 << ": forced scalar " << *ForcedScalar << "\n";
5548 });
5549 Cost += ForcedCost;
5550 }
5551
5552 // Don't apply legacy scalarization costs if nothing remains scalar &
5553 // predicated.
5554 if (!hasReplicatorRegion(Plan))
5555 return Cost;
5556
5557 auto UseVPlanCostModel = [](Instruction *I) -> bool {
5558 switch (I->getOpcode()) {
5559 case Instruction::SDiv:
5560 case Instruction::UDiv:
5561 case Instruction::SRem:
5562 case Instruction::URem:
5563 return true;
5564 default:
5565 return false;
5566 }
5567 };
5568 for (const auto &[Scalarized, ScalarCost] : CostCtx.CM.InstsToScalarize[VF]) {
5569 if (UseVPlanCostModel(Scalarized) ||
5570 CostCtx.skipCostComputation(Scalarized, VF.isVector()))
5571 continue;
5572 CostCtx.SkipCostComputation.insert(Scalarized);
5573 LLVM_DEBUG({
5574 dbgs() << "Cost of " << ScalarCost << " for VF " << VF
5575 << ": profitable to scalarize " << *Scalarized << "\n";
5576 });
5577 Cost += ScalarCost;
5578 }
5579
5580 return Cost;
5581}
5582
5583#ifndef NDEBUG
5584/// Returns the frequency with which \p VPBB executes, as recorded on its
5585/// recipes. All recipes of a block share the same frequency.
5586static std::optional<VPExecutionFrequency>
5588 if (VPBB->empty())
5589 return std::nullopt;
5590 return cast<VPInstruction>(&VPBB->front())->getExecutionFrequency();
5591}
5592#endif
5593
5594InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
5595 VPRegisterUsage *RU) const {
5596 VPCostContext CostCtx(*TLI, Plan, *CM, Config,
5597 /*ReusePrintingSlotTracker=*/true);
5598 InstructionCost Cost = precomputeCosts(Plan, VF, CostCtx);
5599 LLVM_DEBUG(dbgs() << "Precomputed costs for VF " << VF << ": " << Cost
5600 << '\n');
5601
5602 // Now compute and add the VPlan-based cost.
5603 Cost += Plan.cost(VF, CostCtx);
5604
5605 // Add the cost of spills due to excess register usage
5606 if (RU && Config.shouldConsiderRegPressureForVF(VF)) {
5607 InstructionCost SpillCost =
5608 RU->spillCost(TTI, Config.CostKind, ForceTargetNumVectorRegs);
5609 LLVM_DEBUG(dbgs() << "Spill costs for VF " << VF << ": " << SpillCost
5610 << '\n');
5611 Cost += SpillCost;
5612 }
5613
5614#ifndef NDEBUG
5615 unsigned EstimatedWidth =
5616 estimateElementCount(VF, Config.getVScaleForTuning());
5617 LLVM_DEBUG(dbgs() << "Cost for VF " << VF << ": " << Cost
5618 << " (Estimated cost per lane: ");
5619 if (Cost.isValid()) {
5620 APFloat CostPerLane(APFloat::IEEEdouble());
5621 APFloat EstimatedWidthAsAPFloat(APFloat::IEEEdouble());
5622 (void)CostPerLane.convertFromAPInt(APInt(64, (uint64_t)Cost.getValue()),
5623 false, APFloat::rmTowardZero);
5624 (void)EstimatedWidthAsAPFloat.convertFromAPInt(
5625 APInt(64, (uint64_t)EstimatedWidth), false, APFloat::rmTowardZero);
5626 (void)CostPerLane.divide(EstimatedWidthAsAPFloat, APFloat::rmTowardZero);
5627
5628 SmallString<16> Str;
5629 CostPerLane.toString(Str, 3);
5630 LLVM_DEBUG(dbgs() << Str);
5631 } else /* No point dividing an invalid cost - it will still be invalid */
5632 LLVM_DEBUG(dbgs() << "Invalid");
5633 LLVM_DEBUG(dbgs() << ")\n");
5634#endif
5635 return Cost;
5636}
5637
5638std::pair<VectorizationFactor, VPlan *>
5640 if (VPlans.empty())
5641 return {VectorizationFactor::Disabled(), nullptr};
5642 // If there is a single VPlan with a single VF, return it directly.
5643 VPlan &FirstPlan = *VPlans[0];
5644
5645 ElementCount UserVF = Config.getHints().getWidth();
5646 if (VPlans.size() == 1) {
5647 // For outer loops, the plan has a single vector VF determined by the
5648 // heuristic.
5649 assert((FirstPlan.hasScalarVFOnly() || hasPlanWithVF(UserVF) ||
5650 FirstPlan.isOuterLoop()) &&
5651 "must have a single scalar VF, UserVF or an outer loop");
5652 return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
5653 }
5654
5655 if (hasPlanWithVF(UserVF) && hasForcedEpilogueVF() && VPlans.size() == 2) {
5656 assert(VPlans[0]->getSingleVF() == UserVF &&
5657 "expected second plan to be for the forced UserVF");
5658 assert(VPlans[1]->getSingleVF() == EpilogueVectorizationForceVF &&
5659 "expected first plan to be for the forced epilogue VF");
5660 return {VectorizationFactor(UserVF, 0, 0), VPlans[0].get()};
5661 }
5662
5663 LLVM_DEBUG(dbgs() << "LV: Computing best VF using cost kind: "
5664 << (Config.CostKind == TTI::TCK_RecipThroughput
5665 ? "Reciprocal Throughput\n"
5666 : Config.CostKind == TTI::TCK_Latency
5667 ? "Instruction Latency\n"
5668 : Config.CostKind == TTI::TCK_CodeSize ? "Code Size\n"
5669 : Config.CostKind == TTI::TCK_SizeAndLatency
5670 ? "Code Size and Latency\n"
5671 : "Unknown\n"));
5672
5674 assert(FirstPlan.hasVF(ScalarVF) &&
5675 "More than a single plan/VF w/o any plan having scalar VF");
5676
5677 // TODO: Compute scalar cost using VPlan-based cost model.
5678 InstructionCost ScalarCost = CM->expectedCost(ScalarVF);
5679 LLVM_DEBUG(dbgs() << "LV: Scalar loop costs: " << ScalarCost << ".\n");
5680 VectorizationFactor ScalarFactor(ScalarVF, ScalarCost, ScalarCost);
5681 VectorizationFactor BestFactor = ScalarFactor;
5682
5683 bool ForceVectorization =
5684 Config.getHints().getForce() == LoopVectorizeHints::FK_Enabled;
5685 if (ForceVectorization) {
5686 // Ignore scalar width, because the user explicitly wants vectorization.
5687 // Initialize cost to max so that VF = 2 is, at least, chosen during cost
5688 // evaluation.
5689 BestFactor.Cost = InstructionCost::getMax();
5690 }
5691
5692 VPlan *PlanForBestVF = &FirstPlan;
5693 ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), OrigLoop);
5694
5695 for (auto &P : VPlans) {
5696 ArrayRef<ElementCount> VFs(P->vectorFactors().begin(),
5697 P->vectorFactors().end());
5698
5699 // For loops where the Trip Count is below the TailFoldingThreshold, only
5700 // consider the largest VF to result in at most one vector iteration, and at
5701 // most one scalar iteration.
5702 // FIXME: Encode this decision directly in LVPlanner.
5703 if (!ForceVectorization && P->hasScalarTail() && ExactTC.isFixed() &&
5704 ExactTC.getFixedValue() > 0 &&
5705 ExactTC.getFixedValue() <= TTI.getMinTripCountTailFoldingThreshold()) {
5706 VFs = VFs.take_back(1);
5707 }
5708
5710 bool ConsiderRegPressure = any_of(VFs, [this](ElementCount VF) {
5711 return Config.shouldConsiderRegPressureForVF(VF);
5712 });
5714 RUs = calculateRegisterUsageForPlan(*P, VFs, TTI);
5715
5716 for (unsigned I = 0; I < VFs.size(); I++) {
5717 ElementCount VF = VFs[I];
5718 if (VF.isScalar())
5719 continue;
5720 if (!ForceVectorization && !willGenerateVectors(*P, VF, TTI)) {
5721 LLVM_DEBUG(
5722 dbgs()
5723 << "LV: Not considering vector loop of width " << VF
5724 << " because it will not generate any vector instructions.\n");
5725 continue;
5726 }
5727 if (Config.OptForSize && !ForceVectorization && hasReplicatorRegion(*P)) {
5728 LLVM_DEBUG(
5729 dbgs()
5730 << "LV: Not considering vector loop of width " << VF
5731 << " because it would cause replicated blocks to be generated,"
5732 << " which isn't allowed when optimizing for size.\n");
5733 continue;
5734 }
5735
5737 cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
5738 VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
5739
5740 if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
5741 BestFactor = CurrentFactor;
5742 PlanForBestVF = P.get();
5743 }
5744
5745 // If profitable add it to ProfitableVF list.
5746 if (isMoreProfitable(CurrentFactor, ScalarFactor, P->hasScalarTail()))
5747 ProfitableVFs.push_back(CurrentFactor);
5748 }
5749 }
5750
5751 VPlan &BestPlan = *PlanForBestVF;
5752
5753 assert((BestFactor.Width.isScalar() || BestFactor.ScalarCost > 0) &&
5754 "when vectorizing, the scalar cost must be computed.");
5755
5756 LLVM_DEBUG(dbgs() << "LV: Selecting VF: " << BestFactor.Width << ".\n");
5757 return {BestFactor, &BestPlan};
5758}
5759
5761 Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
5763 std::unique_ptr<LoopVectorizationCostModel> CM, VFSelectionContext &Config,
5766 std::function<const BranchProbabilityInfo &()> GetBPI)
5767 : OrigLoop(L), LI(LI), DT(DT), TLI(TLI), TTI(TTI), Legal(Legal),
5768 CM(std::move(CM)), Config(Config), IAI(IAI), PSE(PSE), ORE(ORE),
5769 GetBPI(GetBPI) {}
5770
5772
5774
5776 ElementCount BestVF, unsigned BestUF, VPlan &BestVPlan,
5778 EpilogueVectorizationKind EpilogueVecKind) {
5779 assert(BestVPlan.hasVF(BestVF) &&
5780 "Trying to execute plan with unsupported VF");
5781 assert(BestVPlan.hasUF(BestUF) &&
5782 "Trying to execute plan with unsupported UF");
5783 if (BestVPlan.hasEarlyExit())
5784 ++LoopsEarlyExitVectorized;
5785
5787 *PSE.getSE(), TTI, Config.CostKind, BestVF, BestUF);
5788 // TODO: Move to VPlan transform stage once the transition to the VPlan-based
5789 // cost model is complete for better cost estimates.
5790 RUN_VPLAN_PASS(VPlanTransforms::unrollByUF, BestVPlan, BestUF);
5794 bool HasBranchWeights =
5795 hasBranchWeightMD(*OrigLoop->getLoopLatch()->getTerminator());
5796 if (HasBranchWeights) {
5797 std::optional<unsigned> VScale = Config.getVScaleForTuning();
5799 BestVPlan, BestVF, VScale);
5800 }
5801
5802 if (vputils::findIncomingAliasMask(BestVPlan)) {
5803 assert(BestVPlan.hasTailFolded() && "Expected tail folding to be enabled");
5805 *Legal->getRuntimePointerChecking()->getDiffChecks(),
5806 HasBranchWeights);
5807 ++LoopsPartialAliasVectorized;
5808 }
5809
5810 // Retrieving VectorPH now when it's easier while VPlan still has Regions.
5811 VPBasicBlock *VectorPH = cast<VPBasicBlock>(BestVPlan.getVectorPreheader());
5812
5814 BestVF, BestUF, PSE);
5815 RUN_VPLAN_PASS(VPlanTransforms::optimizeForVFAndUF, BestVPlan, BestVF, BestUF,
5816 PSE);
5818 // Check if scalar epilogue is required, before simplifying constant branches.
5819 const bool RequiresScalarEpilogue = BestVPlan.requiresScalarEpilogue();
5820 if (EpilogueVecKind == EpilogueVectorizationKind::None)
5822 /*OnlyLatches=*/false);
5823 if (BestVPlan.getEntry()->getSingleSuccessor() ==
5824 BestVPlan.getScalarPreheader()) {
5825 // TODO: The vector loop would be dead, should not even try to vectorize.
5826 ORE->emit([&]() {
5827 return OptimizationRemarkAnalysis(DEBUG_TYPE, "VectorizationDead",
5828 OrigLoop->getStartLoc(),
5829 OrigLoop->getHeader())
5830 << "Created vector loop never executes due to insufficient trip "
5831 "count.";
5832 });
5834 }
5835
5837
5839 // Convert the exit condition to AVLNext == 0 for EVL tail folded loops.
5841 // Regions are dissolved after optimizing for VF and UF, which completely
5842 // removes unneeded loop regions first.
5843 const bool HasTailFolded = BestVPlan.hasTailFolded();
5845 // Expand BranchOnTwoConds after dissolution, when latch has direct access to
5846 // its successors.
5848 // Convert loops with variable-length stepping after regions are dissolved.
5850 // Remove dead back-edges for single-iteration loops with BranchOnCond(true).
5851 // Only process loop latches to avoid removing edges from the middle block,
5852 // which may be needed for epilogue vectorization.
5854 /*OnlyLatches=*/true);
5856 VectorPH);
5857 std::optional<uint64_t> MaxRuntimeStep = getMaxRuntimeElementCount(
5858 BestVF * BestUF, *OrigLoop->getHeader()->getParent());
5859
5860 assert((LI->getUniqueLatchExitBlock(*OrigLoop) || RequiresScalarEpilogue) &&
5861 "loops not exiting via the latch without required epilogue?");
5863 VectorPH, HasTailFolded, RequiresScalarEpilogue,
5864 &BestVPlan.getVFxUF(), MaxRuntimeStep);
5866 BestVF);
5867 // Limit expansions to VPInstruction to when not vectorizing the epilogue.
5868 // Currently this code path still relies on code re-using SCEVs expanded
5869 // directly to IR instructions.
5870 if (EpilogueVecKind == EpilogueVectorizationKind::None)
5872 *PSE.getSE());
5875 // Removing branches and incoming values may expose additional simplification
5876 // opportunities.
5878 /*OnlyLatches=*/EpilogueVecKind !=
5881 RUN_VPLAN_PASS(VPlanTransforms::simplifyKnownEVL, BestVPlan, BestVF, PSE);
5882
5883 // 0. Generate SCEV-dependent code in the entry, including TripCount, before
5884 // making any changes to the CFG.
5885 DenseMap<const SCEV *, Value *> ExpandedSCEVs =
5886 RUN_VPLAN_PASS(VPlanTransforms::expandSCEVs, BestVPlan, *PSE.getSE());
5887
5888 // Perform the actual loop transformation.
5889 VPTransformState State(&TTI, BestVF, LI, DT, ILV.AC, ILV.Builder, &BestVPlan,
5890 OrigLoop->getParentLoop());
5891
5892#ifdef EXPENSIVE_CHECKS
5893 assert(DT->verify(DominatorTree::VerificationLevel::Fast));
5894#endif
5895
5896 // 1. Set up the skeleton for vectorization, including vector pre-header and
5897 // middle block. The vector loop is created during VPlan execution.
5898 State.CFG.PrevBB = ILV.createVectorizedLoopSkeleton();
5899 if (VPBasicBlock *ScalarPH = BestVPlan.getScalarPreheader())
5900 replaceVPBBWithIRVPBB(ScalarPH, State.CFG.PrevBB->getSingleSuccessor(),
5901 &BestVPlan);
5903
5904 assert(verifyVPlanIsValid(BestVPlan) && "final VPlan is invalid");
5905
5906 // After vectorization, the exit blocks of the original loop will have
5907 // additional predecessors. Invalidate SCEVs for the exit phis in case SE
5908 // looked through single-entry phis.
5909 ScalarEvolution &SE = *PSE.getSE();
5910 for (VPIRBasicBlock *Exit : BestVPlan.getExitBlocks()) {
5911 if (!Exit->hasPredecessors())
5912 continue;
5913 for (VPRecipeBase &PhiR : Exit->phis())
5915 &cast<VPIRPhi>(PhiR).getIRPhi());
5916 }
5917
5918 // Query whether the target wants loops it vectorizes to remain eligible for
5919 // runtime unrolling. Do this here, on the original loop and before its SCEV
5920 // is forgotten below.
5922 TTI.getUnrollingPreferences(OrigLoop, SE, UP, ORE);
5923 bool UnrollVectorizedLoop = UP.UnrollVectorizedLoop;
5924
5925 // Forget the original loop and block dispositions.
5926 SE.forgetLoop(OrigLoop);
5928
5929 //===------------------------------------------------===//
5930 //
5931 // Notice: any optimization or new instruction that go
5932 // into the code below should also be implemented in
5933 // the cost-model.
5934 //
5935 //===------------------------------------------------===//
5936
5937 // Retrieve loop information before executing the plan, which may remove the
5938 // original loop, if it becomes unreachable.
5939 MDNode *LID = OrigLoop->getLoopID();
5940 unsigned OrigLoopInvocationWeight = 0;
5941 std::optional<unsigned> OrigAverageTripCount =
5942 getLoopEstimatedTripCount(OrigLoop, &OrigLoopInvocationWeight);
5943
5944 BestVPlan.execute(&State);
5945
5946 // 2.6. Maintain Loop Hints
5947 // Keep all loop hints from the original loop on the vector loop (we'll
5948 // replace the vectorizer-specific hints below).
5949 VPBasicBlock *HeaderVPBB = vputils::getFirstLoopHeader(BestVPlan, State.VPDT);
5950 // Add metadata to disable runtime unrolling a scalar loop when there
5951 // are no runtime checks about strides and memory. A scalar loop that is
5952 // rarely used is not worth unrolling.
5953 bool DisableRuntimeUnroll = !ILV.RTChecks.hasChecks() && !BestVF.isScalar();
5955 HeaderVPBB ? LI->getLoopFor(State.CFG.VPBB2IRBB.lookup(HeaderVPBB))
5956 : nullptr,
5957 HeaderVPBB, BestVPlan,
5958 EpilogueVecKind == EpilogueVectorizationKind::Epilogue, LID,
5959 OrigAverageTripCount, OrigLoopInvocationWeight,
5960 estimateElementCount(BestVF * BestUF, Config.getVScaleForTuning()),
5961 DisableRuntimeUnroll, UnrollVectorizedLoop);
5962
5963 // 3. Fix the vectorized code: take care of header phi's, live-outs,
5964 // predication, updating analyses.
5965 ILV.fixVectorizedLoop(State);
5966
5967 // Wrap the generated blocks in VPIRBasicBlocks, so they can be used in the
5968 // epilogue plan.
5969 if (EpilogueVecKind == EpilogueVectorizationKind::MainLoop)
5971 vp_depth_first_shallow(BestVPlan.getEntry()))))
5972 if (!isa<VPIRBasicBlock>(VPBB))
5973 replaceVPBBWithIRVPBB(VPBB, State.CFG.VPBB2IRBB.at(VPBB), &BestVPlan);
5974
5975 return ExpandedSCEVs;
5976}
5977
5978//===--------------------------------------------------------------------===//
5979// EpilogueVectorizerEpilogueLoop
5980//===--------------------------------------------------------------------===//
5981
5982/// This function creates a new scalar preheader, using the previous one as
5983/// entry block to the epilogue VPlan. The minimum iteration check is being
5984/// represented in VPlan.
5986 BasicBlock *NewScalarPH = createScalarPreheader("vec.epilog.");
5987 BasicBlock *OriginalScalarPH = NewScalarPH->getSinglePredecessor();
5988 OriginalScalarPH->setName("vec.epilog.iter.check");
5989 VPIRBasicBlock *NewEntry = Plan.createVPIRBasicBlock(OriginalScalarPH);
5990 VPBasicBlock *OldEntry = Plan.getEntry();
5991 for (auto &R : make_early_inc_range(*OldEntry)) {
5992 // Skip moving VPIRInstructions (including VPIRPhis), which are unmovable by
5993 // defining.
5994 if (isa<VPIRInstruction>(&R))
5995 continue;
5996 R.moveBefore(*NewEntry, NewEntry->end());
5997 }
5998
5999 VPBlockUtils::reassociateBlocks(OldEntry, NewEntry);
6000
6002
6003 // Model the skeleton from the main vector loop in the epilogue plan.
6005 NewEntry);
6006
6007 return OriginalScalarPH;
6008}
6009
6011 return CM.isPredicatedInst(I);
6012}
6013
6015 return CM.TTI.prefersVectorizedAddressing();
6016}
6017
6019 VFRange &Range) {
6020 assert((VPI->getOpcode() == Instruction::Load ||
6021 VPI->getOpcode() == Instruction::Store) &&
6022 "Must be called with either a load or store");
6024
6025 auto WillWiden = [&](ElementCount VF) -> bool {
6027 CM.getWideningDecision(I, VF);
6029 "CM decision should be taken at this point.");
6031 return true;
6032 if (CM.isScalarAfterVectorization(I, VF) ||
6033 CM.isProfitableToScalarize(I, VF))
6034 return false;
6036 };
6037
6039 return nullptr;
6040
6041 // If a mask is not required, drop it - use unmasked version for safe loads.
6042 // TODO: Determine if mask is needed in VPlan.
6043 VPValue *Mask = CM.isMaskRequired(I) ? VPI->getMask() : nullptr;
6044
6045 // Determine if the pointer operand of the access is either consecutive or
6046 // reverse consecutive.
6048 CM.getWideningDecision(I, Range.Start);
6050 bool Consecutive =
6052
6053 VPValue *Ptr = VPI->getOpcode() == Instruction::Load ? VPI->getOperand(0)
6054 : VPI->getOperand(1);
6055 Builder.setInsertPoint(VPI);
6056 if (Consecutive) {
6057 Ptr = Builder.createConsecutiveVectorPointer(Ptr, getLoadStoreType(I),
6058 Reverse, VPI->getDebugLoc());
6059 }
6060
6061 if (Reverse && Mask)
6062 Mask = Builder.createNaryOp(VPInstruction::Reverse, Mask, I->getDebugLoc());
6063
6064 if (VPI->getOpcode() == Instruction::Load) {
6065 auto *Load = cast<LoadInst>(I);
6066 auto *LoadR = Builder.createWidenLoad(*Load, Ptr, Mask, Consecutive, *VPI,
6067 Load->getDebugLoc());
6068 if (Reverse)
6069 return Builder.createNaryOp(VPInstruction::Reverse, LoadR,
6070 LoadR->getDebugLoc());
6071 return LoadR;
6072 }
6073
6075 VPValue *StoredVal = VPI->getOperand(0);
6076 if (Reverse)
6077 StoredVal = Builder.createNaryOp(VPInstruction::Reverse, StoredVal,
6078 Store->getDebugLoc());
6079 return Builder.createWidenStore(*Store, Ptr, StoredVal, Mask, Consecutive,
6080 *VPI, Store->getDebugLoc());
6081}
6082
6083bool VPRecipeBuilder::shouldWiden(Instruction *I, VFRange &Range) const {
6085 "Instruction should have been handled earlier");
6086 // Instruction should be widened, unless it is scalar after vectorization,
6087 // scalarization is profitable or it is predicated.
6088 auto WillScalarize = [this, I](ElementCount VF) -> bool {
6089 return CM.isScalarAfterVectorization(I, VF) ||
6090 CM.isProfitableToScalarize(I, VF) ||
6091 CM.isScalarWithPredication(I, VF);
6092 };
6094 Range);
6095}
6096
6097VPRecipeWithIRFlags *VPRecipeBuilder::tryToWiden(VPInstruction *VPI) {
6098 auto *I = VPI->getUnderlyingInstr();
6099 switch (VPI->getOpcode()) {
6100 default:
6101 return nullptr;
6102 case Instruction::SDiv:
6103 case Instruction::UDiv:
6104 case Instruction::SRem:
6105 case Instruction::URem:
6106 // If not provably safe, use a masked intrinsic.
6107 if (CM.isPredicatedInst(I))
6108 return new VPWidenIntrinsicRecipe(
6110 I->getType(), {}, {}, VPI->getDebugLoc());
6111 [[fallthrough]];
6112 case Instruction::Add:
6113 case Instruction::And:
6114 case Instruction::AShr:
6115 case Instruction::FAdd:
6116 case Instruction::FCmp:
6117 case Instruction::FDiv:
6118 case Instruction::FMul:
6119 case Instruction::FNeg:
6120 case Instruction::FRem:
6121 case Instruction::FSub:
6122 case Instruction::ICmp:
6123 case Instruction::LShr:
6124 case Instruction::Mul:
6125 case Instruction::Or:
6126 case Instruction::Select:
6127 case Instruction::Shl:
6128 case Instruction::Sub:
6129 case Instruction::Xor:
6130 case Instruction::Freeze:
6131 return new VPWidenRecipe(*I, VPI->operandsWithoutMask(), *VPI, *VPI,
6132 VPI->getDebugLoc());
6133 case Instruction::ExtractValue: {
6135 auto *EVI = cast<ExtractValueInst>(I);
6136 assert(EVI->getNumIndices() == 1 && "Expected one extractvalue index");
6137 unsigned Idx = EVI->getIndices()[0];
6138 NewOps.push_back(Plan.getConstantInt(32, Idx));
6139 return new VPWidenRecipe(*I, NewOps, *VPI, *VPI, VPI->getDebugLoc());
6140 }
6141 };
6142}
6143
6145 if (VPI->getOpcode() != Instruction::Store)
6146 return nullptr;
6147
6148 auto HistInfo =
6149 Legal->getHistogramInfo(cast<StoreInst>(VPI->getUnderlyingInstr()));
6150 if (!HistInfo)
6151 return nullptr;
6152
6153 const HistogramInfo *HI = *HistInfo;
6154 // FIXME: Support other operations.
6155 unsigned Opcode = HI->Update->getOpcode();
6156 assert((Opcode == Instruction::Add || Opcode == Instruction::Sub) &&
6157 "Histogram update operation must be an Add or Sub");
6158
6160 // Bucket address.
6161 HGramOps.push_back(VPI->getOperand(1));
6162 // Increment value.
6163 HGramOps.push_back(Plan.getOrAddLiveIn(HI->Update->getOperand(1)));
6164
6165 // In case of predicated execution (due to tail-folding, or conditional
6166 // execution, or both), pass the relevant mask.
6167 if (CM.isMaskRequired(HI->Store))
6168 HGramOps.push_back(VPI->getMask());
6169
6170 return new VPHistogramRecipe(Opcode, HGramOps, cast<VPIRMetadata>(*VPI),
6171 VPI->getDebugLoc());
6172}
6173
6175 VPInstruction *VPI, VPBuilder &FinalRedStoresBuilder) {
6176 StoreInst *SI;
6177 if ((SI = dyn_cast<StoreInst>(VPI->getUnderlyingInstr())) &&
6178 Legal->isInvariantAddressOfReduction(SI->getPointerOperand())) {
6179 // Only create recipe for the final invariant store of the reduction.
6180 if (Legal->isInvariantStoreOfReduction(SI)) {
6181 VPValue *Val = VPI->getOperand(0);
6182 VPValue *Addr = VPI->getOperand(1);
6183 // We need to store the exiting value of the reduction, so use the blend
6184 // if tail folded.
6185 if (auto *Blend = VPlanPatternMatch::findUserOf<VPBlendRecipe>(Val))
6186 Val = Blend;
6187 [[maybe_unused]] auto *Rdx =
6189 assert((isa<VPIRValue>(Val) || !Rdx || Rdx->getBackedgeValue() == Val) &&
6190 "Store of reduction thats not the backedge value?");
6191 auto *Recipe = new VPReplicateRecipe(
6192 SI, {Val, Addr}, true /* IsUniform */, nullptr /*Mask*/, *VPI, *VPI,
6193 VPI->getDebugLoc());
6194 FinalRedStoresBuilder.insert(Recipe);
6195 }
6196 VPI->eraseFromParent();
6197 return true;
6198 }
6199
6200 return false;
6201}
6202
6204 VFRange &Range) {
6205 auto *I = VPI->getUnderlyingInstr();
6207 [&](ElementCount VF) { return CM.isUniformAfterVectorization(I, VF); },
6208 Range);
6209
6210 bool IsPredicated = CM.isPredicatedInst(I);
6211
6212 // Even if the instruction is not marked as uniform, there are certain
6213 // intrinsic calls that can be effectively treated as such, so we check for
6214 // them here. Conservatively, we only do this for scalable vectors, since
6215 // for fixed-width VFs we can always fall back on full scalarization.
6216 if (!IsUniform && Range.Start.isScalable() && isa<IntrinsicInst>(I)) {
6217 switch (cast<IntrinsicInst>(I)->getIntrinsicID()) {
6218 case Intrinsic::assume:
6219 case Intrinsic::lifetime_start:
6220 case Intrinsic::lifetime_end:
6221 // For scalable vectors if one of the operands is variant then we still
6222 // want to mark as uniform, which will generate one instruction for just
6223 // the first lane of the vector. We can't scalarize the call in the same
6224 // way as for fixed-width vectors because we don't know how many lanes
6225 // there are.
6226 //
6227 // The reasons for doing it this way for scalable vectors are:
6228 // 1. For the assume intrinsic generating the instruction for the first
6229 // lane is still be better than not generating any at all. For
6230 // example, the input may be a splat across all lanes.
6231 // 2. For the lifetime start/end intrinsics the pointer operand only
6232 // does anything useful when the input comes from a stack object,
6233 // which suggests it should always be uniform. For non-stack objects
6234 // the effect is to poison the object, which still allows us to
6235 // remove the call.
6236 IsUniform = true;
6237 break;
6238 default:
6239 break;
6240 }
6241 }
6242 VPValue *BlockInMask = nullptr;
6243 if (!IsPredicated) {
6244 // Finalize the recipe for Instr, first if it is not predicated.
6245 LLVM_DEBUG(dbgs() << "LV: Scalarizing:" << *I << "\n");
6246 } else {
6247 LLVM_DEBUG(dbgs() << "LV: Scalarizing and predicating:" << *I << "\n");
6248 // Instructions marked for predication are replicated and a mask operand is
6249 // added initially. Masked replicate recipes will later be placed under an
6250 // if-then construct to prevent side-effects. Generate recipes to compute
6251 // the block mask for this region.
6252 BlockInMask = VPI->getMask();
6253 }
6254
6255 // Note that there is some custom logic to mark some intrinsics as uniform
6256 // manually above for scalable vectors, which this assert needs to account for
6257 // as well.
6258 assert((Range.Start.isScalar() || !IsUniform || !IsPredicated ||
6259 (Range.Start.isScalable() && isa<IntrinsicInst>(I))) &&
6260 "Should not predicate a uniform recipe");
6261 if (IsUniform) {
6263 VPI->getOpcode(), VPI->operandsWithoutMask(), BlockInMask, *VPI, *VPI,
6264 VPI->getDebugLoc(), VPI->getScalarType(), I);
6265 }
6266 auto *Recipe = new VPReplicateRecipe(I, VPI->operandsWithoutMask(),
6267 /*IsSingleScalar=*/false, BlockInMask,
6268 *VPI, *VPI, VPI->getDebugLoc());
6269 return Recipe;
6270}
6271
6274 VFRange &Range) {
6275 assert(!R->isPhi() && "phis must be handled earlier");
6276 auto *VPI = cast<VPInstruction>(R);
6277 assert(VPI->getOpcode() != Instruction::Call &&
6278 "Call should have been handled by makeCallWideningDecisions");
6279
6280 // All widen recipes below deal only with VF > 1.
6282 [&](ElementCount VF) { return VF.isScalar(); }, Range))
6283 return nullptr;
6284
6285 Instruction *Instr = R->getUnderlyingInstr();
6286 assert(!is_contained({Instruction::Load, Instruction::Store},
6287 VPI->getOpcode()) &&
6288 "Should have been handled prior to this!");
6289
6290 // We can only replicate an extractvalue if its operand generates per lane in
6291 // the same block, otherwise we would need to extract a lane from its struct
6292 // operand which is invalid.
6293 if (VPI->getOpcode() == Instruction::ExtractValue &&
6295 if (VPRecipeBase *OpR = VPI->getOperand(0)->getDefiningRecipe())
6297 OpR->getParent() != VPI->getParent())
6298 return tryToWiden(VPI);
6299
6300 if (!shouldWiden(Instr, Range))
6301 return nullptr;
6302
6303 if (VPI->getOpcode() == Instruction::GetElementPtr) {
6304 auto *GEP = cast<GetElementPtrInst>(Instr);
6305 return new VPWidenGEPRecipe(GEP->getSourceElementType(),
6306 VPI->operandsWithoutMask(), *VPI,
6307 VPI->getDebugLoc(), GEP);
6308 }
6309
6310 if (Instruction::isCast(VPI->getOpcode())) {
6311 auto *CI = cast<CastInst>(Instr);
6312 return new VPWidenCastRecipe(CI->getOpcode(), VPI->getOperand(0),
6313 VPI->getScalarType(), CI, *VPI, *VPI,
6314 VPI->getDebugLoc());
6315 }
6316
6317 return tryToWiden(VPI);
6318}
6319
6320// To allow RUN_VPLAN_PASS to print the VPlan after VF/UF independent
6321// optimizations.
6323
6324#ifndef NDEBUG
6325/// Cross-check the execution frequencies recorded in \p Plan against
6326/// BlockFrequencyInfo for the blocks of \p OrigLoop.
6327/// FIXME: Temporary verification aid, to be removed.
6328static bool verifyExecutionFrequenciesMatchBFI(VPlan &Plan, Loop *OrigLoop,
6329 LoopInfo *LI,
6331 // Limited to loops with the latch as only exiting block
6332 if (OrigLoop->getExitingBlock() != OrigLoop->getLoopLatch())
6333 return true;
6334
6335 // Visit the loop body in the same order as recordExecutionFrequencies. Both
6336 // are reverse post-orders of the same CFG, so indices correspond.
6339 assert(Blocks.size() == OrigLoop->getNumBlocks() &&
6340 "loop body and original loop must have the same blocks");
6341
6342 LoopBlocksRPO OrigRPO(OrigLoop);
6343 OrigRPO.perform(LI);
6344
6345 // Only request the expensive BFI once the cheap bail-outs are past.
6346 BlockFrequencyInfo &BFI = CM.getBFI();
6347 uint64_t HeaderFreq = BFI.getBlockFreq(OrigLoop->getHeader()).getFrequency();
6348 if (HeaderFreq == 0)
6349 return true;
6350
6351 // BFI's fixed-point mass propagation loses up to 1 ULP per edge, so bound the
6352 // error by the number of edges in the region.
6353 uint64_t Edges = 0;
6354 for (const VPBasicBlock *VPBB : Blocks)
6355 Edges += VPBB->getNumSuccessors();
6356 uint64_t Tolerance = Edges + BranchProbability::getDenominator() / HeaderFreq;
6357
6358 for (const auto &[VPBB, BB] :
6359 zip_equal(drop_begin(Blocks), drop_begin(OrigRPO))) {
6360 // Nothing to check for blocks without a recorded frequency.
6361 std::optional<VPExecutionFrequency> Freq =
6363 if (!Freq)
6364 continue;
6366
6367 // Clamp to the header's frequency, which BFI's rounding may exceed.
6370 std::min(BBFreq, HeaderFreq), HeaderFreq);
6371 if (AbsoluteDifference(Computed.getNumerator(), Expected.getNumerator()) <=
6372 Tolerance)
6373 continue;
6374
6375 errs() << "Block frequency mismatch for " << VPBB->getName() << ": VPlan "
6376 << Computed << ", BlockFrequencyInfo " << Expected << "\n";
6377 return false;
6378 }
6379 return true;
6380}
6381#endif
6382
6383VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
6384 bool IsInnerLoop = OrigLoop->isInnermost();
6385
6386 // Set up loop versioning for inner loops with memory runtime checks.
6387 // Outer loops don't have LoopAccessInfo since canVectorizeMemory() is not
6388 // called for them.
6389 std::optional<LoopVersioning> LVer;
6390 if (IsInnerLoop) {
6391 const LoopAccessInfo *LAI = Legal->getLAI();
6392 LVer.emplace(*LAI, LAI->getRuntimePointerChecking()->getChecks(), OrigLoop,
6393 LI, DT, PSE.getSE());
6394 if (!LAI->getRuntimePointerChecking()->getChecks().empty() &&
6396 // Only use noalias metadata when using memory checks guaranteeing no
6397 // overlap across all iterations.
6398 LVer->prepareNoAliasMetadata();
6399 }
6400 }
6401
6402 // Create initial base VPlan0, to serve as common starting point for all
6403 // candidates built later for specific VF ranges.
6404 auto VPlan0 = VPlanTransforms::buildVPlan0(
6405 OrigLoop, *LI, Legal->getWidestInductionType(), PSE,
6406 LVer ? &*LVer : nullptr, GetBPI);
6407
6408 VPDominatorTree VPDT(*VPlan0);
6409 if (const LoopAccessInfo *LAI = Legal->getLAI())
6411 LAI->getSymbolicStrides(), VPDT);
6414 if (IsInnerLoop) {
6416 assert(verifyExecutionFrequenciesMatchBFI(*VPlan0, OrigLoop, LI, *CM) &&
6417 "execution frequencies do not match the loop's block frequencies");
6418 }
6419
6420 // Create recipes for header phis. For outer loops, reductions, recurrences
6421 // and in-loop reductions are empty since legality doesn't detect them.
6422 if (!RUN_VPLAN_PASS(
6423 VPlanTransforms::createHeaderPhiRecipes, *VPlan0, PSE, *OrigLoop,
6424 VPDT, Legal->getInductionVars(), Legal->getReductionVars(),
6425 Legal->getFixedOrderRecurrences(), Config.getInLoopReductions(),
6426 Config.getHints().allowReordering())) {
6427 return nullptr;
6428 }
6429
6430 if (const LoopAccessInfo *LAI = Legal->getLAI())
6432 LAI->getSymbolicStrides(), VPDT);
6433
6434 // Add surviving induction predicates to PSE and check constraints.
6435 bool ForceVectorization =
6436 Config.getHints().getForce() == LoopVectorizeHints::FK_Enabled;
6437 bool OptForSize =
6438 !ForceVectorization &&
6439 (CM->EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
6440 CM->EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
6441 unsigned SCEVCheckThreshold = ForceVectorization
6445 OptForSize, SCEVCheckThreshold, ORE, OrigLoop))
6446 return nullptr;
6447
6449
6450 // If we're vectorizing a loop with an uncountable exit, make sure that the
6451 // recipes are safe to handle.
6452 // TODO: Remove this once we can properly check the VPlan itself for both
6453 // the presence of an uncountable exit and the presence of stores in
6454 // the loop inside handleUncountableEarlyExits itself.
6455 if (Legal->hasUncountableEarlyExit()) {
6457 OrigLoop))
6458 return nullptr;
6459
6460 // TODO: Check target preference for style.
6461 UncountableExitStyle EEStyle =
6462 Legal->hasUncountableExitWithSideEffects()
6466 ORE, OrigLoop, PSE, *DT, Legal->getAssumptionCache(),
6467 EEStyle)) {
6468 return nullptr;
6469 }
6470 } else {
6472 }
6473
6475 getDebugLocFromInstOrOperands(Legal->getPrimaryInduction()));
6476 if (CM->foldTailByMasking())
6478
6480
6481 return VPlan0;
6482}
6483
6484void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
6485 ElementCount MaxVF) {
6486 if (ElementCount::isKnownGT(MinVF, MaxVF))
6487 return;
6488
6489 auto MaxVFTimes2 = MaxVF * 2;
6490 for (ElementCount VF = MinVF; ElementCount::isKnownLT(VF, MaxVFTimes2);) {
6491 VFRange SubRange = {VF, MaxVFTimes2};
6492 auto Plan =
6493 tryToBuildVPlan(std::unique_ptr<VPlan>(VPlan1.duplicate()), SubRange);
6494 VF = SubRange.End;
6495
6496 if (!Plan)
6497 continue;
6498
6499 // Now optimize the initial VPlan.
6503 Config.getMinimalBitwidths());
6505 // TODO: try to put addExplicitVectorLength close to addActiveLaneMask
6506 if (CM->foldTailWithEVL()) {
6508 Config.getMaxSafeElements());
6510 }
6511
6512 if (auto P =
6514 VPlans.push_back(std::move(P));
6515
6516 TailFoldingStyle Style = CM->getTailFoldingStyle();
6518 useActiveLaneMask(Style),
6520
6522 assert(verifyVPlanIsValid(*Plan) && "VPlan is invalid");
6523 VPlans.push_back(std::move(Plan));
6524 }
6525}
6526
6527VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
6528 VFRange &Range) {
6529
6530 // For outer loops, the plan only needs basic recipe conversion and induction
6531 // live-out optimization; the full inner-loop recipe building below does not
6532 // apply (no widening decisions, interleave groups, reductions, etc.).
6533 if (Plan->isOuterLoop()) {
6534 for (ElementCount VF : Range)
6535 Plan->addVF(VF);
6537 *Plan, *TLI, PSE, OrigLoop))
6538 return nullptr;
6540 OrigLoop);
6541 return Plan;
6542 }
6543
6544 using namespace llvm::VPlanPatternMatch;
6545 SmallPtrSet<const InterleaveGroup<Instruction> *, 1> InterleaveGroups;
6546
6547 // ---------------------------------------------------------------------------
6548 // Build initial VPlan: Scan the body of the loop in a topological order to
6549 // visit each basic block after having visited its predecessor basic blocks.
6550 // ---------------------------------------------------------------------------
6551
6552 bool RequiresScalarEpilogueCheck =
6554 [this](ElementCount VF) {
6555 return !CM->requiresScalarEpilogue(VF.isVector());
6556 },
6557 Range);
6558 // Update the branch in the middle block if a scalar epilogue is required.
6559 VPBasicBlock *MiddleVPBB = Plan->getMiddleBlock();
6560 if (!RequiresScalarEpilogueCheck && MiddleVPBB->getNumSuccessors() == 2) {
6561 auto *BranchOnCond = cast<VPInstruction>(MiddleVPBB->getTerminator());
6562 assert(MiddleVPBB->getSuccessors()[1] == Plan->getScalarPreheader() &&
6563 "second successor must be scalar preheader");
6564 BranchOnCond->setOperand(0, Plan->getFalse());
6565 }
6566
6567 // Don't use getDecisionAndClampRange here, because we don't know the UF
6568 // so this function is better to be conservative, rather than to split
6569 // it up into different VPlans.
6570 // TODO: Consider using getDecisionAndClampRange here to split up VPlans.
6571 bool IVUpdateMayOverflow = false;
6572 for (ElementCount VF : Range)
6573 IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(CM.get(), VF);
6574
6575 TailFoldingStyle Style = CM->getTailFoldingStyle();
6576 // Use NUW for the induction increment if we proved that it won't overflow in
6577 // the vector loop or when not folding the tail. In the later case, we know
6578 // that the canonical induction increment will not overflow as the vector trip
6579 // count is >= increment and a multiple of the increment.
6580 VPRegionBlock *LoopRegion = Plan->getVectorLoopRegion();
6581 bool HasNUW = !IVUpdateMayOverflow || Style == TailFoldingStyle::None;
6582 if (!HasNUW) {
6583 auto *IVInc =
6584 LoopRegion->getExitingBasicBlock()->getTerminator()->getOperand(0);
6585 assert(match(IVInc,
6586 m_VPInstruction<Instruction::Add>(
6587 m_Specific(LoopRegion->getCanonicalIV()), m_VPValue())) &&
6588 "Did not find the canonical IV increment");
6589 LoopRegion->clearCanonicalIVNUW(cast<VPInstruction>(IVInc));
6590 }
6591
6592 // ---------------------------------------------------------------------------
6593 // Pre-construction: record ingredients whose recipes we'll need to further
6594 // process after constructing the initial VPlan.
6595 // ---------------------------------------------------------------------------
6596
6597 // For each interleave group which is relevant for this (possibly trimmed)
6598 // Range, add it to the set of groups to be later applied to the VPlan and add
6599 // placeholders for its members' Recipes which we'll be replacing with a
6600 // single VPInterleaveRecipe.
6601 for (InterleaveGroup<Instruction> *IG : IAI.getInterleaveGroups()) {
6602 auto ApplyIG = [IG, this](ElementCount VF) -> bool {
6603 bool Result = (VF.isVector() && // Query is illegal for VF == 1
6604 CM->getWideningDecision(IG->getInsertPos(), VF) ==
6606 // For scalable vectors, the interleave factors must be <= 8 since we
6607 // require the (de)interleaveN intrinsics instead of shufflevectors.
6608 assert((!Result || !VF.isScalable() || IG->getFactor() <= 8) &&
6609 "Unsupported interleave factor for scalable vectors");
6610 return Result;
6611 };
6612 if (!getDecisionAndClampRange(ApplyIG, Range))
6613 continue;
6614 InterleaveGroups.insert(IG);
6615 }
6616
6617 // ---------------------------------------------------------------------------
6618 // Construct wide recipes and apply predication for original scalar
6619 // VPInstructions in the loop.
6620 // ---------------------------------------------------------------------------
6621 VPRecipeBuilder RecipeBuilder(*Plan, Legal, *CM, Builder);
6622
6624 Range.Start);
6625
6626 VPCostContext CostCtx(*TLI, *Plan, *CM, Config);
6627
6629 RecipeBuilder, CostCtx);
6630
6632
6634 RecipeBuilder, CostCtx);
6635
6637 PSE);
6638
6639 // Convert remaining VPInstructions to widen or replicate recipes.
6640 // TODO: This legacy code should eventually be migrated to VPlan.
6641 VPBasicBlock *HeaderVPBB = LoopRegion->getEntryBasicBlock();
6642 for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(
6643 vp_depth_first_shallow(HeaderVPBB))) {
6644 // All types but VPInstructions are already widened and don't need extra
6645 // processing. We process VPInstructions below.
6646 assert(
6647 all_of(
6648 make_range(VPBB->getFirstNonPhi(), VPBB->end()),
6649 IsaPred<VPWidenCanonicalIVRecipe, VPBlendRecipe, VPReductionRecipe,
6650 VPReplicateRecipe, VPWidenLoadRecipe, VPWidenStoreRecipe,
6651 VPWidenCallRecipe, VPWidenIntrinsicRecipe,
6652 VPVectorPointerRecipe, VPVectorEndPointerRecipe,
6653 VPHistogramRecipe, VPInstruction>) &&
6654 "Unexpected recipe");
6655 for (VPInstruction &VPI :
6657 // We represent single-scalar casts directly as VPInstructions.
6658 if (Instruction::isCast(VPI.getOpcode()) &&
6660 continue;
6661
6662 // Only VPInstrutions with an underlying value need to be processed.
6663 if (!VPI.getUnderlyingValue())
6664 continue;
6665
6666 Builder.setInsertPoint(&VPI);
6667
6668 VPRecipeBase *Recipe =
6669 RecipeBuilder.tryToCreateWidenNonPhiRecipe(&VPI, Range);
6670 if (!Recipe)
6671 Recipe = RecipeBuilder.handleReplication(&VPI, Range);
6672 Builder.insert(Recipe);
6673
6674 if (Recipe->getNumDefinedValues() == 1) {
6675 VPI.replaceAllUsesWith(Recipe->getVPSingleValue());
6676 } else {
6677 assert(Recipe->getNumDefinedValues() == 0 &&
6678 "Unexpected multidef recipe");
6679 }
6680 VPI.eraseFromParent();
6681 }
6682 }
6683
6684 assert(isa<VPRegionBlock>(LoopRegion) &&
6685 !LoopRegion->getEntryBasicBlock()->empty() &&
6686 "entry block must be set to a VPRegionBlock having a non-empty entry "
6687 "VPBasicBlock");
6688
6690 Range);
6691
6692 // ---------------------------------------------------------------------------
6693 // Transform initial VPlan: Apply previously taken decisions, in order, to
6694 // bring the VPlan to its final state.
6695 // ---------------------------------------------------------------------------
6696
6697 addReductionResultComputation(Plan, Range.Start);
6698
6699 // Optimize FindIV reductions to use sentinel-based approach when possible.
6701 *OrigLoop);
6703 OrigLoop);
6704
6705 // Apply mandatory transformation to handle reductions with multiple in-loop
6706 // uses if possible, bail out otherwise.
6708 OrigLoop))
6709 return nullptr;
6710 // Apply mandatory transformation to handle FP maxnum/minnum reduction with
6711 // NaNs if possible, bail out otherwise.
6713 return nullptr;
6714
6715 // Create whole-vector selects for find-last recurrences.
6717 return nullptr;
6718
6720
6721 // Create partial reduction recipes for scaled reductions and transform
6722 // recipes to abstract recipes if it is legal and beneficial and clamp the
6723 // range for better cost estimation.
6725 Range);
6727 Range);
6728
6729 // Interleave memory: for each Interleave Group we marked earlier as relevant
6730 // for this VPlan, replace the Recipes widening its memory instructions with a
6731 // single VPInterleaveRecipe at its insertion point.
6733 InterleaveGroups, CM->isEpilogueAllowed());
6734
6735 // Convert memory recipes to strided access recipes if the strided access is
6736 // legal and profitable.
6738 *OrigLoop, CostCtx, Range);
6739
6740 // Ensure scalar VF plans only contain VF=1, as required by hasScalarVFOnly.
6741 if (Range.Start.isScalar())
6742 Range.End = Range.Start * 2;
6743
6744 for (ElementCount VF : Range)
6745 Plan->addVF(VF);
6746 Plan->setName("Initial VPlan");
6747
6749
6750 if (CM->maskPartialAliasing())
6752
6753 assert(verifyVPlanIsValid(*Plan) && "VPlan is invalid");
6754 return Plan;
6755}
6756
6757void LoopVectorizationPlanner::addReductionResultComputation(
6758 VPlanPtr &Plan, ElementCount MinVF) {
6759 using namespace VPlanPatternMatch;
6760 VPRegionBlock *VectorLoopRegion = Plan->getVectorLoopRegion();
6761 VPBasicBlock *MiddleVPBB = Plan->getMiddleBlock();
6762 VPBasicBlock *LatchVPBB = VectorLoopRegion->getExitingBasicBlock();
6763 Builder.setInsertPoint(&*std::prev(std::prev(LatchVPBB->end())));
6764 VPBasicBlock::iterator IP = MiddleVPBB->getFirstNonPhi();
6765 VPValue *HeaderMask = Plan->getVectorLoopRegion()->getHeaderMask();
6766 for (VPRecipeBase &R : make_early_inc_range(
6767 Plan->getVectorLoopRegion()->getEntryBasicBlock()->phis())) {
6768 VPReductionPHIRecipe *PhiR = dyn_cast<VPReductionPHIRecipe>(&R);
6769 if (!PhiR)
6770 continue;
6771
6772 // Clean up reductions that have become invariant.
6773 if (PhiR->getBackedgeValue() == PhiR) {
6774 PhiR->replaceAllUsesWith(PhiR->getStartValue());
6775 PhiR->eraseFromParent();
6776 continue;
6777 }
6778
6779 RecurKind RecurrenceKind = PhiR->getRecurrenceKind();
6780 const RecurrenceDescriptor &RdxDesc = Legal->getRecurrenceDescriptor(
6782 Type *PhiTy = PhiR->getScalarType();
6783
6784 // Convert a VPBlendRecipe backedge to a select.
6785 if (auto *Blend = dyn_cast<VPBlendRecipe>(PhiR->getBackedgeValue())) {
6786 if (Blend->getNumIncomingValues() == 2 &&
6787 Blend->getMask(0) == HeaderMask) {
6788 auto *Sel = VPBuilder(Blend).createSelect(
6789 Blend->getMask(0), Blend->getIncomingValue(0),
6790 Blend->getIncomingValue(1), {}, "", *Blend);
6791 Blend->replaceAllUsesWith(Sel);
6792 Blend->eraseFromParent();
6793 }
6794 }
6795
6796 auto *OrigExitingVPV = PhiR->getBackedgeValue();
6797 auto *NewExitingVPV = OrigExitingVPV;
6798
6799 // Remove the predicated select if the target doesn't want it.
6800 VPValue *V;
6801 if (!CM->usePredicatedReductionSelect(RecurrenceKind) &&
6802 match(PhiR->getBackedgeValue(),
6803 m_Select(m_Specific(HeaderMask), m_VPValue(V), m_Specific(PhiR))))
6804 PhiR->setBackedgeValue(V);
6805
6806 // We want code in the middle block to appear to execute on the location of
6807 // the scalar loop's latch terminator because: (a) it is all compiler
6808 // generated, (b) these instructions are always executed after evaluating
6809 // the latch conditional branch, and (c) other passes may add new
6810 // predecessors which terminate on this line. This is the easiest way to
6811 // ensure we don't accidentally cause an extra step back into the loop while
6812 // debugging.
6813 DebugLoc ExitDL = OrigLoop->getLoopLatch()->getTerminator()->getDebugLoc();
6814
6815 // TODO: At the moment ComputeReductionResult also drives creation of the
6816 // bc.merge.rdx phi nodes, hence it needs to be created unconditionally here
6817 // even for in-loop reductions, until the reduction resume value handling is
6818 // also modeled in VPlan.
6819 VPInstruction *FinalReductionResult;
6820 VPBuilder::InsertPointGuard Guard(Builder);
6821 Builder.setInsertPoint(MiddleVPBB, IP);
6822 // For AnyOf reductions, find the select among PhiR's users and convert
6823 // the reduction phi to operate on bools before creating the final
6824 // reduction result.
6825 if (RecurrenceDescriptor::isAnyOfRecurrenceKind(RecurrenceKind)) {
6826 auto *AnyOfSelect = cast<VPSingleDefRecipe>(
6828 VPValue *Start = PhiR->getStartValue();
6829 bool TrueValIsPhi = AnyOfSelect->getOperand(1) == PhiR;
6830 // NewVal is the non-phi operand of the select.
6831 VPValue *NewVal = TrueValIsPhi ? AnyOfSelect->getOperand(2)
6832 : AnyOfSelect->getOperand(1);
6833
6834 // Adjust AnyOf reductions; replace the reduction phi for the selected
6835 // value with a boolean reduction phi node to check if the condition is
6836 // true in any iteration. The final value is selected by the final
6837 // ComputeReductionResult.
6838 VPValue *Cmp = AnyOfSelect->getOperand(0);
6839 // If the compare is checking the reduction PHI node, adjust it to check
6840 // the start value.
6841 if (VPRecipeBase *CmpR = Cmp->getDefiningRecipe())
6842 CmpR->replaceUsesOfWith(PhiR, PhiR->getStartValue());
6843 Builder.setInsertPoint(AnyOfSelect);
6844
6845 // If the true value of the select is the reduction phi, the new value
6846 // is selected if the negated condition is true in any iteration.
6847 if (TrueValIsPhi)
6848 Cmp = Builder.createNot(Cmp);
6849
6850 // Build a fresh i1 chain (phi, or, and i1 versions of any blend/select
6851 // the exiting value flows through).
6852 auto *NewPhiR =
6853 PhiR->cloneWithOperands(Plan->getFalse(), Plan->getFalse());
6854 NewPhiR->insertBefore(PhiR);
6855 VPValue *NewExiting = Builder.createOr(NewPhiR, Cmp);
6856
6857 // The exiting value may flow through a chain of VPBlendRecipes and
6858 // select recipes (VPInstruction, VPWidenRecipe or VPReplicateRecipe with
6859 // Select opcode) before reaching OrigExitingVPV. Clone each chain link
6860 // in topological order so each clone refers to the already-rewritten i1
6861 // operands via Substitutions.
6862 DenseMap<VPValue *, VPValue *> Substitutions = {{AnyOfSelect, NewExiting},
6863 {PhiR, NewPhiR}};
6864 std::function<void(VPSingleDefRecipe *)> CloneChain =
6865 [&](VPSingleDefRecipe *Old) {
6866 if (Substitutions.contains(Old))
6867 return;
6869 for (VPValue *Op : Old->operands()) {
6870 if (isa<VPBlendRecipe>(Op) ||
6872 CloneChain(cast<VPSingleDefRecipe>(Op));
6873 NewOps.push_back(Substitutions.lookup_or(Op, Op));
6874 }
6875 VPSingleDefRecipe *New;
6876 if (auto *B = dyn_cast<VPBlendRecipe>(Old))
6877 New = B->cloneWithOperands(NewOps);
6878 else if (auto *W = dyn_cast<VPWidenRecipe>(Old))
6879 New = W->cloneWithOperands(NewOps);
6880 else if (auto *Rep = dyn_cast<VPReplicateRecipe>(Old))
6881 New = Rep->cloneWithOperands(NewOps);
6882 else
6883 New = cast<VPInstruction>(Old)->cloneWithOperands(NewOps);
6884 New->insertBefore(Old);
6885 Substitutions[Old] = New;
6886 };
6887
6888 if (OrigExitingVPV != AnyOfSelect) {
6889 CloneChain(cast<VPSingleDefRecipe>(OrigExitingVPV));
6890 NewExiting = Substitutions.lookup(OrigExitingVPV);
6891 }
6892 NewPhiR->setOperand(1, NewExiting);
6893 PhiR->replaceAllUsesWith(Plan->getPoison(PhiR->getScalarType()));
6894
6895 Builder.setInsertPoint(MiddleVPBB, IP);
6896 FinalReductionResult =
6897 Builder.createAnyOfReduction(NewExiting, NewVal, Start, ExitDL);
6898 } else {
6899 // If the vector reduction can be performed in a smaller type, we
6900 // truncate then extend the loop exit value to enable InstCombine to
6901 // evaluate the entire expression in the smaller type.
6902 VPValue *ReductionOp = NewExitingVPV;
6903 Instruction::CastOps ExtendOpc = Instruction::CastOpsEnd;
6904 if (MinVF.isVector() && PhiTy != RdxDesc.getRecurrenceType()) {
6905 assert(!PhiR->isInLoop() && "Unexpected truncated inloop reduction!");
6907 "Unexpected truncated min-max recurrence!");
6908 Type *RdxTy = RdxDesc.getRecurrenceType();
6909 ExtendOpc = RdxDesc.isSigned() ? Instruction::SExt : Instruction::ZExt;
6910 {
6911 VPBuilder::InsertPointGuard Guard(Builder);
6912 Builder.setInsertPoint(
6913 NewExitingVPV->getDefiningRecipe()->getParent(),
6914 std::next(NewExitingVPV->getDefiningRecipe()->getIterator()));
6915 ReductionOp =
6916 Builder.createWidenCast(Instruction::Trunc, NewExitingVPV, RdxTy);
6917 VPWidenCastRecipe *Extnd =
6918 Builder.createWidenCast(ExtendOpc, ReductionOp, PhiTy);
6919 if (PhiR->getOperand(1) == NewExitingVPV)
6920 PhiR->setOperand(1, Extnd);
6921 }
6922 }
6923
6924 VPIRFlags Flags(RecurrenceKind, PhiR->isOrdered(), PhiR->isInLoop(),
6925 PhiR->getFastMathFlagsOrNone());
6926 FinalReductionResult = Builder.createNaryOp(
6927 VPInstruction::ComputeReductionResult, {ReductionOp}, Flags, ExitDL);
6928 if (ExtendOpc != Instruction::CastOpsEnd)
6929 FinalReductionResult = Builder.createScalarCast(
6930 ExtendOpc, FinalReductionResult, PhiTy, {});
6931 }
6932
6933 // Update all users outside the vector region. Also replace redundant
6934 // extracts.
6935 for (auto *U : to_vector(OrigExitingVPV->users())) {
6936 auto *Parent = cast<VPRecipeBase>(U)->getParent();
6937 if (FinalReductionResult == U || Parent->getParent())
6938 continue;
6939 // Skip ComputeReductionResult and FindIV reductions when they are not the
6940 // final result.
6941 if (match(U, m_VPInstruction<VPInstruction::ComputeReductionResult>()) ||
6943 match(U, m_VPInstruction<Instruction::ICmp>())))
6944 continue;
6945 U->replaceUsesOfWith(OrigExitingVPV, FinalReductionResult);
6946
6947 // Look through ExtractLastPart.
6949 U = cast<VPInstruction>(U)->getSingleUser();
6950
6953 cast<VPInstruction>(U)->replaceAllUsesWith(FinalReductionResult);
6954 }
6955
6956 RecurKind RK = PhiR->getRecurrenceKind();
6961 VPBuilder PHBuilder(Plan->getVectorPreheader());
6962 VPValue *Iden = Plan->getOrAddLiveIn(
6963 getRecurrenceIdentity(RK, PhiTy, PhiR->getFastMathFlagsOrNone()));
6964 auto *ScaleFactorVPV = Plan->getConstantInt(32, 1);
6965 VPValue *StartV = PHBuilder.createNaryOp(
6967 {PhiR->getStartValue(), Iden, ScaleFactorVPV}, *PhiR);
6968 PhiR->setOperand(0, StartV);
6969 }
6970 }
6971
6973}
6974
6976 VPlan &Plan, GeneratedRTChecks &RTChecks, bool HasBranchWeights) const {
6977 const auto &[SCEVCheckCond, SCEVCheckBlock] = RTChecks.getSCEVChecks();
6978 if (SCEVCheckBlock && SCEVCheckBlock->hasNPredecessors(0)) {
6979 assert((!Config.OptForSize ||
6980 Config.getHints().getForce() == LoopVectorizeHints::FK_Enabled) &&
6981 "Cannot SCEV check stride or overflow when optimizing for size");
6983 SCEVCheckBlock, HasBranchWeights);
6984 }
6985 const auto &[MemCheckCond, MemCheckBlock] = RTChecks.getMemRuntimeChecks();
6986 if (MemCheckBlock && MemCheckBlock->hasNPredecessors(0)) {
6987 // VPlan-native path does not do any analysis for runtime checks
6988 // currently.
6990 "Runtime checks are not supported for outer loops yet");
6991
6992 if (Config.OptForSize) {
6993 assert(
6994 Config.getHints().getForce() == LoopVectorizeHints::FK_Enabled &&
6995 "Cannot emit memory checks when optimizing for size, unless forced "
6996 "to vectorize.");
6997 ORE->emit([&]() {
6998 return OptimizationRemarkAnalysis(DEBUG_TYPE, "VectorizationCodeSize",
6999 OrigLoop->getStartLoc(),
7000 OrigLoop->getHeader())
7001 << "Code-size may be reduced by not forcing "
7002 "vectorization, or by source-code modifications "
7003 "eliminating the need for runtime checks "
7004 "(e.g., adding 'restrict').";
7005 });
7006 }
7007 // VPSCEVExpander expands AddRecs in the plan's entry, not the check block,
7008 // and does not support pointer-typed min/max yet.
7009 auto IsUnsupported = [](const SCEV *S) {
7010 return isa<SCEVAddRecExpr>(S) ||
7011 (isa<SCEVMinMaxExpr>(S) && S->getType()->isPointerTy());
7012 };
7013 // Diff checks are not modelled in VPlan yet, and the VPlan expander cannot
7014 // hoist bounds out of an enclosing loop.
7015 const auto &RtPtrChecking = *Legal->getRuntimePointerChecking();
7016 if (RtPtrChecking.getDiffChecks() || OrigLoop->getParentLoop() ||
7017 any_of(RtPtrChecking.CheckingGroups,
7018 [&](const RuntimeCheckingPtrGroup &CG) {
7019 return SCEVExprContains(CG.Low, IsUnsupported) ||
7020 SCEVExprContains(CG.High, IsUnsupported);
7021 }))
7023 MemCheckCond, MemCheckBlock, HasBranchWeights);
7024
7025 // Erase the temporary IR before recipe expansion can reuse its values.
7026 RTChecks.eraseMemCheckBlock();
7028 RtPtrChecking.getChecks(), *PSE.getSE(),
7029 OrigLoop->getStartLoc(), HasBranchWeights);
7030 }
7031}
7032
7034 VPlan &Plan, ElementCount VF, unsigned UF,
7035 ElementCount MinProfitableTripCount) const {
7036 const uint32_t *BranchWeights =
7037 hasBranchWeightMD(*OrigLoop->getLoopLatch()->getTerminator())
7039 : nullptr;
7041 MinProfitableTripCount, Plan.requiresScalarEpilogue(),
7042 Plan.hasTailFolded(), OrigLoop, BranchWeights,
7043 OrigLoop->getLoopPredecessor()->getTerminator()->getDebugLoc(),
7044 PSE, Plan.getEntry());
7045}
7046
7047// Determine how to lower the epilogue, which depends on 1) optimising
7048// for minimum code-size, 2) tail-folding compiler options, 3) loop
7049// hints forcing tail-folding, and 4) a TTI hook that analyses whether the loop
7050// is suitable for tail-folding.
7051// This function determines epilogue lowering for the main vector loop while
7052// epilogue lowering for the tail-folded epilogue path will be handled
7053// separately in getEpilogueTailLowering.
7054static EpilogueLowering
7056 bool OptForSize, TargetTransformInfo *TTI,
7058 InterleavedAccessInfo *IAI) {
7059 // 1) OptSize takes precedence over all other options, i.e. if this is set,
7060 // don't look at hints or options, and don't request an epilogue.
7061 if (F->hasOptSize() ||
7062 (OptForSize && Hints.getForce() != LoopVectorizeHints::FK_Enabled))
7064
7065 // 2) If set, obey the directives
7066 if (TailFoldingPolicy.getNumOccurrences()) {
7067 switch (TailFoldingPolicy) {
7069 return CM_EpilogueAllowed;
7074 };
7075 }
7076
7077 // 3) If set, obey the hints
7078 switch (Hints.getPredicate()) {
7082 return CM_EpilogueAllowed;
7083 };
7084
7085 // 4) if the TTI hook indicates this is profitable, request tail-folding.
7086 TailFoldingInfo TFI(TLI, &LVL, IAI);
7087 if (TTI->preferTailFoldingOverEpilogue(&TFI))
7089
7090 return CM_EpilogueAllowed;
7091}
7092
7093// Emit a remark if there are stores to floats that required a floating point
7094// extension. If the vectorized loop was generated with floating point there
7095// will be a performance penalty from the conversion overhead and the change in
7096// the vector width.
7099 for (BasicBlock *BB : L->getBlocks()) {
7100 for (Instruction &Inst : *BB) {
7101 if (auto *S = dyn_cast<StoreInst>(&Inst)) {
7102 if (S->getValueOperand()->getType()->isFloatTy())
7103 Worklist.push_back(S);
7104 }
7105 }
7106 }
7107
7108 // Traverse the floating point stores upwards searching, for floating point
7109 // conversions.
7112 while (!Worklist.empty()) {
7113 auto *I = Worklist.pop_back_val();
7114 if (!L->contains(I))
7115 continue;
7116 if (!Visited.insert(I).second)
7117 continue;
7118
7119 // Emit a remark if the floating point store required a floating
7120 // point conversion.
7121 // TODO: More work could be done to identify the root cause such as a
7122 // constant or a function return type and point the user to it.
7123 if (isa<FPExtInst>(I) && EmittedRemark.insert(I).second)
7124 ORE->emit([&]() {
7125 return OptimizationRemarkAnalysis(LV_NAME, "VectorMixedPrecision",
7126 I->getDebugLoc(), L->getHeader())
7127 << "floating point conversion changes vector width. "
7128 << "Mixed floating point precision requires an up/down "
7129 << "cast that will negatively impact performance.";
7130 });
7131
7132 for (Use &Op : I->operands())
7133 if (auto *OpI = dyn_cast<Instruction>(Op))
7134 Worklist.push_back(OpI);
7135 }
7136}
7137
7138/// For loops with uncountable early exits, find the cost of doing work when
7139/// exiting the loop early, such as calculating the final exit values of
7140/// variables used outside the loop.
7141/// TODO: This is currently overly pessimistic because the loop may not take
7142/// the early exit, but better to keep this conservative for now. In future,
7143/// it might be possible to relax this by using branch probabilities.
7145 VPlan &Plan, ElementCount VF) {
7146 InstructionCost Cost = 0;
7147 for (auto *ExitVPBB : Plan.getExitBlocks()) {
7148 for (auto *PredVPBB : ExitVPBB->getPredecessors()) {
7149 // If the predecessor is not the middle.block, then it must be the
7150 // vector.early.exit block, which may contain work to calculate the exit
7151 // values of variables used outside the loop.
7152 if (PredVPBB != Plan.getMiddleBlock()) {
7153 LLVM_DEBUG(dbgs() << "Calculating cost of work in exit block "
7154 << PredVPBB->getName() << ":\n");
7155 Cost += PredVPBB->cost(VF, CostCtx);
7156 }
7157 }
7158 }
7159 return Cost;
7160}
7161
7162/// This function determines whether or not it's still profitable to vectorize
7163/// the loop given the extra work we have to do outside of the loop:
7164/// 1. Perform the runtime checks before entering the loop to ensure it's safe
7165/// to vectorize.
7166/// 2. In the case of loops with uncountable early exits, we may have to do
7167/// extra work when exiting the loop early, such as calculating the final
7168/// exit values of variables used outside the loop.
7169/// 3. The middle block.
7170static bool isOutsideLoopWorkProfitable(GeneratedRTChecks &Checks,
7171 VectorizationFactor &VF, Loop *L,
7173 VPCostContext &CostCtx, VPlan &Plan,
7174 EpilogueLowering SEL,
7175 std::optional<unsigned> VScale) {
7176 InstructionCost RtC = Checks.getCost();
7177 if (!RtC.isValid())
7178 return false;
7179
7180 // When interleaving only scalar and vector cost will be equal, which in turn
7181 // would lead to a divide by 0. Fall back to hard threshold.
7182 if (VF.Width.isScalar()) {
7183 // TODO: Should we rename VectorizeMemoryCheckThreshold?
7185 LLVM_DEBUG(
7186 dbgs()
7187 << "LV: Interleaving only is not profitable due to runtime checks\n");
7188 return false;
7189 }
7190 return true;
7191 }
7192
7193 // The scalar cost should only be 0 when vectorizing with a user specified
7194 // VF/IC. In those cases, runtime checks should always be generated.
7195 uint64_t ScalarC = VF.ScalarCost.getValue();
7196 if (ScalarC == 0)
7197 return true;
7198
7199 InstructionCost TotalCost = RtC;
7200 // Add on the cost of any work required in the vector early exit block, if
7201 // one exists.
7202 TotalCost += calculateEarlyExitCost(CostCtx, Plan, VF.Width);
7203 TotalCost += Plan.getMiddleBlock()->cost(VF.Width, CostCtx);
7204
7205 // First, compute the minimum iteration count required so that the vector
7206 // loop outperforms the scalar loop.
7207 // The total cost of the scalar loop is
7208 // ScalarC * TC
7209 // where
7210 // * TC is the actual trip count of the loop.
7211 // * ScalarC is the cost of a single scalar iteration.
7212 //
7213 // The total cost of the vector loop is
7214 // TotalCost + VecC * (TC / VF) + EpiC
7215 // where
7216 // * TotalCost is the sum of the costs cost of
7217 // - the generated runtime checks, i.e. RtC
7218 // - performing any additional work in the vector.early.exit block for
7219 // loops with uncountable early exits.
7220 // - the middle block, if ExpectedTC <= VF.Width.
7221 // * VecC is the cost of a single vector iteration.
7222 // * TC is the actual trip count of the loop
7223 // * VF is the vectorization factor
7224 // * EpiCost is the cost of the generated epilogue, including the cost
7225 // of the remaining scalar operations.
7226 //
7227 // Vectorization is profitable once the total vector cost is less than the
7228 // total scalar cost:
7229 // TotalCost + VecC * (TC / VF) + EpiC < ScalarC * TC
7230 //
7231 // Now we can compute the minimum required trip count TC as
7232 // VF * (TotalCost + EpiC) / (ScalarC * VF - VecC) < TC
7233 //
7234 // For now we assume the epilogue cost EpiC = 0 for simplicity. Note that
7235 // the computations are performed on doubles, not integers and the result
7236 // is rounded up, hence we get an upper estimate of the TC.
7237 unsigned IntVF = estimateElementCount(VF.Width, VScale);
7238 uint64_t Div = ScalarC * IntVF - VF.Cost.getValue();
7239 uint64_t MinTC1 =
7240 Div == 0 ? 0 : divideCeil(TotalCost.getValue() * IntVF, Div);
7241
7242 // Second, compute a minimum iteration count so that the cost of the
7243 // runtime checks is only a fraction of the total scalar loop cost. This
7244 // adds a loop-dependent bound on the overhead incurred if the runtime
7245 // checks fail. In case the runtime checks fail, the cost is RtC + ScalarC
7246 // * TC. To bound the runtime check to be a fraction 1/X of the scalar
7247 // cost, compute
7248 // RtC < ScalarC * TC * (1 / X) ==> RtC * X / ScalarC < TC
7249 uint64_t MinTC2 = divideCeil(RtC.getValue() * 10, ScalarC);
7250
7251 // Now pick the larger minimum. If it is not a multiple of VF and an epilogue
7252 // is allowed, choose the next closest multiple of VF. This should partly
7253 // compensate for ignoring the epilogue cost.
7254 uint64_t MinTC = std::max(MinTC1, MinTC2);
7255 if (SEL == CM_EpilogueAllowed)
7256 MinTC = alignTo(MinTC, IntVF);
7258
7259 LLVM_DEBUG(
7260 dbgs() << "LV: Minimum required TC for runtime checks to be profitable:"
7261 << VF.MinProfitableTripCount << "\n");
7262
7263 // Skip vectorization if the expected trip count is less than the minimum
7264 // required trip count.
7265 if (auto ExpectedTC = getSmallBestKnownTC(PSE, L)) {
7266 if (ElementCount::isKnownLT(*ExpectedTC, VF.MinProfitableTripCount)) {
7267 LLVM_DEBUG(dbgs() << "LV: Vectorization is not beneficial: expected "
7268 "trip count < minimum profitable VF ("
7269 << *ExpectedTC << " < " << VF.MinProfitableTripCount
7270 << ")\n");
7271
7272 return false;
7273 }
7274 }
7275 return true;
7276}
7277
7279 : InterleaveOnlyWhenForced(Opts.InterleaveOnlyWhenForced ||
7281 VectorizeOnlyWhenForced(Opts.VectorizeOnlyWhenForced ||
7283
7284/// Prepare \p MainPlan for vectorizing the main vector loop during epilogue
7285/// vectorization.
7288 using namespace VPlanPatternMatch;
7289 // When vectorizing the epilogue, FindFirstIV & FindLastIV reductions can
7290 // introduce multiple uses of undef/poison. If the reduction start value may
7291 // be undef or poison it needs to be frozen and the frozen start has to be
7292 // used when computing the reduction result. We also need to use the frozen
7293 // value in the resume phi generated by the main vector loop, as this is also
7294 // used to compute the reduction result after the epilogue vector loop.
7295 auto AddFreezeForFindLastIVReductions = [](VPlan &Plan,
7296 bool UpdateResumePhis) {
7297 VPBuilder Builder(Plan.getEntry());
7298 for (VPInstruction &VPI :
7300 VPValue *OrigStart;
7301 if (!matchFindIVResult(&VPI, m_VPValue(), m_VPValue(OrigStart)))
7302 continue;
7304 continue;
7305 VPInstruction *Freeze = Builder.createFreeze(OrigStart, {}, "fr");
7306 VPI.setOperand(2, Freeze);
7307 if (UpdateResumePhis)
7308 OrigStart->replaceUsesWithIf(
7309 Freeze, [](VPUser &U, unsigned) { return isa<VPPhi>(&U); });
7310 }
7311 };
7312 AddFreezeForFindLastIVReductions(MainPlan, true);
7313 AddFreezeForFindLastIVReductions(EpiPlan, false);
7314
7315 VPValue *VectorTC = nullptr;
7316 auto *Term =
7318 [[maybe_unused]] bool MatchedTC =
7319 match(Term, m_BranchOnCount(m_VPValue(), m_VPValue(VectorTC)));
7320 assert(MatchedTC && "must match vector trip count");
7321
7322 // If there is a suitable resume value for the canonical induction in the
7323 // scalar (which will become vector) epilogue loop, use it and move it to the
7324 // beginning of the scalar preheader. Otherwise create it below.
7325 VPBasicBlock *MainScalarPH = MainPlan.getScalarPreheader();
7326 auto ResumePhiIter =
7327 find_if(MainScalarPH->phis(), [VectorTC](VPRecipeBase &R) {
7328 return match(&R, m_VPInstruction<Instruction::PHI>(m_Specific(VectorTC),
7329 m_ZeroInt()));
7330 });
7331 VPPhi *ResumePhi = nullptr;
7332 if (ResumePhiIter == MainScalarPH->phis().end()) {
7334 "canonical IV must exist");
7335 Type *Ty = VectorTC->getScalarType();
7336 VPBuilder ScalarPHBuilder(MainScalarPH, MainScalarPH->begin());
7337 ResumePhi = ScalarPHBuilder.createScalarPhi(
7338 {VectorTC, MainPlan.getZero(Ty)}, {}, "vec.epilog.resume.val");
7339 } else {
7340 ResumePhi = cast<VPPhi>(&*ResumePhiIter);
7341 ResumePhi->setName("vec.epilog.resume.val");
7342 if (&MainScalarPH->front() != ResumePhi)
7343 ResumePhi->moveBefore(*MainScalarPH, MainScalarPH->begin());
7344 }
7345
7346 // Create a ResumeForEpilogue for the canonical IV resume and its bypass value
7347 // as the first non-phi, to keep them alive for the epilogue.
7348 VPBuilder ResumeBuilder(MainScalarPH);
7350 {ResumePhi, ResumePhi->getOperand(1)});
7351
7352 // Create ResumeForEpilogue instructions for the resume phis of the
7353 // VPIRPhis and their bypass values in the scalar header of the main plan and
7354 // return them so they can be used as resume values when vectorizing the
7355 // epilogue.
7356 return to_vector(
7357 map_range(MainPlan.getScalarHeader()->phis(), [&](VPRecipeBase &R) {
7358 assert(isa<VPIRPhi>(R) &&
7359 "only VPIRPhis expected in the scalar header");
7360 VPValue *MainResumePhi = R.getOperand(0);
7361 VPValue *Bypass = MainResumePhi->getDefiningRecipe()->getOperand(1);
7362 return ResumeBuilder.createNaryOp(VPInstruction::ResumeForEpilogue,
7363 {MainResumePhi, Bypass});
7364 }));
7365}
7366
7367/// Prepare \p Plan for vectorizing the epilogue loop. That is, re-use expanded
7368/// SCEVs from \p ExpandedSCEVs and set resume values for header recipes.
7370 VPlan &MainPlan, VPlan &Plan, Loop *L, const SCEV2ValueTy &ExpandedSCEVs,
7373 ArrayRef<VPInstruction *> ResumeValues) {
7374 // Build a map from the scalar-header PHI to the ResumeForEpilogue markers
7375 // from the main plan.
7376 // TODO: Replace the IR PHI key.
7377 DenseMap<PHINode *, VPInstruction *> IRPhiToResumeForEpi;
7378 for (auto [HeaderPhi, ResumeForEpi] :
7379 zip_equal(MainPlan.getScalarHeader()->phis(), ResumeValues))
7380 IRPhiToResumeForEpi[&cast<VPIRPhi>(HeaderPhi).getIRPhi()] = ResumeForEpi;
7381 VPRegionBlock *VectorLoop = Plan.getVectorLoopRegion();
7382 VPBasicBlock *Header = VectorLoop->getEntryBasicBlock();
7383 Header->setName("vec.epilog.vector.body");
7384
7385 VPValue *IV = VectorLoop->getCanonicalIV();
7386 // When vectorizing the epilogue loop, the canonical induction needs to start
7387 // at the resume value from the main vector loop. Find the resume value
7388 // created during execution of the main VPlan. Add this resume value as an
7389 // offset to the canonical IV of the epilogue loop.
7390 using namespace llvm::PatternMatch;
7391 VPInstruction *ResumeForEpilogue =
7393 Value *EPResumeVal = ResumeForEpilogue->getUnderlyingValue();
7394 if (auto *ResumePhi = dyn_cast<PHINode>(EPResumeVal)) {
7395 for (Value *Inc : ResumePhi->incoming_values()) {
7396 if (match(Inc, m_SpecificInt(0)))
7397 continue;
7398 assert(!EPI.VectorTripCount &&
7399 "Must only have a single non-zero incoming value");
7400 EPI.VectorTripCount = Inc;
7401 }
7402 // If we didn't find a non-zero vector trip count, all incoming values
7403 // must be zero, which also means the vector trip count is zero.
7404 if (!EPI.VectorTripCount) {
7405 assert(ResumePhi->getNumIncomingValues() > 0 &&
7406 all_of(ResumePhi->incoming_values(), match_fn(m_SpecificInt(0))) &&
7407 "all incoming values must be 0");
7408 EPI.VectorTripCount = ResumePhi->getIncomingValue(0);
7409 }
7410 } else {
7411 EPI.VectorTripCount = EPResumeVal;
7412 }
7413 VPValue *VPV = Plan.getOrAddLiveIn(EPResumeVal);
7414 assert(all_of(IV->users(),
7415 [](const VPUser *U) {
7416 if (isa<VPScalarIVStepsRecipe, VPDerivedIVRecipe>(U))
7417 return true;
7418 unsigned Opc = cast<VPInstruction>(U)->getOpcode();
7419 return Instruction::isCast(Opc) || Opc == Instruction::Add;
7420 }) &&
7421 "the canonical IV should only be used by its increment or "
7422 "ScalarIVSteps when resetting the start value");
7423 VPBuilder Builder(Header, Header->getFirstNonPhi());
7424 VPInstruction *Add = Builder.createAdd(IV, VPV);
7425 // Replace all users of the canonical IV and its increment with the offset
7426 // version, except for the Add itself and the canonical IV increment.
7428 assert(Increment && "Must have a canonical IV increment at this point");
7429 IV->replaceUsesWithIf(Add, [Add, Increment](VPUser &U, unsigned) {
7430 return &U != Add && &U != Increment;
7431 });
7432 VPInstruction *OffsetIVInc =
7434 Increment->replaceAllUsesWith(OffsetIVInc);
7435 OffsetIVInc->setOperand(0, Increment);
7436
7438
7439 // Resume values must be created in the vector preheader.
7440 VPBasicBlock *VectorPH = Plan.getVectorPreheader();
7441 VPBuilder PHBuilder(VectorPH, VectorPH->getFirstNonPhi());
7442
7443 // Ensure that the start values for all header phi recipes are updated before
7444 // vectorizing the epilogue loop.
7445 for (VPRecipeBase &R : Header->phis()) {
7446 VPValue *ResumeVPV = nullptr;
7447 // TODO: Move setting of resume values to prepareToExecute.
7448 if (auto *ReductionPhi = dyn_cast<VPReductionPHIRecipe>(&R)) {
7449 // Find the reduction result by searching users of the phi or its backedge
7450 // value.
7451 auto IsReductionResult = [](VPRecipeBase *R) {
7452 auto *VPI = dyn_cast<VPInstruction>(R);
7453 return VPI && VPI->getOpcode() == VPInstruction::ComputeReductionResult;
7454 };
7455 auto *RdxResult = cast<VPInstruction>(
7456 vputils::findRecipe(ReductionPhi->getBackedgeValue(), IsReductionResult));
7457 assert(RdxResult && "expected to find reduction result");
7458
7459 VPInstruction *ResumeForEpi = IRPhiToResumeForEpi.at(
7460 cast<PHINode>(ReductionPhi->getUnderlyingInstr()));
7461
7462 // Check for FindIV pattern by looking for icmp user of RdxResult.
7463 // The pattern is: select(icmp ne RdxResult, Sentinel), RdxResult, Start
7464 using namespace VPlanPatternMatch;
7465 VPValue *SentinelVPV = nullptr;
7466 bool IsFindIV = any_of(RdxResult->users(), [&](VPUser *U) {
7467 return match(U, VPlanPatternMatch::m_SpecificICmp(
7468 ICmpInst::ICMP_NE, m_Specific(RdxResult),
7469 m_VPValue(SentinelVPV)));
7470 });
7471
7472 RecurKind RK = ReductionPhi->getRecurrenceKind();
7473 ResumeVPV = Plan.getOrAddLiveIn(ResumeForEpi->getUnderlyingValue());
7474 if (RecurrenceDescriptor::isAnyOfRecurrenceKind(RK) || IsFindIV) {
7475 VPValue *BypassOp = ResumeForEpi->getOperand(1);
7476 assert((isa<VPIRValue>(BypassOp) ||
7478 BypassOp,
7480 "expected live-in or Freeze");
7481 VPValue *StartV = Plan.getOrAddLiveIn(BypassOp->getUnderlyingValue());
7483 // VPReductionPHIRecipes for AnyOf reductions expect a boolean as
7484 // start value; compare the final value from the main vector loop
7485 // to the start value.
7486 ResumeVPV = PHBuilder.createICmp(CmpInst::ICMP_NE, ResumeVPV, StartV);
7487 } else {
7488 assert(isa<VPIRValue>(SentinelVPV) &&
7489 "sentinel must be a live-in to be used in the preheader");
7490 VPValue *OrigStart;
7492 BypassOp, VPlanPatternMatch::m_Freeze(m_VPValue(OrigStart))))
7493 ToFrozen[Plan.getOrAddLiveIn(cast<VPIRValue>(OrigStart))] = StartV;
7494
7495 // Adjust resume: select(icmp eq ResumeVPV, StartV), Sentinel,
7496 // ResumeVPV
7497 VPValue *Cmp =
7498 PHBuilder.createICmp(CmpInst::ICMP_EQ, ResumeVPV, StartV);
7499 ResumeVPV = PHBuilder.createSelect(Cmp, SentinelVPV, ResumeVPV);
7500 }
7501 // TODO: materializeBroadcasts does not cover values in the vector
7502 // preheader.
7503 ReductionPhi->setStartValue(
7504 PHBuilder.createNaryOp(VPInstruction::Broadcast, ResumeVPV));
7505 continue;
7506 } else {
7507 auto *PhiR = dyn_cast<VPReductionPHIRecipe>(&R);
7508 if (auto *VPI = dyn_cast<VPInstruction>(PhiR->getStartValue())) {
7510 "unexpected start value");
7511 // Partial sub-reductions always start at 0 and account for the
7512 // reduction start value in a final subtraction. Update it to use the
7513 // resume value from the main vector loop.
7514 if (PhiR->getVFScaleFactor() > 1 &&
7516 PhiR->getRecurrenceKind())) {
7517 auto *Sub = cast<VPInstruction>(RdxResult->getSingleUser());
7518 assert((Sub->getOpcode() == Instruction::Sub ||
7519 Sub->getOpcode() == Instruction::FSub) &&
7520 "Unexpected opcode");
7521 assert(isa<VPIRValue>(Sub->getOperand(0)) &&
7522 "Expected operand to match the original start value of the "
7523 "reduction");
7524 // For integer sub-reductions, verify start value is zero.
7525 // For FP sub-reductions, verify start value is negative zero.
7526 [[maybe_unused]] auto StartValueIsIdentity = [&] {
7527 Value *IdentityValue = getRecurrenceIdentity(
7528 PhiR->getRecurrenceKind(), ResumeVPV->getScalarType(),
7529 PhiR->getFastMathFlagsOrNone());
7530 auto *StartValue = dyn_cast<VPIRValue>(VPI->getOperand(0));
7531 return StartValue && StartValue->getValue() == IdentityValue;
7532 };
7533 assert(StartValueIsIdentity() &&
7534 "Expected start value for partial sub-reduction to be zero "
7535 "(or negative zero)");
7536
7537 Sub->setOperand(0, ResumeVPV);
7538 } else
7539 VPI->setOperand(0, ResumeVPV);
7540 continue;
7541 }
7542 }
7543 } else {
7544 // Retrieve the induction resume value via ResumeForEpilogue.
7545 PHINode *IndPhi = cast<VPWidenInductionRecipe>(&R)->getPHINode();
7546 ResumeVPV = Plan.getOrAddLiveIn(
7547 IRPhiToResumeForEpi.at(IndPhi)->getUnderlyingValue());
7548 }
7549 assert(ResumeVPV && "Must have a resume value");
7550 cast<VPHeaderPHIRecipe>(&R)->setStartValue(ResumeVPV);
7551 }
7552
7553 // For some VPValues in the epilogue plan we must re-use the generated IR
7554 // values from the main plan. Replace them with live-in VPValues.
7555 // TODO: This is a workaround needed for epilogue vectorization and it
7556 // should be removed once induction resume value creation is done
7557 // directly in VPlan.
7558 for (auto &R : make_early_inc_range(*Plan.getEntry())) {
7559 // Re-use frozen values from the main plan for Freeze VPInstructions in the
7560 // epilogue plan. This ensures all users use the same frozen value.
7561 auto *VPI = dyn_cast<VPInstruction>(&R);
7562 if (VPI && VPI->getOpcode() == Instruction::Freeze) {
7563 VPI->replaceAllUsesWith(ToFrozen.lookup(VPI->getOperand(0)));
7564 continue;
7565 }
7566
7567 // Re-use the trip count and steps expanded for the main loop, as
7568 // skeleton creation needs it as a value that dominates both the scalar
7569 // and vector epilogue loops
7570 auto *ExpandR = dyn_cast<VPExpandSCEVRecipe>(&R);
7571 if (!ExpandR)
7572 continue;
7573 assert(ExpandedSCEVs.contains(ExpandR->getSCEV()) &&
7574 "Epilogue plan needs a SCEV not expanded for the main loop");
7575 VPValue *ExpandedVal =
7576 Plan.getOrAddLiveIn(ExpandedSCEVs.lookup(ExpandR->getSCEV()));
7577 ExpandR->replaceAllUsesWith(ExpandedVal);
7578 if (Plan.getTripCount() == ExpandR)
7579 Plan.resetTripCount(ExpandedVal);
7580 ExpandR->eraseFromParent();
7581 }
7582
7583 auto VScale = Config.getVScaleForTuning();
7584 unsigned MainLoopStep =
7585 estimateElementCount(EPI.MainLoopVF * EPI.MainLoopUF, VScale);
7586 unsigned EpilogueLoopStep = estimateElementCount(EPI.EpilogueVF, VScale);
7589 EPI.EpilogueVF, MainLoopStep, EpilogueLoopStep, SE);
7590}
7591
7592static void
7594 ArrayRef<VPInstruction *> ResumeValues) {
7595 auto *ScalarPH = cast<VPIRBasicBlock>(BestEpiPlan.getScalarPreheader());
7596 BasicBlock *PH = ScalarPH->getIRBasicBlock();
7597 if (ScalarPH->hasPredecessors()) {
7598 // Fix resume values for inductions and reductions from the additional
7599 // bypass block using the incoming values from the main loop's resume phis.
7600 // ResumeValues correspond 1:1 with the scalar loop header phis.
7601 for (auto [ResumeV, HeaderPhi] :
7602 zip(ResumeValues, BestEpiPlan.getScalarHeader()->phis())) {
7603 auto *HeaderPhiR = cast<VPIRPhi>(&HeaderPhi);
7604 auto *EpiResumePhi =
7605 cast<PHINode>(HeaderPhiR->getIRPhi().getIncomingValueForBlock(PH));
7606 if (EpiResumePhi->getBasicBlockIndex(BypassBlock) == -1)
7607 continue;
7608 auto *MainResumePhi = cast<PHINode>(ResumeV->getUnderlyingValue());
7609 EpiResumePhi->setIncomingValueForBlock(
7610 BypassBlock, MainResumePhi->getIncomingValueForBlock(BypassBlock));
7611 }
7612 }
7613}
7614
7615/// Connect the epilogue vector loop generated for \p EpiPlan to the main vector
7616/// loop, after both plans have executed, updating the branch from the iteration
7617/// count check of the main loop, as well as updating various phis.
7619 VPIRBasicBlock *VecEpilogueIterCheckVPBB,
7620 ArrayRef<VPInstruction *> ResumeValues) {
7621 ArrayRef<VPBlockBase *> Preds = VecEpilogueIterCheckVPBB->getPredecessors();
7622 BasicBlock *MainLoopIterationCountCheck =
7623 cast<VPIRBasicBlock>(Preds.front())->getIRBasicBlock();
7624 BasicBlock *VecEpilogueIterationCountCheck =
7625 VecEpilogueIterCheckVPBB->getIRBasicBlock();
7626 BasicBlock *VecEpiloguePreHeader =
7627 cast<CondBrInst>(VecEpilogueIterationCountCheck->getTerminator())
7628 ->getSuccessor(1);
7629 DomTreeUpdater DTU(DT, DomTreeUpdater::UpdateStrategy::Eager);
7630
7631 MainLoopIterationCountCheck->getTerminator()->replaceSuccessorWith(
7632 VecEpilogueIterationCountCheck, VecEpiloguePreHeader);
7633 DTU.applyUpdates({{DominatorTree::Delete, MainLoopIterationCountCheck,
7634 VecEpilogueIterationCountCheck},
7635 {DominatorTree::Insert, MainLoopIterationCountCheck,
7636 VecEpiloguePreHeader}});
7637
7638 // The vec.epilog.iter.check block may contain Phi nodes from inductions
7639 // or reductions which merge control-flow from the latch block and the
7640 // middle block. Update the incoming values here and move the Phi into the
7641 // preheader.
7642 SmallVector<PHINode *, 4> PhisInBlock(
7643 llvm::make_pointer_range(VecEpilogueIterationCountCheck->phis()));
7644
7645 for (PHINode *Phi : PhisInBlock) {
7646 Phi->moveBefore(VecEpiloguePreHeader->getFirstNonPHIIt());
7647 Phi->replaceIncomingBlockWith(
7648 VecEpilogueIterationCountCheck->getSinglePredecessor(),
7649 VecEpilogueIterationCountCheck);
7650 }
7651
7652 // VecEpilogueIterationCountCheck conditionally skips over the epilogue loop
7653 // after executing the main loop. We need to update the resume values of
7654 // inductions and reductions during epilogue vectorization.
7655 fixScalarResumeValuesFromBypass(VecEpilogueIterationCountCheck, EpiPlan,
7656 ResumeValues);
7657
7658 // Remove dead phis that were moved to the epilogue preheader but are unused
7659 // (e.g., resume phis for inductions not widened in the epilogue vector loop).
7660 for (PHINode &Phi : make_early_inc_range(VecEpiloguePreHeader->phis()))
7661 if (Phi.use_empty())
7662 Phi.eraseFromParent();
7663}
7664
7666 assert((EnableVPlanNativePath || L->isInnermost()) &&
7667 "VPlan-native path is not enabled. Only process inner loops.");
7668
7669 LLVM_DEBUG(dbgs() << "\nLV: Checking a loop in '"
7670 << L->getHeader()->getParent()->getName() << "' from "
7671 << L->getLocStr() << "\n");
7672
7673 LoopVectorizeHints Hints(L, InterleaveOnlyWhenForced, *ORE, TTI);
7674
7675 LLVM_DEBUG(
7676 dbgs() << "LV: Loop hints:"
7677 << " force="
7679 ? "disabled"
7681 ? "enabled"
7682 : "?"))
7683 << " width=" << Hints.getWidth()
7684 << " interleave=" << Hints.getInterleave() << "\n");
7685
7686 // Function containing loop
7687 Function *F = L->getHeader()->getParent();
7688
7689 // Looking at the diagnostic output is the only way to determine if a loop
7690 // was vectorized (other than looking at the IR or machine code), so it
7691 // is important to generate an optimization remark for each loop. Most of
7692 // these messages are generated as OptimizationRemarkAnalysis. Remarks
7693 // generated as OptimizationRemark and OptimizationRemarkMissed are
7694 // less verbose reporting vectorized loops and unvectorized loops that may
7695 // benefit from vectorization, respectively.
7696
7697 if (!Hints.allowVectorization(F, L, VectorizeOnlyWhenForced)) {
7698 LLVM_DEBUG(dbgs() << "LV: Loop hints prevent vectorization.\n");
7699 return false;
7700 }
7701
7702 PredicatedScalarEvolution PSE(*SE, *L);
7703
7704 // Query this against the original loop and save it here because the profile
7705 // of the original loop header may change as the transformation happens.
7706 bool OptForSize = llvm::shouldOptimizeForSize(
7707 L->getHeader(), PSI,
7708 PSI && PSI->hasProfileSummary() ? &GetBFI() : nullptr,
7710
7711 // Check if it is legal to vectorize the loop.
7712 LoopVectorizationRequirements Requirements;
7713 LoopVectorizationLegality LVL(L, PSE, DT, TTI, TLI, F, *LAIs, LI, ORE,
7714 &Requirements, &Hints, DB, AC,
7715 /*AllowRuntimeSCEVChecks=*/!OptForSize, AA);
7717 LLVM_DEBUG(dbgs() << "LV: Not vectorizing: Cannot prove legality.\n");
7718 Hints.emitRemarkWithHints();
7719 return false;
7720 }
7721
7722 bool IsInnerLoop = L->isInnermost();
7723
7724 // Outer loops require a computable trip count.
7725 if (!IsInnerLoop && isa<SCEVCouldNotCompute>(PSE.getBackedgeTakenCount())) {
7726 LLVM_DEBUG(dbgs() << "LV: cannot compute the outer-loop trip count\n");
7727 return false;
7728 }
7729
7730 if (LVL.hasUncountableEarlyExit()) {
7732 reportVectorizationFailure("Auto-vectorization of loops with uncountable "
7733 "early exit is not enabled",
7734 "UncountableEarlyExitLoopsDisabled", ORE, L);
7735 return false;
7736 }
7739 reportVectorizationFailure("Auto-vectorization of loops with uncountable "
7740 "early exit and side effects is not enabled",
7741 "UncountableEarlyExitSideEffectLoopsDisabled",
7742 ORE, L);
7743 return false;
7744 }
7745 }
7746
7747 InterleavedAccessInfo IAI(PSE, L, DT, LI, LVL.getLAI(), OptForSize);
7748 bool UseInterleaved =
7749 IsInnerLoop && TTI->enableInterleavedAccessVectorization();
7750
7751 // If an override option has been passed in for interleaved accesses, use it.
7752 if (EnableInterleavedMemAccesses.getNumOccurrences() > 0)
7753 UseInterleaved = IsInnerLoop && EnableInterleavedMemAccesses;
7754
7755 // Analyze interleaved memory accesses.
7756 if (UseInterleaved)
7758
7759 if (LVL.hasUncountableEarlyExit()) {
7760 BasicBlock *LoopLatch = L->getLoopLatch();
7761 if (IAI.requiresScalarEpilogue() ||
7762 any_of(LVL.getCountableExitingBlocks(), not_equal_to(LoopLatch))) {
7763 reportVectorizationFailure("Auto-vectorization of early exit loops "
7764 "requiring a scalar epilogue is unsupported",
7765 "UncountableEarlyExitUnsupported", ORE, L);
7766 return false;
7767 }
7768 }
7769
7770 // Check the function attributes and profiles to find out if this function
7771 // should be optimized for size.
7772 EpilogueLowering SEL =
7773 getEpilogueLowering(F, L, Hints, OptForSize, TTI, TLI, LVL, &IAI);
7774
7775 // Check the loop for a trip count threshold: vectorize loops with a tiny trip
7776 // count by optimizing for size, to minimize overheads.
7777 auto ExpectedTC = getSmallBestKnownTC(PSE, L);
7778 if (ExpectedTC && ExpectedTC->isFixed() &&
7779 ExpectedTC->getFixedValue() < TinyTripCountVectorThreshold) {
7780 LLVM_DEBUG(dbgs() << "LV: Found a loop with a very small trip count. "
7781 << "This loop is worth vectorizing only if no scalar "
7782 << "iteration overheads are incurred.");
7784 LLVM_DEBUG(dbgs() << " But vectorizing was explicitly forced.\n");
7785 else {
7786 LLVM_DEBUG(dbgs() << "\n");
7787 // Tail-folded loops are efficient even when the loop
7788 // iteration count is low. However, setting the epilogue policy to
7789 // `CM_EpilogueNotAllowedLowTripLoop` prevents vectorizing loops
7790 // with runtime checks. It's more effective to let
7791 // `isOutsideLoopWorkProfitable` determine if vectorization is
7792 // beneficial for the loop. If the trip count is below the target's
7793 // minimum for tail-folding, the tail cannot be folded, so treat it like
7794 // any other low trip count loop.
7795 if (SEL != CM_EpilogueNotNeededFoldTail ||
7796 ExpectedTC->getFixedValue() <=
7797 TTI->getMinTripCountTailFoldingThreshold())
7799 }
7800 }
7801
7802 // Check the function attributes to see if implicit floats or vectors are
7803 // allowed.
7804 if (F->hasFnAttribute(Attribute::NoImplicitFloat)) {
7806 "Can't vectorize when the NoImplicitFloat attribute is used",
7807 "loop not vectorized due to NoImplicitFloat attribute",
7808 "NoImplicitFloat", ORE, L);
7809 Hints.emitRemarkWithHints();
7810 return false;
7811 }
7812
7813 // Check if the target supports potentially unsafe FP vectorization.
7814 // FIXME: Add a check for the type of safety issue (denormal, signaling)
7815 // for the target we're vectorizing for, to make sure none of the
7816 // additional fp-math flags can help.
7817 if (Hints.isPotentiallyUnsafe() &&
7818 TTI->isFPVectorizationPotentiallyUnsafe()) {
7820 "Potentially unsafe FP op prevents vectorization",
7821 "loop not vectorized due to unsafe FP support.", "UnsafeFP", ORE, L);
7822 Hints.emitRemarkWithHints();
7823 return false;
7824 }
7825
7826 bool AllowOrderedReductions;
7827 // If the flag is set, use that instead and override the TTI behaviour.
7828 if (ForceOrderedReductions.getNumOccurrences() > 0)
7829 AllowOrderedReductions = ForceOrderedReductions;
7830 else
7831 AllowOrderedReductions = TTI->enableOrderedReductions();
7832 if (!LVL.canVectorizeFPMath(AllowOrderedReductions)) {
7833 ORE->emit([&]() {
7834 auto *ExactFPMathInst = Requirements.getExactFPInst();
7835 return OptimizationRemarkAnalysisFPCommute(DEBUG_TYPE, "CantReorderFPOps",
7836 ExactFPMathInst->getDebugLoc(),
7837 ExactFPMathInst->getParent())
7838 << "loop not vectorized: cannot prove it is safe to reorder "
7839 "floating-point operations";
7840 });
7841 LLVM_DEBUG(dbgs() << "LV: loop not vectorized: cannot prove it is safe to "
7842 "reorder floating-point operations\n");
7843 Hints.emitRemarkWithHints();
7844 return false;
7845 }
7846
7847 // Use the cost model.
7848 VFSelectionContext Config(*TTI, &LVL, L, *F, PSE, DB, ORE, &Hints,
7849 OptForSize);
7850 // Use the planner for vectorization.
7852 L, LI, DT, TLI, *TTI, &LVL,
7853 std::make_unique<LoopVectorizationCostModel>(
7854 SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE, GetBFI, F, IAI, Config),
7855 Config, IAI, PSE, ORE, GetBPI);
7856
7857 EpilogueLowering EpilogueTailLoweringStatus =
7858 getEpilogueTailLowering(LVP.getCostModel(), L, ORE, LVL, Hints, TTI);
7859 if (EpilogueTailLoweringStatus ==
7861 // TODO: Apply tail-folding on the vectorized epilogue loop.
7862 LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is not supported yet\n");
7864 "The epilogue-tail-folding policy prefer-fold-tail is not supported "
7865 "yet, fall back to a normal epilogue",
7866 "UnsupportedEpilogueTailFoldingPolicy", ORE, L);
7867 }
7868
7869 // Get user vectorization factor and interleave count.
7870 ElementCount UserVF = Hints.getWidth();
7871 unsigned UserIC = Hints.getInterleave();
7872 // Outer loops don't have LoopAccessInfo, so skip the safety check and reset
7873 // UserIC (interleaving is not supported for outer loops).
7874 if (!IsInnerLoop)
7875 UserIC = 0;
7876 else if (UserIC > 1 && !LVL.isSafeForAnyVectorWidth())
7877 UserIC = 1;
7878
7879 // Plan how to best vectorize.
7880 LVP.plan(UserVF, UserIC);
7881 auto [VF, BestPlanPtr] = LVP.computeBestVF();
7882 unsigned IC = 1;
7883
7884 // For VPlan build stress testing of outer loops, bail after plan
7885 // construction.
7886 if (!IsInnerLoop && VPlanBuildOuterloopStressTest)
7887 return false;
7888
7889 if (IsInnerLoop && ORE->allowExtraAnalysis(LV_NAME))
7891
7892 assert((IsInnerLoop || !LVP.getCostModel().maskPartialAliasing()) &&
7893 "Did not expect to alias-mask outer loop");
7894
7895 GeneratedRTChecks Checks(PSE, DT, LI, TTI, Config.CostKind,
7897 if (IsInnerLoop && LVP.hasPlanWithVF(VF.Width)) {
7898 // Select the interleave count.
7899 IC = LVP.selectInterleaveCount(*BestPlanPtr, VF.Width, VF.Cost);
7900
7901 unsigned SelectedIC = UserIC > 0 ? UserIC : IC;
7902 // Optimistically generate runtime checks if they are needed. Drop them if
7903 // they turn out to not be profitable.
7904 if (VF.Width.isVector() || SelectedIC > 1) {
7905 Checks.create(L, *LVL.getLAI(), PSE.getPredicate(), VF.Width, SelectedIC,
7906 *ORE);
7907
7908 // Bail out early if either the SCEV or memory runtime checks are known to
7909 // fail. In that case, the vector loop would never execute.
7910 using namespace llvm::PatternMatch;
7911 if (Checks.getSCEVChecks().first &&
7912 match(Checks.getSCEVChecks().first, m_One()))
7913 return false;
7914 if (Checks.getMemRuntimeChecks().first &&
7915 match(Checks.getMemRuntimeChecks().first, m_One()))
7916 return false;
7917 }
7918
7919 // Check if it is profitable to vectorize with runtime checks.
7920 bool ForceVectorization =
7922 VPCostContext CostCtx(*TLI, *BestPlanPtr, LVP.getCostModel(), Config,
7923 /*ReusePrintingSlotTracker=*/true);
7924 if (!ForceVectorization &&
7925 !isOutsideLoopWorkProfitable(Checks, VF, L, PSE, CostCtx, *BestPlanPtr,
7926 SEL, Config.getVScaleForTuning())) {
7927 ORE->emit([&]() {
7929 DEBUG_TYPE, "CantReorderMemOps", L->getStartLoc(),
7930 L->getHeader())
7931 << "loop not vectorized: cannot prove it is safe to reorder "
7932 "memory operations";
7933 });
7934 LLVM_DEBUG(dbgs() << "LV: Too many memory checks needed.\n");
7935 Hints.emitRemarkWithHints();
7936 return false;
7937 }
7938 }
7939
7940 // Identify the diagnostic messages that should be produced.
7941 std::pair<StringRef, std::string> VecDiagMsg, IntDiagMsg;
7942 bool VectorizeLoop = true, InterleaveLoop = true;
7943 if (VF.Width.isScalar()) {
7944 LLVM_DEBUG(dbgs() << "LV: Vectorization is possible but not beneficial.\n");
7945 VecDiagMsg = {
7946 "VectorizationNotBeneficial",
7947 "the cost-model indicates that vectorization is not beneficial"};
7948 VectorizeLoop = false;
7949 }
7950
7951 if (UserIC == 1 && Hints.getInterleave() > 1) {
7953 "UserIC should only be ignored due to unsafe dependencies");
7954 LLVM_DEBUG(dbgs() << "LV: Ignoring user-specified interleave count.\n");
7955 IntDiagMsg = {"InterleavingUnsafe",
7956 "Ignoring user-specified interleave count due to possibly "
7957 "unsafe dependencies in the loop."};
7958 InterleaveLoop = false;
7959 } else if (!LVP.hasPlanWithVF(VF.Width) && UserIC > 1) {
7960 // Tell the user interleaving was avoided up-front, despite being explicitly
7961 // requested.
7962 LLVM_DEBUG(dbgs() << "LV: Ignoring UserIC, because vectorization and "
7963 "interleaving should be avoided up front\n");
7964 IntDiagMsg = {"InterleavingAvoided",
7965 "Ignoring UserIC, because interleaving was avoided up front"};
7966 InterleaveLoop = false;
7967 } else if (IC == 1 && UserIC <= 1) {
7968 // Tell the user interleaving is not beneficial.
7969 LLVM_DEBUG(dbgs() << "LV: Interleaving is not beneficial.\n");
7970 IntDiagMsg = {
7971 "InterleavingNotBeneficial",
7972 "the cost-model indicates that interleaving is not beneficial"};
7973 InterleaveLoop = false;
7974 if (UserIC == 1) {
7975 IntDiagMsg.first = "InterleavingNotBeneficialAndDisabled";
7976 IntDiagMsg.second +=
7977 " and is explicitly disabled or interleave count is set to 1";
7978 }
7979 } else if (IC > 1 && UserIC == 1) {
7980 // Tell the user interleaving is beneficial, but it explicitly disabled.
7981 LLVM_DEBUG(dbgs() << "LV: Interleaving is beneficial but is explicitly "
7982 "disabled.\n");
7983 IntDiagMsg = {"InterleavingBeneficialButDisabled",
7984 "the cost-model indicates that interleaving is beneficial "
7985 "but is explicitly disabled or interleave count is set to 1"};
7986 InterleaveLoop = false;
7987 }
7988
7989 // If there is a histogram in the loop, do not just interleave without
7990 // vectorizing. The order of operations will be incorrect without the
7991 // histogram intrinsics, which are only used for recipes with VF > 1.
7992 if (!VectorizeLoop && InterleaveLoop && LVL.hasHistograms()) {
7993 LLVM_DEBUG(dbgs() << "LV: Not interleaving without vectorization due "
7994 << "to histogram operations.\n");
7995 IntDiagMsg = {
7996 "HistogramPreventsScalarInterleaving",
7997 "Unable to interleave without vectorization due to constraints on "
7998 "the order of histogram operations"};
7999 InterleaveLoop = false;
8000 }
8001
8002 // Override IC if user provided an interleave count.
8003 IC = UserIC > 0 ? UserIC : IC;
8004
8005 if (LVP.getCostModel().maskPartialAliasing()) {
8006 LLVM_DEBUG(
8007 dbgs()
8008 << "LV: Not interleaving due to partial aliasing vectorization.\n");
8009 IntDiagMsg = {
8010 "PartialAliasingVectorization",
8011 "Unable to interleave due to partial aliasing vectorization."};
8012 InterleaveLoop = false;
8013 IC = 1;
8014 }
8015
8016 // FIXME: Enable interleaving for EE-with-side-effects.
8017 if (InterleaveLoop && LVL.hasUncountableExitWithSideEffects()) {
8018 LLVM_DEBUG(dbgs() << "LV: Not interleaving due to EE with side effects.\n");
8019 IntDiagMsg = {"EEWithSideEffectsPreventsInterleaving",
8020 "Unable to interleave due to early exit with side effects."};
8021 InterleaveLoop = false;
8022 IC = 1;
8023 }
8024
8025 // Emit diagnostic messages, if any.
8026 if (!VectorizeLoop && !InterleaveLoop) {
8027 // Do not vectorize or interleaving the loop.
8028 ORE->emit([&]() {
8029 return OptimizationRemarkMissed(LV_NAME, VecDiagMsg.first,
8030 L->getStartLoc(), L->getHeader())
8031 << VecDiagMsg.second;
8032 });
8033 ORE->emit([&]() {
8034 return OptimizationRemarkMissed(LV_NAME, IntDiagMsg.first,
8035 L->getStartLoc(), L->getHeader())
8036 << IntDiagMsg.second;
8037 });
8038 return false;
8039 }
8040
8041 if (!VectorizeLoop && InterleaveLoop) {
8042 LLVM_DEBUG(dbgs() << "LV: Interleave Count is " << IC << '\n');
8043 ORE->emit([&]() {
8044 return OptimizationRemarkAnalysis(LV_NAME, VecDiagMsg.first,
8045 L->getStartLoc(), L->getHeader())
8046 << VecDiagMsg.second;
8047 });
8048 } else if (VectorizeLoop && !InterleaveLoop) {
8049 LLVM_DEBUG(dbgs() << "LV: Found a vectorizable loop (" << VF.Width
8050 << ") in " << L->getLocStr() << '\n');
8051 ORE->emit([&]() {
8052 return OptimizationRemarkAnalysis(LV_NAME, IntDiagMsg.first,
8053 L->getStartLoc(), L->getHeader())
8054 << IntDiagMsg.second;
8055 });
8056 } else if (VectorizeLoop && InterleaveLoop) {
8057 LLVM_DEBUG(dbgs() << "LV: Found a vectorizable loop (" << VF.Width
8058 << ") in " << L->getLocStr() << '\n');
8059 LLVM_DEBUG(dbgs() << "LV: Interleave Count is " << IC << '\n');
8060 }
8061
8062 // Report the vectorization decision.
8063 if (VF.Width.isScalar()) {
8064 using namespace ore;
8065 assert(IC > 1);
8066 ORE->emit([&]() {
8067 return OptimizationRemark(LV_NAME, "Interleaved", L->getStartLoc(),
8068 L->getHeader())
8069 << "interleaved loop (interleaved count: "
8070 << NV("InterleaveCount", IC) << ")";
8071 });
8072 } else {
8073 // Report the vectorization decision.
8074 reportVectorization(ORE, L, VF.Width, IC);
8075 }
8076 if (ORE->allowExtraAnalysis(LV_NAME))
8078
8079 // If we decided that it is *legal* to interleave or vectorize the loop, then
8080 // do it.
8081
8082 // Whether a scalar epilogue may be created is decided by the epilogue
8083 // lowering policy.
8084 // TODO: Also move check to be based on VPlan.
8085 bool ScalarEpilogueAllowed = LVP.getCostModel().isEpilogueAllowed();
8086
8087 // Destroy the cost model before executing any plan, so that code generation
8088 // cannot rely on cost-modeling decisions.
8089 LVP.clearCostModel();
8090
8091 VPlan &BestPlan = *BestPlanPtr;
8092 // Consider vectorizing the epilogue too if it's profitable.
8093 std::unique_ptr<VPlan> EpiPlan =
8094 LVP.selectBestEpiloguePlan(BestPlan, VF.Width, IC, ScalarEpilogueAllowed);
8095 bool HasBranchWeights =
8096 hasBranchWeightMD(*L->getLoopLatch()->getTerminator());
8097 if (EpiPlan) {
8098 VPlan &BestEpiPlan = *EpiPlan;
8099 VPlan &BestMainPlan = BestPlan;
8100 ElementCount EpilogueVF = BestEpiPlan.getSingleVF();
8101
8102 // The first pass vectorizes the main loop and creates a scalar epilogue
8103 // to be vectorized by executing the plan (potentially with a different
8104 // factor) again shortly afterwards.
8105 BestEpiPlan.getMiddleBlock()->setName("vec.epilog.middle.block");
8106 BestEpiPlan.getVectorPreheader()->setName("vec.epilog.ph");
8107 SmallVector<VPInstruction *> ResumeValues =
8108 preparePlanForMainVectorLoop(BestMainPlan, BestEpiPlan);
8109 EpilogueLoopVectorizationInfo EPI(VF.Width, IC, EpilogueVF);
8110
8111 // Add minimum iteration check for the epilogue plan, followed by runtime
8112 // checks for the main plan.
8113 LVP.addMinimumIterationCheck(BestMainPlan, EPI.EpilogueVF, /*UF=*/1,
8115 LVP.attachRuntimeChecks(BestMainPlan, Checks, HasBranchWeights);
8118 EPI.MainLoopVF, EPI.MainLoopUF, BestMainPlan.requiresScalarEpilogue(),
8119 L, HasBranchWeights ? MinItersBypassWeights : nullptr,
8120 L->getLoopPredecessor()->getTerminator()->getDebugLoc(), PSE);
8121
8122 LLVM_DEBUG({
8123 dbgs() << "Create Skeleton for epilogue vectorized loop (first pass)\n"
8124 << "Main Loop VF:" << EPI.MainLoopVF
8125 << ", Main Loop UF:" << EPI.MainLoopUF
8126 << ", Epilogue Loop VF:" << EPI.EpilogueVF
8127 << ", Epilogue Loop UF:1\n";
8128 });
8129 InnerLoopVectorizer MainILV(L, PSE, LI, DT, TTI, AC, EPI.MainLoopVF,
8130 EPI.MainLoopUF, Checks, BestMainPlan);
8131 auto ExpandedSCEVs = LVP.executePlan(
8132 EPI.MainLoopVF, EPI.MainLoopUF, BestMainPlan, MainILV, DT,
8134 ++LoopsVectorized;
8136 dbgs() << "intermediate fn:\n" << *L->getHeader()->getParent() << "\n";
8137 });
8138
8139 BasicBlock *EntryBB =
8140 cast<VPIRBasicBlock>(BestMainPlan.getEntry())->getIRBasicBlock();
8141 EntryBB->setName("iter.check");
8142
8143 // Second pass vectorizes the epilogue and adjusts the control flow
8144 // edges from the first pass.
8145 EpilogueVectorizerEpilogueLoop EpilogILV(L, PSE, LI, DT, TTI, AC,
8146 EPI.EpilogueVF, /*UnrollFactor=*/1,
8147 Checks, BestEpiPlan, BestMainPlan);
8148 preparePlanForEpilogueVectorLoop(BestMainPlan, BestEpiPlan, L,
8149 ExpandedSCEVs, EPI, LVP, Config,
8150 *PSE.getSE(), ResumeValues);
8152 LLVM_DEBUG({
8153 dbgs() << "Create Skeleton for epilogue vectorized loop (second pass)\n"
8154 << "Epilogue Loop VF:" << EPI.EpilogueVF
8155 << ", Epilogue Loop UF:1\n";
8156 });
8157 LVP.executePlan(
8158 EPI.EpilogueVF, /*BestUF=*/1, BestEpiPlan, EpilogILV, DT,
8161 dbgs() << "final fn:\n" << *L->getHeader()->getParent() << "\n";
8162 });
8163 connectEpilogueVectorLoop(BestEpiPlan, DT,
8165 ResumeValues);
8166 ++LoopsEpilogueVectorized;
8167 } else {
8168 InnerLoopVectorizer LB(L, PSE, LI, DT, TTI, AC, VF.Width, IC, Checks,
8169 BestPlan);
8170 LVP.addMinimumIterationCheck(BestPlan, VF.Width, IC,
8171 VF.MinProfitableTripCount);
8172 LVP.attachRuntimeChecks(BestPlan, Checks, HasBranchWeights);
8173
8174 if (!IsInnerLoop)
8175 LLVM_DEBUG(dbgs() << "Vectorizing outer loop in \"" << F->getName()
8176 << "\"\n");
8177 LVP.executePlan(VF.Width, IC, BestPlan, LB, DT);
8178 ++LoopsVectorized;
8179 }
8180
8181 assert(DT->verify(DominatorTree::VerificationLevel::Fast) &&
8182 "DT not preserved correctly");
8183
8184 return true;
8185}
8186
8188 CFGChanged = false;
8189
8190 // Don't attempt if
8191 // 1. the target claims to have no vector registers, and
8192 // 2. interleaving won't help ILP.
8193 //
8194 // The second condition is necessary because, even if the target has no
8195 // vector registers, loop vectorization may still enable scalar
8196 // interleaving.
8197 if (!TTI->getNumberOfRegisters(TTI->getRegisterClassForType(true)) &&
8198 (TTI->getMaxInterleaveFactor(ElementCount::getFixed(1), false) < 2 ||
8199 TTI->getMaxInterleaveFactor(ElementCount::getFixed(1), true) < 2))
8200 return LoopVectorizeResult(false, false);
8201
8202 bool Changed = false;
8203
8204 // The vectorizer requires loops to be in simplified form.
8205 // Since simplification may add new inner loops, it has to run before the
8206 // legality and profitability checks. This means running the loop vectorizer
8207 // will simplify all loops, regardless of whether anything end up being
8208 // vectorized.
8209 for (const auto &L : *LI)
8210 Changed |= CFGChanged |=
8211 simplifyLoop(L, DT, LI, SE, AC, nullptr, false /* PreserveLCSSA */);
8212
8213 // Build up a worklist of inner-loops to vectorize. This is necessary as
8214 // the act of vectorizing or partially unrolling a loop creates new loops
8215 // and can invalidate iterators across the loops.
8216 SmallVector<Loop *, 8> Worklist;
8217
8218 for (Loop *L : *LI)
8219 collectSupportedLoops(*L, LI, ORE, Worklist);
8220
8221 LoopsAnalyzed += Worklist.size();
8222
8223 // Now walk the identified inner loops.
8224 while (!Worklist.empty()) {
8225 Loop *L = Worklist.pop_back_val();
8226
8227 // For the inner loops we actually process, form LCSSA to simplify the
8228 // transform.
8229 Changed |= formLCSSARecursively(*L, *DT, LI, SE);
8230
8232
8233 if (Changed) {
8234 LAIs->clear();
8235
8236#ifndef NDEBUG
8237 if (VerifySCEV)
8238 SE->verify();
8239#endif
8240 }
8241 }
8242
8243 // Verify once per function rather than once per processed loop, which would
8244 // make the pass quadratic in the number of loops.
8245 assert((!Changed || !verifyFunction(F, &dbgs())) &&
8246 "Invalid IR produced by LoopVectorize");
8247
8248 // Process each loop nest in the function.
8250}
8251
8254 LI = &AM.getResult<LoopAnalysis>(F);
8255 // There are no loops in the function. Return before computing other
8256 // expensive analyses.
8257 if (LI->empty())
8258 return PreservedAnalyses::all();
8267 AA = &AM.getResult<AAManager>(F);
8268
8269 auto &MAMProxy = AM.getResult<ModuleAnalysisManagerFunctionProxy>(F);
8270 PSI = MAMProxy.getCachedResult<ProfileSummaryAnalysis>(*F.getParent());
8271 // CycleInfo cached by an earlier pass is invalidated when the CFG changes.
8272 // Both BlockFrequencyAnalysis and BranchProbabilityAnalysis depend on it, so
8273 // drop the stale result before either is (re-)computed.
8274 auto ClearStaleCycleInfo = [this, &AM, &F] {
8277 };
8278 GetBFI = [&AM, &F, ClearStaleCycleInfo]() -> BlockFrequencyInfo & {
8279 ClearStaleCycleInfo();
8281 };
8282 GetBPI = [&AM, &F, ClearStaleCycleInfo]() -> const BranchProbabilityInfo & {
8283 ClearStaleCycleInfo();
8285 };
8286 LoopVectorizeResult Result = runImpl(F);
8287 if (!Result.MadeAnyChange)
8288 return PreservedAnalyses::all();
8290
8291 if (isAssignmentTrackingEnabled(*F.getParent())) {
8292 for (auto &BB : F)
8294 }
8295
8296 PA.preserve<LoopAnalysis>();
8300
8301 if (Result.MadeCFGChange) {
8302 // Making CFG changes likely means a loop got vectorized. Indicate that
8303 // extra simplification passes should be run.
8304 // TODO: MadeCFGChanges is not a prefect proxy. Extra passes should only
8305 // be run if runtime checks have been added.
8308 } else {
8310 }
8311 return PA;
8312}
8313
8315 raw_ostream &OS, function_ref<StringRef(StringRef)> MapClassName2PassName) {
8316 static_cast<PassInfoMixin<LoopVectorizePass> *>(this)->printPipeline(
8317 OS, MapClassName2PassName);
8318
8319 OS << '<';
8320 OS << (InterleaveOnlyWhenForced ? "" : "no-") << "interleave-forced-only;";
8321 OS << (VectorizeOnlyWhenForced ? "" : "no-") << "vectorize-forced-only;";
8322 OS << '>';
8323}
for(const MachineOperand &MO :llvm::drop_begin(OldMI.operands(), Desc.getNumOperands()))
static unsigned getIntrinsicID(const SDNode *N)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
AMDGPU Lower Kernel Arguments
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static bool isEqual(const Function &Caller, const Function &Callee)
This file contains the simple types necessary to represent the attributes associated with functions a...
static const Function * getParent(const Value *V)
This is the interface for LLVM's primary stateless and local alias analysis.
static bool IsEmptyBlock(MachineBasicBlock *MBB)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
#define clEnumValN(ENUMVAL, FLAGNAME, DESC)
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static InstructionCost getCost(Instruction &Inst, TTI::TargetCostKind CostKind, TargetTransformInfo &TTI)
Definition CostModel.cpp:73
This file declares an analysis pass that computes CycleInfo for LLVM IR, specialized from GenericCycl...
This file defines the DenseMap class.
#define DEBUG_TYPE
This is the interface for a simple mod/ref and alias analysis over globals.
Hexagon Common GEP
This file provides various utilities for inspecting and working with the control flow graph in LLVM I...
Module.h This file contains the declarations for the Module class.
This defines the Use class.
static bool hasNoUnsignedWrap(BinaryOperator &I)
This file defines an InstructionCost class that is used when calculating the cost of an instruction,...
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static cl::opt< ElementCount, true > VectorizationFactor("force-vector-width", cl::Hidden, cl::desc("Sets the SIMD width. Zero is autoselect."), cl::location(VectorizerParams::VectorizationFactor))
This header provides classes for managing per-loop analyses.
static const char * VerboseDebug
#define LV_NAME
This file defines the LoopVectorizationLegality class.
static cl::opt< bool > ConsiderRegPressure("vectorizer-consider-reg-pressure", cl::init(false), cl::Hidden, cl::desc("Discard VFs if their register pressure is too high."))
This file provides a LoopVectorizationPlanner class.
static void collectSupportedLoops(Loop &L, LoopInfo *LI, OptimizationRemarkEmitter *ORE, SmallVectorImpl< Loop * > &V)
static cl::opt< unsigned > EpilogueVectorizationMinVF("epilogue-vectorization-minimum-VF", cl::Hidden, cl::desc("Only loops with vectorization factor equal to or larger than " "the specified value are considered for epilogue vectorization."))
static unsigned getMaxTCFromNonZeroRange(PredicatedScalarEvolution &PSE, Loop *L)
Get the maximum trip count for L from the SCEV unsigned range, excluding zero from the range.
static Type * maybeVectorizeType(Type *Ty, ElementCount VF)
static ElementCount getSmallConstantTripCount(ScalarEvolution *SE, const Loop *L)
A version of ScalarEvolution::getSmallConstantTripCount that returns an ElementCount to include loops...
static cl::opt< unsigned > TinyTripCountVectorThreshold("vectorizer-min-trip-count", cl::init(16), cl::Hidden, cl::desc("Loops with a constant trip count that is smaller than this " "value are vectorized only if no scalar iteration overheads " "are incurred."))
Loops with a known constant trip count below this number are vectorized only if no scalar iteration o...
static cl::opt< unsigned > PragmaVectorizeSCEVCheckThreshold("pragma-vectorize-scev-check-threshold", cl::init(128), cl::Hidden, cl::desc("The maximum number of SCEV checks allowed with a " "vectorize(enable) pragma"))
static void preparePlanForEpilogueVectorLoop(VPlan &MainPlan, VPlan &Plan, Loop *L, const SCEV2ValueTy &ExpandedSCEVs, EpilogueLoopVectorizationInfo &EPI, LoopVectorizationPlanner &LVP, VFSelectionContext &Config, ScalarEvolution &SE, ArrayRef< VPInstruction * > ResumeValues)
Prepare Plan for vectorizing the epilogue loop.
static cl::opt< cl::boolOrDefault > ForceMaskedDivRem("force-widen-divrem-via-masked-intrinsic", cl::Hidden, cl::desc("Override cost based masked intrinsic widening " "for div/rem instructions"))
static void legacyCSE(BasicBlock *BB)
FIXME: This legacy common-subexpression-elimination routine is scheduled for removal,...
static VPIRBasicBlock * replaceVPBBWithIRVPBB(VPBasicBlock *VPBB, BasicBlock *IRBB, VPlan *Plan=nullptr)
Replace VPBB with a VPIRBasicBlock wrapping IRBB.
static void fixScalarResumeValuesFromBypass(BasicBlock *BypassBlock, VPlan &BestEpiPlan, ArrayRef< VPInstruction * > ResumeValues)
static Intrinsic::ID getMaskedDivRemIntrinsic(unsigned Opcode)
static DebugLoc getDebugLocFromInstOrOperands(Instruction *I)
Look for a meaningful debug location on the instruction or its operands.
static cl::opt< unsigned > LowTripCountLoopBodySizeLimit("low-trip-count-loop-body-size-limit", cl::init(20), cl::Hidden, cl::desc("Minimum number of instructions to vectorize loops with trip " "counts below tail folding threshold"))
TailFoldingPolicyTy
Option tail-folding-policy controls the tail-folding strategy and lists all available options.
static bool useActiveLaneMaskForControlFlow(TailFoldingStyle Style)
static std::optional< VPExecutionFrequency > getRecordedExecutionFrequency(const VPBasicBlock *VPBB)
Returns the frequency with which VPBB executes, as recorded on its recipes.
static cl::opt< TailFoldingPolicyTy > EpilogueTailFoldingPolicy("epilogue-tail-folding-policy", cl::Hidden, cl::desc("Epilogue-tail-folding preferences over creating an epilogue loop."), cl::values(clEnumValN(TailFoldingPolicyTy::None, "dont-fold-tail", "Don't tail-fold loops."), clEnumValN(TailFoldingPolicyTy::PreferFoldTail, "prefer-fold-tail", "prefer tail-folding, otherwise create an epilogue when " "appropriate.")))
static cl::opt< bool > EnableEarlyExitVectorization("enable-early-exit-vectorization", cl::init(true), cl::Hidden, cl::desc("Enable vectorization of early exit loops with uncountable exits."))
static unsigned estimateElementCount(ElementCount VF, std::optional< unsigned > VScale)
This function attempts to return a value that represents the ElementCount at runtime.
static bool hasVectorLibraryVariantFor(const CallInst &CI, ElementCount VF, bool MaskRequired, const TargetLibraryInfo *TLI)
Returns true iff CI has a library vector variant usable at VF.
static constexpr uint32_t MinItersBypassWeights[]
static cl::opt< unsigned > ForceTargetNumScalarRegs("force-target-num-scalar-regs", cl::init(0), cl::Hidden, cl::desc("A flag that overrides the target's number of scalar registers."))
static SmallVector< VPInstruction * > preparePlanForMainVectorLoop(VPlan &MainPlan, VPlan &EpiPlan)
Prepare MainPlan for vectorizing the main vector loop during epilogue vectorization.
static cl::opt< unsigned > SmallLoopCost("small-loop-cost", cl::init(20), cl::Hidden, cl::desc("The cost of a loop that is considered 'small' by the interleaver."))
static cl::opt< bool > ForcePartialAliasingVectorization("force-partial-aliasing-vectorization", cl::init(false), cl::Hidden, cl::desc("Replace pointer diff checks with alias masks."))
static Function * getVectorLibraryVariantFor(const CallInst &CI, ElementCount VF, bool MaskRequired, const TargetLibraryInfo *TLI)
Returns the vector library variant function of CI usable at VF, respecting MaskRequired,...
static cl::opt< unsigned > ForceTargetNumVectorRegs("force-target-num-vector-regs", cl::init(0), cl::Hidden, cl::desc("A flag that overrides the target's number of vector registers."))
static bool isExplicitVecOuterLoop(Loop *OuterLp, OptimizationRemarkEmitter *ORE)
static cl::opt< bool > EnableIndVarRegisterHeur("enable-ind-var-reg-heur", cl::init(true), cl::Hidden, cl::desc("Count the induction variable only once when interleaving"))
static bool hasForcedEpilogueVF()
static cl::opt< TailFoldingStyle > ForceTailFoldingStyle("force-tail-folding-style", cl::desc("Force the tail folding style"), cl::init(TailFoldingStyle::None), cl::values(clEnumValN(TailFoldingStyle::None, "none", "Disable tail folding"), clEnumValN(TailFoldingStyle::Data, "data", "Create lane mask for data only, using active.lane.mask intrinsic"), clEnumValN(TailFoldingStyle::DataWithoutLaneMask, "data-without-lane-mask", "Create lane mask with compare/stepvector"), clEnumValN(TailFoldingStyle::DataAndControlFlow, "data-and-control", "Create lane mask using active.lane.mask intrinsic, and use " "it for both data and control flow"), clEnumValN(TailFoldingStyle::DataWithEVL, "data-with-evl", "Use predicated EVL instructions for tail folding. If EVL " "is unsupported, fallback to data-without-lane-mask.")))
static cl::opt< bool > EnableVPlanNativePath("enable-vplan-native-path", cl::Hidden, cl::desc("Enable VPlan-native vectorization path with " "support for outer loop vectorization."))
static void printOptimizedVPlan(VPlan &)
static cl::opt< bool > EnableEpilogueVectorization("enable-epilogue-vectorization", cl::init(true), cl::Hidden, cl::desc("Enable vectorization of epilogue loops."))
static cl::opt< bool > PreferPredicatedReductionSelect("prefer-predicated-reduction-select", cl::init(false), cl::Hidden, cl::desc("Prefer predicating a reduction operation over an after loop select."))
static const SCEV * getAddressAccessSCEV(Value *Ptr, PredicatedScalarEvolution &PSE, const Loop *TheLoop)
Gets the address access SCEV for Ptr, if it should be used for cost modeling according to isAddressSC...
static cl::opt< bool > EnableLoadStoreRuntimeInterleave("enable-loadstore-runtime-interleave", cl::init(true), cl::Hidden, cl::desc("Enable runtime interleaving until load/store ports are saturated"))
static cl::opt< bool > LoopVectorizeWithBlockFrequency("loop-vectorize-with-block-frequency", cl::init(true), cl::Hidden, cl::desc("Enable the use of the block frequency analysis to access PGO " "heuristics minimizing code growth in cold regions and being more " "aggressive in hot regions."))
static EpilogueLowering getEpilogueTailLowering(const LoopVectorizationCostModel &MainCM, const Loop *L, OptimizationRemarkEmitter *ORE, LoopVectorizationLegality &LVL, const LoopVectorizeHints &Hints, TargetTransformInfo *TTI)
Determine how to lower the epilogue for the vector epilogue loop.
static bool useActiveLaneMask(TailFoldingStyle Style)
static bool hasReplicatorRegion(VPlan &Plan)
static std::optional< ElementCount > getSmallBestKnownTC(PredicatedScalarEvolution &PSE, Loop *L, bool CanUseConstantMax=true, bool CanExcludeZeroTrips=false, bool ComputeUpperBoundOnly=false)
Returns "best known" trip count, which is either a valid positive trip count or std::nullopt when an ...
static bool isIndvarOverflowCheckKnownFalse(const LoopVectorizationCostModel *Cost, ElementCount VF, std::optional< unsigned > UF=std::nullopt)
For the given VF and UF and maximum trip count computed for the loop, return whether the induction va...
static void addFullyUnrolledInstructionsToIgnore(Loop *L, const LoopVectorizationLegality::InductionList &IL, SmallPtrSetImpl< Instruction * > &InstsToIgnore)
Knowing that loop L executes a single vector iteration, add instructions that will get simplified and...
static bool hasFindLastReductionPhi(VPlan &Plan)
Returns true if the VPlan contains a VPReductionPHIRecipe with FindLast recurrence kind.
static cl::opt< bool > EnableInterleavedMemAccesses("enable-interleaved-mem-accesses", cl::init(false), cl::Hidden, cl::desc("Enable vectorization on interleaved memory accesses in a loop"))
static cl::opt< unsigned > VectorizeSCEVCheckThreshold("vectorize-scev-check-threshold", cl::init(16), cl::Hidden, cl::desc("The maximum number of SCEV checks allowed."))
static cl::opt< bool > EnableMaskedInterleavedMemAccesses("enable-masked-interleaved-mem-accesses", cl::init(false), cl::Hidden, cl::desc("Enable vectorization on masked interleaved memory accesses in a loop"))
An interleave-group may need masking if it resides in a block that needs predication,...
static cl::opt< bool > ForceOrderedReductions("force-ordered-reductions", cl::init(false), cl::Hidden, cl::desc("Enable the vectorisation of loops with in-order (strict) " "FP reductions"))
static cl::opt< bool > EnableEarlyExitVectorizationWithSideEffects("enable-early-exit-vectorization-with-side-effects", cl::init(false), cl::Hidden, cl::desc("Enable vectorization of early exit loops with uncountable exits " "and side effects"))
static bool verifyExecutionFrequenciesMatchBFI(VPlan &Plan, Loop *OrigLoop, LoopInfo *LI, LoopVectorizationCostModel &CM)
Cross-check the execution frequencies recorded in Plan against BlockFrequencyInfo for the blocks of O...
static cl::opt< TailFoldingPolicyTy > TailFoldingPolicy("tail-folding-policy", cl::init(TailFoldingPolicyTy::None), cl::Hidden, cl::desc("Tail-folding preferences over creating an epilogue loop."), cl::values(clEnumValN(TailFoldingPolicyTy::None, "dont-fold-tail", "Don't tail-fold loops."), clEnumValN(TailFoldingPolicyTy::PreferFoldTail, "prefer-fold-tail", "prefer tail-folding, otherwise create an epilogue when " "appropriate."), clEnumValN(TailFoldingPolicyTy::MustFoldTail, "must-fold-tail", "always tail-fold, don't attempt vectorization if " "tail-folding fails.")))
static bool isOutsideLoopWorkProfitable(GeneratedRTChecks &Checks, VectorizationFactor &VF, Loop *L, PredicatedScalarEvolution &PSE, VPCostContext &CostCtx, VPlan &Plan, EpilogueLowering SEL, std::optional< unsigned > VScale)
This function determines whether or not it's still profitable to vectorize the loop given the extra w...
static InstructionCost calculateEarlyExitCost(VPCostContext &CostCtx, VPlan &Plan, ElementCount VF)
For loops with uncountable early exits, find the cost of doing work when exiting the loop early,...
static cl::opt< unsigned > ForceTargetMaxVectorInterleaveFactor("force-target-max-vector-interleave", cl::init(0), cl::Hidden, cl::desc("A flag that overrides the target's max interleave factor for " "vectorized loops."))
static bool useMaskedInterleavedAccesses(const TargetTransformInfo &TTI)
static EpilogueLowering getEpilogueLowering(Function *F, Loop *L, LoopVectorizeHints &Hints, bool OptForSize, TargetTransformInfo *TTI, TargetLibraryInfo *TLI, LoopVectorizationLegality &LVL, InterleavedAccessInfo *IAI)
static cl::opt< unsigned > MaxNestedScalarReductionIC("max-nested-scalar-reduction-interleave", cl::init(2), cl::Hidden, cl::desc("The maximum interleave count to use when interleaving a scalar " "reduction in a nested loop."))
static cl::opt< unsigned > ForceTargetMaxScalarInterleaveFactor("force-target-max-scalar-interleave", cl::init(0), cl::Hidden, cl::desc("A flag that overrides the target's max interleave factor for " "scalar loops."))
static void checkMixedPrecision(Loop *L, OptimizationRemarkEmitter *ORE)
static cl::opt< ElementCount > EpilogueVectorizationForceVF("epilogue-vectorization-force-VF", cl::init(ElementCount::getFixed(1)), cl::Hidden, cl::desc("When epilogue vectorization is enabled, and a value greater than " "1 is specified, forces the given VF for all applicable epilogue " "loops. Note: This allows all scalable VFs >= vscale x 1."))
static void connectEpilogueVectorLoop(VPlan &EpiPlan, DominatorTree *DT, VPIRBasicBlock *VecEpilogueIterCheckVPBB, ArrayRef< VPInstruction * > ResumeValues)
Connect the epilogue vector loop generated for EpiPlan to the main vector loop, after both plans have...
static bool willGenerateVectors(VPlan &Plan, ElementCount VF, const TargetTransformInfo &TTI)
Check if any recipe of Plan will generate a vector value, which will be assigned a vector register.
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
This file implements a map that provides insertion order iteration.
This file contains the declarations for metadata subclasses.
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t IntrinsicInst * II
#define P(N)
This file contains the declarations for profiling metadata utility functions.
const SmallVectorImpl< MachineOperand > & Cond
SI Fold Operands
Func getContext().diagnose(DiagnosticInfoUnsupported(Func
This file contains some templates that are useful if you are working with the STL at all.
#define OP(OPC)
Definition Instruction.h:46
This file defines the SmallPtrSet class.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
#define LLVM_DEBUG(...)
Definition Debug.h:119
#define DEBUG_WITH_TYPE(TYPE,...)
DEBUG_WITH_TYPE macro - This macro should be used by passes to emit debug information.
Definition Debug.h:72
This pass exposes codegen information to IR-level passes.
LocallyHashedType DenseMapInfo< LocallyHashedType >::Empty
This file implements the TypeSwitch template, which mimics a switch() statement whose cases are type ...
This file contains the declarations of different VPlan-related auxiliary helpers.
This file provides utility VPlan to VPlan transformations.
#define RUN_VPLAN_PASS(PASS,...)
#define RUN_VPLAN_PASS_NO_VERIFY(PASS,...)
This file declares the class VPlanVerifier, which contains utility functions to check the consistency...
This file contains the declarations of the Vectorization Plan base classes:
Value * RHS
Value * LHS
static const uint32_t IV[8]
Definition blake3_impl.h:83
A manager for alias analyses.
static constexpr roundingMode rmTowardZero
Definition APFloat.h:365
static const fltSemantics & IEEEdouble()
Definition APFloat.h:305
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1560
unsigned getActiveBits() const
Compute the number of active bits in the value.
Definition APInt.h:1532
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
Definition APInt.h:376
bool ult(const APInt &RHS) const
Unsigned less than comparison.
Definition APInt.h:1115
void clearAnalysis(IRUnitT &IR)
Directly clear a cached analysis for an IR unit.
PassT::Result * getCachedResult(IRUnitT &IR) const
Get the cached result of an analysis pass for a given IR unit.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
const T & front() const
Get the first element.
Definition ArrayRef.h:144
size_t size() const
Get the array size.
Definition ArrayRef.h:141
ArrayRef< T > take_back(size_t N=1) const
Return a copy of *this with only the last N elements.
Definition ArrayRef.h:225
A function analysis which provides an AssumptionCache.
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
iterator_range< const_phi_iterator > phis() const
Returns a range that iterates over the phis in the basic block.
Definition BasicBlock.h:515
const Function * getParent() const
Return the enclosing method, or null if none.
Definition BasicBlock.h:213
LLVM_ABI InstListType::const_iterator getFirstNonPHIIt() const
Returns an iterator to the first instruction in this block that is not a PHINode instruction.
LLVM_ABI const BasicBlock * getSinglePredecessor() const
Return the predecessor of this block if it has a single predecessor block.
LLVM_ABI const BasicBlock * getSingleSuccessor() const
Return the successor of this block if it has a single successor.
LLVM_ABI LLVMContext & getContext() const
Get the context in which this basic block lives.
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
Definition BasicBlock.h:237
Analysis pass which computes BlockFrequencyInfo.
BlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate IR basic block frequen...
LLVM_ABI BlockFrequency getBlockFreq(const BasicBlock *BB) const
getblockFreq - Return block frequency.
uint64_t getFrequency() const
Returns the frequency as a fixpoint number scaled by the entry frequency.
Analysis pass which computes BranchProbabilityInfo.
Analysis providing branch probability information.
static LLVM_ABI BranchProbability getBranchProbability(uint64_t Numerator, uint64_t Denominator)
static uint32_t getDenominator()
uint32_t getNumerator() const
Represents analyses that only rely on functions' control flow.
Definition Analysis.h:73
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
bool isNoBuiltin() const
Return true if the call should not be treated as a call to a builtin.
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
iterator_range< User::op_iterator > args()
Iteration adapter for range-for loops.
This class represents a function call, abstracting a target machine's calling convention.
static Type * makeCmpResultType(Type *opnd_type)
Create a result type for fcmp/icmp.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ ICMP_NE
not equal
Definition InstrTypes.h:762
Conditional Branch instruction.
BasicBlock * getSuccessor(unsigned i) const
This is the shared class of boolean and integer constants.
Definition Constants.h:87
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
This class represents a range of values.
LLVM_ABI APInt getUnsignedMax() const
Return the largest unsigned value contained in the ConstantRange.
Analysis pass which computes a CycleInfo.
A debug info location.
Definition DebugLoc.h:126
static DebugLoc getTemporary()
Definition DebugLoc.h:152
static DebugLoc getUnknown()
Definition DebugLoc.h:153
An analysis that produces DemandedBits for a function.
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
Definition DenseMap.h:773
iterator find(const_arg_type_t< KeyT > Val)
Definition DenseMap.h:782
void insert_range(Range &&R)
Inserts range of 'std::pair<KeyT, ValueT>' values into the map.
Definition DenseMap.h:910
ValueT & at(const_arg_type_t< KeyT > Val)
Return the entry for the specified key, or abort if no such entry exists.
Definition DenseMap.h:827
ValueT lookup_or(const_arg_type_t< KeyT > Val, U &&Default) const
Definition DenseMap.h:819
iterator end()
Definition DenseMap.h:702
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
Definition DenseMap.h:809
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
Definition DenseMap.h:872
Implements a dense probed hash-table based set.
Definition DenseSet.h:281
Analysis pass which computes a DominatorTree.
Definition Dominators.h:241
void changeImmediateDominator(DomTreeNodeBase< NodeT > *N, DomTreeNodeBase< NodeT > *NewIDom)
changeImmediateDominator - This method is used to update the dominator tree information when a node's...
void eraseNode(NodeT *BB)
eraseNode - Removes a node from the dominator tree.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
Definition Dominators.h:122
constexpr bool isVector() const
One or more elements.
Definition TypeSize.h:320
static constexpr ElementCount getScalable(ScalarTy MinVal)
Definition TypeSize.h:308
static constexpr ElementCount getFixed(ScalarTy MinVal)
Definition TypeSize.h:305
static constexpr ElementCount get(ScalarTy MinVal, bool Scalable)
Definition TypeSize.h:311
constexpr bool isScalar() const
Exactly one element.
Definition TypeSize.h:316
A specialized derived class of inner loop vectorizer that performs vectorization of epilogue loops in...
BasicBlock * createVectorizedLoopSkeleton() final
Implements the interface for creating a vectorized skeleton using the epilogue loop strategy (i....
EpilogueVectorizerEpilogueLoop(Loop *OrigLoop, PredicatedScalarEvolution &PSE, LoopInfo *LI, DominatorTree *DT, const TargetTransformInfo *TTI, AssumptionCache *AC, ElementCount VecWidth, unsigned UnrollFactor, GeneratedRTChecks &Checks, VPlan &Plan, VPlan &MainPlan)
Tagged union holding either a T or a Error.
Definition Error.h:485
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
Class to represent function types.
param_iterator param_begin() const
param_iterator param_end() const
FunctionType * getFunctionType() const
Returns the FunctionType for me.
Definition Function.h:212
void applyUpdates(ArrayRef< UpdateT > Updates)
Submit updates to all available trees.
Common base class shared among various IRBuilders.
Definition IRBuilder.h:114
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
Definition IRBuilder.h:2901
A struct for saving information about induction variables.
ArrayRef< Instruction * > getCastInsts() const
Returns an ArrayRef to the type cast instructions in the induction update chain, that are redundant w...
@ IK_PtrInduction
Pointer induction var. Step = C.
InnerLoopVectorizer vectorizes loops which contain only one basic block to a specified vectorization ...
const TargetTransformInfo * TTI
Target Transform Info.
friend class LoopVectorizationPlanner
PredicatedScalarEvolution & PSE
A wrapper around ScalarEvolution used to add runtime SCEV checks.
LoopInfo * LI
Loop Info.
DominatorTree * DT
Dominator Tree.
InnerLoopVectorizer(Loop *OrigLoop, PredicatedScalarEvolution &PSE, LoopInfo *LI, DominatorTree *DT, const TargetTransformInfo *TTI, AssumptionCache *AC, ElementCount VecWidth, unsigned UnrollFactor, GeneratedRTChecks &RTChecks, VPlan &Plan)
void fixVectorizedLoop(VPTransformState &State)
Fix the vectorized code, taking care of header phi's, and more.
virtual BasicBlock * createVectorizedLoopSkeleton()
Creates a basic block for the scalar preheader.
AssumptionCache * AC
Assumption Cache.
IRBuilder Builder
The builder that we use.
VPBasicBlock * VectorPHVPBB
The vector preheader block of Plan, used as target for check blocks introduced during skeleton creati...
unsigned UF
The vectorization unroll factor to use.
GeneratedRTChecks & RTChecks
Structure to hold information about generated runtime checks, responsible for cleaning the checks,...
virtual ~InnerLoopVectorizer()=default
ElementCount VF
The vectorization SIMD factor to use.
Loop * OrigLoop
The original loop.
BasicBlock * createScalarPreheader(StringRef Prefix)
Create and return a new IR basic block for the scalar preheader whose name is prefixed with Prefix.
static InstructionCost getInvalid(CostType Val=0)
static InstructionCost getMax()
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
bool isCast() const
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void moveBefore(InstListType::iterator InsertPos)
Unlink this instruction from its current basic block and insert it into the basic block that MovePos ...
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
LLVM_ABI void replaceSuccessorWith(BasicBlock *OldBB, BasicBlock *NewBB)
Replace specified successor OldBB to point at the provided block.
iterator_range< user_iterator > users()
const char * getOpcodeName() const
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
LLVM_ABI APInt getMask() const
For example, this is 0xFF for an 8 bit integer, 0xFFFF for i16, etc.
Definition Type.cpp:362
The group of interleaved loads/stores sharing the same stride and close to each other.
auto members() const
Return an iterator range over the non-null members of this group, in index order.
InstTy * getInsertPos() const
uint32_t getNumMembers() const
Drive the analysis of interleaved memory accesses in the loop.
bool requiresScalarEpilogue() const
Returns true if an interleaved group that may access memory out-of-bounds requires a scalar epilogue ...
bool hasGroups() const
Returns true if we have any interleave groups.
LLVM_ABI void analyzeInterleaving(bool EnableMaskedInterleavedGroup)
Analyze the interleaved accesses and collect them in interleave groups.
An instruction for reading from memory.
Type * getPointerOperandType() const
This analysis provides dependence information for the memory accesses of a loop.
const RuntimePointerChecking * getRuntimePointerChecking() const
unsigned getNumRuntimePointerChecks() const
Number of memchecks required to prove independence of otherwise may-alias pointers.
const SymbolicStrideMap & getSymbolicStrides() const
If an access has a symbolic strides, this maps the pointer value to the stride symbol.
Analysis pass that exposes the LoopInfo for a function.
Definition LoopInfo.h:594
BlockT * getLoopLatch() const
If there is a single latch block for this loop, return it.
bool isInnermost() const
Return true if the loop does not contain any (natural) loops.
unsigned getNumBlocks() const
Get the number of blocks in this loop in constant time.
BlockT * getHeader() const
iterator_range< block_iterator > blocks() const
BlockT * getExitingBlock() const
If getExitingBlocks would return exactly one block, return that block.
Store the result of a depth first search within basic blocks contained by a single loop.
RPOIterator beginRPO() const
Reverse iterate over the cached postorder blocks.
LLVM_ABI void perform(const LoopInfo *LI)
Traverse the loop blocks and store the DFS result.
RPOIterator endRPO() const
Wrapper class to LoopBlocksDFS that provides a standard begin()/end() interface for the DFS reverse p...
void perform(const LoopInfo *LI)
Traverse the loop blocks and store the DFS result.
void removeBlock(BlockT *BB)
This method completely removes BB from all data structures, including all of the Loop objects it is n...
LoopVectorizationCostModel - estimates the expected speedups due to vectorization.
bool isPredicatedInst(Instruction *I) const
Returns true if I is an instruction that needs to be predicated at runtime.
void collectValuesToIgnore()
Collect values we want to ignore in the cost model.
BlockFrequencyInfo * BFI
The BlockFrequencyInfo returned from GetBFI.
BlockFrequencyInfo & getBFI()
Returns the BlockFrequencyInfo for the function if cached, otherwise fetches it via GetBFI.
bool isForcedScalar(Instruction *I, ElementCount VF) const
Returns true if I has been forced to be scalarized at VF.
bool isUniformAfterVectorization(Instruction *I, ElementCount VF) const
Returns true if I is known to be uniform after vectorization.
void collectNonVectorizedAndSetWideningDecisions(ElementCount VF)
Collect values that will not be widened, including Uniforms, Scalars, and Instructions to Scalarize f...
static constexpr StringLiteral getInstWideningStr(InstWidening W)
bool isMaskRequired(Instruction *I) const
Wrapper function for LoopVectorizationLegality::isMaskRequired, that passes the Instruction I and if ...
PredicatedScalarEvolution & PSE
Predicated scalar evolution analysis.
const TargetTransformInfo & TTI
Vector target information.
LoopVectorizationLegality * Legal
Vectorization legality.
uint64_t getPredBlockCostDivisor(TargetTransformInfo::TargetCostKind CostKind, const BasicBlock *BB)
A helper function that returns how much we should divide the cost of a predicated block by.
std::optional< InstWidening > memoryInstructionCanBeWidened(Instruction *I, ElementCount VF)
If I is a memory instruction with a consecutive pointer that can be widened, returns the widening kin...
InstructionCost getInstructionCost(Instruction *I, ElementCount VF)
Returns the execution time cost of an instruction for a given vector width.
bool interleavedAccessCanBeWidened(Instruction *I, ElementCount VF) const
Returns true if I is a memory instruction in an interleaved-group of memory accesses that can be vect...
const TargetLibraryInfo * TLI
Target Library Info.
const InterleaveGroup< Instruction > * getInterleavedAccessGroup(Instruction *Instr) const
Get the interleaved access group that Instr belongs to.
InstructionCost getVectorIntrinsicCost(CallInst *CI, ElementCount VF) const
Estimate cost of an intrinsic call instruction CI if it were vectorized with factor VF.
bool maskPartialAliasing() const
Returns true if all loop blocks should have partial aliases masked.
bool isScalarAfterVectorization(Instruction *I, ElementCount VF) const
Returns true if I is known to be scalar after vectorization.
bool isOptimizableIVTruncate(Instruction *I, ElementCount VF)
Return True if instruction I is an optimizable truncate whose operand is an induction variable.
bool isLegalGatherOrScatter(Instruction *I, ElementCount VF) const
Returns true if the target machine supports gather or scatter for I's data type and alignment.
FixedScalableVFPair computeMaxVF(ElementCount UserVF, unsigned UserIC)
Loop * TheLoop
The loop that we evaluate.
InterleavedAccessInfo & InterleaveInfo
The interleave access information contains groups of interleaved accesses with the same stride and cl...
SmallPtrSet< const Value *, 16 > ValuesToIgnore
Values to ignore in the cost model.
LoopVectorizationCostModel(EpilogueLowering SEL, Loop *L, PredicatedScalarEvolution &PSE, LoopInfo *LI, LoopVectorizationLegality *Legal, const TargetTransformInfo &TTI, const TargetLibraryInfo *TLI, AssumptionCache *AC, OptimizationRemarkEmitter *ORE, std::function< BlockFrequencyInfo &()> GetBFI, const Function *F, InterleavedAccessInfo &IAI, VFSelectionContext &Config)
void invalidateCostModelingDecisions()
Invalidates decisions already taken by the cost model.
bool isAccessInterleaved(Instruction *Instr) const
Check if Instr belongs to any interleaved access group.
void setTailFoldingStyle(bool IsScalableVF, unsigned UserIC)
Selects and saves TailFoldingStyle.
OptimizationRemarkEmitter * ORE
Interface to emit optimization remarks.
LoopInfo * LI
Loop Info analysis.
bool requiresScalarEpilogue(bool IsVectorizing) const
Returns true if we're required to use a scalar epilogue for at least the final iteration of the origi...
SmallPtrSet< const Value *, 16 > VecValuesToIgnore
Values to ignore in the cost model when VF > 1.
bool useEmulatedMaskMemRefHack(Instruction *I, ElementCount VF) const
Returns true if an artificially high cost for emulated masked memrefs should be used.
bool isLegalMaskedLoadOrStore(Instruction *I, ElementCount VF) const
Returns true if the target machine supports masked loads or stores for I's data type and alignment.
bool isProfitableToScalarize(Instruction *I, ElementCount VF) const
void setWideningDecision(const InterleaveGroup< Instruction > *Grp, ElementCount VF, InstWidening W, InstructionCost Cost)
Save vectorization decision W and Cost taken by the cost model for interleaving group Grp and vector ...
bool isEpilogueAllowed() const
Returns true if an epilogue is allowed (e.g., not prevented by optsize or a loop hint annotation).
bool canTruncateToMinimalBitwidth(Instruction *I, ElementCount VF) const
bool shouldConsiderInvariant(Value *Op)
Returns true if Op should be considered invariant and if it is trivially hoistable.
bool foldTailByMasking() const
Returns true if all loop blocks should be masked to fold tail loop.
bool foldTailWithEVL() const
Returns true if VP intrinsics with explicit vector length support should be generated in the tail fol...
bool blockNeedsPredicationForAnyReason(BasicBlock *BB) const
Returns true if the instructions in this block requires predication for any reason,...
AssumptionCache * AC
Assumption cache.
void setWideningDecision(Instruction *I, ElementCount VF, InstWidening W, InstructionCost Cost)
Save vectorization decision W and Cost taken by the cost model for instruction I and vector width VF.
InstWidening
Decision that was taken during cost calculation for memory instruction.
@ CM_InvalidatedDecision
A widening decision that has been invalidated after replacing the corresponding recipe during VPlan t...
bool usePredicatedReductionSelect(RecurKind RecurrenceKind) const
Returns true if the predicated reduction select should be used to set the incoming value for the redu...
std::pair< InstructionCost, InstructionCost > getDivRemSpeculationCost(Instruction *I, ElementCount VF)
Return the costs for our two available strategies for lowering a div/rem operation which requires spe...
InstructionCost getVectorCallCost(CallInst *CI, ElementCount VF) const
Estimate cost of a call instruction CI if it were vectorized with factor VF.
bool isScalarWithPredication(Instruction *I, ElementCount VF)
Returns true if I is an instruction which requires predication and for which our chosen predication s...
std::function< BlockFrequencyInfo &()> GetBFI
A function to lazily fetch BlockFrequencyInfo.
InstructionCost expectedCost(ElementCount VF)
Returns the expected execution cost.
void setCostBasedWideningDecision(ElementCount VF)
Memory access instruction may be vectorized in more than one way.
bool isDivRemScalarWithPredication(InstructionCost ScalarCost, InstructionCost MaskedCost) const
Given costs for both strategies, return true if the scalar predication lowering should be used for di...
InstWidening getWideningDecision(Instruction *I, ElementCount VF) const
Return the cost model decision for the given instruction I and vector width VF.
InstructionCost getWideningCost(Instruction *I, ElementCount VF)
Return the vectorization cost for the given instruction I and vector width VF.
TailFoldingStyle getTailFoldingStyle() const
Returns the TailFoldingStyle that is best for the current loop.
void collectInstsToScalarize(ElementCount VF)
Collects the instructions to scalarize for each predicated instruction in the loop.
LoopVectorizationLegality checks if it is legal to vectorize a loop, and to what vectorization factor...
MapVector< PHINode *, InductionDescriptor > InductionList
InductionList saves induction variables and maps them to the induction descriptor.
RecurrenceSet & getFixedOrderRecurrences()
Return the fixed-order recurrences found in the loop.
LLVM_ABI bool canVectorize(bool UseVPlanNativePath)
Returns true if it is legal to vectorize this loop.
bool hasUncountableExitWithSideEffects() const
Returns true if this is an early exit loop with state-changing or potentially-faulting operations and...
LLVM_ABI bool canVectorizeFPMath(bool EnableStrictReductions)
Returns true if it is legal to vectorize the FP math operations in this loop.
const SmallVector< BasicBlock *, 4 > & getCountableExitingBlocks() const
Returns all exiting blocks with a countable exit, i.e.
const ReductionList & getReductionVars() const
Returns the reduction variables found in the loop.
bool hasUncountableEarlyExit() const
Returns true if the loop has uncountable early exits, i.e.
bool hasHistograms() const
Returns a list of all known histogram operations in the loop.
const LoopAccessInfo * getLAI() const
Planner drives the vectorization process after having passed Legality checks.
DenseMap< const SCEV *, Value * > executePlan(ElementCount VF, unsigned UF, VPlan &BestPlan, InnerLoopVectorizer &LB, DominatorTree *DT, EpilogueVectorizationKind EpilogueVecKind=EpilogueVectorizationKind::None)
EpilogueVectorizationKind
Generate the IR code for the vectorized loop captured in VPlan BestPlan according to the best selecte...
@ MainLoop
Vectorizing the main loop of epilogue vectorization.
void clearCostModel()
Destroy the cost model.
VPlan & getPlanFor(ElementCount VF) const
Return the VPlan for VF.
Definition VPlan.cpp:1652
void updateLoopMetadataAndProfileInfo(Loop *VectorLoop, VPBasicBlock *HeaderVPBB, const VPlan &Plan, bool VectorizingEpilogue, MDNode *OrigLoopID, std::optional< unsigned > OrigAverageTripCount, unsigned OrigLoopInvocationWeight, unsigned EstimatedVFxUF, bool DisableRuntimeUnroll, bool UnrollVectorizedLoop)
Update loop metadata and profile info for both the scalar remainder loop and VectorLoop,...
Definition VPlan.cpp:1703
LoopVectorizationCostModel & getCostModel()
Return the cost model. Must not be called after clearCostModel().
void attachRuntimeChecks(VPlan &Plan, GeneratedRTChecks &RTChecks, bool HasBranchWeights) const
Attach the runtime checks of RTChecks to Plan.
unsigned selectInterleaveCount(VPlan &Plan, ElementCount VF, InstructionCost LoopCost)
void emitInvalidCostRemarks(OptimizationRemarkEmitter *ORE)
Emit remarks for recipes with invalid costs in the available VPlans.
LoopVectorizationPlanner(Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI, const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal, std::unique_ptr< LoopVectorizationCostModel > CM, VFSelectionContext &Config, InterleavedAccessInfo &IAI, PredicatedScalarEvolution &PSE, OptimizationRemarkEmitter *ORE, std::function< const BranchProbabilityInfo &()> GetBPI)
static bool getDecisionAndClampRange(const std::function< bool(ElementCount)> &Predicate, VFRange &Range)
Test a Predicate on a Range of VF's.
Definition VPlan.cpp:1638
void printPlans(raw_ostream &O)
Definition VPlan.cpp:1807
std::unique_ptr< VPlan > selectBestEpiloguePlan(VPlan &MainPlan, ElementCount MainLoopVF, unsigned IC, bool ScalarEpilogueAllowed)
void plan(ElementCount UserVF, unsigned UserIC)
Build VPlans for the specified UserVF and UserIC if they are non-zero or all applicable candidate VFs...
void addMinimumIterationCheck(VPlan &Plan, ElementCount VF, unsigned UF, ElementCount MinProfitableTripCount) const
Create a check to Plan to see if the vector loop should be executed based on its trip count.
bool hasPlanWithVF(ElementCount VF) const
Look through the existing plans and return true if we have one with vectorization factor VF.
std::pair< VectorizationFactor, VPlan * > computeBestVF()
Compute and return the most profitable vectorization factor and the corresponding best VPlan.
This holds vectorization requirements that must be verified late in the process.
Utility class for getting and setting loop vectorizer hints in the form of loop metadata.
LLVM_ABI bool allowVectorization(Function *F, Loop *L, bool VectorizeOnlyWhenForced) const
LLVM_ABI void emitRemarkWithHints() const
Dumps all the hint information.
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
Metadata node.
Definition Metadata.h:1081
bool empty() const
Definition MapVector.h:79
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Definition MapVector.h:126
Function * getFunction(StringRef Name) const
Look up the specified function in the module symbol table.
Definition Module.cpp:235
Diagnostic information for optimization analysis remarks related to pointer aliasing.
Diagnostic information for optimization analysis remarks related to floating-point non-commutativity.
Diagnostic information for optimization analysis remarks.
The optimization diagnostic interface.
LLVM_ABI void emit(DiagnosticInfoOptimizationBase &OptDiag)
Output the remark via the diagnostic handler and to the optimization record file.
Diagnostic information for missed-optimization remarks.
Diagnostic information for applied optimization remarks.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
ScalarEvolution * getSE() const
Returns the ScalarEvolution analysis used.
LLVM_ABI const SCEVPredicate & getPredicate() const
LLVM_ABI unsigned getSmallConstantMaxTripCount()
Returns the upper bound of the loop trip count as a normal unsigned value, or 0 if the trip count is ...
LLVM_ABI const SCEV * getBackedgeTakenCount()
Get the (predicated) backedge count for the analyzed loop.
LLVM_ABI const SCEV * getSCEV(Value *V)
Returns the SCEV expression of V, in the context of the current SCEV predicate.
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
PreservedAnalyses & preserveSet()
Mark an analysis set as preserved.
Definition Analysis.h:151
PreservedAnalyses & preserve()
Mark an analysis as preserved.
Definition Analysis.h:132
An analysis pass based on the new PM to deliver ProfileSummaryInfo.
The RecurrenceDescriptor is used to identify recurrences variables in a loop.
Type * getRecurrenceType() const
Returns the type of the recurrence.
const SmallPtrSet< Instruction *, 8 > & getCastInsts() const
Returns a reference to the instructions used for type-promoting the recurrence.
static bool isFindLastRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
static bool isAnyOfRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
static LLVM_ABI bool isSubRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is for a sub operation.
bool isSigned() const
Returns true if all source operands of the recurrence are SExtInsts.
static bool isFindIVRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
static bool isMinMaxRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is any min/max kind.
Holds information about the memory runtime legality checks to verify that a group of pointers do not ...
std::optional< ArrayRef< PointerDiffInfo > > getDiffChecks() const
const SmallVectorImpl< RuntimePointerCheck > & getChecks() const
Returns the checks that generateChecks created.
This class uses information about analyze scalars to rewrite expressions in canonical form.
ScalarEvolution * getSE()
bool isInsertedInstruction(Instruction *I) const
Return true if the specified instruction was inserted by the code rewriter.
LLVM_ABI Value * expandCodeForPredicate(const SCEVPredicate *Pred, Instruction *Loc)
Generates a code sequence that evaluates this predicate.
LLVM_ABI void eraseDeadInstructions(Value *Root)
Remove inserted instructions that are dead, e.g.
virtual bool isAlwaysTrue() const =0
Returns true if the predicate is always true.
This class represents an analyzed expression in the program.
LLVM_ABI bool isZero() const
Return true if the expression is a constant zero.
Type * getType() const
Return the LLVM type of this SCEV expression.
Analysis pass that exposes the ScalarEvolution for a function.
The main scalar evolution driver.
LLVM_ABI const SCEV * getElementCount(Type *Ty, ElementCount EC, SCEVFlags Flags=SCEV::FlagNone)
LLVM_ABI const SCEV * getURemExpr(SCEVUse LHS, SCEVUse RHS)
Represents an unsigned remainder expression based on unsigned division.
LLVM_ABI const SCEV * getBackedgeTakenCount(const Loop *L, ExitCountKind Kind=Exact)
If the specified loop has a predictable backedge-taken count, return it, otherwise return a SCEVCould...
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
LLVM_ABI const SCEV * getSCEV(Value *V)
Return a SCEV expression for the full generality of the specified expression.
LLVM_ABI const SCEV * getTripCountFromExitCount(const SCEV *ExitCount)
A version of getTripCountFromExitCount below which always picks an evaluation type which can not resu...
const SCEV * getOne(Type *Ty)
Return a SCEV for the constant 1 of a specific type.
LLVM_ABI void forgetLoop(const Loop *L)
This method should be called by the client when it has changed a loop in a way that may effect Scalar...
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
LLVM_ABI SCEVUse getAddExpr(SmallVectorImpl< SCEVUse > &Ops, SCEVFlagsPair Flags={}, unsigned Depth=0)
Get a canonical add expression, or something simpler if possible.
ConstantRange getUnsignedRange(const SCEV *S)
Determine the unsigned range for a particular SCEV.
LLVM_ABI void forgetValue(Value *V)
This method should be called by the client when it has changed a value in a way that may effect its v...
LLVM_ABI void forgetBlockAndLoopDispositions(Value *V=nullptr)
Called when the client has changed the disposition of values in a loop or block.
const SCEV * getMinusOne(Type *Ty)
Return a SCEV for the constant -1 of a specific type.
LLVM_ABI void forgetLcssaPhiWithNewPredecessor(Loop *L, PHINode *V)
Forget LCSSA phi node V of loop L to which a new predecessor was added, such that it may no longer be...
LLVM_ABI SCEVUse getMulExpr(SmallVectorImpl< SCEVUse > &Ops, SCEVFlagsPair Flags={}, unsigned Depth=0)
Get a canonical multiply expression, or something simpler if possible.
LLVM_ABI unsigned getSmallConstantTripCount(const Loop *L)
Returns the exact trip count of the loop if we can compute it, and the result is a small constant.
APInt getUnsignedRangeMax(const SCEV *S)
Determine the max of the unsigned range for a particular SCEV.
LLVM_ABI bool isKnownPredicate(CmpPredicate Pred, SCEVUse LHS, SCEVUse RHS)
Test if the given expression is known to satisfy the condition described by Pred, LHS,...
LLVM_ABI const SCEV * applyLoopGuards(const SCEV *Expr, const Loop *L)
Try to apply information from loop guards for L to Expr.
This class represents the LLVM 'select' instruction.
A vector that has set insertion semantics.
Definition SetVector.h:57
size_type size() const
Determine the number of elements in the SetVector.
Definition SetVector.h:103
void insert_range(Range &&R)
Definition SetVector.h:182
size_type count(const_arg_type key) const
Count the number of elements of a given key in the SetVector.
Definition SetVector.h:268
bool contains(const_arg_type key) const
Check if the SetVector contains the given key.
Definition SetVector.h:258
bool insert(const value_type &X)
Insert a new element into the SetVector.
Definition SetVector.h:157
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
A SetVector that performs no allocations if smaller than a certain size.
Definition SetVector.h:345
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
A wrapper around a string literal that serves as a proxy for constructing global tables of StringRefs...
Definition StringRef.h:888
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Analysis pass providing the TargetTransformInfo.
Analysis pass providing the TargetLibraryInfo.
Provides information about what library functions are available for the current target.
This pass provides access to the codegen interfaces that are needed for IR-level transformations.
LLVM_ABI bool supportsEfficientVectorElementLoadStore() const
If target has efficient vector element load/store instructions, it can return true here so that inser...
LLVM_ABI bool prefersVectorizedAddressing() const
Return true if target doesn't mind addresses in vectors.
LLVM_ABI InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const
LLVM_ABI InstructionCost getOperandsScalarizationOverhead(ArrayRef< Type * > Tys, TTI::TargetCostKind CostKind, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const
Estimate the overhead of scalarizing operands with the given types.
LLVM_ABI InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, OperandValueInfo OpdInfo={OK_AnyValue, OP_None}, const Instruction *I=nullptr) const
LLVM_ABI InstructionCost getShuffleCost(ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask={}, int Index=0, VectorType *SubTp=nullptr, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const
static LLVM_ABI OperandValueInfo getOperandInfo(const Value *V)
Collect properties of V used in cost analysis, e.g. OP_PowerOf2.
LLVM_ABI InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const
TargetCostKind
The kind of cost model.
@ TCK_RecipThroughput
Reciprocal throughput.
@ TCK_CodeSize
Instruction code size.
@ TCK_SizeAndLatency
The weighted sum of size and latency.
@ TCK_Latency
The latency of instruction.
LLVM_ABI InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
LLVM_ABI InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const
llvm::VectorInstrContext VectorInstrContext
@ TCC_Free
Expected to fold away in lowering.
LLVM_ABI InstructionCost getInstructionCost(const User *U, ArrayRef< const Value * > Operands, TargetCostKind CostKind) const
Estimate the cost of a given IR user when lowered.
LLVM_ABI InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const
LLVM_ABI InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const
Estimate the overhead of scalarizing an instruction.
@ SK_Splice
Concatenates elements from the first input vector with elements of the second input vector.
@ SK_Broadcast
Broadcast element 0 to all other elements.
@ SK_Reverse
Reverse the order of the vector.
CastContextHint
Represents a hint about the context in which a cast is used.
@ Reversed
The cast is used with a reversed load/store.
@ Masked
The cast is used with a masked load/store.
@ None
The cast is not used with a load/store of any kind.
@ Normal
The cast is used with a normal load/store.
@ Interleave
The cast is used with an interleaved load/store.
@ GatherScatter
The cast is used with a gather/scatter.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
This class implements a switch-like dispatch statement for a value of 'T' using dyn_cast functionalit...
Definition TypeSwitch.h:89
TypeSwitch< T, ResultT > & Case(CallableT &&caseFn)
Add a case on the given type.
Definition TypeSwitch.h:98
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
Definition Type.cpp:272
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:296
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
static SmallVector< VFInfo, 8 > getMappings(const CallInst &CI)
Retrieve all the VFInfo instances associated to the CallInst CI.
Definition VectorUtils.h:76
Holds state needed to make cost decisions before computing costs per-VF, including the maximum VFs.
const TTI::TargetCostKind CostKind
The kind of cost that we are calculating.
bool isEpilogueVectorizationProfitable(ElementCount VF, unsigned IC) const
Returns true if epilogue vectorization is considered profitable for a main loop with vectorization fa...
std::optional< unsigned > getVScaleForTuning() const
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
Definition VPlan.h:4414
RecipeListTy::iterator iterator
Instruction iterators...
Definition VPlan.h:4441
iterator end()
Definition VPlan.h:4451
iterator begin()
Recipe iterator methods.
Definition VPlan.h:4449
iterator_range< iterator > phis()
Returns an iterator range over the PHI-like recipes in the block.
Definition VPlan.h:4502
InstructionCost cost(ElementCount VF, VPCostContext &Ctx) override
Return the cost of this VPBasicBlock.
Definition VPlan.cpp:749
iterator getFirstNonPhi()
Return the position of the first non-phi node recipe in the block.
Definition VPlan.cpp:233
const VPRecipeBase & front() const
Definition VPlan.h:4461
VPRecipeBase * getTerminator()
If the block has multiple successors, return the branch recipe terminating the block.
Definition VPlan.cpp:619
bool empty() const
Definition VPlan.h:4460
const VPBasicBlock * getExitingBasicBlock() const
Definition VPlan.cpp:203
void setName(const Twine &newName)
Definition VPlan.h:188
const VPBlocksTy & getPredecessors() const
Definition VPlan.h:230
VPlan * getPlan()
Definition VPlan.h:199
const VPBasicBlock * getEntryBasicBlock() const
Definition VPlan.cpp:188
VPBlockBase * getSingleSuccessor() const
Definition VPlan.h:235
static auto blocksAs(T &&Range)
Return an iterator range over Range with each block cast to BlockTy.
Definition VPlanUtils.h:421
static void reassociateBlocks(VPBlockBase *Old, VPBlockBase *New)
Reassociate all the blocks connected to Old so that they now point to New.
Definition VPlanUtils.h:384
static auto blocksOnly(T &&Range)
Return an iterator range over Range which only includes BlockTy blocks.
Definition VPlanUtils.h:414
static std::pair< VPBasicBlock *, VPBasicBlock * > getPlainCFGHeaderAndLatch(const VPlan &Plan)
Returns the header and latch of the outermost loop of Plan in plain CFG form (before regions are form...
static VPBuilderBase getToInsertAfter(VPRecipeBase *R)
VPPhi * createScalarPhi(ArrayRef< VPValue * > IncomingValues, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", std::optional< VPIRFlags > Flags=std::nullopt, Type *ResultTy=nullptr)
Create a phi with IncomingValues, using the default flags for the result type, unless Flags is set.
T * insert(T *R)
Insert R at the current insertion point. Returns R unchanged.
VPInstruction * createAdd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", VPRecipeWithIRFlags::WrapFlagsTy WrapFlags={false, false})
VPInstruction * createSelect(VPValue *Cond, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", std::optional< VPIRFlags > Flags=std::nullopt)
Create a select of TrueVal and FalseVal based on Cond, using the default flags for the result type,...
static VPSingleDefRecipe * createSingleScalarOp(unsigned Opcode, ArrayRef< VPValue * > Operands, VPValue *Mask, const VPIRFlags &Flags, const VPIRMetadata &Metadata, DebugLoc DL, Type *ResultTy, Instruction *UV)
VPInstruction * createNaryOp(unsigned Opcode, ArrayRef< VPValue * > Operands, Instruction *Inst=nullptr, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
Create an N-ary operation with Opcode, Operands and set Inst as its underlying Instruction.
VPInstruction * createICmp(CmpInst::Predicate Pred, VPValue *A, VPValue *B, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
Create a new ICmp VPInstruction with predicate Pred and operands A and B.
unsigned getNumDefinedValues() const
Returns the number of values defined by the VPDef.
Definition VPlanValue.h:576
VPValue * getVPSingleValue()
Returns the only VPValue defined by the VPDef.
Definition VPlanValue.h:549
A pure virtual base class for all recipes modeling header phis, including phis for first order recurr...
Definition VPlan.h:2448
virtual VPValue * getBackedgeValue()
Returns the incoming value from the loop backedge.
Definition VPlan.h:2495
void setBackedgeValue(VPValue *V)
Update the incoming value from the loop backedge.
Definition VPlan.h:2498
VPValue * getStartValue()
Returns the start value of the phi, if one is set.
Definition VPlan.h:2484
A recipe representing a sequence of load -> update -> store as part of a histogram operation.
Definition VPlan.h:2170
A special type of VPBasicBlock that wraps an existing IR basic block.
Definition VPlan.h:4567
BasicBlock * getIRBasicBlock() const
Definition VPlan.h:4591
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
This is a concrete Recipe that models a single VPlan-level instruction.
Definition VPlan.h:1300
iterator_range< operand_iterator > operandsWithoutMask()
Returns an iterator range over the operands excluding the mask operand if present.
Definition VPlan.h:1563
@ ResumeForEpilogue
Explicit user for the resume phi of the canonical induction in the main VPlan, used by the epilogue v...
Definition VPlan.h:1403
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
Definition VPlan.h:1396
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
Definition VPlan.h:1354
unsigned getOpcode() const
Definition VPlan.h:1485
void setName(StringRef NewName)
Set the symbolic name for the VPInstruction.
Definition VPlan.h:1595
VPValue * getMask() const
Returns the mask for the VPInstruction.
Definition VPlan.h:1557
VPInterleaveRecipe is a recipe for transforming an interleave group of load or stores into one wide l...
Definition VPlan.h:3137
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
Definition VPlan.h:403
VPBasicBlock * getParent()
Definition VPlan.h:475
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
Definition VPlan.h:553
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
VPRecipeBase * tryToCreateWidenNonPhiRecipe(VPSingleDefRecipe *R, VFRange &Range)
Create and return a widened recipe for a non-phi recipe R if one can be created within the given VF R...
VPHistogramRecipe * widenIfHistogram(VPInstruction *VPI)
If VPI represents a histogram operation (as determined by LoopVectorizationLegality) make that safe f...
bool prefersVectorizedAddressing() const
Returns true if the target prefers vectorized addressing.
VPRecipeBase * tryToWidenMemory(VPInstruction *VPI, VFRange &Range)
Check if the load or store instruction VPI should widened for Range.Start and potentially masked.
bool replaceWithFinalIfReductionStore(VPInstruction *VPI, VPBuilder &FinalRedStoresBuilder)
If VPI is a store of a reduction into an invariant address, delete it.
VPSingleDefRecipe * handleReplication(VPInstruction *VPI, VFRange &Range)
Build a replicating or single-scalar recipe for VPI.
bool isPredicatedInst(Instruction *I) const
Returns true if I needs to be predicated (i.e.
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
Definition VPlanValue.h:351
A recipe for handling reduction phis.
Definition VPlan.h:2863
bool isOrdered() const
Returns true, if the phi is part of an ordered reduction.
Definition VPlan.h:2923
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
Definition VPlan.h:2907
bool isInLoop() const
Returns true if the phi is part of an in-loop reduction.
Definition VPlan.h:2926
VPReductionPHIRecipe * cloneWithOperands(VPValue *Start, VPValue *BackedgeValue)
Definition VPlan.h:2889
RecurKind getRecurrenceKind() const
Returns the recurrence kind of the reduction.
Definition VPlan.h:2920
A recipe to represent inloop, ordered or partial reduction operations.
Definition VPlan.h:3230
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
Definition VPlan.h:4639
const VPBlockBase * getEntry() const
Definition VPlan.h:4683
void clearCanonicalIVNUW(VPInstruction *Increment)
Unsets NUW for the canonical IV increment Increment, for loop regions.
Definition VPlan.h:4806
VPRegionValue * getCanonicalIV()
Return the canonical induction variable of the region, null for replicating regions.
Definition VPlan.h:4759
VPReplicateRecipe replicates a given instruction producing multiple scalar copies of the original sca...
Definition VPlan.h:3397
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Definition VPlan.h:611
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
Definition VPlan.h:681
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
Definition VPlanValue.h:398
operand_range operands()
Definition VPlanValue.h:471
void setOperand(unsigned I, VPValue *New)
Definition VPlanValue.h:444
VPValue * getOperand(unsigned N) const
Definition VPlanValue.h:439
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Definition VPlanValue.h:50
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Definition VPlan.cpp:147
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
Definition VPlan.cpp:141
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
Definition VPlan.cpp:128
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
Definition VPlanValue.h:75
void replaceAllUsesWith(VPValue *New)
Definition VPlan.cpp:1464
void replaceUsesWithIf(VPValue *New, llvm::function_ref< bool(VPUser &U, unsigned Idx)> ShouldReplace)
Go through the uses list for this VPValue and make each use point to New if the callback ShouldReplac...
Definition VPlan.cpp:1470
VPWidenCastRecipe is a recipe to create vector cast instructions.
Definition VPlan.h:1885
A recipe for handling GEP instructions.
Definition VPlan.h:2218
VPWidenRecipe is a recipe for producing a widened instruction using the opcode and operands of the re...
Definition VPlan.h:1818
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
Definition VPlan.h:4826
bool hasVF(ElementCount VF) const
Definition VPlan.h:5062
ElementCount getSingleVF() const
Returns the single VF of the plan, asserting that the plan has exactly one VF.
Definition VPlan.h:5075
VPBasicBlock * getEntry()
Definition VPlan.h:4922
VPValue * getTripCount() const
The trip count of the original loop.
Definition VPlan.h:4994
VPSymbolicValue & getVFxUF()
Returns VF * UF of the vector loop region.
Definition VPlan.h:5034
bool hasUF(unsigned UF) const
Definition VPlan.h:5087
ArrayRef< VPIRBasicBlock * > getExitBlocks() const
Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of the original scalar loop.
Definition VPlan.h:4988
VPIRValue * getOrAddLiveIn(Value *V)
Gets the live-in VPIRValue for V or adds a new live-in (if none exists yet) for V.
Definition VPlan.h:5112
VPIRValue * getZero(Type *Ty)
Return a VPIRValue wrapping the null value of type Ty.
Definition VPlan.h:5138
LLVM_ABI_FOR_TEST VPRegionBlock * getVectorLoopRegion()
Returns the VPRegionBlock of the vector loop.
Definition VPlan.cpp:1042
bool hasEarlyExit() const
Returns true if the VPlan is based on a loop with an early exit.
Definition VPlan.h:5245
InstructionCost cost(ElementCount VF, VPCostContext &Ctx)
Return the cost of this plan.
Definition VPlan.cpp:1024
LLVM_ABI_FOR_TEST bool isOuterLoop() const
Returns true if this VPlan is for an outer loop, i.e., its vector loop region contains a nested loop ...
Definition VPlan.cpp:1066
void resetTripCount(VPValue *NewTripCount)
Resets the trip count for the VPlan.
Definition VPlan.h:5008
VPBasicBlock * getMiddleBlock()
Returns the 'middle' block of the plan, that is the block that selects whether to execute the scalar ...
Definition VPlan.h:4964
VPBasicBlock * getVectorPreheader() const
Returns the preheader of the vector loop region, if one exists, or null otherwise.
Definition VPlan.h:4927
bool requiresScalarEpilogue() const
Returns true if the plan requires a scalar epilogue after the vector loop.
Definition VPlan.h:4950
VPSymbolicValue & getUF()
Returns the UF of the vector loop region.
Definition VPlan.h:5031
bool hasScalarVFOnly() const
Definition VPlan.h:5080
VPBasicBlock * getScalarPreheader() const
Return the VPBasicBlock for the preheader of the scalar loop.
Definition VPlan.h:4978
void execute(VPTransformState *State)
Generate the IR code for this VPlan.
Definition VPlan.cpp:917
bool hasTailFolded() const
Returns true if the vector loop region is tail-folded.
Definition VPlan.h:4943
VPIRBasicBlock * getScalarHeader() const
Return the VPIRBasicBlock wrapping the header of the scalar loop.
Definition VPlan.h:4984
LLVM_ABI_FOR_TEST VPlan * duplicate()
Clone the current VPlan, update all VPValues of the new VPlan and cloned recipes to refer to the clon...
Definition VPlan.cpp:1207
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
LLVM_ABI void setName(const Twine &Name)
Change the name of the value.
Definition Value.cpp:394
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
Definition Value.cpp:553
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
std::pair< iterator, bool > insert(const ValueT &V)
Definition DenseSet.h:209
bool contains(const_arg_type_t< ValueT > V) const
Check if the set contains the given element.
Definition DenseSet.h:182
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
static constexpr bool isKnownLE(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:230
constexpr bool isNonZero() const
Definition TypeSize.h:155
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:216
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
Definition TypeSize.h:171
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
constexpr bool isZero() const
Definition TypeSize.h:153
static constexpr bool isKnownGT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:223
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
An efficient, type-erasing, non-owning reference to a callable.
const ParentTy * getParent() const
Definition ilist_node.h:34
self_iterator getIterator()
Definition ilist_node.h:123
IteratorT end() const
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
A raw_ostream that writes to an std::string.
CallInst * Call
Changed
This provides a very simple, boring adaptor for a begin and end iterator into a range type.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:83
@ Legal
The operation is expected to be selectable directly by the target, and no transformation is necessary...
void reportVectorizationFailure(const StringRef DebugMsg, const StringRef OREMsg, const StringRef ORETag, OptimizationRemarkEmitter *ORE, const Loop *TheLoop, Instruction *I=nullptr)
Reports a vectorization failure: print DebugMsg for debugging purposes along with the corresponding o...
void reportVectorizationInfo(const StringRef Msg, const StringRef ORETag, OptimizationRemarkEmitter *ORE, const Loop *TheLoop, Instruction *I=nullptr, DebugLoc DL={})
Reports an informative message: print Msg for debugging purposes as well as an optimization remark.
void reportVectorization(OptimizationRemarkEmitter *ORE, Loop *TheLoop, ElementCount VFWidth, unsigned IC)
Report successful vectorization of the loop.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
bool match(Val *V, const Pattern &P)
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
auto m_Value()
Match an arbitrary value and ignore it.
auto m_LogicalOr()
Matches L || R where L and R are arbitrary values.
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
bind_cst_ty m_scev_APInt(const APInt *&C)
Match an SCEV constant and bind it to an APInt.
match_bind< const SCEVMulExpr > m_scev_Mul(const SCEVMulExpr *&V)
bool match(const SCEV *S, const Pattern &P)
SCEVBinaryExpr_match< SCEVMulExpr, Op0_t, Op1_t, SCEV::FlagNone, true > m_scev_c_Mul(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< Instruction::Freeze, Op0_t > m_Freeze(const Op0_t &Op0)
bool matchFindIVResult(VPInstruction *VPI, Op0_t ReducedIV, Op1_t Start)
Match FindIV result pattern: select(icmp ne ComputeReductionResult(ReducedIV), Sentinel),...
VPInstruction_match< VPInstruction::ExtractLastLane, Op0_t > m_ExtractLastLane(const Op0_t &Op0)
VPInstruction_match< VPInstruction::BranchOnCount > m_BranchOnCount()
auto m_VPValue()
Match an arbitrary VPValue and ignore it.
VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > m_ExtractLastPart(const Op0_t &Op0)
VPRecipeBase * findUserOf(VPValue *V, const MatchT &P)
If V is used by a recipe matching pattern P, return it.
bool match(Val *V, const Pattern &P)
match_bind< VPInstruction > m_VPInstruction(VPInstruction *&V)
Match a VPInstruction, capturing if we match.
VPInstruction_match< VPInstruction::ExtractLane, Op0_t, Op1_t > m_ExtractLane(const Op0_t &Op0, const Op1_t &Op1)
ValuesClass values(OptsTy... Options)
Helper to build a ValuesClass by forwarding a variable number of arguments as an initializer list to ...
initializer< Ty > init(const Ty &Val)
Add a small namespace to avoid name clashes with the classes used in the streaming interface.
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
InstructionCost getScalarizationOverhead(const TargetTransformInfo &TTI, bool ReVec, Type *ScalarTy, VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, const TTI::TargetCostKind CostKind, bool ForPoisonSrc, ArrayRef< Value * > VL, TTI::VectorInstrContext VIC)
This is similar to TargetTransformInfo::getScalarizationOverhead, but if ScalarTy is a FixedVectorTyp...
BranchProbability getExecutionProbability(BlockFrequency Freq)
Returns Freq as a BranchProbability, relative to the full mass.
bool isSingleScalar(const VPValue *VPV)
Returns true if VPV is a single scalar, either because it produces the same value for all lanes or on...
VPBasicBlock * getFirstLoopHeader(VPlan &Plan, VPDominatorTree &VPDT)
Returns the header block of the first, top-level loop, or null if none exist.
bool isAddressSCEVForCost(const SCEV *Addr, ScalarEvolution &SE, const Loop *L)
Returns true if Addr is an address SCEV that can be passed to TTI::getAddressComputationCost,...
VPInstruction * findCanonicalIVIncrement(VPlan &Plan)
Find the canonical IV increment of Plan's vector loop region.
bool onlyFirstLaneUsed(const VPValue *Def)
Returns true if only the first lane of Def is used.
VPValue * findIncomingAliasMask(const VPlan &Plan)
Finds the incoming alias-mask within the vector preheader.
bool doesGeneratePerAllLanes(const VPRecipeBase *R)
Returns true if R produces scalar values for all VF lanes.
VPRecipeBase * findRecipe(VPValue *Start, PredT Pred)
Search Start's users for a recipe satisfying Pred, looking through recipes with definitions.
Definition VPlanUtils.h:146
LLVM_ABI_FOR_TEST const SCEV * getSCEVExprForVPValue(const VPValue *V, PredicatedScalarEvolution &PSE, const Loop *L=nullptr)
Return the SCEV expression for V.
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI bool simplifyLoop(Loop *L, DominatorTree *DT, LoopInfo *LI, ScalarEvolution *SE, AssumptionCache *AC, MemorySSAUpdater *MSSAU, bool PreserveLCSSA)
Simplify each loop in a loop nest recursively.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
SmallVector< VPBasicBlock * > vp_rpo_plain_cfg_loop_body(VPBasicBlock *Header)
Returns the VPBasicBlocks forming the loop body of a plain (pre-region) VPlan in reverse post-order s...
Definition VPlanCFG.h:262
detail::zippy< detail::zip_shortest, T, U, Args... > zip(T &&t, U &&u, Args &&...args)
zip iterator for two or more iteratable types.
Definition STLExtras.h:846
constexpr auto not_equal_to(T &&Arg)
Functor variant of std::not_equal_to that can be used as a UnaryPredicate in functional algorithms li...
Definition STLExtras.h:2196
LLVM_ABI Value * addRuntimeChecks(Instruction *Loc, Loop *TheLoop, const SmallVectorImpl< RuntimePointerCheck > &PointerChecks, SCEVExpander &Expander, bool HoistRuntimeChecks=false)
Add code that checks at runtime if the accessed arrays in PointerChecks overlap.
auto cast_if_present(const Y &Val)
cast_if_present<X> - Functionally identical to cast, except that a null value is accepted.
Definition Casting.h:683
LLVM_ABI bool RemoveRedundantDbgInstrs(BasicBlock *BB)
Try to remove redundant dbg.value instructions from given basic block.
LLVM_ABI_FOR_TEST cl::opt< bool > VerifyEachVPlan
LLVM_ABI std::optional< unsigned > getLoopEstimatedTripCount(Loop *L, unsigned *EstimatedLoopInvocationWeight=nullptr)
Return either:
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
unsigned getLoadStoreAddressSpace(const Value *I)
A helper function that returns the address space of the pointer operand of load or store instruction.
LLVM_ABI Intrinsic::ID getVectorIntrinsicIDForCall(const CallInst *CI, const TargetLibraryInfo *TLI)
Returns intrinsic ID for call.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
Definition STLExtras.h:856
InstructionCost Cost
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
LLVM_ABI bool verifyFunction(const Function &F, raw_ostream *OS=nullptr)
Check a function for errors, useful for use when debugging a pass.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
VPBuilderBase<> VPBuilder
Definition VPlan.h:67
OuterAnalysisManagerProxy< ModuleAnalysisManager, Function > ModuleAnalysisManagerFunctionProxy
Provide the ModuleAnalysisManager to Function proxy.
Value * getRuntimeVF(IRBuilderBase &B, Type *Ty, ElementCount VF)
Return the runtime value for VF.
LLVM_ABI bool formLCSSARecursively(Loop &L, const DominatorTree &DT, const LoopInfo *LI, ScalarEvolution *SE)
Put a loop nest into LCSSA form.
Definition LCSSA.cpp:469
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
Definition Casting.h:732
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
cl::opt< bool > VPlanBuildOuterloopStressTest
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
Definition STLExtras.h:2224
LLVM_ABI bool shouldOptimizeForSize(const MachineFunction *MF, ProfileSummaryInfo *PSI, const MachineBlockFrequencyInfo *BFI, PGSOQueryType QueryType=PGSOQueryType::Other)
Returns true if machine function MF is suggested to be size-optimized based on the profile.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:649
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
iterator_range< df_iterator< VPBlockShallowTraversalWrapper< VPBlockBase * > > > vp_depth_first_shallow(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order.
Definition VPlanCFG.h:250
LLVM_ABI bool VerifySCEV
LLVM_ABI_FOR_TEST cl::opt< bool > VPlanPrintAfterAll
LLVM_ABI bool isSafeToSpeculativelyExecute(const Instruction *I, const Instruction *CtxI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr, const TargetLibraryInfo *TLI=nullptr, bool UseVariableInfo=true, bool IgnoreUBImplyingAttrs=true)
Return true if the instruction does not have any effects besides calculating the result and does not ...
bool isa_and_nonnull(const Y &Val)
Definition Casting.h:676
iterator_range< df_iterator< VPBlockDeepTraversalWrapper< VPBlockBase * > > > vp_depth_first_deep(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order while traversing t...
Definition VPlanCFG.h:285
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
Definition STLExtras.h:366
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
auto make_isa_range(RangeT &&Range)
Return a range over Range containing only elements for which isa<T> holds, casting each of them to T.
Definition STLExtras.h:567
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
void collectEphemeralRecipesForVPlan(VPlan &Plan, DenseSet< VPRecipeBase * > &EphRecipes)
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
bool containsIrreducibleCFG(RPOTraversalT &RPOTraversal, const LoopInfoT &LI)
Return true if the control flow in RPOTraversal is irreducible.
Definition CFG.h:154
std::optional< uint64_t > getMaxRuntimeElementCount(ElementCount EC, const Function &F)
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
cl::opt< unsigned > ForceTargetInstructionCost("force-target-instruction-cost", cl::init(0), cl::Hidden, cl::desc("A flag that overrides the target's expected cost for " "an instruction to a single constant value. Mostly " "useful for getting consistent testing."))
Definition VPlan.cpp:58
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1652
bool hasIrregularType(Type *Ty, const DataLayout &DL)
A helper function that returns true if the given type is irregular.
UncountableExitStyle
Different methods of handling early exits.
Definition VPlan.h:83
@ ReadOnly
No side effects to worry about, so we can process any uncountable exits in the loop and branch either...
Definition VPlan.h:87
@ MaskedHandleExitInScalarLoop
All memory operations other than the load(s) required to determine whether an uncountable exit occurr...
Definition VPlan.h:92
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
LLVM_ABI cl::opt< bool > EnableLoopVectorization
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
SmallVector< VPRegisterUsage, 8 > calculateRegisterUsageForPlan(VPlan &Plan, ArrayRef< ElementCount > VFs, const TargetTransformInfo &TTI)
Estimate the register usage for Plan and vectorization factors in VFs by calculating the highest numb...
LLVM_ABI_FOR_TEST cl::list< std::string > VPlanPrintAfterPasses
LLVM_ABI bool wouldInstructionBeTriviallyDead(const Instruction *I, const TargetLibraryInfo *TLI=nullptr)
Return true if the result produced by the instruction would have no side effects if it was not used.
Definition Local.cpp:409
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
Type * toVectorizedTy(Type *Ty, ElementCount EC)
A helper for converting to vectorized types.
T * find_singleton(R &&Range, Predicate P, bool AllowRepeats=false)
Return the single value in Range that satisfies P(<member of Range> *, AllowRepeats)->T * returning n...
Definition STLExtras.h:1853
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
bool canVectorizeTy(Type *Ty)
Returns true if Ty is a valid vector element type, void, or an unpacked literal struct where all elem...
TargetTransformInfo TTI
@ CM_EpilogueNotAllowedLowTripLoop
@ CM_EpilogueNotNeededFoldTail
@ CM_EpilogueNotAllowedFoldTail
@ CM_EpilogueNotAllowedOptSize
@ CM_EpilogueAllowed
LLVM_ABI bool isAssignmentTrackingEnabled(const Module &M)
Return true if assignment tracking is enabled for module M.
LLVM_ABI_FOR_TEST cl::list< std::string > VPlanPrintBeforePasses
RecurKind
These are the kinds of recurrences that we support.
@ Sub
Subtraction of integers.
@ Add
Sum of integers.
LLVM_ABI Value * getRecurrenceIdentity(RecurKind K, Type *Tp, FastMathFlags FMF)
Given information about an recurrence kind, return the identity for the @llvm.vector....
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
cl::opt< unsigned > NumberOfStoresToPredicate("vectorize-num-stores-pred", cl::init(1), cl::Hidden, cl::desc("Max number of stores to be predicated behind an if."))
The number of stores in a loop that are allowed to need predication.
Definition VPlan.cpp:59
constexpr T AbsoluteDifference(U X, V Y)
Subtract two unsigned integers, X and Y, of type T and return the absolute value of the result.
Definition MathExtras.h:595
DWARFExpression::Operation Op
LLVM_ABI bool isGuaranteedNotToBeUndefOrPoison(const Value *V, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Return true if this function can prove that V does not have undef bits and is never poison.
ArrayRef(const T &OneElt) -> ArrayRef< T >
auto sum_of(R &&Range, E Init=E{0})
Returns the sum of all values in Range with Init initial value.
Definition STLExtras.h:1733
OutputIt move(R &&Range, OutputIt Out)
Provide wrappers to std::move which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1933
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI_FOR_TEST cl::opt< bool > VPlanPrintBeforeAll
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
auto predecessors(const MachineBasicBlock *BB)
iterator_range< pointer_iterator< WrappedIteratorT > > make_pointer_range(RangeT &&Range)
Definition iterator.h:368
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
ArrayRef< Type * > getContainedTypes(Type *const &Ty)
Returns the types contained in Ty.
bool pred_empty(const BasicBlock *BB)
Definition CFG.h:107
@ None
Don't use tail folding.
@ DataWithEVL
Use predicated EVL instructions for tail-folding.
@ DataAndControlFlow
Use predicate to control both data and control flow.
@ DataWithoutLaneMask
Same as Data, but avoids using the get.active.lane.mask intrinsic to calculate the mask and instead i...
@ Data
Use predicate only to mask operations on data in the loop.
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
LLVM_ABI bool hasBranchWeightMD(const Instruction &I)
Checks if an instructions has Branch Weight Metadata.
hash_code hash_combine(const Ts &...args)
Combine values into a single hash_code.
Definition Hashing.h:307
@ Increment
Incrementally increasing token ID.
Definition AllocToken.h:26
@ Enabled
Convert any .debug_str_offsets tables to DWARF64 if needed.
Definition DWP.h:31
@ Disabled
Don't do any conversion of .debug_str_offsets tables.
Definition DWP.h:30
T bit_floor(T Value)
Returns the largest integral power of two no greater than Value if Value is nonzero.
Definition bit.h:347
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
std::unique_ptr< VPlan > VPlanPtr
Definition VPlan.h:78
LLVM_ABI Value * addDiffRuntimeChecks(Instruction *Loc, ArrayRef< PointerDiffInfo > Checks, SCEVExpander &Expander, ElementCount VF, unsigned IC)
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
Definition Casting.h:866
LLVM_ABI_FOR_TEST bool verifyVPlanIsValid(const VPlan &Plan)
Verify invariants for general VPlans.
hash_code hash_combine_range(InputIteratorT first, InputIteratorT last)
Compute a hash_code for a sequence of values.
Definition Hashing.h:287
LLVM_ABI_FOR_TEST cl::opt< bool > VPlanPrintVectorRegionScope
LLVM_ABI cl::opt< bool > EnableLoopInterleaving
Implement std::hash so that hash_code can be used in STL containers.
Definition BitVector.h:878
A special type used by analysis passes to provide an address that identifies that particular analysis...
Definition Analysis.h:29
static LLVM_ABI void collectEphemeralValues(const Loop *L, AssumptionCache *AC, SmallPtrSetImpl< const Value * > &EphValues)
Collect a loop's ephemeral values (those used only by an assume or similar intrinsics in the loop).
Encapsulate information regarding vectorization of a loop and its epilogue.
EpilogueLoopVectorizationInfo(ElementCount MVF, unsigned MUF, ElementCount EVF)
A class that represents two vectorization factors (initialized with 0 by default).
static FixedScalableVFPair getNone()
This holds details about a histogram operation – a load -> update -> store sequence where each lane i...
TargetLibraryInfo * TLI
LLVM_ABI LoopVectorizeResult runImpl(Function &F)
LLVM_ABI bool processLoop(Loop *L)
ProfileSummaryInfo * PSI
LoopAccessInfoManager * LAIs
std::function< const BranchProbabilityInfo &()> GetBPI
LLVM_ABI void printPipeline(raw_ostream &OS, function_ref< StringRef(StringRef)> MapClassName2PassName)
LLVM_ABI LoopVectorizePass(LoopVectorizeOptions Opts={})
ScalarEvolution * SE
AssumptionCache * AC
LLVM_ABI PreservedAnalyses run(Function &F, FunctionAnalysisManager &AM)
OptimizationRemarkEmitter * ORE
std::function< BlockFrequencyInfo &()> GetBFI
TargetTransformInfo * TTI
Storage for information about made changes.
A marker analysis to determine if extra passes should be run after loop vectorization.
static LLVM_ABI AnalysisKey Key
Parameters that control the generic loop unrolling transformation.
bool UnrollVectorizedLoop
Disable runtime unrolling by default for vectorized loops.
Holds the VFShape for a specific scalar to vector function mapping.
A range of powers-of-2 vectorization factors with fixed start and adjustable end.
ElementCount End
Struct to hold various analysis needed for cost computations.
LLVMContext & LLVMCtx
const VFSelectionContext & Config
LoopVectorizationCostModel & CM
VPCostContext(const TargetLibraryInfo &TLI, const VPlan &Plan, LoopVectorizationCostModel &CM, VFSelectionContext &Config, bool ReusePrintingSlotTracker=false)
bool skipCostComputation(Instruction *UI, bool IsVector) const
Return true if the cost for UI shouldn't be computed, e.g.
InstructionCost getLegacyCost(Instruction *UI, ElementCount VF) const
Return the cost for UI with VF using the legacy cost model as fallback until computing the cost of al...
bool isMaskRequired(Instruction *I) const
Forwards to LoopVectorizationCostModel::isMaskRequired.
void invalidateWideningDecision(Instruction *I, ElementCount VF)
Mark the widening decision for I at VF as invalidated since a VPlan transform replaced the original r...
PredicatedScalarEvolution & PSE
bool willBeScalarized(Instruction *I, ElementCount VF) const
Returns true if I is known to be scalarized at VF.
static bool executesAtMostOnce(const VPlan &Plan, ElementCount VF)
Returns true if the vector loop body of Plan is known to execute at most once at VF,...
TargetTransformInfo::TargetCostKind CostKind
const TargetLibraryInfo & TLI
const TargetTransformInfo & TTI
SmallPtrSet< Instruction *, 8 > SkipCostComputation
A pure-virtual common base class for recipes defining a single VPValue and using IR flags.
Definition VPlan.h:1113
A struct that represents some properties of the register usage of a loop.
InstructionCost spillCost(const TargetTransformInfo &TTI, TargetTransformInfo::TargetCostKind CostKind, unsigned OverrideMaxNumRegs=0) const
Calculate the estimated cost of any spills due to using more registers than the number available for ...
VPTransformState holds information passed down when "executing" a VPlan, needed for generating the ou...
A recipe for widening load operations, using the address to load from and an optional mask.
Definition VPlan.h:3814
A recipe for widening store operations, using the stored value, the address to store to and an option...
Definition VPlan.h:3919
static void simplifyLiveInsWithSCEV(VPlan &Plan, PredicatedScalarEvolution &PSE)
Check Plan's live-ins and replace them with constants, if they can be simplified via SCEV.
static void expandSCEVsToVPInstructions(VPlan &Plan, ScalarEvolution &SE)
Expand VPExpandSCEVRecipes in Plan's entry block to VPInstructions.
static void materializeBroadcasts(VPlan &Plan)
Add explicit broadcasts for live-ins and VPValues defined in Plan's entry block if they are used as v...
static void materializePacksAndUnpacks(VPlan &Plan)
Add explicit Build[Struct]Vector recipes to Pack multiple scalar values into vectors and Unpack recip...
static void createInterleaveGroups(VPlan &Plan, const SmallPtrSetImpl< const InterleaveGroup< Instruction > * > &InterleaveGroups, const bool &EpilogueAllowed)
static LLVM_ABI_FOR_TEST bool handleUncountableEarlyExits(VPlan &Plan, OptimizationRemarkEmitter *ORE, Loop *TheLoop, PredicatedScalarEvolution &PSE, DominatorTree &DT, AssumptionCache *AC, UncountableExitStyle Style)
Update Plan to account for uncountable early exits by introducing appropriate branching logic in the ...
static bool simplifyKnownEVL(VPlan &Plan, ElementCount VF, PredicatedScalarEvolution &PSE)
Try to simplify VPInstruction::ExplicitVectorLength recipes when the AVL is known to be <= VF,...
static void introduceMasksAndLinearize(VPlan &Plan)
Predicate and linearize the control-flow in the only loop region of Plan.
static void materializeFactors(VPlan &Plan, VPBasicBlock *VectorPH, ElementCount VF)
Materialize UF, VF and VFxUF to be computed explicitly using VPInstructions.
static void foldTailByMasking(VPlan &Plan)
Adapts the vector loop region for tail folding by introducing a header mask and conditionally executi...
static void materializeBackedgeTakenCount(VPlan &Plan, VPBasicBlock *VectorPH)
Materialize the backedge-taken count to be computed explicitly using VPInstructions.
static LLVM_ABI_FOR_TEST bool tryToConvertVPInstructionsToVPRecipes(VPlan &Plan, const TargetLibraryInfo &TLI, PredicatedScalarEvolution &PSE, Loop *OuterLoop)
Replaces the VPInstructions in Plan with corresponding widen recipes.
static bool handleMultiUseReductions(VPlan &Plan, OptimizationRemarkEmitter *ORE, Loop *TheLoop)
Try to legalize reductions with multiple in-loop uses.
static void convertToVariableLengthStep(VPlan &Plan)
Transform loops with variable-length stepping after region dissolution.
static void materializeHeaderMask(VPlan &Plan, bool UseActiveLaneMask, bool UseActiveLaneMaskForControlFlow)
Materialize the abstract header mask of the loop region into concrete recipes: an active-lane-mask if...
static void recordExecutionFrequencies(VPlan &Plan)
Add execution frequencies to each recipe in the loop body of Plan.
static void addBranchWeightToMiddleTerminator(VPlan &Plan, ElementCount VF, std::optional< unsigned > VScaleForTuning)
Add branch weight metadata, if the Plan's middle block is terminated by a BranchOnCond recipe.
static std::unique_ptr< VPlan > narrowInterleaveGroups(VPlan &Plan, const TargetTransformInfo &TTI)
Try to find a single VF among Plan's VFs for which all interleave groups (with known minimum VF eleme...
static bool handleFindLastReductions(VPlan &Plan)
Check if Plan contains any FindLast reductions.
static void createInLoopReductionRecipes(VPlan &Plan, ElementCount MinVF)
Create VPReductionRecipes for in-loop reductions.
static void materializeAliasMaskCheckBlock(VPlan &Plan, ArrayRef< PointerDiffInfo > DiffChecks, bool HasBranchWeights)
Materializes the alias mask within a check block before the loop.
static void modelGeneratedMainLoopBlocks(VPlan &EpiPlan, VPlan &MainPlan, VPIRBasicBlock *EnteredFrom)
Model the blocks the executed MainPlan generated for the main vector loop in EpiPlan during epilogue ...
static void unrollByUF(VPlan &Plan, unsigned UF)
Explicitly unroll Plan by UF.
static DenseMap< const SCEV *, Value * > expandSCEVs(VPlan &Plan, ScalarEvolution &SE)
Expand remaining VPExpandSCEVRecipes in Plan's entry block using SCEVExpander.
static void convertToConcreteRecipes(VPlan &Plan)
Lower abstract recipes to concrete ones, that can be codegen'd.
static LLVM_ABI_FOR_TEST void createLoopRegions(VPlan &Plan, DebugLoc DL)
Replace loops in Plan's flat CFG with VPRegionBlocks, turning Plan's flat CFG into a hierarchical CFG...
static void makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range, VPRecipeBuilder &RecipeBuilder, VPCostContext &CostCtx)
Convert load/store VPInstructions in Plan into widened or replicate recipes.
static LLVM_ABI_FOR_TEST void addMiddleCheck(VPlan &Plan)
If a check is needed to guard executing the scalar epilogue loop, it will be added to the middle bloc...
static void narrowInductionTruncates(VPlan &Plan, VFRange &Range, const TargetTransformInfo &TTI, PredicatedScalarEvolution &PSE)
Replace truncates of a wide induction, or of that induction's increment, by a VPWidenIntOrFpInduction...
static LLVM_ABI_FOR_TEST bool createHeaderPhiRecipes(VPlan &Plan, PredicatedScalarEvolution &PSE, Loop &OrigLoop, const VPDominatorTree &VPDT, const MapVector< PHINode *, InductionDescriptor > &Inductions, const MapVector< PHINode *, RecurrenceDescriptor > &Reductions, const SmallPtrSetImpl< const PHINode * > &FixedOrderRecurrences, const SmallPtrSetImpl< PHINode * > &InLoopReductions, bool AllowReordering)
Replace VPPhi recipes in Plan's header with corresponding VPHeaderPHIRecipe subclasses for inductions...
static void expandBranchOnTwoConds(VPlan &Plan)
Expand BranchOnTwoConds instructions into explicit CFG with BranchOnCond instructions.
static void materializeVectorTripCount(VPlan &Plan, VPBasicBlock *VectorPHVPBB, bool TailByMasking, bool RequiresScalarEpilogue, VPValue *Step, std::optional< uint64_t > MaxRuntimeStep=std::nullopt)
Materialize vector trip count computations to a set of VPInstructions.
static void hoistPredicatedLoads(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L)
Hoist predicated loads from the same address to the loop entry block, if they are guaranteed to execu...
static void attachAliasMaskToHeaderMask(VPlan &Plan)
Attaches the alias-mask to the existing header-mask.
static void optimizeFindIVReductions(VPlan &Plan, PredicatedScalarEvolution &PSE, Loop &L)
Optimize FindLast reductions selecting IVs (or expressions of IVs) by converting them to FindIV reduc...
static void convertToAbstractRecipes(VPlan &Plan, VPCostContext &Ctx, VFRange &Range)
This function converts initial recipes to the abstract recipes and clamps Range based on cost model f...
static void materializeConstantVectorTripCount(VPlan &Plan, ElementCount BestVF, unsigned BestUF, PredicatedScalarEvolution &PSE)
static void makeScalarizationDecisions(VPlan &Plan, VFRange &Range)
Make VPlan-based scalarization decision prior to delegating to the ones made by the legacy CM.
static void replaceWideCanonicalIVWithWideIV(VPlan &Plan, ScalarEvolution &SE, const TargetTransformInfo &TTI, TargetTransformInfo::TargetCostKind CostKind, ElementCount VF, unsigned UF)
Replace a VPWidenCanonicalIVRecipe if it is present in Plan, with a VPWidenIntOrFpInductionRecipe,...
static LLVM_ABI_FOR_TEST std::unique_ptr< VPlan > buildVPlan0(Loop *TheLoop, LoopInfo &LI, Type *InductionTy, PredicatedScalarEvolution &PSE, LoopVersioning *LVer=nullptr, function_ref< const BranchProbabilityInfo &()> GetBPI=nullptr)
Create a base VPlan0, serving as the common starting point for all later candidates.
static void optimizeInductionLiveOutUsers(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L)
If there's a single exit block, optimize its phi recipes that use exiting IV values by feeding them p...
static void addExplicitVectorLength(VPlan &Plan, const std::optional< unsigned > &MaxEVLSafeElements)
Add a VPCurrentIterationPHIRecipe and related recipes to Plan and replaces all uses of the canonical ...
static void makeCallWideningDecisions(VPlan &Plan, VFRange &Range, VPRecipeBuilder &RecipeBuilder, VPCostContext &CostCtx)
Convert call VPInstructions in Plan into widened call, vector intrinsic or replicate recipes based on...
static void adjustFirstOrderRecurrenceMiddleUsers(VPlan &Plan, VFRange &Range)
Adjust first-order recurrence users in the middle block: create penultimate element extracts for LCSS...
static void optimizeEVLMasks(VPlan &Plan)
Optimize recipes which use an EVL-based header mask to VP intrinsics, for example:
static bool handleMaxMinNumReductions(VPlan &Plan)
Check if Plan contains any FMaxNum or FMinNum reductions.
static void removeDeadRecipes(VPlan &Plan)
Remove dead recipes from Plan.
static void attachCheckBlock(VPlan &Plan, Value *Cond, BasicBlock *CheckBlock, bool AddBranchWeights)
static LLVM_ABI_FOR_TEST void handleCountableEarlyExits(VPlan &Plan)
Disconnect countable early exits from the loop.
static void sinkPredicatedStores(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L)
Sink predicated stores to the same address with complementary predicates (P and NOT P) to an uncondit...
static bool finalizeSCEVPredicates(VPlan &Plan, PredicatedScalarEvolution &PSE, bool OptForSize, unsigned SCEVCheckThreshold, OptimizationRemarkEmitter *ORE, Loop *TheLoop)
Finalize SCEV predicates by adding induction predicates from Plan to PSE and checking constraints.
static void replicateByVF(VPlan &Plan, ElementCount VF)
Replace replicating VPReplicateRecipe, VPScalarIVStepsRecipe and VPInstruction in Plan with VF single...
static bool removeBranchOnConst(VPlan &Plan, bool OnlyLatches=false)
Remove BranchOnCond recipes with true or false conditions together with removing dead edges to their ...
static void convertToStridedAccesses(VPlan &Plan, PredicatedScalarEvolution &PSE, Loop &L, VPCostContext &Ctx, VFRange &Range)
Transform widen memory recipes into strided access recipes when legal and profitable.
static void addIterationCountCheckBlock(VPlan &Plan, ElementCount VF, unsigned UF, bool RequiresScalarEpilogue, Loop *OrigLoop, const uint32_t *MinItersBypassWeights, DebugLoc DL, PredicatedScalarEvolution &PSE)
Add a new check block before the vector preheader to Plan to check if the main vector loop should be ...
static void clearReductionWrapFlags(VPlan &Plan)
Clear NSW/NUW flags from reduction instructions if necessary.
static void createPartialReductions(VPlan &Plan, VPCostContext &CostCtx, VFRange &Range)
Detect and create partial reduction recipes for scaled or unordered reductions in Plan.
static void addMinimumIterationCheck(VPlan &Plan, ElementCount VF, unsigned UF, ElementCount MinProfitableTripCount, bool RequiresScalarEpilogue, bool TailFolded, Loop *OrigLoop, const uint32_t *MinItersBypassWeights, DebugLoc DL, PredicatedScalarEvolution &PSE, VPBasicBlock *CheckBlock)
static void cse(VPlan &Plan)
Perform common-subexpression-elimination on Plan.
static void replaceSymbolicStrides(VPlan &Plan, PredicatedScalarEvolution &PSE, const SymbolicStrideMap &StridesMap, const VPDominatorTree &VPDT)
Replace symbolic strides from StridesMap in Plan with constants when possible.
static LLVM_ABI_FOR_TEST void optimize(VPlan &Plan)
Apply VPlan-to-VPlan optimizations to Plan, including induction recipe optimizations,...
static void dissolveLoopRegions(VPlan &Plan)
Replace loop regions with explicit CFG.
static void truncateToMinimalBitwidths(VPlan &Plan, const MapVector< Instruction *, uint64_t > &MinBWs)
Insert truncates and extends for any truncated recipe.
static void dropPoisonGeneratingRecipes(VPlan &Plan)
Drop poison flags from recipes that may generate a poison value that is used after vectorization,...
static void optimizeForVFAndUF(VPlan &Plan, ElementCount BestVF, unsigned BestUF, PredicatedScalarEvolution &PSE)
Optimize Plan based on BestVF and BestUF.
static void convertEVLExitCond(VPlan &Plan)
Replaces the exit condition from (branch-on-cond eq CanonicalIVInc, VectorTripCount) to (branch-on-co...
static bool splitCombinedExits(VPlan &Plan, PredicatedScalarEvolution &PSE, Loop *TheLoop)
If a single exit has multiple conditions combined together, split them and create new exiting blocks.
static void addMinimumVectorEpilogueIterationCheck(VPlan &Plan, Value *VectorTripCount, bool RequiresScalarEpilogue, ElementCount EpilogueVF, unsigned MainLoopStep, unsigned EpilogueLoopStep, ScalarEvolution &SE)
Add a check to Plan to see if the epilogue vector loop should be executed.
static void attachMemoryChecks(VPlan &Plan, ArrayRef< RuntimePointerCheck > Checks, ScalarEvolution &SE, DebugLoc DL, bool AddBranchWeights)
Generate Checks as recipes and attach the check block to Plan.
static void combineRecipes(VPlan &Plan)
Perform instcombine-like simplifications on recipes in Plan.
TODO: The following VectorizationFactor was pulled out of LoopVectorizationCostModel class.
InstructionCost Cost
Cost of the loop with that width.
ElementCount MinProfitableTripCount
The minimum trip count required to make vectorization profitable, e.g.
ElementCount Width
Vector width with best cost.
InstructionCost ScalarCost
Cost of the scalar loop.
static VectorizationFactor Disabled()
Width 1 means no vectorization, cost 0 means uncomputed cost.
static LLVM_ABI unsigned VectorizeMemoryCheckThreshold
The maximum allowed number of runtime memory checks.
static LLVM_ABI bool HoistRuntimeChecks