[llvm] [LV] Add VPlan planning context and result (PR #226415)

Kiran Chandramohan via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 25 02:39:55 PDT 2026


https://github.com/kiranchandramohan created https://github.com/llvm/llvm-project/pull/226415

Introduce VPlanPlanningContext to group the cost model with the candidates
produced during one planning run, and VPlanPlanningResult to retain VPlans
and profitable VFs after planning completes. Move plan lookup and
profitability state out of the planner, thread the context through planning
APIs, and finalize it before code generation so the cost model cannot be
used after planning. Make epilogue selection consume an explicit candidate
result, preparing for independent main and epilogue planning runs.

Assisted-by: Codex

>From 8c06a5001fd4f545933bab9a513d1e51c331d040 Mon Sep 17 00:00:00 2001
From: Kiran Chandramohan <kiran.chandramohan at arm.com>
Date: Fri, 25 Sep 2026 01:24:45 +0200
Subject: [PATCH 1/3] [LV] Derive VPlans from VFs in planner cost APIs

Remove the redundant VPlan arguments from cost() and
selectInterleaveCount(). Both APIs already receive a VF, and the planner
maintains a unique plan for each VF, so resolve the plan through
getPlanFor(). This prevents callers from supplying inconsistent plan/VF
pairs and prepares plan lookup to be scoped to a planning context.

Assisted-by: Codex
---
 .../Vectorize/LoopVectorizationPlanner.h           |  7 +++----
 llvm/lib/Transforms/Vectorize/LoopVectorize.cpp    | 14 ++++++++------
 2 files changed, 11 insertions(+), 10 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index bb83ef7dec883..59630de654311 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -878,7 +878,7 @@ class LoopVectorizationPlanner {
   /// A builder used to construct the current plan.
   VPBuilder Builder;
 
-  /// Computes the cost of \p Plan for vectorization factor \p VF.
+  /// Computes the cost of the VPlan associated with \p VF.
   ///
   /// The current implementation requires access to the
   /// LoopVectorizationLegality to handle inductions and reductions, which is
@@ -886,7 +886,7 @@ class LoopVectorizationPlanner {
   ///
   /// TODO: Move to VPlan::cost once the use of LoopVectorizationLegality has
   /// been retired.
-  InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU) const;
+  InstructionCost cost(ElementCount VF, VPRegisterUsage *RU) const;
 
   /// Precompute costs for certain instructions using the legacy cost model. The
   /// function is used to bring up the VPlan-based cost model to initially avoid
@@ -932,8 +932,7 @@ class LoopVectorizationPlanner {
   /// If interleave count has been specified by metadata it will be returned.
   /// Otherwise, the interleave count is computed and returned. VF and LoopCost
   /// are the selected vectorization factor and the cost of the selected VF.
-  unsigned selectInterleaveCount(VPlan &Plan, ElementCount VF,
-                                 InstructionCost LoopCost);
+  unsigned selectInterleaveCount(ElementCount VF, InstructionCost LoopCost);
 
   /// Generate the IR code for the vectorized loop captured in VPlan \p BestPlan
   /// according to the best selected \p VF and  \p UF.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 2f1fc4398654a..972d583dad01e 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3696,8 +3696,9 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
 }
 
 unsigned
-LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
+LoopVectorizationPlanner::selectInterleaveCount(ElementCount VF,
                                                 InstructionCost LoopCost) {
+  VPlan &Plan = getPlanFor(VF);
   // -- The interleave heuristics --
   // We interleave the loop in order to expose ILP and reduce the loop overhead.
   // There are many micro-architectural considerations that we can't predict
@@ -3751,7 +3752,7 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
     if (VF.isScalar())
       LoopCost = CM->expectedCost(VF);
     else
-      LoopCost = cost(Plan, VF, &R);
+      LoopCost = cost(VF, &R);
     assert(LoopCost.isValid() && "Expected to have chosen a VF with valid cost");
 
     // Loop body is free and there is no need for interleaving.
@@ -5418,7 +5419,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
         // For scalar VF, skip VPlan cost check as VPlan cost is designed for
         // vector VFs only.
         if (UserVF.isScalar() ||
-            cost(*VPlans.front(), UserVF, /*RU=*/nullptr).isValid()) {
+            cost(UserVF, /*RU=*/nullptr).isValid()) {
           LLVM_DEBUG(dbgs() << "LV: Using user VF " << UserVF << ".\n");
           LLVM_DEBUG(printPlans(dbgs()));
           return;
@@ -5593,8 +5594,9 @@ getRecordedExecutionFrequency(const VPBasicBlock *VPBB) {
 }
 #endif
 
-InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
+InstructionCost LoopVectorizationPlanner::cost(ElementCount VF,
                                                VPRegisterUsage *RU) const {
+  VPlan &Plan = getPlanFor(VF);
   VPCostContext CostCtx(*TLI, Plan, *CM, Config,
                         /*ReusePrintingSlotTracker=*/true);
   InstructionCost Cost = precomputeCosts(Plan, VF, CostCtx);
@@ -5729,7 +5731,7 @@ LoopVectorizationPlanner::computeBestVF() {
       }
 
       InstructionCost Cost =
-          cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
+          cost(VF, ConsiderRegPressure ? &RUs[I] : nullptr);
       VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
 
       if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
@@ -7916,7 +7918,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                            LVP.getCostModel().maskPartialAliasing());
   if (IsInnerLoop && LVP.hasPlanWithVF(VF.Width)) {
     // Select the interleave count.
-    IC = LVP.selectInterleaveCount(*BestPlanPtr, VF.Width, VF.Cost);
+    IC = LVP.selectInterleaveCount(VF.Width, VF.Cost);
 
     unsigned SelectedIC = std::max(IC, UserIC);
     //  Optimistically generate runtime checks if they are needed. Drop them if

>From f224f74c44c5e85d9cdd0cdeb9efea5f5b6b7fbc Mon Sep 17 00:00:00 2001
From: Kiran Chandramohan <kiran.chandramohan at arm.com>
Date: Fri, 25 Sep 2026 01:24:51 +0200
Subject: [PATCH 2/3] [LV] Remove redundant InterleavedAccessInfo from the
 planner

The cost model already retains the InterleavedAccessInfo used to make its
widening decisions. Use that reference when building interleave recipes
and remove the duplicate reference and constructor parameter from
LoopVectorizationPlanner. This keeps interleave decisions associated with
the active cost model.

Assisted-by: Codex
---
 .../Transforms/Vectorize/LoopVectorizationPlanner.h   |  7 ++-----
 llvm/lib/Transforms/Vectorize/LoopVectorize.cpp       | 11 +++++------
 2 files changed, 7 insertions(+), 11 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 59630de654311..2681a4f2f5b0c 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -860,9 +860,6 @@ class LoopVectorizationPlanner {
   /// VF selection state independent of cost-modeling decisions.
   VFSelectionContext &Config;
 
-  /// The interleaved access analysis.
-  InterleavedAccessInfo &IAI;
-
   PredicatedScalarEvolution &PSE;
 
   OptimizationRemarkEmitter *ORE;
@@ -899,8 +896,8 @@ class LoopVectorizationPlanner {
       Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
       const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
       std::unique_ptr<LoopVectorizationCostModel> CM,
-      VFSelectionContext &Config, InterleavedAccessInfo &IAI,
-      PredicatedScalarEvolution &PSE, OptimizationRemarkEmitter *ORE,
+      VFSelectionContext &Config, PredicatedScalarEvolution &PSE,
+      OptimizationRemarkEmitter *ORE,
       std::function<const BranchProbabilityInfo &()> GetBPI);
 
   ~LoopVectorizationPlanner();
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 972d583dad01e..ee617c1dbc582 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5758,12 +5758,10 @@ LoopVectorizationPlanner::LoopVectorizationPlanner(
     Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
     const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
     std::unique_ptr<LoopVectorizationCostModel> CM, VFSelectionContext &Config,
-    InterleavedAccessInfo &IAI, PredicatedScalarEvolution &PSE,
-    OptimizationRemarkEmitter *ORE,
+    PredicatedScalarEvolution &PSE, OptimizationRemarkEmitter *ORE,
     std::function<const BranchProbabilityInfo &()> GetBPI)
     : OrigLoop(L), LI(LI), DT(DT), TLI(TLI), TTI(TTI), Legal(Legal),
-      CM(std::move(CM)), Config(Config), IAI(IAI), PSE(PSE), ORE(ORE),
-      GetBPI(GetBPI) {}
+      CM(std::move(CM)), Config(Config), PSE(PSE), ORE(ORE), GetBPI(GetBPI) {}
 
 LoopVectorizationPlanner::~LoopVectorizationPlanner() = default;
 
@@ -6630,7 +6628,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // Range, add it to the set of groups to be later applied to the VPlan and add
   // placeholders for its members' Recipes which we'll be replacing with a
   // single VPInterleaveRecipe.
-  for (InterleaveGroup<Instruction> *IG : IAI.getInterleaveGroups()) {
+  for (InterleaveGroup<Instruction> *IG :
+       CM->InterleaveInfo.getInterleaveGroups()) {
     auto ApplyIG = [IG, this](ElementCount VF) -> bool {
       bool Result = (VF.isVector() && // Query is illegal for VF == 1
                      CM->getWideningDecision(IG->getInsertPos(), VF) ==
@@ -7874,7 +7873,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
       L, LI, DT, TLI, *TTI, &LVL,
       std::make_unique<LoopVectorizationCostModel>(
           SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE, GetBFI, F, IAI, Config),
-      Config, IAI, PSE, ORE, GetBPI);
+      Config, PSE, ORE, GetBPI);
 
   EpilogueLowering EpilogueTailLoweringStatus =
       getEpilogueTailLowering(LVP.getCostModel(), L, ORE, LVL, Hints, TTI);

>From e0f1dcfd015e30486242bb6aabcd6cfabfdf6dc4 Mon Sep 17 00:00:00 2001
From: Kiran Chandramohan <kiran.chandramohan at arm.com>
Date: Fri, 25 Sep 2026 01:24:57 +0200
Subject: [PATCH 3/3] [LV] Add VPlan planning context and result

Introduce VPlanPlanningContext to group the cost model with the candidates
produced during one planning run, and VPlanPlanningResult to retain VPlans
and profitable VFs after planning completes. Move plan lookup and
profitability state out of the planner, thread the context through planning
APIs, and finalize it before code generation so the cost model cannot be
used after planning. Make epilogue selection consume an explicit candidate
result, preparing for independent main and epilogue planning runs.

Assisted-by: Codex
---
 .../Vectorize/LoopVectorizationPlanner.h      | 114 +++++----
 .../Transforms/Vectorize/LoopVectorize.cpp    | 241 ++++++++++--------
 llvm/lib/Transforms/Vectorize/VPlan.cpp       |  11 +-
 3 files changed, 217 insertions(+), 149 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 2681a4f2f5b0c..199096bffff38 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -833,6 +833,52 @@ class VFSelectionContext {
   }
 };
 
+/// The plans and profitability results produced by a completed planning run.
+class VPlanPlanningResult {
+  friend class LoopVectorizationPlanner;
+
+  SmallVector<VPlanPtr, 4> VPlans;
+  SmallVector<VectorizationFactor, 8> ProfitableVFs;
+
+public:
+  VPlanPlanningResult() = default;
+  VPlanPlanningResult(VPlanPlanningResult &&) = default;
+  VPlanPlanningResult &operator=(VPlanPlanningResult &&) = default;
+  VPlanPlanningResult(const VPlanPlanningResult &) = delete;
+  VPlanPlanningResult &operator=(const VPlanPlanningResult &) = delete;
+
+  /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
+  /// for each VF.
+  VPlan &getPlanFor(ElementCount VF) const;
+
+  /// Return true if there is a plan for \p VF.
+  bool hasPlanWithVF(ElementCount VF) const;
+};
+
+/// State shared by VPlan construction and cost-modeling for one planning run.
+///
+/// Keeping the cost model and the plans whose decisions it produced in the
+/// same context makes the dependency explicit when multiple cost models are
+/// used for the same loop. Finalizing the context destroys the cost model and
+/// returns the plans and profitability results needed by code generation.
+class VPlanPlanningContext {
+  friend class LoopVectorizationPlanner;
+
+  std::unique_ptr<LoopVectorizationCostModel> CM;
+  VPlanPlanningResult Result;
+
+public:
+  explicit VPlanPlanningContext(std::unique_ptr<LoopVectorizationCostModel> CM);
+  ~VPlanPlanningContext();
+
+  LoopVectorizationCostModel &getCostModel();
+
+  const VPlanPlanningResult &getResult() const { return Result; }
+
+  /// End planning, destroy the cost model, and return the completed result.
+  VPlanPlanningResult finalize() &&;
+};
+
 /// Planner drives the vectorization process after having passed
 /// Legality checks.
 class LoopVectorizationPlanner {
@@ -854,10 +900,8 @@ class LoopVectorizationPlanner {
   /// The legality analysis.
   LoopVectorizationLegality *Legal;
 
-  /// The profitability analysis. Cleared after making cost based decisions.
-  std::unique_ptr<LoopVectorizationCostModel> CM;
-
-  /// VF selection state independent of cost-modeling decisions.
+  /// Loop-scoped VF selection state shared by planning contexts for this loop.
+  /// Context-specific values are recomputed by plan() before use.
   VFSelectionContext &Config;
 
   PredicatedScalarEvolution &PSE;
@@ -867,11 +911,6 @@ class LoopVectorizationPlanner {
   /// Lazily fetch BranchProbabilityInfo, independent of BlockFrequencyInfo.
   std::function<const BranchProbabilityInfo &()> GetBPI;
 
-  SmallVector<VPlanPtr, 4> VPlans;
-
-  /// Profitable vector factors.
-  SmallVector<VectorizationFactor, 8> ProfitableVFs;
-
   /// A builder used to construct the current plan.
   VPBuilder Builder;
 
@@ -883,7 +922,8 @@ class LoopVectorizationPlanner {
   ///
   /// TODO: Move to VPlan::cost once the use of LoopVectorizationLegality has
   /// been retired.
-  InstructionCost cost(ElementCount VF, VPRegisterUsage *RU) const;
+  InstructionCost cost(VPlanPlanningContext &Ctx, ElementCount VF,
+                       VPRegisterUsage *RU) const;
 
   /// Precompute costs for certain instructions using the legacy cost model. The
   /// function is used to bring up the VPlan-based cost model to initially avoid
@@ -895,41 +935,27 @@ class LoopVectorizationPlanner {
   LoopVectorizationPlanner(
       Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
       const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
-      std::unique_ptr<LoopVectorizationCostModel> CM,
       VFSelectionContext &Config, PredicatedScalarEvolution &PSE,
       OptimizationRemarkEmitter *ORE,
       std::function<const BranchProbabilityInfo &()> GetBPI);
 
-  ~LoopVectorizationPlanner();
-
-  /// Return the cost model. Must not be called after clearCostModel().
-  LoopVectorizationCostModel &getCostModel() {
-    assert(CM && "Cost model has already been cleared");
-    return *CM;
-  }
-
-  /// Destroy the cost model.
-  void clearCostModel();
-
   /// Build VPlans for the specified \p UserVF and \p UserIC if they are
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
   /// interleaving should be avoided up-front, no plans are generated.
-  void plan(ElementCount UserVF, unsigned UserIC);
-
-  /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
-  /// for each VF.
-  VPlan &getPlanFor(ElementCount VF) const;
+  void plan(VPlanPlanningContext &Ctx, ElementCount UserVF, unsigned UserIC);
 
   /// Compute and return the most profitable vectorization factor and the
   /// corresponding best VPlan. Also collect all profitable VFs in
   /// ProfitableVFs.
-  std::pair<VectorizationFactor, VPlan *> computeBestVF();
+  std::pair<VectorizationFactor, VPlan *>
+  computeBestVF(VPlanPlanningContext &Ctx);
 
   /// \return The desired interleave count.
   /// If interleave count has been specified by metadata it will be returned.
   /// Otherwise, the interleave count is computed and returned. VF and LoopCost
   /// are the selected vectorization factor and the cost of the selected VF.
-  unsigned selectInterleaveCount(ElementCount VF, InstructionCost LoopCost);
+  unsigned selectInterleaveCount(VPlanPlanningContext &Ctx, ElementCount VF,
+                                 InstructionCost LoopCost);
 
   /// Generate the IR code for the vectorized loop captured in VPlan \p BestPlan
   /// according to the best selected \p VF and  \p UF.
@@ -952,16 +978,9 @@ class LoopVectorizationPlanner {
                   EpilogueVectorizationKind::None);
 
 #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
-  void printPlans(raw_ostream &O);
+  void printPlans(const VPlanPlanningContext &Ctx, raw_ostream &O);
 #endif
 
-  /// Look through the existing plans and return true if we have one with
-  /// vectorization factor \p VF.
-  bool hasPlanWithVF(ElementCount VF) const {
-    return any_of(VPlans,
-                  [&](const VPlanPtr &Plan) { return Plan->hasVF(VF); });
-  }
-
   /// Test a \p Predicate on a \p Range of VF's. Return the value of applying
   /// \p Predicate on Range.Start, possibly decreasing Range.End such that the
   /// returned value holds for the entire \p Range.
@@ -974,13 +993,13 @@ class LoopVectorizationPlanner {
   /// Returns nullptr if epilogue vectorization is not supported or not
   /// profitable for the loop. \p ScalarEpilogueAllowed indicates whether the
   /// epilogue lowering policy permits creating a scalar epilogue at all.
-  std::unique_ptr<VPlan> selectBestEpiloguePlan(VPlan &MainPlan,
-                                                ElementCount MainLoopVF,
-                                                unsigned IC,
-                                                bool ScalarEpilogueAllowed);
+  std::unique_ptr<VPlan> selectBestEpiloguePlan(
+      const VPlanPlanningResult &EpilogueCandidates, VPlan &MainPlan,
+      ElementCount MainLoopVF, unsigned IC, bool ScalarEpilogueAllowed);
 
   /// Emit remarks for recipes with invalid costs in the available VPlans.
-  void emitInvalidCostRemarks(OptimizationRemarkEmitter *ORE);
+  void emitInvalidCostRemarks(VPlanPlanningContext &Ctx,
+                              OptimizationRemarkEmitter *ORE);
 
   /// Create a check to \p Plan to see if the vector loop should be executed
   /// based on its trip count.
@@ -1011,7 +1030,7 @@ class LoopVectorizationPlanner {
   /// Build an initial VPlan, with HCFG wrapping the original scalar loop and
   /// scalar transformations applied. Returns null if an initial VPlan cannot
   /// be built.
-  VPlanPtr tryToBuildVPlan1();
+  VPlanPtr tryToBuildVPlan1(VPlanPlanningContext &Ctx);
 
   /// Build a VPlan using VPRecipes according to the information gathered by
   /// Legal and VPlan-based analysis. For outer loops, performs basic recipe
@@ -1021,18 +1040,21 @@ class LoopVectorizationPlanner {
   /// maximum VF for which no plan could be built. Each VPlan is built starting
   /// from a copy of \p InitialPlan, which is a plain CFG VPlan wrapping the
   /// original scalar loop.
-  VPlanPtr tryToBuildVPlan(VPlanPtr InitialPlan, VFRange &Range);
+  VPlanPtr tryToBuildVPlan(VPlanPlanningContext &Ctx, VPlanPtr InitialPlan,
+                           VFRange &Range);
 
   /// Build VPlans for power-of-2 VF's between \p MinVF and \p MaxVF inclusive,
   /// based on \p VPlan1 and according to the information gathered by Legal
   /// when it checked if it is legal to vectorize the loop.
-  void buildVPlans(VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF);
+  void buildVPlans(VPlanPlanningContext &Ctx, VPlan &VPlan1, ElementCount MinVF,
+                   ElementCount MaxVF);
 
   /// Add ComputeReductionResult recipes to the middle block to compute the
   /// final reduction results. Add Select recipes to the latch block when
   /// folding tail, to feed ComputeReductionResult with the last or penultimate
   /// iteration values according to the header mask.
-  void addReductionResultComputation(VPlanPtr &Plan, ElementCount MinVF);
+  void addReductionResultComputation(VPlanPlanningContext &Ctx, VPlanPtr &Plan,
+                                     ElementCount MinVF);
 
   /// Returns true if the per-lane cost of VectorizationFactor A is lower than
   /// that of B.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index ee617c1dbc582..f4a659c16ccb2 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -1520,6 +1520,25 @@ class LoopVectorizationCostModel {
   /// Values to ignore in the cost model when VF > 1.
   SmallPtrSet<const Value *, 16> VecValuesToIgnore;
 };
+
+VPlanPlanningContext::VPlanPlanningContext(
+    std::unique_ptr<LoopVectorizationCostModel> CM)
+    : CM(std::move(CM)) {
+  assert(this->CM && "expected a cost model");
+}
+
+VPlanPlanningContext::~VPlanPlanningContext() = default;
+
+LoopVectorizationCostModel &VPlanPlanningContext::getCostModel() {
+  assert(CM && "cost model has already been cleared");
+  return *CM;
+}
+
+VPlanPlanningResult VPlanPlanningContext::finalize() && {
+  assert(CM && "planning context has already been finalized");
+  CM.reset();
+  return std::move(Result);
+}
 } // end namespace llvm
 
 namespace {
@@ -3162,7 +3181,9 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
 }
 
 void LoopVectorizationPlanner::emitInvalidCostRemarks(
-    OptimizationRemarkEmitter *ORE) {
+    VPlanPlanningContext &Ctx, OptimizationRemarkEmitter *ORE) {
+  auto &VPlans = Ctx.Result.VPlans;
+  LoopVectorizationCostModel &CM = Ctx.getCostModel();
   using RecipeVFPair = std::pair<VPRecipeBase *, ElementCount>;
   SmallVector<RecipeVFPair> InvalidCosts;
   for (const auto &Plan : VPlans) {
@@ -3174,7 +3195,7 @@ void LoopVectorizationPlanner::emitInvalidCostRemarks(
       if (VF.isScalar())
         continue;
 
-      VPCostContext CostCtx(*TLI, *Plan, *CM, Config,
+      VPCostContext CostCtx(*TLI, *Plan, CM, Config,
                             /*ReusePrintingSlotTracker=*/true);
       precomputeCosts(*Plan, VF, CostCtx);
       auto Iter = vp_depth_first_deep(Plan->getVectorLoopRegion()->getEntry());
@@ -3513,8 +3534,9 @@ bool VFSelectionContext::isEpilogueVectorizationProfitable(
 }
 
 std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
-    VPlan &MainPlan, ElementCount MainLoopVF, unsigned IC,
-    bool ScalarEpilogueAllowed) {
+    const VPlanPlanningResult &EpilogueCandidates, VPlan &MainPlan,
+    ElementCount MainLoopVF, unsigned IC, bool ScalarEpilogueAllowed) {
+  const auto &ProfitableVFs = EpilogueCandidates.ProfitableVFs;
   if (!EnableEpilogueVectorization) {
     LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization is disabled.\n");
     return nullptr;
@@ -3554,9 +3576,10 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     }
 
     LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization factor is forced.\n");
-    if (hasPlanWithVF(EpilogueVectorizationForceVF)) {
+    if (EpilogueCandidates.hasPlanWithVF(EpilogueVectorizationForceVF)) {
       std::unique_ptr<VPlan> Clone(
-          getPlanFor(EpilogueVectorizationForceVF).duplicate());
+          EpilogueCandidates.getPlanFor(EpilogueVectorizationForceVF)
+              .duplicate());
       Clone->setVF(EpilogueVectorizationForceVF);
       return Clone;
     }
@@ -3647,10 +3670,10 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
   VPlan *BestPlan = nullptr;
   for (auto &NextVF : ProfitableVFs) {
     // Skip candidate VFs without a corresponding VPlan.
-    if (!hasPlanWithVF(NextVF.Width))
+    if (!EpilogueCandidates.hasPlanWithVF(NextVF.Width))
       continue;
 
-    VPlan &CurrentPlan = getPlanFor(NextVF.Width);
+    VPlan &CurrentPlan = EpilogueCandidates.getPlanFor(NextVF.Width);
     ElementCount EffectiveVF = GetEffectiveVF(CurrentPlan, NextVF.Width);
     // Skip fixed vector VFs > than the estimated runtime VF, or any VF > than
     // the VF of the main loop.
@@ -3696,9 +3719,11 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
 }
 
 unsigned
-LoopVectorizationPlanner::selectInterleaveCount(ElementCount VF,
+LoopVectorizationPlanner::selectInterleaveCount(VPlanPlanningContext &Ctx,
+                                                ElementCount VF,
                                                 InstructionCost LoopCost) {
-  VPlan &Plan = getPlanFor(VF);
+  LoopVectorizationCostModel &CM = Ctx.getCostModel();
+  VPlan &Plan = Ctx.Result.getPlanFor(VF);
   // -- The interleave heuristics --
   // We interleave the loop in order to expose ILP and reduce the loop overhead.
   // There are many micro-architectural considerations that we can't predict
@@ -3716,7 +3741,7 @@ LoopVectorizationPlanner::selectInterleaveCount(ElementCount VF,
   // Do not interleave tail-folded loops, as the overhead of multiple
   // instructions to calculate the predicate is likely not beneficial.
   // If an epilogue is not allowed for any other reason, do not interleave.
-  if (!CM->isEpilogueAllowed())
+  if (!CM.isEpilogueAllowed())
     return 1;
 
   if (any_of(Plan.getVectorLoopRegion()->getEntryBasicBlock()->phis(),
@@ -3750,10 +3775,11 @@ LoopVectorizationPlanner::selectInterleaveCount(ElementCount VF,
   // then we calculate the cost of VF here.
   if (LoopCost == 0) {
     if (VF.isScalar())
-      LoopCost = CM->expectedCost(VF);
+      LoopCost = CM.expectedCost(VF);
     else
-      LoopCost = cost(VF, &R);
-    assert(LoopCost.isValid() && "Expected to have chosen a VF with valid cost");
+      LoopCost = cost(Ctx, VF, &R);
+    assert(LoopCost.isValid() &&
+           "Expected to have chosen a VF with valid cost");
 
     // Loop body is free and there is no need for interleaving.
     if (LoopCost == 0)
@@ -3833,7 +3859,7 @@ LoopVectorizationPlanner::selectInterleaveCount(ElementCount VF,
   auto BestKnownTC =
       getSmallBestKnownTC(PSE, OrigLoop,
                           /*CanUseConstantMax=*/true,
-                          /*CanExcludeZeroTrips=*/CM->isEpilogueAllowed());
+                          /*CanExcludeZeroTrips=*/CM.isEpilogueAllowed());
 
   // For fixed length VFs treat a scalable trip count as unknown.
   if (BestKnownTC && (BestKnownTC->isFixed() || VF.isScalable())) {
@@ -5345,11 +5371,14 @@ void LoopVectorizationCostModel::collectValuesToIgnore() {
   }
 }
 
-void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
-  CM->collectValuesToIgnore();
-  Config.collectElementTypesForWidening(&CM->ValuesToIgnore);
+void LoopVectorizationPlanner::plan(VPlanPlanningContext &Ctx,
+                                    ElementCount UserVF, unsigned UserIC) {
+  LoopVectorizationCostModel &CM = Ctx.getCostModel();
+  auto &VPlans = Ctx.Result.VPlans;
+  CM.collectValuesToIgnore();
+  Config.collectElementTypesForWidening(&CM.ValuesToIgnore);
 
-  FixedScalableVFPair MaxFactors = CM->computeMaxVF(UserVF, UserIC);
+  FixedScalableVFPair MaxFactors = CM.computeMaxVF(UserVF, UserIC);
   if (!MaxFactors) // Cases that should not to be vectorized nor interleaved.
     return;
 
@@ -5360,7 +5389,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   if (MaxFactors.FixedVF.isVector() || MaxFactors.ScalableVF.isVector())
     Legal->collectUnitStridePredicates();
 
-  auto VPlan1 = tryToBuildVPlan1();
+  auto VPlan1 = tryToBuildVPlan1(Ctx);
   if (!VPlan1)
     return;
 
@@ -5369,8 +5398,8 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
     // plan for that VF only.
     ElementCount VF =
         MaxFactors.FixedVF ? MaxFactors.FixedVF : MaxFactors.ScalableVF;
-    buildVPlans(*VPlan1, VF, VF);
-    LLVM_DEBUG(printPlans(dbgs()));
+    buildVPlans(Ctx, *VPlan1, VF, VF);
+    LLVM_DEBUG(printPlans(Ctx, dbgs()));
     return;
   }
 
@@ -5379,20 +5408,20 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   Config.computeMinimalBitwidths();
 
   // Invalidate interleave groups if all blocks of loop will be predicated.
-  if (CM->blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
+  if (CM.blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
       !useMaskedInterleavedAccesses(TTI)) {
     LLVM_DEBUG(
         dbgs()
         << "LV: Invalidate all interleaved groups due to fold-tail by masking "
            "which requires masked-interleaved support.\n");
-    if (CM->InterleaveInfo.invalidateGroups())
+    if (CM.InterleaveInfo.invalidateGroups())
       // Invalidating interleave groups also requires invalidating all decisions
       // based on them, which includes widening decisions and uniform and scalar
       // values.
-      CM->invalidateCostModelingDecisions();
+      CM.invalidateCostModelingDecisions();
   }
 
-  if (CM->foldTailByMasking())
+  if (CM.foldTailByMasking())
     Legal->prepareToFoldTailByMasking();
 
   ElementCount MaxUserVF =
@@ -5407,21 +5436,21 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
              "VF needs to be a power of two");
       // Collect the instructions (and their associated costs) that will be more
       // profitable to scalarize.
-      CM->collectNonVectorizedAndSetWideningDecisions(UserVF);
-      buildVPlans(*VPlan1, UserVF, UserVF);
+      CM.collectNonVectorizedAndSetWideningDecisions(UserVF);
+      buildVPlans(Ctx, *VPlan1, UserVF, UserVF);
       ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
       if (EpilogueUserVF.isVector() &&
           ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
-        CM->collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
-        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF);
+        CM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
+        buildVPlans(Ctx, *VPlan1, EpilogueUserVF, EpilogueUserVF);
       }
       if (!VPlans.empty() && VPlans.front()->getSingleVF() == UserVF) {
         // For scalar VF, skip VPlan cost check as VPlan cost is designed for
         // vector VFs only.
         if (UserVF.isScalar() ||
-            cost(UserVF, /*RU=*/nullptr).isValid()) {
+            cost(Ctx, UserVF, /*RU=*/nullptr).isValid()) {
           LLVM_DEBUG(dbgs() << "LV: Using user VF " << UserVF << ".\n");
-          LLVM_DEBUG(printPlans(dbgs()));
+          LLVM_DEBUG(printPlans(Ctx, dbgs()));
           return;
         }
       }
@@ -5442,13 +5471,14 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
 
   for (const auto &VF : VFCandidates) {
     // Collect Uniform and Scalar instructions after vectorization with VF.
-    CM->collectNonVectorizedAndSetWideningDecisions(VF);
+    CM.collectNonVectorizedAndSetWideningDecisions(VF);
   }
 
-  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF);
-  buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF);
+  buildVPlans(Ctx, *VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF);
+  buildVPlans(Ctx, *VPlan1, ElementCount::getScalable(1),
+              MaxFactors.ScalableVF);
 
-  LLVM_DEBUG(printPlans(dbgs()));
+  LLVM_DEBUG(printPlans(Ctx, dbgs()));
 }
 
 VPCostContext::VPCostContext(const TargetLibraryInfo &TLI, const VPlan &Plan,
@@ -5594,10 +5624,11 @@ getRecordedExecutionFrequency(const VPBasicBlock *VPBB) {
 }
 #endif
 
-InstructionCost LoopVectorizationPlanner::cost(ElementCount VF,
+InstructionCost LoopVectorizationPlanner::cost(VPlanPlanningContext &Ctx,
+                                               ElementCount VF,
                                                VPRegisterUsage *RU) const {
-  VPlan &Plan = getPlanFor(VF);
-  VPCostContext CostCtx(*TLI, Plan, *CM, Config,
+  VPlan &Plan = Ctx.Result.getPlanFor(VF);
+  VPCostContext CostCtx(*TLI, Plan, Ctx.getCostModel(), Config,
                         /*ReusePrintingSlotTracker=*/true);
   InstructionCost Cost = precomputeCosts(Plan, VF, CostCtx);
 
@@ -5633,7 +5664,10 @@ InstructionCost LoopVectorizationPlanner::cost(ElementCount VF,
 }
 
 std::pair<VectorizationFactor, VPlan *>
-LoopVectorizationPlanner::computeBestVF() {
+LoopVectorizationPlanner::computeBestVF(VPlanPlanningContext &Ctx) {
+  auto &VPlans = Ctx.Result.VPlans;
+  auto &ProfitableVFs = Ctx.Result.ProfitableVFs;
+  LoopVectorizationCostModel &CM = Ctx.getCostModel();
   if (VPlans.empty())
     return {VectorizationFactor::Disabled(), nullptr};
   // If there is a single VPlan with a single VF, return it directly.
@@ -5643,13 +5677,15 @@ LoopVectorizationPlanner::computeBestVF() {
   if (VPlans.size() == 1) {
     // For outer loops, the plan has a single vector VF determined by the
     // heuristic.
-    assert((FirstPlan.hasScalarVFOnly() || hasPlanWithVF(UserVF) ||
+    assert((FirstPlan.hasScalarVFOnly() ||
+            Ctx.Result.hasPlanWithVF(UserVF) ||
             FirstPlan.isOuterLoop()) &&
            "must have a single scalar VF, UserVF or an outer loop");
     return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
   }
 
-  if (hasPlanWithVF(UserVF) && hasForcedEpilogueVF() && VPlans.size() == 2) {
+  if (Ctx.Result.hasPlanWithVF(UserVF) && hasForcedEpilogueVF() &&
+      VPlans.size() == 2) {
     assert(VPlans[0]->getSingleVF() == UserVF &&
            "expected second plan to be for the forced UserVF");
     assert(VPlans[1]->getSingleVF() == EpilogueVectorizationForceVF &&
@@ -5672,7 +5708,7 @@ LoopVectorizationPlanner::computeBestVF() {
          "More than a single plan/VF w/o any plan having scalar VF");
 
   // TODO: Compute scalar cost using VPlan-based cost model.
-  InstructionCost ScalarCost = CM->expectedCost(ScalarVF);
+  InstructionCost ScalarCost = CM.expectedCost(ScalarVF);
   LLVM_DEBUG(dbgs() << "LV: Scalar loop costs: " << ScalarCost << ".\n");
   VectorizationFactor ScalarFactor(ScalarVF, ScalarCost, ScalarCost);
   VectorizationFactor BestFactor = ScalarFactor;
@@ -5731,7 +5767,7 @@ LoopVectorizationPlanner::computeBestVF() {
       }
 
       InstructionCost Cost =
-          cost(VF, ConsiderRegPressure ? &RUs[I] : nullptr);
+          cost(Ctx, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
       VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
 
       if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
@@ -5757,15 +5793,11 @@ LoopVectorizationPlanner::computeBestVF() {
 LoopVectorizationPlanner::LoopVectorizationPlanner(
     Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
     const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
-    std::unique_ptr<LoopVectorizationCostModel> CM, VFSelectionContext &Config,
-    PredicatedScalarEvolution &PSE, OptimizationRemarkEmitter *ORE,
+    VFSelectionContext &Config, PredicatedScalarEvolution &PSE,
+    OptimizationRemarkEmitter *ORE,
     std::function<const BranchProbabilityInfo &()> GetBPI)
     : OrigLoop(L), LI(LI), DT(DT), TLI(TLI), TTI(TTI), Legal(Legal),
-      CM(std::move(CM)), Config(Config), PSE(PSE), ORE(ORE), GetBPI(GetBPI) {}
-
-LoopVectorizationPlanner::~LoopVectorizationPlanner() = default;
-
-void LoopVectorizationPlanner::clearCostModel() { CM.reset(); }
+      Config(Config), PSE(PSE), ORE(ORE), GetBPI(GetBPI) {}
 
 DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
     ElementCount BestVF, unsigned BestUF, VPlan &BestVPlan,
@@ -6414,7 +6446,8 @@ static bool verifyExecutionFrequenciesMatchBFI(VPlan &Plan, Loop *OrigLoop,
 }
 #endif
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(VPlanPlanningContext &Ctx) {
+  LoopVectorizationCostModel &CM = Ctx.getCostModel();
   bool IsInnerLoop = OrigLoop->isInnermost();
 
   // Set up loop versioning for inner loops with memory runtime checks.
@@ -6447,7 +6480,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
   RUN_VPLAN_PASS(VPlanTransforms::removeDeadRecipes, *VPlan0);
   if (IsInnerLoop) {
     RUN_VPLAN_PASS(VPlanTransforms::recordExecutionFrequencies, *VPlan0);
-    assert(verifyExecutionFrequenciesMatchBFI(*VPlan0, OrigLoop, LI, *CM) &&
+    assert(verifyExecutionFrequenciesMatchBFI(*VPlan0, OrigLoop, LI, CM) &&
            "execution frequencies do not match the loop's block frequencies");
   }
 
@@ -6470,8 +6503,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
       Config.getHints().getForce() == LoopVectorizeHints::FK_Enabled;
   bool OptForSize =
       !ForceVectorization &&
-      (CM->EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
-       CM->EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
+      (CM.EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
+       CM.EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
   unsigned SCEVCheckThreshold = ForceVectorization
                                     ? PragmaVectorizeSCEVCheckThreshold
                                     : VectorizeSCEVCheckThreshold;
@@ -6503,7 +6536,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
 
   RUN_VPLAN_PASS(VPlanTransforms::createLoopRegions, *VPlan0,
                  getDebugLocFromInstOrOperands(Legal->getPrimaryInduction()));
-  if (CM->foldTailByMasking())
+  if (CM.foldTailByMasking())
     RUN_VPLAN_PASS(VPlanTransforms::foldTailByMasking, *VPlan0);
 
   RUN_VPLAN_PASS(VPlanTransforms::introduceMasksAndLinearize, *VPlan0);
@@ -6511,16 +6544,19 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
   return VPlan0;
 }
 
-void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
+void LoopVectorizationPlanner::buildVPlans(VPlanPlanningContext &Ctx,
+                                           VPlan &VPlan1, ElementCount MinVF,
                                            ElementCount MaxVF) {
+  LoopVectorizationCostModel &CM = Ctx.getCostModel();
+  auto &VPlans = Ctx.Result.VPlans;
   if (ElementCount::isKnownGT(MinVF, MaxVF))
     return;
 
   auto MaxVFTimes2 = MaxVF * 2;
   for (ElementCount VF = MinVF; ElementCount::isKnownLT(VF, MaxVFTimes2);) {
     VFRange SubRange = {VF, MaxVFTimes2};
-    auto Plan =
-        tryToBuildVPlan(std::unique_ptr<VPlan>(VPlan1.duplicate()), SubRange);
+    auto Plan = tryToBuildVPlan(Ctx, std::unique_ptr<VPlan>(VPlan1.duplicate()),
+                                SubRange);
     VF = SubRange.End;
 
     if (!Plan)
@@ -6533,7 +6569,7 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
                    Config.getMinimalBitwidths());
     RUN_VPLAN_PASS(VPlanTransforms::optimize, *Plan);
     // TODO: try to put addExplicitVectorLength close to addActiveLaneMask
-    if (CM->foldTailWithEVL()) {
+    if (CM.foldTailWithEVL()) {
       RUN_VPLAN_PASS(VPlanTransforms::addExplicitVectorLength, *Plan,
                      Config.getMaxSafeElements());
       RUN_VPLAN_PASS(VPlanTransforms::optimizeEVLMasks, *Plan);
@@ -6543,7 +6579,7 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
             RUN_VPLAN_PASS(VPlanTransforms::narrowInterleaveGroups, *Plan, TTI))
       VPlans.push_back(std::move(P));
 
-    TailFoldingStyle Style = CM->getTailFoldingStyle();
+    TailFoldingStyle Style = CM.getTailFoldingStyle();
     RUN_VPLAN_PASS(VPlanTransforms::materializeHeaderMask, *Plan,
                    useActiveLaneMask(Style),
                    useActiveLaneMaskForControlFlow(Style));
@@ -6554,8 +6590,10 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
   }
 }
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPlanningContext &Ctx,
+                                                   VPlanPtr Plan,
                                                    VFRange &Range) {
+  LoopVectorizationCostModel &CM = Ctx.getCostModel();
 
   // For outer loops, the plan only needs basic recipe conversion and induction
   // live-out optimization; the full inner-loop recipe building below does not
@@ -6581,8 +6619,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
 
   bool RequiresScalarEpilogueCheck =
       LoopVectorizationPlanner::getDecisionAndClampRange(
-          [this](ElementCount VF) {
-            return !CM->requiresScalarEpilogue(VF.isVector());
+          [&CM](ElementCount VF) {
+            return !CM.requiresScalarEpilogue(VF.isVector());
           },
           Range);
   // Update the branch in the middle block if a scalar epilogue is required.
@@ -6600,9 +6638,9 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // TODO: Consider using getDecisionAndClampRange here to split up VPlans.
   bool IVUpdateMayOverflow = false;
   for (ElementCount VF : Range)
-    IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(CM.get(), VF);
+    IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(&CM, VF);
 
-  TailFoldingStyle Style = CM->getTailFoldingStyle();
+  TailFoldingStyle Style = CM.getTailFoldingStyle();
   // Use NUW for the induction increment if we proved that it won't overflow in
   // the vector loop or when not folding the tail. In the later case, we know
   // that the canonical induction increment will not overflow as the vector trip
@@ -6629,10 +6667,10 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // placeholders for its members' Recipes which we'll be replacing with a
   // single VPInterleaveRecipe.
   for (InterleaveGroup<Instruction> *IG :
-       CM->InterleaveInfo.getInterleaveGroups()) {
-    auto ApplyIG = [IG, this](ElementCount VF) -> bool {
+       CM.InterleaveInfo.getInterleaveGroups()) {
+    auto ApplyIG = [IG, &CM](ElementCount VF) -> bool {
       bool Result = (VF.isVector() && // Query is illegal for VF == 1
-                     CM->getWideningDecision(IG->getInsertPos(), VF) ==
+                     CM.getWideningDecision(IG->getInsertPos(), VF) ==
                          LoopVectorizationCostModel::CM_Interleave);
       // For scalable vectors, the interleave factors must be <= 8 since we
       // require the (de)interleaveN intrinsics instead of shufflevectors.
@@ -6649,12 +6687,12 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // Construct wide recipes and apply predication for original scalar
   // VPInstructions in the loop.
   // ---------------------------------------------------------------------------
-  VPRecipeBuilder RecipeBuilder(*Plan, Legal, *CM, Builder);
+  VPRecipeBuilder RecipeBuilder(*Plan, Legal, CM, Builder);
 
   RUN_VPLAN_PASS(VPlanTransforms::createInLoopReductionRecipes, *Plan,
                  Range.Start);
 
-  VPCostContext CostCtx(*TLI, *Plan, *CM, Config);
+  VPCostContext CostCtx(*TLI, *Plan, CM, Config);
 
   RUN_VPLAN_PASS(VPlanTransforms::makeMemOpWideningDecisions, *Plan, Range,
                  RecipeBuilder, CostCtx);
@@ -6725,7 +6763,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // bring the VPlan to its final state.
   // ---------------------------------------------------------------------------
 
-  addReductionResultComputation(Plan, Range.Start);
+  addReductionResultComputation(Ctx, Plan, Range.Start);
 
   // Optimize FindIV reductions to use sentinel-based approach when possible.
   RUN_VPLAN_PASS(VPlanTransforms::optimizeFindIVReductions, *Plan, PSE,
@@ -6761,7 +6799,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // for this VPlan, replace the Recipes widening its memory instructions with a
   // single VPInterleaveRecipe at its insertion point.
   RUN_VPLAN_PASS(VPlanTransforms::createInterleaveGroups, *Plan,
-                 InterleaveGroups, CM->isEpilogueAllowed());
+                 InterleaveGroups, CM.isEpilogueAllowed());
 
   // Convert memory recipes to strided access recipes if the strided access is
   // legal and profitable.
@@ -6778,7 +6816,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
 
   RUN_VPLAN_PASS(VPlanTransforms::dropPoisonGeneratingRecipes, *Plan);
 
-  if (CM->maskPartialAliasing())
+  if (CM.maskPartialAliasing())
     RUN_VPLAN_PASS(VPlanTransforms::attachAliasMaskToHeaderMask, *Plan);
 
   assert(verifyVPlanIsValid(*Plan) && "VPlan is invalid");
@@ -6786,7 +6824,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
 }
 
 void LoopVectorizationPlanner::addReductionResultComputation(
-    VPlanPtr &Plan, ElementCount MinVF) {
+    VPlanPlanningContext &Ctx, VPlanPtr &Plan, ElementCount MinVF) {
+  LoopVectorizationCostModel &CM = Ctx.getCostModel();
   using namespace VPlanPatternMatch;
   VPRegionBlock *VectorLoopRegion = Plan->getVectorLoopRegion();
   VPBasicBlock *MiddleVPBB = Plan->getMiddleBlock();
@@ -6829,7 +6868,7 @@ void LoopVectorizationPlanner::addReductionResultComputation(
 
     // Remove the predicated select if the target doesn't want it.
     VPValue *V;
-    if (!CM->usePredicatedReductionSelect(RecurrenceKind) &&
+    if (!CM.usePredicatedReductionSelect(RecurrenceKind) &&
         match(PhiR->getBackedgeValue(),
               m_Select(m_Specific(HeaderMask), m_VPValue(V), m_Specific(PhiR))))
       PhiR->setBackedgeValue(V);
@@ -7868,15 +7907,14 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   // Use the cost model.
   VFSelectionContext Config(*TTI, &LVL, L, *F, PSE, DB, ORE, &Hints,
                             OptForSize);
+  VPlanPlanningContext PlanningCtx(std::make_unique<LoopVectorizationCostModel>(
+      SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE, GetBFI, F, IAI, Config));
   // Use the planner for vectorization.
-  LoopVectorizationPlanner LVP(
-      L, LI, DT, TLI, *TTI, &LVL,
-      std::make_unique<LoopVectorizationCostModel>(
-          SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE, GetBFI, F, IAI, Config),
-      Config, PSE, ORE, GetBPI);
-
-  EpilogueLowering EpilogueTailLoweringStatus =
-      getEpilogueTailLowering(LVP.getCostModel(), L, ORE, LVL, Hints, TTI);
+  LoopVectorizationPlanner LVP(L, LI, DT, TLI, *TTI, &LVL, Config, PSE, ORE,
+                               GetBPI);
+
+  EpilogueLowering EpilogueTailLoweringStatus = getEpilogueTailLowering(
+      PlanningCtx.getCostModel(), L, ORE, LVL, Hints, TTI);
   if (EpilogueTailLoweringStatus ==
       EpilogueLowering::CM_EpilogueNotNeededFoldTail) {
     // TODO: Apply tail-folding on the vectorized epilogue loop.
@@ -7898,8 +7936,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     UserIC = 1;
 
   // Plan how to best vectorize.
-  LVP.plan(UserVF, UserIC);
-  auto [VF, BestPlanPtr] = LVP.computeBestVF();
+  LVP.plan(PlanningCtx, UserVF, UserIC);
+  auto [VF, BestPlanPtr] = LVP.computeBestVF(PlanningCtx);
   unsigned IC = 1;
 
   // For VPlan build stress testing of outer loops, bail after plan
@@ -7908,16 +7946,16 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     return false;
 
   if (IsInnerLoop && ORE->allowExtraAnalysis(LV_NAME))
-    LVP.emitInvalidCostRemarks(ORE);
+    LVP.emitInvalidCostRemarks(PlanningCtx, ORE);
 
-  assert((IsInnerLoop || !LVP.getCostModel().maskPartialAliasing()) &&
+  assert((IsInnerLoop || !PlanningCtx.getCostModel().maskPartialAliasing()) &&
          "Did not expect to alias-mask outer loop");
 
   GeneratedRTChecks Checks(PSE, DT, LI, TTI, Config.CostKind,
-                           LVP.getCostModel().maskPartialAliasing());
-  if (IsInnerLoop && LVP.hasPlanWithVF(VF.Width)) {
+                           PlanningCtx.getCostModel().maskPartialAliasing());
+  if (IsInnerLoop && PlanningCtx.getResult().hasPlanWithVF(VF.Width)) {
     // Select the interleave count.
-    IC = LVP.selectInterleaveCount(VF.Width, VF.Cost);
+    IC = LVP.selectInterleaveCount(PlanningCtx, VF.Width, VF.Cost);
 
     unsigned SelectedIC = std::max(IC, UserIC);
     //  Optimistically generate runtime checks if they are needed. Drop them if
@@ -7940,7 +7978,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     // Check if it is profitable to vectorize with runtime checks.
     bool ForceVectorization =
         Hints.getForce() == LoopVectorizeHints::FK_Enabled;
-    VPCostContext CostCtx(*TLI, *BestPlanPtr, LVP.getCostModel(), Config,
+    VPCostContext CostCtx(*TLI, *BestPlanPtr, PlanningCtx.getCostModel(),
+                          Config,
                           /*ReusePrintingSlotTracker=*/true);
     if (!ForceVectorization &&
         !isOutsideLoopWorkProfitable(Checks, VF, L, PSE, CostCtx, *BestPlanPtr,
@@ -7977,7 +8016,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                   "Ignoring user-specified interleave count due to possibly "
                   "unsafe dependencies in the loop."};
     InterleaveLoop = false;
-  } else if (!LVP.hasPlanWithVF(VF.Width) && UserIC > 1) {
+  } else if (!PlanningCtx.getResult().hasPlanWithVF(VF.Width) && UserIC > 1) {
     // Tell the user interleaving was avoided up-front, despite being explicitly
     // requested.
     LLVM_DEBUG(dbgs() << "LV: Ignoring UserIC, because vectorization and "
@@ -8023,7 +8062,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   // Override IC if user provided an interleave count.
   IC = UserIC > 0 ? UserIC : IC;
 
-  if (LVP.getCostModel().maskPartialAliasing()) {
+  if (PlanningCtx.getCostModel().maskPartialAliasing()) {
     LLVM_DEBUG(
         dbgs()
         << "LV: Not interleaving due to partial aliasing vectorization.\n");
@@ -8103,16 +8142,16 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   // Whether a scalar epilogue may be created is decided by the epilogue
   // lowering policy.
   // TODO: Also move check to be based on VPlan.
-  bool ScalarEpilogueAllowed = LVP.getCostModel().isEpilogueAllowed();
-
-  // Destroy the cost model before executing any plan, so that code generation
-  // cannot rely on cost-modeling decisions.
-  LVP.clearCostModel();
+  bool ScalarEpilogueAllowed = PlanningCtx.getCostModel().isEpilogueAllowed();
 
   VPlan &BestPlan = *BestPlanPtr;
+  // Finalize planning and destroy the cost model before executing any plan, so
+  // that code generation cannot rely on cost-modeling decisions.
+  VPlanPlanningResult PlanningResult = std::move(PlanningCtx).finalize();
+
   // Consider vectorizing the epilogue too if it's profitable.
-  std::unique_ptr<VPlan> EpiPlan =
-      LVP.selectBestEpiloguePlan(BestPlan, VF.Width, IC, ScalarEpilogueAllowed);
+  std::unique_ptr<VPlan> EpiPlan = LVP.selectBestEpiloguePlan(
+      PlanningResult, BestPlan, VF.Width, IC, ScalarEpilogueAllowed);
   bool HasBranchWeights =
       hasBranchWeightMD(*L->getLoopLatch()->getTerminator());
   if (EpiPlan) {
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.cpp b/llvm/lib/Transforms/Vectorize/VPlan.cpp
index 80d2704fa6139..9192d53fca2a1 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlan.cpp
@@ -1679,7 +1679,7 @@ VPBuilder::createConsecutiveVectorPointer(VPValue *Ptr, Type *SourceElementTy,
   return createVectorPointer(Ptr, SourceElementTy, StrideOne, Flags, DL);
 }
 
-VPlan &LoopVectorizationPlanner::getPlanFor(ElementCount VF) const {
+VPlan &VPlanPlanningResult::getPlanFor(ElementCount VF) const {
   assert(count_if(VPlans,
                   [VF](const VPlanPtr &Plan) { return Plan->hasVF(VF); }) ==
              1 &&
@@ -1692,6 +1692,11 @@ VPlan &LoopVectorizationPlanner::getPlanFor(ElementCount VF) const {
   llvm_unreachable("No plan found!");
 }
 
+bool VPlanPlanningResult::hasPlanWithVF(ElementCount VF) const {
+  return any_of(VPlans,
+                [VF](const VPlanPtr &Plan) { return Plan->hasVF(VF); });
+}
+
 static void addRuntimeUnrollDisableMetaData(Loop *L) {
   SmallVector<Metadata *, 4> MDs;
   // Reserve first location for self reference to the LoopID metadata node.
@@ -1834,7 +1839,9 @@ void LoopVectorizationPlanner::updateLoopMetadataAndProfileInfo(
 }
 
 #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
-void LoopVectorizationPlanner::printPlans(raw_ostream &O) {
+void LoopVectorizationPlanner::printPlans(const VPlanPlanningContext &Ctx,
+                                          raw_ostream &O) {
+  const auto &VPlans = Ctx.Result.VPlans;
   if (VPlans.empty()) {
     O << "LV: No VPlans built.\n";
     return;



More information about the llvm-commits mailing list