[llvm] [LV] codegen for tail-folded epilogue loop (PR #208764)

Hassnaa Hamdi via llvm-commits llvm-commits at lists.llvm.org
Mon Aug 24 08:48:29 PDT 2026


https://github.com/hassnaaHamdi updated https://github.com/llvm/llvm-project/pull/208764

>From aabd26d88babb0167a653ee70ff260ecc8eae8ea Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Fri, 10 Jul 2026 16:06:34 +0100
Subject: [PATCH 01/11] [LV][EpilogueTailFolding] hack patch to start by
 codegen

---
 .../Vectorize/LoopVectorizationPlanner.h      |  24 +--
 .../Transforms/Vectorize/LoopVectorize.cpp    | 186 +++++++++++++-----
 llvm/lib/Transforms/Vectorize/VPlan.cpp       |  25 ++-
 llvm/lib/Transforms/Vectorize/VPlan.h         |   9 +
 .../Vectorize/VPlanConstruction.cpp           |  18 +-
 .../AArch64/fold-epilogue-tail.ll             |  63 ++++++
 .../LoopVectorize/fold-epilogue-tail.ll       |  11 +-
 7 files changed, 269 insertions(+), 67 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index d488607a0c7dc..cad4eb60674c5 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -905,16 +905,17 @@ class LoopVectorizationPlanner {
   /// Build VPlans for the specified \p UserVF and \p UserIC if they are
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
   /// interleaving should be avoided up-front, no plans are generated.
-  void plan(ElementCount UserVF, unsigned UserIC);
+  void plan(ElementCount UserVF, unsigned UserIC, bool IsEpilogueTFEnabled);
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
   /// for each VF.
-  VPlan &getPlanFor(ElementCount VF) const;
+  VPlan &getPlanFor(ElementCount VF, bool TF) const;
 
   /// Compute and return the most profitable vectorization factor and the
   /// corresponding best VPlan. Also collect all profitable VFs in
   /// ProfitableVFs.
-  std::pair<VectorizationFactor, VPlan *> computeBestVF();
+  std::pair<VectorizationFactor, VPlan *>
+  computeBestVF(bool IsEpilogueTFEnabled);
 
   /// \return The desired interleave count.
   /// If interleave count has been specified by metadata it will be returned.
@@ -940,6 +941,7 @@ class LoopVectorizationPlanner {
   DenseMap<const SCEV *, Value *>
   executePlan(ElementCount VF, unsigned UF, VPlan &BestPlan,
               InnerLoopVectorizer &LB, DominatorTree *DT,
+              bool IsEpilogueTFEnabled,
               EpilogueVectorizationKind EpilogueVecKind =
                   EpilogueVectorizationKind::None);
 
@@ -949,10 +951,7 @@ class LoopVectorizationPlanner {
 
   /// Look through the existing plans and return true if we have one with
   /// vectorization factor \p VF.
-  bool hasPlanWithVF(ElementCount VF) const {
-    return any_of(VPlans,
-                  [&](const VPlanPtr &Plan) { return Plan->hasVF(VF); });
-  }
+  bool hasPlanWithVF(ElementCount VF, bool TF) const;
 
   /// Test a \p Predicate on a \p Range of VF's. Return the value of applying
   /// \p Predicate on Range.Start, possibly decreasing Range.End such that the
@@ -965,8 +964,10 @@ class LoopVectorizationPlanner {
   /// VF narrowed to the chosen factor. The returned plan is a duplicate.
   /// Returns nullptr if epilogue vectorization is not supported or not
   /// profitable for the loop.
-  std::unique_ptr<VPlan>
-  selectBestEpiloguePlan(VPlan &MainPlan, ElementCount MainLoopVF, unsigned IC);
+  std::unique_ptr<VPlan> selectBestEpiloguePlan(VPlan &MainPlan,
+                                                ElementCount MainLoopVF,
+                                                unsigned IC,
+                                                bool IsEpilogueTFEnabled);
 
   /// Emit remarks for recipes with invalid costs in the available VPlans.
   void emitInvalidCostRemarks(OptimizationRemarkEmitter *ORE);
@@ -1004,7 +1005,7 @@ class LoopVectorizationPlanner {
   /// Build an initial VPlan, with HCFG wrapping the original scalar loop and
   /// scalar transformations applied. Returns null if an initial VPlan cannot
   /// be built.
-  VPlanPtr tryToBuildVPlan1();
+  VPlanPtr tryToBuildVPlan1(bool IsEpilogueTFEnabled);
 
   /// Build a VPlan using VPRecipes according to the information gathered by
   /// Legal and VPlan-based analysis. For outer loops, performs basic recipe
@@ -1019,7 +1020,8 @@ class LoopVectorizationPlanner {
   /// Build VPlans for power-of-2 VF's between \p MinVF and \p MaxVF inclusive,
   /// based on \p VPlan1 and according to the information gathered by Legal
   /// when it checked if it is legal to vectorize the loop.
-  void buildVPlans(VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF);
+  void buildVPlans(VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF,
+                   bool IsEpilogueTFEnabled);
 
   /// Add ComputeReductionResult recipes to the middle block to compute the
   /// final reduction results. Add Select recipes to the latch block when
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 3c65eda187a6e..b57640797deb4 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3151,6 +3151,8 @@ void LoopVectorizationPlanner::emitInvalidCostRemarks(
   using RecipeVFPair = std::pair<VPRecipeBase *, ElementCount>;
   SmallVector<RecipeVFPair> InvalidCosts;
   for (const auto &Plan : VPlans) {
+    if (!Plan->isCompatibleWithTF(CM.foldTailByMasking()))
+      continue;
     for (ElementCount VF : Plan->vectorFactors()) {
       // The VPlan-based cost model is designed for computing vector cost.
       // Querying VPlan-based cost model with a scarlar VF will cause some
@@ -3453,7 +3455,8 @@ bool LoopVectorizationCostModel::isEpilogueVectorizationProfitable(
 }
 
 std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
-    VPlan &MainPlan, ElementCount MainLoopVF, unsigned IC) {
+    VPlan &MainPlan, ElementCount MainLoopVF, unsigned IC,
+    bool IsEpilogueTFEnabled) {
   if (!EnableEpilogueVectorization) {
     LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization is disabled.\n");
     return nullptr;
@@ -3493,9 +3496,10 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     }
 
     LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization factor is forced.\n");
-    if (hasPlanWithVF(EpilogueVectorizationForceVF)) {
+    if (hasPlanWithVF(EpilogueVectorizationForceVF, CM.foldTailByMasking())) {
       std::unique_ptr<VPlan> Clone(
-          getPlanFor(EpilogueVectorizationForceVF).duplicate());
+          getPlanFor(EpilogueVectorizationForceVF, CM.foldTailByMasking())
+              .duplicate());
       Clone->setVF(EpilogueVectorizationForceVF);
       return Clone;
     }
@@ -3586,10 +3590,12 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
   VPlan *BestPlan = nullptr;
   for (auto &NextVF : ProfitableVFs) {
     // Skip candidate VFs without a corresponding VPlan.
-    if (!hasPlanWithVF(NextVF.Width))
+    if (!hasPlanWithVF(NextVF.Width,
+                       CM.foldTailByMasking() || IsEpilogueTFEnabled))
       continue;
 
-    VPlan &CurrentPlan = getPlanFor(NextVF.Width);
+    VPlan &CurrentPlan =
+        getPlanFor(NextVF.Width, CM.foldTailByMasking() || IsEpilogueTFEnabled);
     ElementCount EffectiveVF = GetEffectiveVF(CurrentPlan, NextVF.Width);
     // Skip fixed vector VFs > than the estimated runtime VF, or any VF > than
     // the VF of the main loop.
@@ -5487,7 +5493,8 @@ void LoopVectorizationCostModel::collectValuesToIgnore() {
   }
 }
 
-void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
+void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
+                                    bool IsEpilogueTFEnabled) {
   CM.collectValuesToIgnore();
   Config.collectElementTypesForWidening(&CM.ValuesToIgnore);
 
@@ -5502,7 +5509,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   if (MaxFactors.FixedVF.isVector() || MaxFactors.ScalableVF.isVector())
     Legal->collectUnitStridePredicates();
 
-  auto VPlan1 = tryToBuildVPlan1();
+  auto VPlan1 = tryToBuildVPlan1(/*IsEpilogueTFEnabled*/ false);
   if (!VPlan1)
     return;
 
@@ -5511,7 +5518,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
     // plan for that VF only.
     ElementCount VF =
         MaxFactors.FixedVF ? MaxFactors.FixedVF : MaxFactors.ScalableVF;
-    buildVPlans(*VPlan1, VF, VF);
+    buildVPlans(*VPlan1, VF, VF, /*IsEpilogueTFEnabled*/ false);
     LLVM_DEBUG(printPlans(dbgs()));
     return;
   }
@@ -5550,12 +5557,13 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
       // Collect the instructions (and their associated costs) that will be more
       // profitable to scalarize.
       CM.collectNonVectorizedAndSetWideningDecisions(UserVF);
-      buildVPlans(*VPlan1, UserVF, UserVF);
+      buildVPlans(*VPlan1, UserVF, UserVF, /*IsEpilogueTFEnabled*/ false);
       ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
       if (EpilogueUserVF.isVector() &&
           ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
         CM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
-        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF);
+        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF,
+                    /*IsEpilogueTFEnabled*/ false);
       }
       if (!VPlans.empty() && VPlans.front()->getSingleVF() == UserVF) {
         // For scalar VF, skip VPlan cost check as VPlan cost is designed for
@@ -5587,10 +5595,27 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
     CM.collectNonVectorizedAndSetWideningDecisions(VF);
   }
 
-  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF);
-  buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF);
-
+  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF,
+              /*IsEpilogueTFEnabled*/ false);
+  buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF,
+              /*IsEpilogueTFEnabled*/ false);
   LLVM_DEBUG(printPlans(dbgs()));
+
+  // Build tail-folded vplans when IsEpilogueTFEnabled is enabled:
+  if (IsEpilogueTFEnabled) {
+    auto TFVPlan1 = tryToBuildVPlan1(/*IsEpilogueTFEnabled*/ true);
+    if (!TFVPlan1)
+      return;
+    buildVPlans(*TFVPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF,
+                /*IsEpilogueTFEnabled*/ true);
+    buildVPlans(*TFVPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF,
+                /*IsEpilogueTFEnabled*/ true);
+    LLVM_DEBUG(dbgs() << "LV: Tail-folded vplans:\n");
+    for (auto &vplan : VPlans) {
+      if (vplan->isCompatibleWithTF(true))
+        LLVM_DEBUG(vplan->dump());
+    }
+  }
 }
 
 VPCostContext::VPCostContext(const TargetLibraryInfo &TLI, const VPlan &Plan,
@@ -5824,7 +5849,7 @@ InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
 }
 
 std::pair<VectorizationFactor, VPlan *>
-LoopVectorizationPlanner::computeBestVF() {
+LoopVectorizationPlanner::computeBestVF(bool IsEpilogueTFEnabled) {
   if (VPlans.empty())
     return {VectorizationFactor::Disabled(), nullptr};
   // If there is a single VPlan with a single VF, return it directly.
@@ -5834,13 +5859,14 @@ LoopVectorizationPlanner::computeBestVF() {
   if (VPlans.size() == 1) {
     // For outer loops, the plan has a single vector VF determined by the
     // heuristic.
-    assert((FirstPlan.hasScalarVFOnly() || hasPlanWithVF(UserVF) ||
+    assert((FirstPlan.hasScalarVFOnly() ||
+            hasPlanWithVF(UserVF, CM.foldTailByMasking()) ||
             FirstPlan.isOuterLoop()) &&
            "must have a single scalar VF, UserVF or an outer loop");
     return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
   }
 
-  if (hasPlanWithVF(UserVF) && hasForcedEpilogueVF()) {
+  if (hasPlanWithVF(UserVF, CM.foldTailByMasking()) && hasForcedEpilogueVF()) {
     assert(VPlans.size() == 2 && "Must have exactly 2 VPlans built");
     assert(VPlans[0]->getSingleVF() == UserVF &&
            "expected second plan to be for the forced UserVF");
@@ -5914,9 +5940,11 @@ LoopVectorizationPlanner::computeBestVF() {
           cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
       VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
 
-      if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
-        BestFactor = CurrentFactor;
-        PlanForBestVF = P.get();
+      if (P->isCompatibleWithTF(CM.foldTailByMasking())) {
+        if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
+          BestFactor = CurrentFactor;
+          PlanForBestVF = P.get();
+        }
       }
 
       // If profitable add it to ProfitableVF list.
@@ -5936,7 +5964,7 @@ LoopVectorizationPlanner::computeBestVF() {
 
 DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
     ElementCount BestVF, unsigned BestUF, VPlan &BestVPlan,
-    InnerLoopVectorizer &ILV, DominatorTree *DT,
+    InnerLoopVectorizer &ILV, DominatorTree *DT, bool IsEpilogueTFEnabled,
     EpilogueVectorizationKind EpilogueVecKind) {
   assert(BestVPlan.hasVF(BestVF) &&
          "Trying to execute plan with unsupported VF");
@@ -6537,7 +6565,7 @@ VPRecipeBuilder::tryToCreateWidenNonPhiRecipe(VPSingleDefRecipe *R,
 // optimizations.
 static void printOptimizedVPlan(VPlan &) {}
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(bool IsEpilogueTFEnabled) {
   bool IsInnerLoop = OrigLoop->isInnermost();
 
   // Set up loop versioning for inner loops with memory runtime checks.
@@ -6616,7 +6644,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
 
   RUN_VPLAN_PASS(VPlanTransforms::createLoopRegions, *VPlan0,
                  getDebugLocFromInstOrOperands(Legal->getPrimaryInduction()));
-  if (CM.foldTailByMasking())
+  if (CM.foldTailByMasking() || IsEpilogueTFEnabled)
     RUN_VPLAN_PASS(VPlanTransforms::foldTailByMasking, *VPlan0);
   RUN_VPLAN_PASS(VPlanTransforms::introduceMasksAndLinearize, *VPlan0);
 
@@ -6624,7 +6652,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
 }
 
 void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
-                                           ElementCount MaxVF) {
+                                           ElementCount MaxVF,
+                                           bool IsEpilogueTFEnabled) {
   if (ElementCount::isKnownGT(MinVF, MaxVF))
     return;
 
@@ -6656,6 +6685,8 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
       VPlans.push_back(std::move(P));
 
     TailFoldingStyle Style = CM.getTailFoldingStyle();
+    if (IsEpilogueTFEnabled)
+      Style = TailFoldingStyle::Data;
     RUN_VPLAN_PASS(VPlanTransforms::materializeHeaderMask, *Plan,
                    useActiveLaneMask(Style),
                    useActiveLaneMaskForControlFlow(Style));
@@ -7231,7 +7262,8 @@ getEpilogueLowering(Function *F, Loop *L, LoopVectorizeHints &Hints,
 /// otherwise CM_EpilogueAllowed.
 static EpilogueLowering
 getEpilogueTailLowering(const LoopVectorizationCostModel &MainCM, const Loop *L,
-                        OptimizationRemarkEmitter *ORE) {
+                        OptimizationRemarkEmitter *ORE,
+                        LoopVectorizationLegality &LVL) {
   // Epilogue TF is only enabled when explicitly requested via command line.
   if (!EpilogueTailFoldingPolicy.getNumOccurrences() ||
       EpilogueTailFoldingPolicy != TailFoldingPolicyTy::PreferFoldTail)
@@ -7253,6 +7285,13 @@ getEpilogueTailLowering(const LoopVectorizationCostModel &MainCM, const Loop *L,
     return CM_EpilogueAllowed;
   }
 
+  if (LVL.hasUncountableEarlyExit()) {
+    LLVM_DEBUG(dbgs() << "LV: Epilogue tail-folding can't be applied because "
+                         " of loop has early exit\n"
+                         "LV: Fall back to a normal epilogue\n");
+    return CM_EpilogueAllowed;
+  }
+
   // If having epilogue is NOT allowed, then no epilogue to apply TF for.
   if (!MainCM.isEpilogueAllowed()) {
     LLVM_DEBUG(dbgs() << "LV: No epilogue to apply tail-folding for.\n"
@@ -7793,6 +7832,8 @@ fixScalarResumeValuesFromBypass(BasicBlock *BypassBlock, Loop *L,
     for (auto [ResumeV, HeaderPhi] :
          zip(ResumeValues, BestEpiPlan.getScalarHeader()->phis())) {
       auto *HeaderPhiR = cast<VPIRPhi>(&HeaderPhi);
+      if (!isa<PHINode>(HeaderPhiR->getIRPhi().getIncomingValueForBlock(PH)))
+        continue;
       auto *EpiResumePhi =
           cast<PHINode>(HeaderPhiR->getIRPhi().getIncomingValueForBlock(PH));
       if (EpiResumePhi->getBasicBlockIndex(BypassBlock) == -1)
@@ -7809,15 +7850,15 @@ fixScalarResumeValuesFromBypass(BasicBlock *BypassBlock, Loop *L,
 /// and runtime checks of the main loop, as well as updating various phis. \p
 /// InstsToMove contains instructions that need to be moved to the preheader of
 /// the epilogue vector loop.
-static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
+static void connectEpilogueVectorLoop(VPlan &MainPlan, VPlan &EpiPlan, Loop *L,
                                       EpilogueLoopVectorizationInfo &EPI,
-                                      DominatorTree *DT,
+                                      DominatorTree *DT, LoopInfo *LI,
                                       GeneratedRTChecks &Checks,
                                       ArrayRef<Instruction *> InstsToMove,
-                                      ArrayRef<VPInstruction *> ResumeValues) {
+                                      ArrayRef<VPInstruction *> ResumeValues,
+                                      bool IsEpilogueTFEnabled) {
   BasicBlock *VecEpilogueIterationCountCheck =
       cast<VPIRBasicBlock>(EpiPlan.getEntry())->getIRBasicBlock();
-
   BasicBlock *VecEpiloguePreHeader =
       cast<CondBrInst>(VecEpilogueIterationCountCheck->getTerminator())
           ->getSuccessor(1);
@@ -7841,7 +7882,11 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
 
   BasicBlock *ScalarPH =
       cast<VPIRBasicBlock>(EpiPlan.getScalarPreheader())->getIRBasicBlock();
-  RedirectEdge(EPI.EpilogueIterationCountCheck, ScalarPH);
+  // With a tail-folded epilogue there is no scalar remainder to bail
+  // to, even a trip count too small for the epilogue VF is handled safely by
+  // the masked epilogue vector loop, so skip straight to its preheader.
+  RedirectEdge(EPI.EpilogueIterationCountCheck,
+               IsEpilogueTFEnabled ? VecEpiloguePreHeader : ScalarPH);
 
   // Adjust the terminators of runtime check blocks and phis using them.
   BasicBlock *SCEVCheckBlock = Checks.getSCEVChecks().second;
@@ -7872,11 +7917,20 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
           return EPI.EpilogueIterationCountCheck == IncB;
         }))
       continue;
-    for (BasicBlock *BB :
-         {EPI.EpilogueIterationCountCheck, SCEVCheckBlock, MemCheckBlock}) {
+    for (BasicBlock *BB : {SCEVCheckBlock, MemCheckBlock}) {
       if (BB)
         Phi->removeIncomingValue(BB);
     }
+    // When the epilogue is tail-folded, EpilogueIterationCountCheck
+    // (iter.check) is redirected to branch straight into the vector epilogue
+    // preheader (see the IsEpilogueTFEnabled redirect above), so it is now a
+    // genuine predecessor and its incoming value must be kept rather than
+    // stripped.
+    // TODO: revisit for reduction phis, whose resume value on this bypass
+    // edge may need dedicated handling rather than reusing the value already
+    // present here.
+    if (!IsEpilogueTFEnabled)
+      Phi->removeIncomingValue(EPI.EpilogueIterationCountCheck);
   }
 
   auto IP = VecEpiloguePreHeader->getFirstNonPHIIt();
@@ -7894,6 +7948,46 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
   for (PHINode &Phi : make_early_inc_range(VecEpiloguePreHeader->phis()))
     if (Phi.use_empty())
       Phi.eraseFromParent();
+
+  if (IsEpilogueTFEnabled) {
+    // The epilogue vector loop is tail-folded, so it can safely handle
+    // any remaining trip count, including zero, via masking.
+    // vec.epilog.iter.check's own min-iters check was therefore built with a
+    // compile-time-known-false condition (see
+    // addMinimumVectorEpilogueIterationCheck) that never needs to bail out to
+    // a scalar remainder. Fold it into an unconditional branch into the
+    // vector epilogue preheader.
+    auto *Br =
+        cast<CondBrInst>(VecEpilogueIterationCountCheck->getTerminator());
+    [[maybe_unused]] auto *CondC = dyn_cast<ConstantInt>(Br->getCondition());
+    assert(CondC && CondC->isZero() &&
+           "expected vec.epilog.iter.check's branch condition to be a "
+           "compile-time false constant when the epilogue is tail-folded");
+    BasicBlock *DeadSucc = Br->getSuccessor(0);
+    UncondBrInst::Create(VecEpiloguePreHeader, Br->getIterator());
+    Br->eraseFromParent();
+    DTU.applyUpdates(
+        {{DominatorTree::Delete, VecEpilogueIterationCountCheck, DeadSucc}});
+  }
+
+  if (IsEpilogueTFEnabled && !SCEVCheckBlock && !MemCheckBlock) {
+    // No runtime check still needs a scalar fallback, and every trip-count
+    // bypass edge above has been redirected away from the scalar preheader:
+    // the scalar loop is now entirely unreachable. Delete it outright, along
+    // with its LoopInfo entry, instead of leaving it behind as dead code.
+    // This must run last, since fixScalarResumeValuesFromBypass (above) and
+    // the DT/function verification in processLoop (below) still need `L` and
+    // ScalarPH to be valid up to this point.
+    assert(pred_empty(ScalarPH) &&
+           "scalar preheader should have no predecessors left");
+    SmallVector<BasicBlock *> Blocks(L->block_begin(),
+                                     L->block_end());
+    Blocks.push_back(ScalarPH);
+    LI->erase(L);
+    for (auto *BB : Blocks)
+      LI->removeBlock(BB);
+    DeleteDeadBlocks(Blocks, &DTU);
+  }
 }
 
 bool LoopVectorizePass::processLoop(Loop *L) {
@@ -8084,15 +8178,13 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                                Hints, ORE);
 
   EpilogueLowering EpilogueTailLoweringStatus =
-      getEpilogueTailLowering(CM, L, ORE);
+      getEpilogueTailLowering(CM, L, ORE, LVL);
+  bool IsEpilogueTFEnabled = false;
   if (EpilogueTailLoweringStatus ==
       EpilogueLowering::CM_EpilogueNotNeededFoldTail) {
     // TODO: Apply tail-folding on the vectorized epilogue loop.
-    LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is not supported yet\n");
-    reportVectorizationInfo(
-        "The epilogue-tail-folding policy prefer-fold-tail is not supported "
-        "yet, fall back to a normal epilogue",
-        "UnsupportedEpilogueTailFoldingPolicy", ORE, L);
+    LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is enabled\n");
+    IsEpilogueTFEnabled = true;
   }
 
   // Get user vectorization factor and interleave count.
@@ -8106,8 +8198,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     UserIC = 1;
 
   // Plan how to best vectorize.
-  LVP.plan(UserVF, UserIC);
-  auto [VF, BestPlanPtr] = LVP.computeBestVF();
+  LVP.plan(UserVF, UserIC, IsEpilogueTFEnabled);
+  auto [VF, BestPlanPtr] = LVP.computeBestVF(IsEpilogueTFEnabled);
   unsigned IC = 1;
 
   // For VPlan build stress testing of outer loops, bail after plan
@@ -8123,7 +8215,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   GeneratedRTChecks Checks(PSE, DT, LI, TTI, Config.CostKind,
                            CM.maskPartialAliasing());
-  if (IsInnerLoop && LVP.hasPlanWithVF(VF.Width)) {
+  if (IsInnerLoop && LVP.hasPlanWithVF(VF.Width, CM.foldTailByMasking())) {
     // Select the interleave count.
     IC = LVP.selectInterleaveCount(*BestPlanPtr, VF.Width, VF.Cost);
 
@@ -8185,7 +8277,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                   "Ignoring user-specified interleave count due to possibly "
                   "unsafe dependencies in the loop."};
     InterleaveLoop = false;
-  } else if (!LVP.hasPlanWithVF(VF.Width) && UserIC > 1) {
+  } else if (!LVP.hasPlanWithVF(VF.Width, CM.foldTailByMasking()) &&
+             UserIC > 1) {
     // Tell the user interleaving was avoided up-front, despite being explicitly
     // requested.
     LLVM_DEBUG(dbgs() << "LV: Ignoring UserIC, because vectorization and "
@@ -8311,7 +8404,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   VPlan &BestPlan = *BestPlanPtr;
   // Consider vectorizing the epilogue too if it's profitable.
   std::unique_ptr<VPlan> EpiPlan =
-      LVP.selectBestEpiloguePlan(BestPlan, VF.Width, IC);
+      LVP.selectBestEpiloguePlan(BestPlan, VF.Width, IC, IsEpilogueTFEnabled);
   bool HasBranchWeights =
       hasBranchWeightMD(*L->getLoopLatch()->getTerminator());
   if (EpiPlan) {
@@ -8344,6 +8437,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                                        BestMainPlan);
     auto ExpandedSCEVs = LVP.executePlan(
         EPI.MainLoopVF, EPI.MainLoopUF, BestMainPlan, MainILV, DT,
+        /*IsEpilogueTFEnabled*/ false,
         LoopVectorizationPlanner::EpilogueVectorizationKind::MainLoop);
     ++LoopsVectorized;
 
@@ -8375,9 +8469,10 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     LVP.attachRuntimeChecks(BestEpiPlan, Checks, HasBranchWeights);
     LVP.executePlan(
         EPI.EpilogueVF, EPI.EpilogueUF, BestEpiPlan, EpilogILV, DT,
+        IsEpilogueTFEnabled,
         LoopVectorizationPlanner::EpilogueVectorizationKind::Epilogue);
-    connectEpilogueVectorLoop(BestEpiPlan, L, EPI, DT, Checks, InstsToMove,
-                              ResumeValues);
+    connectEpilogueVectorLoop(BestMainPlan, BestEpiPlan, L, EPI, DT, LI, Checks,
+                              InstsToMove, ResumeValues, IsEpilogueTFEnabled);
     ++LoopsEpilogueVectorized;
   } else {
     InnerLoopVectorizer LB(L, PSE, LI, DT, TTI, AC, VF.Width, IC, Checks,
@@ -8389,7 +8484,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     if (!IsInnerLoop)
       LLVM_DEBUG(dbgs() << "Vectorizing outer loop in \"" << F->getName()
                         << "\"\n");
-    LVP.executePlan(VF.Width, IC, BestPlan, LB, DT);
+    LVP.executePlan(VF.Width, IC, BestPlan, LB, DT,
+                    /*IsEpilogueTFEnabled*/ false);
     ++LoopsVectorized;
   }
 
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.cpp b/llvm/lib/Transforms/Vectorize/VPlan.cpp
index 5b53312c3ebda..d2dadf3f00317 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlan.cpp
@@ -1364,6 +1364,13 @@ VPIRBasicBlock *VPlan::createVPIRBasicBlock(BasicBlock *IRBB) {
   return VPIRBB;
 }
 
+bool VPlan::isCompatibleWithTF(bool TF) {
+  auto *VLR = getVectorLoopRegion();
+  assert(VLR && "Vector loop region got eliminated\n");
+  bool HasHeaderMask = (VLR->getHeaderMask() != nullptr);
+  return HasHeaderMask == TF;
+}
+
 #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
 
 Twine VPlanPrinter::getUID(const VPBlockBase *Block) {
@@ -1733,19 +1740,29 @@ VPBuilder::createConsecutiveVectorPointer(VPValue *Ptr, Type *SourceElementTy,
   return createVectorPointer(Ptr, SourceElementTy, StrideOne, Flags, DL);
 }
 
-VPlan &LoopVectorizationPlanner::getPlanFor(ElementCount VF) const {
+VPlan &LoopVectorizationPlanner::getPlanFor(ElementCount VF, bool TF) const {
   assert(count_if(VPlans,
-                  [VF](const VPlanPtr &Plan) { return Plan->hasVF(VF); }) ==
-             1 &&
+                  [VF, TF](const VPlanPtr &Plan) {
+                    LLVM_DEBUG(dbgs() << "LV: given VF: " << VF << " and TF: "
+                                      << TF << " equivalent vplan: ";
+                               Plan->dump());
+                    return Plan->hasVF(VF) && Plan->isCompatibleWithTF(TF);
+                  }) == 1 &&
          "Multiple VPlans for VF.");
 
   for (const VPlanPtr &Plan : VPlans) {
-    if (Plan->hasVF(VF))
+    if (Plan->hasVF(VF) && Plan->isCompatibleWithTF(TF))
       return *Plan.get();
   }
   llvm_unreachable("No plan found!");
 }
 
+bool LoopVectorizationPlanner::hasPlanWithVF(ElementCount VF, bool TF) const {
+  return any_of(VPlans, [VF, TF](const VPlanPtr &Plan) {
+    return Plan->hasVF(VF) && Plan->isCompatibleWithTF(TF);
+  });
+}
+
 static void addRuntimeUnrollDisableMetaData(Loop *L) {
   SmallVector<Metadata *, 4> MDs;
   // Reserve first location for self reference to the LoopID metadata node.
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 814b77a96e825..b5d994b3d2f05 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -5223,6 +5223,15 @@ class VPlan {
            is_contained(ScalarPH->getPredecessors(), getMiddleBlock());
   }
 
+  /// Returns true if this VPlan's tail-folding matches \p TF.
+  /// A plan is tail-folded if it has a header mask, or if it has no scalar
+  /// tail while the TC is not evenly divisible by VF (in which case the tail
+  /// must be folded by the vector loop itself).
+  ///
+  /// This function is used to disambiguate VPlan lookup by VF when multiple
+  /// plans share the same VF but differ in tail-folding status.
+  bool isCompatibleWithTF(bool TF);
+
   /// The type of the canonical induction variable of the vector loop.
   Type *getIndexType() const { return VF.getType(); }
 };
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 421478c85c188..f0309b35d5c03 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1576,9 +1576,25 @@ void VPlanTransforms::addMinimumVectorEpilogueIterationCheck(
     ElementCount EpilogueVF, unsigned EpilogueUF, unsigned MainLoopStep,
     unsigned EpilogueLoopStep, ScalarEvolution &SE) {
   // Add the minimum iteration check for the epilogue vector loop.
+  VPBuilder Builder(cast<VPBasicBlock>(Plan.getEntry()));
+
+  if (Plan.isCompatibleWithTF(/*TF*/ true)) {
+    Builder.createNaryOp(VPInstruction::BranchOnCond, Plan.getFalse());
+    return;
+  }
+  // if (Plan.isCompatibleWithTF(/*TF*/ true)) {
+  //   VPBasicBlock *EntryVPBB = cast<VPBasicBlock>(Plan.getEntry());
+  //   // Successor 0 is the "taken" (scalar) edge, successor 1 is the vector
+  //   // path — see BranchOnCond's execute() and removeBranchOnConst's
+  //   // RemovedIdx convention. Drop the scalar edge directly, leaving Entry
+  //   // with a single successor (which needs no BranchOnCond terminator at
+  //   // all — VPlan codegen emits a plain unconditional `br` for that case).
+  //   VPBlockUtils::disconnectBlocks(EntryVPBB, EntryVPBB->getSuccessors()[0]);
+  //   return;
+  // }
+
   VPValue *TC = Plan.getTripCount();
   Value *TripCount = TC->getLiveInIRValue();
-  VPBuilder Builder(cast<VPBasicBlock>(Plan.getEntry()));
   VPValue *VFxUF = Builder.createExpandSCEV(SE.getElementCount(
       TripCount->getType(), (EpilogueVF * EpilogueUF), SCEV::FlagNUW));
   VPValue *Count = Builder.createSub(TC, Plan.getOrAddLiveIn(VectorTripCount),
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
new file mode 100644
index 0000000000000..495b280990aec
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -0,0 +1,63 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail -mtriple=aarch64-linux-gnu -S %s | FileCheck %s
+
+
+define void @test_epilogue_tf(ptr %A, i64 %n) {
+; CHECK-LABEL: define void @test_epilogue_tf(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = and i64 [[N]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 16
+; CHECK-NEXT:    store <16 x i8> splat (i8 1), ptr [[TMP0]], align 1
+; CHECK-NEXT:    store <16 x i8> splat (i8 1), ptr [[TMP1]], align 1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-NEXT:    [[TMP2:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP2]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[N_RND_UP:%.*]] = add i64 [[N]], 7
+; CHECK-NEXT:    [[N_MOD_VF2:%.*]] = and i64 [[N_RND_UP]], 7
+; CHECK-NEXT:    [[N_VEC3:%.*]] = sub i64 [[N_RND_UP]], [[N_MOD_VF2]]
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT5:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX4]]
+; CHECK-NEXT:    store <8 x i8> splat (i8 1), ptr [[TMP3]], align 1
+; CHECK-NEXT:    [[INDEX_NEXT5]] = add nuw i64 [[INDEX4]], 8
+; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT5]], [[N_VEC3]]
+; CHECK-NEXT:    br i1 [[TMP4]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+  %arrayidx = getelementptr inbounds i8, ptr %A, i64 %iv
+  store i8 1, ptr %arrayidx, align 1
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exitcond = icmp ne i64 %iv.next, %n
+  br i1 %exitcond, label %for.body, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index 590907a5aff79..6bb1f7349413a 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -1,15 +1,14 @@
 ; REQUIRES: asserts
-; RUN: opt -S < %s -p loop-vectorize -debug-only=loop-vectorize --disable-output \
-; RUN: -epilogue-tail-folding-policy=prefer-fold-tail -pass-remarks-analysis=loop-vectorize 2>&1 | FileCheck %s
+; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail\
+; RUN: -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s
 
-; RUN: opt -S < %s -p loop-vectorize -debug-only=loop-vectorize -enable-epilogue-vectorization=false \
-; RUN: --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -pass-remarks-analysis=loop-vectorize 2>&1 \
+; RUN: opt -S -p loop-vectorize -debug -enable-epilogue-vectorization=false \
+; RUN: --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -pass-remarks-analysis=loop-vectorize < %s 2>&1 \
 ; RUN: | FileCheck %s --check-prefix=CHECK-DISABLED-EPILOG
 
 define void @test_epilogue_tf(ptr %A, i64 %n) {
 ; CHECK-LABEL: LV: Checking a loop in 'test_epilogue_tf'
-; CHECK: LV: epilogue tail-folding is not supported yet
-; CHECK: remark: <unknown>:0:0: The epilogue-tail-folding policy prefer-fold-tail is not supported yet, fall back to a normal epilogue
+; CHECK: LV: epilogue tail-folding is enabled
 ;
 entry:
   br label %for.body

>From 370b72e0c53e296ae7722c6bfd3ed93d1dc64d86 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Tue, 21 Jul 2026 15:10:26 +0100
Subject: [PATCH 02/11] rebase and resolve review comments - limit the feature
 to be enabled only for forced epilogue VF

---
 .../Vectorize/LoopVectorizationPlanner.h      | 10 +--
 .../Transforms/Vectorize/LoopVectorize.cpp    | 72 ++++++++-----------
 llvm/lib/Transforms/Vectorize/VPlan.cpp       | 25 ++-----
 llvm/lib/Transforms/Vectorize/VPlan.h         |  9 ---
 .../Vectorize/VPlanConstruction.cpp           | 12 +---
 .../AArch64/fold-epilogue-tail.ll             |  3 +-
 .../LoopVectorize/fold-epilogue-tail.ll       | 42 +++++++++--
 7 files changed, 78 insertions(+), 95 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index cad4eb60674c5..ea1820c751e75 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -909,13 +909,12 @@ class LoopVectorizationPlanner {
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
   /// for each VF.
-  VPlan &getPlanFor(ElementCount VF, bool TF) const;
+  VPlan &getPlanFor(ElementCount VF) const;
 
   /// Compute and return the most profitable vectorization factor and the
   /// corresponding best VPlan. Also collect all profitable VFs in
   /// ProfitableVFs.
-  std::pair<VectorizationFactor, VPlan *>
-  computeBestVF(bool IsEpilogueTFEnabled);
+  std::pair<VectorizationFactor, VPlan *> computeBestVF();
 
   /// \return The desired interleave count.
   /// If interleave count has been specified by metadata it will be returned.
@@ -951,7 +950,10 @@ class LoopVectorizationPlanner {
 
   /// Look through the existing plans and return true if we have one with
   /// vectorization factor \p VF.
-  bool hasPlanWithVF(ElementCount VF, bool TF) const;
+  bool hasPlanWithVF(ElementCount VF) const {
+    return any_of(VPlans,
+                  [&](const VPlanPtr &Plan) { return Plan->hasVF(VF); });
+  }
 
   /// Test a \p Predicate on a \p Range of VF's. Return the value of applying
   /// \p Predicate on Range.Start, possibly decreasing Range.End such that the
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index b57640797deb4..92c86ad6ee570 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3151,8 +3151,6 @@ void LoopVectorizationPlanner::emitInvalidCostRemarks(
   using RecipeVFPair = std::pair<VPRecipeBase *, ElementCount>;
   SmallVector<RecipeVFPair> InvalidCosts;
   for (const auto &Plan : VPlans) {
-    if (!Plan->isCompatibleWithTF(CM.foldTailByMasking()))
-      continue;
     for (ElementCount VF : Plan->vectorFactors()) {
       // The VPlan-based cost model is designed for computing vector cost.
       // Querying VPlan-based cost model with a scarlar VF will cause some
@@ -3496,10 +3494,9 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     }
 
     LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization factor is forced.\n");
-    if (hasPlanWithVF(EpilogueVectorizationForceVF, CM.foldTailByMasking())) {
+    if (hasPlanWithVF(EpilogueVectorizationForceVF)) {
       std::unique_ptr<VPlan> Clone(
-          getPlanFor(EpilogueVectorizationForceVF, CM.foldTailByMasking())
-              .duplicate());
+          getPlanFor(EpilogueVectorizationForceVF).duplicate());
       Clone->setVF(EpilogueVectorizationForceVF);
       return Clone;
     }
@@ -3590,12 +3587,10 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
   VPlan *BestPlan = nullptr;
   for (auto &NextVF : ProfitableVFs) {
     // Skip candidate VFs without a corresponding VPlan.
-    if (!hasPlanWithVF(NextVF.Width,
-                       CM.foldTailByMasking() || IsEpilogueTFEnabled))
+    if (!hasPlanWithVF(NextVF.Width))
       continue;
 
-    VPlan &CurrentPlan =
-        getPlanFor(NextVF.Width, CM.foldTailByMasking() || IsEpilogueTFEnabled);
+    VPlan &CurrentPlan = getPlanFor(NextVF.Width);
     ElementCount EffectiveVF = GetEffectiveVF(CurrentPlan, NextVF.Width);
     // Skip fixed vector VFs > than the estimated runtime VF, or any VF > than
     // the VF of the main loop.
@@ -5562,8 +5557,13 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
       if (EpilogueUserVF.isVector() &&
           ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
         CM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
-        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF,
-                    /*IsEpilogueTFEnabled*/ false);
+        if (IsEpilogueTFEnabled) {
+          auto TFVPlan1 = tryToBuildVPlan1(/*IsEpilogueTFEnabled*/ true);
+          buildVPlans(*TFVPlan1, EpilogueUserVF, EpilogueUserVF,
+                      /*IsEpilogueTFEnabled*/ true);
+        } else
+          buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF,
+                      /*IsEpilogueTFEnabled*/ false);
       }
       if (!VPlans.empty() && VPlans.front()->getSingleVF() == UserVF) {
         // For scalar VF, skip VPlan cost check as VPlan cost is designed for
@@ -5600,22 +5600,6 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
   buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF,
               /*IsEpilogueTFEnabled*/ false);
   LLVM_DEBUG(printPlans(dbgs()));
-
-  // Build tail-folded vplans when IsEpilogueTFEnabled is enabled:
-  if (IsEpilogueTFEnabled) {
-    auto TFVPlan1 = tryToBuildVPlan1(/*IsEpilogueTFEnabled*/ true);
-    if (!TFVPlan1)
-      return;
-    buildVPlans(*TFVPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF,
-                /*IsEpilogueTFEnabled*/ true);
-    buildVPlans(*TFVPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF,
-                /*IsEpilogueTFEnabled*/ true);
-    LLVM_DEBUG(dbgs() << "LV: Tail-folded vplans:\n");
-    for (auto &vplan : VPlans) {
-      if (vplan->isCompatibleWithTF(true))
-        LLVM_DEBUG(vplan->dump());
-    }
-  }
 }
 
 VPCostContext::VPCostContext(const TargetLibraryInfo &TLI, const VPlan &Plan,
@@ -5849,7 +5833,7 @@ InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
 }
 
 std::pair<VectorizationFactor, VPlan *>
-LoopVectorizationPlanner::computeBestVF(bool IsEpilogueTFEnabled) {
+LoopVectorizationPlanner::computeBestVF() {
   if (VPlans.empty())
     return {VectorizationFactor::Disabled(), nullptr};
   // If there is a single VPlan with a single VF, return it directly.
@@ -5859,14 +5843,13 @@ LoopVectorizationPlanner::computeBestVF(bool IsEpilogueTFEnabled) {
   if (VPlans.size() == 1) {
     // For outer loops, the plan has a single vector VF determined by the
     // heuristic.
-    assert((FirstPlan.hasScalarVFOnly() ||
-            hasPlanWithVF(UserVF, CM.foldTailByMasking()) ||
+    assert((FirstPlan.hasScalarVFOnly() || hasPlanWithVF(UserVF) ||
             FirstPlan.isOuterLoop()) &&
            "must have a single scalar VF, UserVF or an outer loop");
     return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
   }
 
-  if (hasPlanWithVF(UserVF, CM.foldTailByMasking()) && hasForcedEpilogueVF()) {
+  if (hasPlanWithVF(UserVF) && hasForcedEpilogueVF()) {
     assert(VPlans.size() == 2 && "Must have exactly 2 VPlans built");
     assert(VPlans[0]->getSingleVF() == UserVF &&
            "expected second plan to be for the forced UserVF");
@@ -5940,11 +5923,9 @@ LoopVectorizationPlanner::computeBestVF(bool IsEpilogueTFEnabled) {
           cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
       VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
 
-      if (P->isCompatibleWithTF(CM.foldTailByMasking())) {
-        if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
-          BestFactor = CurrentFactor;
-          PlanForBestVF = P.get();
-        }
+      if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
+        BestFactor = CurrentFactor;
+        PlanForBestVF = P.get();
       }
 
       // If profitable add it to ProfitableVF list.
@@ -7263,7 +7244,8 @@ getEpilogueLowering(Function *F, Loop *L, LoopVectorizeHints &Hints,
 static EpilogueLowering
 getEpilogueTailLowering(const LoopVectorizationCostModel &MainCM, const Loop *L,
                         OptimizationRemarkEmitter *ORE,
-                        LoopVectorizationLegality &LVL) {
+                        LoopVectorizationLegality &LVL,
+                        LoopVectorizeHints &Hints) {
   // Epilogue TF is only enabled when explicitly requested via command line.
   if (!EpilogueTailFoldingPolicy.getNumOccurrences() ||
       EpilogueTailFoldingPolicy != TailFoldingPolicyTy::PreferFoldTail)
@@ -7277,6 +7259,13 @@ getEpilogueTailLowering(const LoopVectorizationCostModel &MainCM, const Loop *L,
     return CM_EpilogueAllowed;
   }
 
+  if (!hasForcedEpilogueVF() || !Hints.getWidth()) {
+    reportVectorizationInfo("For now, Epilogue tail-folding can't be "
+                            "applied without forced epilogue/main loop VF\n",
+                            "UnsupportedEpilogueTailFoldingPolicy", ORE, L);
+    return CM_EpilogueAllowed;
+  }
+
   // If scalar epilogue is explicitly required, we can't apply TF.
   if (MainCM.requiresScalarEpilogue(/*IsVectorizing*/ true)) {
     LLVM_DEBUG(dbgs() << "LV: Epilogue tail-folding can't be applied because "
@@ -8178,7 +8167,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                                Hints, ORE);
 
   EpilogueLowering EpilogueTailLoweringStatus =
-      getEpilogueTailLowering(CM, L, ORE, LVL);
+      getEpilogueTailLowering(CM, L, ORE, LVL, Hints);
   bool IsEpilogueTFEnabled = false;
   if (EpilogueTailLoweringStatus ==
       EpilogueLowering::CM_EpilogueNotNeededFoldTail) {
@@ -8199,7 +8188,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   // Plan how to best vectorize.
   LVP.plan(UserVF, UserIC, IsEpilogueTFEnabled);
-  auto [VF, BestPlanPtr] = LVP.computeBestVF(IsEpilogueTFEnabled);
+  auto [VF, BestPlanPtr] = LVP.computeBestVF();
   unsigned IC = 1;
 
   // For VPlan build stress testing of outer loops, bail after plan
@@ -8215,7 +8204,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   GeneratedRTChecks Checks(PSE, DT, LI, TTI, Config.CostKind,
                            CM.maskPartialAliasing());
-  if (IsInnerLoop && LVP.hasPlanWithVF(VF.Width, CM.foldTailByMasking())) {
+  if (IsInnerLoop && LVP.hasPlanWithVF(VF.Width)) {
     // Select the interleave count.
     IC = LVP.selectInterleaveCount(*BestPlanPtr, VF.Width, VF.Cost);
 
@@ -8277,8 +8266,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                   "Ignoring user-specified interleave count due to possibly "
                   "unsafe dependencies in the loop."};
     InterleaveLoop = false;
-  } else if (!LVP.hasPlanWithVF(VF.Width, CM.foldTailByMasking()) &&
-             UserIC > 1) {
+  } else if (!LVP.hasPlanWithVF(VF.Width) && UserIC > 1) {
     // Tell the user interleaving was avoided up-front, despite being explicitly
     // requested.
     LLVM_DEBUG(dbgs() << "LV: Ignoring UserIC, because vectorization and "
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.cpp b/llvm/lib/Transforms/Vectorize/VPlan.cpp
index d2dadf3f00317..5b53312c3ebda 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlan.cpp
@@ -1364,13 +1364,6 @@ VPIRBasicBlock *VPlan::createVPIRBasicBlock(BasicBlock *IRBB) {
   return VPIRBB;
 }
 
-bool VPlan::isCompatibleWithTF(bool TF) {
-  auto *VLR = getVectorLoopRegion();
-  assert(VLR && "Vector loop region got eliminated\n");
-  bool HasHeaderMask = (VLR->getHeaderMask() != nullptr);
-  return HasHeaderMask == TF;
-}
-
 #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
 
 Twine VPlanPrinter::getUID(const VPBlockBase *Block) {
@@ -1740,29 +1733,19 @@ VPBuilder::createConsecutiveVectorPointer(VPValue *Ptr, Type *SourceElementTy,
   return createVectorPointer(Ptr, SourceElementTy, StrideOne, Flags, DL);
 }
 
-VPlan &LoopVectorizationPlanner::getPlanFor(ElementCount VF, bool TF) const {
+VPlan &LoopVectorizationPlanner::getPlanFor(ElementCount VF) const {
   assert(count_if(VPlans,
-                  [VF, TF](const VPlanPtr &Plan) {
-                    LLVM_DEBUG(dbgs() << "LV: given VF: " << VF << " and TF: "
-                                      << TF << " equivalent vplan: ";
-                               Plan->dump());
-                    return Plan->hasVF(VF) && Plan->isCompatibleWithTF(TF);
-                  }) == 1 &&
+                  [VF](const VPlanPtr &Plan) { return Plan->hasVF(VF); }) ==
+             1 &&
          "Multiple VPlans for VF.");
 
   for (const VPlanPtr &Plan : VPlans) {
-    if (Plan->hasVF(VF) && Plan->isCompatibleWithTF(TF))
+    if (Plan->hasVF(VF))
       return *Plan.get();
   }
   llvm_unreachable("No plan found!");
 }
 
-bool LoopVectorizationPlanner::hasPlanWithVF(ElementCount VF, bool TF) const {
-  return any_of(VPlans, [VF, TF](const VPlanPtr &Plan) {
-    return Plan->hasVF(VF) && Plan->isCompatibleWithTF(TF);
-  });
-}
-
 static void addRuntimeUnrollDisableMetaData(Loop *L) {
   SmallVector<Metadata *, 4> MDs;
   // Reserve first location for self reference to the LoopID metadata node.
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index b5d994b3d2f05..814b77a96e825 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -5223,15 +5223,6 @@ class VPlan {
            is_contained(ScalarPH->getPredecessors(), getMiddleBlock());
   }
 
-  /// Returns true if this VPlan's tail-folding matches \p TF.
-  /// A plan is tail-folded if it has a header mask, or if it has no scalar
-  /// tail while the TC is not evenly divisible by VF (in which case the tail
-  /// must be folded by the vector loop itself).
-  ///
-  /// This function is used to disambiguate VPlan lookup by VF when multiple
-  /// plans share the same VF but differ in tail-folding status.
-  bool isCompatibleWithTF(bool TF);
-
   /// The type of the canonical induction variable of the vector loop.
   Type *getIndexType() const { return VF.getType(); }
 };
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index f0309b35d5c03..30c7f16ac88aa 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1578,20 +1578,10 @@ void VPlanTransforms::addMinimumVectorEpilogueIterationCheck(
   // Add the minimum iteration check for the epilogue vector loop.
   VPBuilder Builder(cast<VPBasicBlock>(Plan.getEntry()));
 
-  if (Plan.isCompatibleWithTF(/*TF*/ true)) {
+  if (Plan.hasTailFolded()) {
     Builder.createNaryOp(VPInstruction::BranchOnCond, Plan.getFalse());
     return;
   }
-  // if (Plan.isCompatibleWithTF(/*TF*/ true)) {
-  //   VPBasicBlock *EntryVPBB = cast<VPBasicBlock>(Plan.getEntry());
-  //   // Successor 0 is the "taken" (scalar) edge, successor 1 is the vector
-  //   // path — see BranchOnCond's execute() and removeBranchOnConst's
-  //   // RemovedIdx convention. Drop the scalar edge directly, leaving Entry
-  //   // with a single successor (which needs no BranchOnCond terminator at
-  //   // all — VPlan codegen emits a plain unconditional `br` for that case).
-  //   VPBlockUtils::disconnectBlocks(EntryVPBB, EntryVPBB->getSuccessors()[0]);
-  //   return;
-  // }
 
   VPValue *TC = Plan.getTripCount();
   Value *TripCount = TC->getLiveInIRValue();
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index 495b280990aec..a94cf64668ff6 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -1,5 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail -mtriple=aarch64-linux-gnu -S %s | FileCheck %s
+; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -mtriple=aarch64-linux-gnu -S %s | FileCheck %s
 
 
 define void @test_epilogue_tf(ptr %A, i64 %n) {
diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index 6bb1f7349413a..2407c5e28c892 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -1,13 +1,19 @@
 ; REQUIRES: asserts
 ; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail\
-; RUN: -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s
+; RUN: -pass-remarks-analysis=loop-vectorize -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s
+
+; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -force-vector-width=16 \
+; RUN:  -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s --check-prefix=CHECK-NO-FORCED-MAIN-VF
+
+; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -epilogue-vectorization-force-VF=8  \
+; RUN: -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s --check-prefix=CHECK-NO-FORCED-EPILOGUE-VF
 
 ; RUN: opt -S -p loop-vectorize -debug -enable-epilogue-vectorization=false \
-; RUN: --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -pass-remarks-analysis=loop-vectorize < %s 2>&1 \
-; RUN: | FileCheck %s --check-prefix=CHECK-DISABLED-EPILOG
+; RUN: --disable-output -force-vector-width=16 -epilogue-vectorization-force-VF=8 -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s --check-prefix=CHECK-DISABLED-EPILOG
 
 define void @test_epilogue_tf(ptr %A, i64 %n) {
-; CHECK-LABEL: LV: Checking a loop in 'test_epilogue_tf'
+; CHECK-LABEL: Checking a loop in 'test_epilogue_tf'
 ; CHECK: LV: epilogue tail-folding is enabled
 ;
 entry:
@@ -25,8 +31,30 @@ exit:
   ret void
 }
 
+define void @test_epilogue_tf_no_fv(ptr %A, i64 %n) {
+; CHECK-NO-FORCED-MAIN-VF-LABEL: Checking a loop in 'test_epilogue_tf_no_fv'
+; CHECK-NO-FORCED-MAIN-VF: remark: <unknown>:0:0: For now, Epilogue tail-folding can't be applied without forced epilogue/main loop VF
+
+; CHECK-NO-FORCED-EPILOGUE-VF-LABEL: Checking a loop in 'test_epilogue_tf_no_fv'
+; CHECK-NO-FORCED-EPILOGUE-VF: remark: <unknown>:0:0: For now, Epilogue tail-folding can't be applied without forced epilogue/main loop VF
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+  %arrayidx = getelementptr inbounds i8, ptr %A, i64 %iv
+  store i8 1, ptr %arrayidx, align 1
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exitcond = icmp ne i64 %iv.next, %n
+  br i1 %exitcond, label %for.body, label %exit
+
+exit:
+  ret void
+}
+
 define void @epilogue_is_disabled(ptr %a, i64 %n) {
-; CHECK-DISABLED-EPILOG-LABEL: LV: Checking a loop in 'epilogue_is_disabled'
+; CHECK-DISABLED-EPILOG-LABEL: Checking a loop in 'epilogue_is_disabled'
 ; CHECK-DISABLED-EPILOG: remark: <unknown>:0:0: Options conflict, epilogue vectorization is disallowed while epilogue tail-folding allowed!
 ;
 entry:
@@ -45,7 +73,7 @@ for.end:
 }
 
 define i16 @require_scalar_epilogue(ptr %dst, i64 %x) {
-; CHECK-LABEL: LV: Checking a loop in 'require_scalar_epilogue'
+; CHECK-LABEL: Checking a loop in 'require_scalar_epilogue'
 ; CHECK: LV: Epilogue tail-folding can't be applied because scalar epilogue is required
 ; CHECK-NEXT: LV: Fall back to a normal epilogue
 ;
@@ -74,7 +102,7 @@ exit.2:
 }
 
 define i32 @opt_for_size(ptr %p, i32 %n) optsize {
-; CHECK-LABEL: LV: Checking a loop in 'opt_for_size'
+; CHECK-LABEL: Checking a loop in 'opt_for_size'
 ; CHECK: LV: No epilogue to apply tail-folding for.
 ; CHECK-NEXT: LV: Fall back to a normal epilogue
 ;

>From 14bd2f7325ad32d956f31b77eb20eed5be62fe69 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Thu, 23 Jul 2026 11:57:23 +0100
Subject: [PATCH 03/11] Enable correct costs for tail-folded epilogue. To
 achieve that, I pass the epilogue CM to cost/vplan-related functions.

---
 .../Vectorize/LoopVectorizationPlanner.h      |  26 +-
 .../Transforms/Vectorize/LoopVectorize.cpp    | 243 ++++++++++++------
 .../AArch64/fold-epilogue-tail.ll             |  37 +--
 3 files changed, 195 insertions(+), 111 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index ea1820c751e75..9e72b3da45cd5 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -884,7 +884,8 @@ class LoopVectorizationPlanner {
   ///
   /// TODO: Move to VPlan::cost once the use of LoopVectorizationLegality has
   /// been retired.
-  InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU) const;
+  InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU,
+                       LoopVectorizationCostModel &EnabledCM) const;
 
   /// Precompute costs for certain instructions using the legacy cost model. The
   /// function is used to bring up the VPlan-based cost model to initially avoid
@@ -905,7 +906,14 @@ class LoopVectorizationPlanner {
   /// Build VPlans for the specified \p UserVF and \p UserIC if they are
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
   /// interleaving should be avoided up-front, no plans are generated.
-  void plan(ElementCount UserVF, unsigned UserIC, bool IsEpilogueTFEnabled);
+  void plan(ElementCount UserVF, unsigned UserIC);
+
+  /// Build VPlans for the specified \p EpilogueUserVF and \p IC if they are
+  /// non-zero or all applicable candidate VFs otherwise. If vectorization and
+  /// tail-folding should be avoided up-front, no plans are generated.
+  bool planForEpilogueTF(ElementCount UserVF, unsigned UserIC,
+                         ElementCount EpilogueUserVF,
+                         LoopVectorizationCostModel &EpilogueCM);
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
   /// for each VF.
@@ -940,7 +948,6 @@ class LoopVectorizationPlanner {
   DenseMap<const SCEV *, Value *>
   executePlan(ElementCount VF, unsigned UF, VPlan &BestPlan,
               InnerLoopVectorizer &LB, DominatorTree *DT,
-              bool IsEpilogueTFEnabled,
               EpilogueVectorizationKind EpilogueVecKind =
                   EpilogueVectorizationKind::None);
 
@@ -966,10 +973,8 @@ class LoopVectorizationPlanner {
   /// VF narrowed to the chosen factor. The returned plan is a duplicate.
   /// Returns nullptr if epilogue vectorization is not supported or not
   /// profitable for the loop.
-  std::unique_ptr<VPlan> selectBestEpiloguePlan(VPlan &MainPlan,
-                                                ElementCount MainLoopVF,
-                                                unsigned IC,
-                                                bool IsEpilogueTFEnabled);
+  std::unique_ptr<VPlan>
+  selectBestEpiloguePlan(VPlan &MainPlan, ElementCount MainLoopVF, unsigned IC);
 
   /// Emit remarks for recipes with invalid costs in the available VPlans.
   void emitInvalidCostRemarks(OptimizationRemarkEmitter *ORE);
@@ -1007,7 +1012,7 @@ class LoopVectorizationPlanner {
   /// Build an initial VPlan, with HCFG wrapping the original scalar loop and
   /// scalar transformations applied. Returns null if an initial VPlan cannot
   /// be built.
-  VPlanPtr tryToBuildVPlan1(bool IsEpilogueTFEnabled);
+  VPlanPtr tryToBuildVPlan1(LoopVectorizationCostModel &EnabledCM);
 
   /// Build a VPlan using VPRecipes according to the information gathered by
   /// Legal and VPlan-based analysis. For outer loops, performs basic recipe
@@ -1017,13 +1022,14 @@ class LoopVectorizationPlanner {
   /// maximum VF for which no plan could be built. Each VPlan is built starting
   /// from a copy of \p InitialPlan, which is a plain CFG VPlan wrapping the
   /// original scalar loop.
-  VPlanPtr tryToBuildVPlan(VPlanPtr InitialPlan, VFRange &Range);
+  VPlanPtr tryToBuildVPlan(VPlanPtr InitialPlan, VFRange &Range,
+                           LoopVectorizationCostModel &EnabledCM);
 
   /// Build VPlans for power-of-2 VF's between \p MinVF and \p MaxVF inclusive,
   /// based on \p VPlan1 and according to the information gathered by Legal
   /// when it checked if it is legal to vectorize the loop.
   void buildVPlans(VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF,
-                   bool IsEpilogueTFEnabled);
+                   LoopVectorizationCostModel &EnabledCM);
 
   /// Add ComputeReductionResult recipes to the middle block to compute the
   /// final reduction results. Add Select recipes to the latch block when
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 92c86ad6ee570..06385385f631f 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3453,8 +3453,7 @@ bool LoopVectorizationCostModel::isEpilogueVectorizationProfitable(
 }
 
 std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
-    VPlan &MainPlan, ElementCount MainLoopVF, unsigned IC,
-    bool IsEpilogueTFEnabled) {
+    VPlan &MainPlan, ElementCount MainLoopVF, unsigned IC) {
   if (!EnableEpilogueVectorization) {
     LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization is disabled.\n");
     return nullptr;
@@ -3694,7 +3693,7 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
     if (VF.isScalar())
       LoopCost = CM.expectedCost(VF);
     else
-      LoopCost = cost(Plan, VF, &R);
+      LoopCost = cost(Plan, VF, &R, CM);
     assert(LoopCost.isValid() && "Expected to have chosen a VF with valid cost");
 
     // Loop body is free and there is no need for interleaving.
@@ -5488,8 +5487,7 @@ void LoopVectorizationCostModel::collectValuesToIgnore() {
   }
 }
 
-void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
-                                    bool IsEpilogueTFEnabled) {
+void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   CM.collectValuesToIgnore();
   Config.collectElementTypesForWidening(&CM.ValuesToIgnore);
 
@@ -5504,7 +5502,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
   if (MaxFactors.FixedVF.isVector() || MaxFactors.ScalableVF.isVector())
     Legal->collectUnitStridePredicates();
 
-  auto VPlan1 = tryToBuildVPlan1(/*IsEpilogueTFEnabled*/ false);
+  auto VPlan1 = tryToBuildVPlan1(CM);
   if (!VPlan1)
     return;
 
@@ -5513,7 +5511,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
     // plan for that VF only.
     ElementCount VF =
         MaxFactors.FixedVF ? MaxFactors.FixedVF : MaxFactors.ScalableVF;
-    buildVPlans(*VPlan1, VF, VF, /*IsEpilogueTFEnabled*/ false);
+    buildVPlans(*VPlan1, VF, VF, CM);
     LLVM_DEBUG(printPlans(dbgs()));
     return;
   }
@@ -5552,24 +5550,19 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
       // Collect the instructions (and their associated costs) that will be more
       // profitable to scalarize.
       CM.collectNonVectorizedAndSetWideningDecisions(UserVF);
-      buildVPlans(*VPlan1, UserVF, UserVF, /*IsEpilogueTFEnabled*/ false);
+      buildVPlans(*VPlan1, UserVF, UserVF, CM);
+
       ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
       if (EpilogueUserVF.isVector() &&
           ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
         CM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
-        if (IsEpilogueTFEnabled) {
-          auto TFVPlan1 = tryToBuildVPlan1(/*IsEpilogueTFEnabled*/ true);
-          buildVPlans(*TFVPlan1, EpilogueUserVF, EpilogueUserVF,
-                      /*IsEpilogueTFEnabled*/ true);
-        } else
-          buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF,
-                      /*IsEpilogueTFEnabled*/ false);
+        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, CM);
       }
       if (!VPlans.empty() && VPlans.front()->getSingleVF() == UserVF) {
         // For scalar VF, skip VPlan cost check as VPlan cost is designed for
         // vector VFs only.
         if (UserVF.isScalar() ||
-            cost(*VPlans.front(), UserVF, /*RU=*/nullptr).isValid()) {
+            cost(*VPlans.front(), UserVF, /*RU=*/nullptr, CM).isValid()) {
           LLVM_DEBUG(dbgs() << "LV: Using user VF " << UserVF << ".\n");
           LLVM_DEBUG(printPlans(dbgs()));
           return;
@@ -5595,13 +5588,73 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
     CM.collectNonVectorizedAndSetWideningDecisions(VF);
   }
 
-  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF,
-              /*IsEpilogueTFEnabled*/ false);
-  buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF,
-              /*IsEpilogueTFEnabled*/ false);
+  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF, CM);
+  buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF, CM);
+
   LLVM_DEBUG(printPlans(dbgs()));
 }
 
+bool LoopVectorizationPlanner::planForEpilogueTF(
+    ElementCount UserVF, unsigned UserIC, ElementCount EpilogueUserVF,
+    LoopVectorizationCostModel &EpilogueCM) {
+  if (VPlans.empty())
+    return false;
+  if (!OrigLoop->isInnermost())
+    return false;
+
+  if (!EpilogueUserVF.isVector() ||
+      ElementCount::isKnownGE(EpilogueUserVF, UserVF))
+    return false;
+
+  EpilogueCM.ValuesToIgnore.insert_range(CM.ValuesToIgnore);
+  EpilogueCM.VecValuesToIgnore.insert_range(CM.VecValuesToIgnore);
+
+  FixedScalableVFPair MaxFactors =
+      EpilogueCM.computeMaxVF(EpilogueUserVF, UserIC);
+  if (!MaxFactors ||
+      !EpilogueCM.foldTailByMasking()) // Cases that should not to be vectorized
+                                       // nor tail-folded.
+    return false;
+
+  auto VPlan1 = tryToBuildVPlan1(EpilogueCM);
+  if (!VPlan1)
+    return false;
+
+  // Invalidate interleave groups if all blocks of loop will be predicated.
+  if (EpilogueCM.blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
+      !useMaskedInterleavedAccesses(TTI)) {
+    LLVM_DEBUG(
+        dbgs() << "LV: [EpilogueTF] Invalidate all interleaved groups due to "
+                  "fold-tail "
+                  "by masking which requires masked-interleaved support.\n");
+    if (EpilogueCM.InterleaveInfo.invalidateGroups())
+      // Invalidating interleave groups also requires invalidating all decisions
+      // based on them, which includes widening decisions and uniform and scalar
+      // values.
+      EpilogueCM.invalidateCostModelingDecisions();
+  }
+
+  if (EpilogueCM.foldTailByMasking())
+    Legal->prepareToFoldTailByMasking();
+
+  // Collect the instructions (and their associated costs) that will be more
+  // profitable to scalarize.
+  EpilogueCM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
+
+  assert(VPlans.size() == 2 &&
+         "For tail-folded epilogue, VPlans size is expected to be 2");
+  // remove last vplan which should be the epilogue plan to replace it by our
+  // tail-folded vplan:
+  assert(VPlans.back()->getSingleVF() == EpilogueUserVF &&
+         "For tail-folded epilogue, first vplan is expected to have "
+         "EpilogueUserVF");
+  VPlans.pop_back();
+  buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, EpilogueCM);
+
+  cost(*VPlans.back(), EpilogueUserVF, /*RU=*/nullptr, EpilogueCM);
+  return true;
+}
+
 VPCostContext::VPCostContext(const TargetLibraryInfo &TLI, const VPlan &Plan,
                              LoopVectorizationCostModel &CM,
                              VFSelectionContext &Config,
@@ -5796,8 +5849,8 @@ LoopVectorizationPlanner::precomputeCosts(VPlan &Plan, ElementCount VF,
 }
 
 InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
-                                               VPRegisterUsage *RU) const {
-  VPCostContext CostCtx(*TLI, Plan, CM, Config,
+                                               VPRegisterUsage *RU, LoopVectorizationCostModel &EnabledCM) const {
+  VPCostContext CostCtx(*TLI, Plan, EnabledCM, Config,
                         /*ReusePrintingSlotTracker=*/true);
   InstructionCost Cost = precomputeCosts(Plan, VF, CostCtx);
 
@@ -5920,7 +5973,7 @@ LoopVectorizationPlanner::computeBestVF() {
       }
 
       InstructionCost Cost =
-          cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
+          cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr, CM);
       VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
 
       if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
@@ -5945,7 +5998,7 @@ LoopVectorizationPlanner::computeBestVF() {
 
 DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
     ElementCount BestVF, unsigned BestUF, VPlan &BestVPlan,
-    InnerLoopVectorizer &ILV, DominatorTree *DT, bool IsEpilogueTFEnabled,
+    InnerLoopVectorizer &ILV, DominatorTree *DT,
     EpilogueVectorizationKind EpilogueVecKind) {
   assert(BestVPlan.hasVF(BestVF) &&
          "Trying to execute plan with unsupported VF");
@@ -6546,7 +6599,8 @@ VPRecipeBuilder::tryToCreateWidenNonPhiRecipe(VPSingleDefRecipe *R,
 // optimizations.
 static void printOptimizedVPlan(VPlan &) {}
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(bool IsEpilogueTFEnabled) {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(
+    LoopVectorizationCostModel &EnabledCM) {
   bool IsInnerLoop = OrigLoop->isInnermost();
 
   // Set up loop versioning for inner loops with memory runtime checks.
@@ -6596,8 +6650,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(bool IsEpilogueTFEnabled) {
   bool ForceVectorization = Hints.getForce() == LoopVectorizeHints::FK_Enabled;
   bool OptForSize =
       !ForceVectorization &&
-      (CM.EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
-       CM.EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
+      (EnabledCM.EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
+       EnabledCM.EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
   unsigned SCEVCheckThreshold = ForceVectorization
                                     ? PragmaVectorizeSCEVCheckThreshold
                                     : VectorizeSCEVCheckThreshold;
@@ -6625,24 +6679,24 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(bool IsEpilogueTFEnabled) {
 
   RUN_VPLAN_PASS(VPlanTransforms::createLoopRegions, *VPlan0,
                  getDebugLocFromInstOrOperands(Legal->getPrimaryInduction()));
-  if (CM.foldTailByMasking() || IsEpilogueTFEnabled)
+  if (EnabledCM.foldTailByMasking())
     RUN_VPLAN_PASS(VPlanTransforms::foldTailByMasking, *VPlan0);
   RUN_VPLAN_PASS(VPlanTransforms::introduceMasksAndLinearize, *VPlan0);
 
   return VPlan0;
 }
 
-void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
-                                           ElementCount MaxVF,
-                                           bool IsEpilogueTFEnabled) {
+void LoopVectorizationPlanner::buildVPlans(
+    VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF,
+    LoopVectorizationCostModel &EnabledCM) {
   if (ElementCount::isKnownGT(MinVF, MaxVF))
     return;
 
   auto MaxVFTimes2 = MaxVF * 2;
   for (ElementCount VF = MinVF; ElementCount::isKnownLT(VF, MaxVFTimes2);) {
     VFRange SubRange = {VF, MaxVFTimes2};
-    auto Plan =
-        tryToBuildVPlan(std::unique_ptr<VPlan>(VPlan1.duplicate()), SubRange);
+    auto Plan = tryToBuildVPlan(std::unique_ptr<VPlan>(VPlan1.duplicate()),
+                                SubRange, EnabledCM);
     VF = SubRange.End;
 
     if (!Plan)
@@ -6655,7 +6709,7 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
                    Config.getMinimalBitwidths());
     RUN_VPLAN_PASS(VPlanTransforms::optimize, *Plan);
     // TODO: try to put addExplicitVectorLength close to addActiveLaneMask
-    if (CM.foldTailWithEVL()) {
+    if (EnabledCM.foldTailWithEVL()) {
       RUN_VPLAN_PASS(VPlanTransforms::addExplicitVectorLength, *Plan,
                      Config.getMaxSafeElements());
       RUN_VPLAN_PASS(VPlanTransforms::optimizeEVLMasks, *Plan);
@@ -6665,9 +6719,7 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
             RUN_VPLAN_PASS(VPlanTransforms::narrowInterleaveGroups, *Plan, TTI))
       VPlans.push_back(std::move(P));
 
-    TailFoldingStyle Style = CM.getTailFoldingStyle();
-    if (IsEpilogueTFEnabled)
-      Style = TailFoldingStyle::Data;
+    TailFoldingStyle Style = EnabledCM.getTailFoldingStyle();
     RUN_VPLAN_PASS(VPlanTransforms::materializeHeaderMask, *Plan,
                    useActiveLaneMask(Style),
                    useActiveLaneMaskForControlFlow(Style));
@@ -6678,8 +6730,8 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
   }
 }
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
-                                                   VFRange &Range) {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
+    VPlanPtr Plan, VFRange &Range, LoopVectorizationCostModel &EnabledCM) {
 
   // For outer loops, the plan only needs basic recipe conversion and induction
   // live-out optimization; the full inner-loop recipe building below does not
@@ -6705,8 +6757,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
 
   bool RequiresScalarEpilogueCheck =
       LoopVectorizationPlanner::getDecisionAndClampRange(
-          [this](ElementCount VF) {
-            return !CM.requiresScalarEpilogue(VF.isVector());
+          [EnabledCM](ElementCount VF) {
+            return !EnabledCM.requiresScalarEpilogue(VF.isVector());
           },
           Range);
   // Update the branch in the middle block if a scalar epilogue is required.
@@ -6724,9 +6776,9 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // TODO: Consider using getDecisionAndClampRange here to split up VPlans.
   bool IVUpdateMayOverflow = false;
   for (ElementCount VF : Range)
-    IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(&CM, VF);
+    IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(&EnabledCM, VF);
 
-  TailFoldingStyle Style = CM.getTailFoldingStyle();
+  TailFoldingStyle Style = EnabledCM.getTailFoldingStyle();
   // Use NUW for the induction increment if we proved that it won't overflow in
   // the vector loop or when not folding the tail. In the later case, we know
   // that the canonical induction increment will not overflow as the vector trip
@@ -6752,10 +6804,11 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // Range, add it to the set of groups to be later applied to the VPlan and add
   // placeholders for its members' Recipes which we'll be replacing with a
   // single VPInterleaveRecipe.
-  for (InterleaveGroup<Instruction> *IG : IAI.getInterleaveGroups()) {
-    auto ApplyIG = [IG, this](ElementCount VF) -> bool {
+  for (InterleaveGroup<Instruction> *IG :
+       EnabledCM.InterleaveInfo.getInterleaveGroups()) {
+    auto ApplyIG = [IG, EnabledCM](ElementCount VF) -> bool {
       bool Result = (VF.isVector() && // Query is illegal for VF == 1
-                     CM.getWideningDecision(IG->getInsertPos(), VF) ==
+                     EnabledCM.getWideningDecision(IG->getInsertPos(), VF) ==
                          LoopVectorizationCostModel::CM_Interleave);
       // For scalable vectors, the interleave factors must be <= 8 since we
       // require the (de)interleaveN intrinsics instead of shufflevectors.
@@ -6772,7 +6825,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // Construct wide recipes and apply predication for original scalar
   // VPInstructions in the loop.
   // ---------------------------------------------------------------------------
-  VPRecipeBuilder RecipeBuilder(*Plan, Legal, CM, Builder);
+  VPRecipeBuilder RecipeBuilder(*Plan, Legal, EnabledCM, Builder);
 
   // Scan the body of the loop in a topological order to visit each basic block
   // after having visited its predecessor basic blocks.
@@ -6783,7 +6836,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   RUN_VPLAN_PASS(VPlanTransforms::createInLoopReductionRecipes, *Plan,
                  Range.Start);
 
-  VPCostContext CostCtx(*TLI, *Plan, CM, Config);
+  VPCostContext CostCtx(*TLI, *Plan, EnabledCM, Config);
 
   RUN_VPLAN_PASS(VPlanTransforms::makeMemOpWideningDecisions, *Plan, Range,
                  RecipeBuilder, CostCtx);
@@ -6883,7 +6936,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // range for better cost estimation.
   // TODO: Enable following transform when the EVL-version of extended-reduction
   // and mulacc-reduction are implemented.
-  if (!CM.foldTailWithEVL()) {
+  if (!EnabledCM.foldTailWithEVL()) {
     RUN_VPLAN_PASS(VPlanTransforms::createPartialReductions, *Plan, CostCtx,
                    Range);
     RUN_VPLAN_PASS(VPlanTransforms::convertToAbstractRecipes, *Plan, CostCtx,
@@ -6894,7 +6947,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // for this VPlan, replace the Recipes widening its memory instructions with a
   // single VPInterleaveRecipe at its insertion point.
   RUN_VPLAN_PASS(VPlanTransforms::createInterleaveGroups, *Plan,
-                 InterleaveGroups, CM.isEpilogueAllowed());
+                 InterleaveGroups, EnabledCM.isEpilogueAllowed());
 
   // Convert memory recipes to strided access recipes if the strided access is
   // legal and profitable.
@@ -6911,7 +6964,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
 
   RUN_VPLAN_PASS(VPlanTransforms::dropPoisonGeneratingRecipes, *Plan);
 
-  if (CM.maskPartialAliasing())
+  if (EnabledCM.maskPartialAliasing())
     RUN_VPLAN_PASS(VPlanTransforms::attachAliasMaskToHeaderMask, *Plan);
 
   assert(verifyVPlanIsValid(*Plan) && "VPlan is invalid");
@@ -7623,7 +7676,8 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
   VPValue *VPV = Plan.getOrAddLiveIn(EPResumeVal);
   assert(all_of(IV->users(),
                 [](const VPUser *U) {
-                  if (isa<VPScalarIVStepsRecipe, VPDerivedIVRecipe>(U))
+                  if (isa<VPScalarIVStepsRecipe, VPDerivedIVRecipe,
+                          VPWidenCanonicalIVRecipe>(U))
                     return true;
                   unsigned Opc = cast<VPInstruction>(U)->getOpcode();
                   return Instruction::isCast(Opc) || Opc == Instruction::Add;
@@ -7637,7 +7691,7 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
   auto *Increment = vputils::findCanonicalIVIncrement(Plan);
   assert(Increment && "Must have a canonical IV increment at this point");
   IV->replaceUsesWithIf(Add, [Add, Increment](VPUser &U, unsigned) {
-    return &U != Add && &U != Increment;
+    return &U != Add && &U != Increment && !isa<VPWidenCanonicalIVRecipe>(&U);
   });
   VPInstruction *OffsetIVInc =
       VPBuilder::getToInsertAfter(Increment).createAdd(Increment, VPV);
@@ -7748,6 +7802,23 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
           continue;
         }
       }
+    } else if (isa<VPActiveLaneMaskPHIRecipe>(&R)) {
+      // The active-lane-mask phi's entry value was computed assuming the
+      // epilogue vector loop starts at 0. Rebuild it using the resume value
+      // instead, so the mask reflects how many elements the main vector loop
+      // already processed.
+      VPBuilder EntryBuilder(Plan.getVectorPreheader());
+      Type *CanIVTy = VectorLoop->getCanonicalIVType();
+      VPValue *ALMMultiplier = Plan.getConstantInt(CanIVTy, 1);
+      auto *EntryIncrement = EntryBuilder.createOverflowingOp(
+          VPInstruction::CanonicalIVIncrementForPart, {VPV, &Plan.getVF()}, {},
+          R.getDebugLoc(), "index.part.next");
+      auto *EntryALM = EntryBuilder.createNaryOp(
+          VPInstruction::ActiveLaneMask,
+          {EntryIncrement, Plan.getTripCount(), ALMMultiplier}, R.getDebugLoc(),
+          "active.lane.mask.entry");
+      cast<VPHeaderPHIRecipe>(&R)->setStartValue(EntryALM);
+      continue;
     } else {
       // Retrieve the induction resume value via ResumeForEpilogue.
       PHINode *IndPhi = cast<VPWidenInductionRecipe>(&R)->getPHINode();
@@ -7839,7 +7910,7 @@ fixScalarResumeValuesFromBypass(BasicBlock *BypassBlock, Loop *L,
 /// and runtime checks of the main loop, as well as updating various phis. \p
 /// InstsToMove contains instructions that need to be moved to the preheader of
 /// the epilogue vector loop.
-static void connectEpilogueVectorLoop(VPlan &MainPlan, VPlan &EpiPlan, Loop *L,
+static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
                                       EpilogueLoopVectorizationInfo &EPI,
                                       DominatorTree *DT, LoopInfo *LI,
                                       GeneratedRTChecks &Checks,
@@ -7913,8 +7984,7 @@ static void connectEpilogueVectorLoop(VPlan &MainPlan, VPlan &EpiPlan, Loop *L,
     // When the epilogue is tail-folded, EpilogueIterationCountCheck
     // (iter.check) is redirected to branch straight into the vector epilogue
     // preheader (see the IsEpilogueTFEnabled redirect above), so it is now a
-    // genuine predecessor and its incoming value must be kept rather than
-    // stripped.
+    // predecessor and its incoming value must be kept rather than stripped.
     // TODO: revisit for reduction phis, whose resume value on this bypass
     // edge may need dedicated handling rather than reusing the value already
     // present here.
@@ -7957,25 +8027,18 @@ static void connectEpilogueVectorLoop(VPlan &MainPlan, VPlan &EpiPlan, Loop *L,
     Br->eraseFromParent();
     DTU.applyUpdates(
         {{DominatorTree::Delete, VecEpilogueIterationCountCheck, DeadSucc}});
-  }
 
-  if (IsEpilogueTFEnabled && !SCEVCheckBlock && !MemCheckBlock) {
-    // No runtime check still needs a scalar fallback, and every trip-count
-    // bypass edge above has been redirected away from the scalar preheader:
-    // the scalar loop is now entirely unreachable. Delete it outright, along
-    // with its LoopInfo entry, instead of leaving it behind as dead code.
-    // This must run last, since fixScalarResumeValuesFromBypass (above) and
-    // the DT/function verification in processLoop (below) still need `L` and
-    // ScalarPH to be valid up to this point.
-    assert(pred_empty(ScalarPH) &&
-           "scalar preheader should have no predecessors left");
-    SmallVector<BasicBlock *> Blocks(L->block_begin(),
-                                     L->block_end());
-    Blocks.push_back(ScalarPH);
-    LI->erase(L);
-    for (auto *BB : Blocks)
-      LI->removeBlock(BB);
-    DeleteDeadBlocks(Blocks, &DTU);
+    if (!SCEVCheckBlock && !MemCheckBlock) {
+      // Delete the scalar loop as it's dead right
+      assert(pred_empty(ScalarPH) &&
+             "scalar preheader should have no predecessors left");
+      SmallVector<BasicBlock *> Blocks(L->block_begin(), L->block_end());
+      Blocks.push_back(ScalarPH);
+      LI->erase(L);
+      for (auto *BB : Blocks)
+        LI->removeBlock(BB);
+      DeleteDeadBlocks(Blocks, &DTU);
+    }
   }
 }
 
@@ -8169,11 +8232,19 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   EpilogueLowering EpilogueTailLoweringStatus =
       getEpilogueTailLowering(CM, L, ORE, LVL, Hints);
   bool IsEpilogueTFEnabled = false;
+  std::optional<InterleavedAccessInfo> TailFoldingCMIAI;
+  std::optional<LoopVectorizationCostModel> EpilogueTailFoldingCM;
   if (EpilogueTailLoweringStatus ==
       EpilogueLowering::CM_EpilogueNotNeededFoldTail) {
-    // TODO: Apply tail-folding on the vectorized epilogue loop.
     LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is enabled\n");
     IsEpilogueTFEnabled = true;
+    TailFoldingCMIAI.emplace(PSE, L, DT, LI, LVL.getLAI(), OptForSize);
+    if (UseInterleaved && useMaskedInterleavedAccesses(*TTI))
+      TailFoldingCMIAI->analyzeInterleaving(
+          /*useMaskedInterleavedAccesses*/ true);
+    EpilogueTailFoldingCM.emplace(CM_EpilogueNotNeededFoldTail, L, PSE, LI,
+                                  &LVL, *TTI, TLI, AC, ORE, GetBFI, F, &Hints,
+                                  *TailFoldingCMIAI, Config);
   }
 
   // Get user vectorization factor and interleave count.
@@ -8187,7 +8258,16 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     UserIC = 1;
 
   // Plan how to best vectorize.
-  LVP.plan(UserVF, UserIC, IsEpilogueTFEnabled);
+  LVP.plan(UserVF, UserIC);
+  if (IsEpilogueTFEnabled)
+    if (!LVP.planForEpilogueTF(UserVF, /*UserIC*/ 1,
+                               EpilogueVectorizationForceVF,
+                               *EpilogueTailFoldingCM)) {
+      // we can't apply epilogue TF:
+      EpilogueTailFoldingCM.reset();
+      IsEpilogueTFEnabled = false;
+    }
+
   auto [VF, BestPlanPtr] = LVP.computeBestVF();
   unsigned IC = 1;
 
@@ -8392,7 +8472,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   VPlan &BestPlan = *BestPlanPtr;
   // Consider vectorizing the epilogue too if it's profitable.
   std::unique_ptr<VPlan> EpiPlan =
-      LVP.selectBestEpiloguePlan(BestPlan, VF.Width, IC, IsEpilogueTFEnabled);
+      LVP.selectBestEpiloguePlan(BestPlan, VF.Width, IC);
   bool HasBranchWeights =
       hasBranchWeightMD(*L->getLoopLatch()->getTerminator());
   if (EpiPlan) {
@@ -8425,7 +8505,6 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                                        BestMainPlan);
     auto ExpandedSCEVs = LVP.executePlan(
         EPI.MainLoopVF, EPI.MainLoopUF, BestMainPlan, MainILV, DT,
-        /*IsEpilogueTFEnabled*/ false,
         LoopVectorizationPlanner::EpilogueVectorizationKind::MainLoop);
     ++LoopsVectorized;
 
@@ -8457,10 +8536,9 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     LVP.attachRuntimeChecks(BestEpiPlan, Checks, HasBranchWeights);
     LVP.executePlan(
         EPI.EpilogueVF, EPI.EpilogueUF, BestEpiPlan, EpilogILV, DT,
-        IsEpilogueTFEnabled,
         LoopVectorizationPlanner::EpilogueVectorizationKind::Epilogue);
-    connectEpilogueVectorLoop(BestMainPlan, BestEpiPlan, L, EPI, DT, LI, Checks,
-                              InstsToMove, ResumeValues, IsEpilogueTFEnabled);
+    connectEpilogueVectorLoop(BestEpiPlan, L, EPI, DT, LI, Checks, InstsToMove,
+                              ResumeValues, IsEpilogueTFEnabled);
     ++LoopsEpilogueVectorized;
   } else {
     InnerLoopVectorizer LB(L, PSE, LI, DT, TTI, AC, VF.Width, IC, Checks,
@@ -8472,8 +8550,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     if (!IsInnerLoop)
       LLVM_DEBUG(dbgs() << "Vectorizing outer loop in \"" << F->getName()
                         << "\"\n");
-    LVP.executePlan(VF.Width, IC, BestPlan, LB, DT,
-                    /*IsEpilogueTFEnabled*/ false);
+    LVP.executePlan(VF.Width, IC, BestPlan, LB, DT);
     ++LoopsVectorized;
   }
 
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index a94cf64668ff6..746e4495e059e 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -1,11 +1,11 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -mtriple=aarch64-linux-gnu -S %s | FileCheck %s
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug -mtriple=aarch64-linux-gnu -mcpu=neoverse-v1 -S %s | FileCheck %s
 
 
 define void @test_epilogue_tf(ptr %A, i64 %n) {
 ; CHECK-LABEL: define void @test_epilogue_tf(
-; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ITER_CHECK:.*]]:
 ; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
@@ -13,18 +13,18 @@ define void @test_epilogue_tf(ptr %A, i64 %n) {
 ; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
 ; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[N_MOD_VF:%.*]] = and i64 [[N]], 31
-; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
-; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 16
-; CHECK-NEXT:    store <16 x i8> splat (i8 1), ptr [[TMP0]], align 1
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP1]], i64 16
 ; CHECK-NEXT:    store <16 x i8> splat (i8 1), ptr [[TMP1]], align 1
+; CHECK-NEXT:    store <16 x i8> splat (i8 1), ptr [[TMP2]], align 1
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
-; CHECK-NEXT:    [[TMP2:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP2]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
@@ -32,17 +32,18 @@ define void @test_epilogue_tf(ptr %A, i64 %n) {
 ; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
 ; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[N_RND_UP:%.*]] = add i64 [[N]], 7
-; CHECK-NEXT:    [[N_MOD_VF2:%.*]] = and i64 [[N_RND_UP]], 7
-; CHECK-NEXT:    [[N_VEC3:%.*]] = sub i64 [[N_RND_UP]], [[N_MOD_VF2]]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
 ; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
 ; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT5:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX4]]
-; CHECK-NEXT:    store <8 x i8> splat (i8 1), ptr [[TMP3]], align 1
-; CHECK-NEXT:    [[INDEX_NEXT5]] = add nuw i64 [[INDEX4]], 8
-; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT5]], [[N_VEC3]]
-; CHECK-NEXT:    br i1 [[TMP4]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX2]]
+; CHECK-NEXT:    call void @llvm.masked.store.v8i8.p0(<8 x i8> splat (i8 1), ptr align 1 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP6:%.*]] = xor i1 [[TMP5]], true
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    br label %[[EXIT]]
 ; CHECK:       [[EXIT]]:

>From 14f8c916f16acb4becd83bcc20c74229c2ebfa36 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Thu, 30 Jul 2026 17:41:23 +0100
Subject: [PATCH 04/11] resolve review comments

---
 .../Transforms/Vectorize/LoopVectorize.cpp    | 72 +++++++++----------
 .../Vectorize/VPlanConstruction.cpp           |  2 +
 .../AArch64/fold-epilogue-tail.ll             | 33 ++++++++-
 .../LoopVectorize/fold-epilogue-tail.ll       | 56 ++++++++++++++-
 4 files changed, 124 insertions(+), 39 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 06385385f631f..08e1c568209cd 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5550,6 +5550,9 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
       // Collect the instructions (and their associated costs) that will be more
       // profitable to scalarize.
       CM.collectNonVectorizedAndSetWideningDecisions(UserVF);
+      // Build the main-loop VPlan firstly because if epilogue tail-folding is
+      // enabled, it will be built later, so we keep the epilogue vplans at the
+      // end.
       buildVPlans(*VPlan1, UserVF, UserVF, CM);
 
       ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
@@ -5597,56 +5600,61 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
 bool LoopVectorizationPlanner::planForEpilogueTF(
     ElementCount UserVF, unsigned UserIC, ElementCount EpilogueUserVF,
     LoopVectorizationCostModel &EpilogueCM) {
-  if (VPlans.empty())
+  if (VPlans.empty()) {
+    LLVM_DEBUG(dbgs() << "LV: no vplans have been built for main loop VF, bail "
+                         "out of epilogue tail-folding\n");
     return false;
+  }
   if (!OrigLoop->isInnermost())
     return false;
 
   if (!EpilogueUserVF.isVector() ||
-      ElementCount::isKnownGE(EpilogueUserVF, UserVF))
+      (estimateElementCount(EpilogueUserVF, Config.getVScaleForTuning()) >=
+       estimateElementCount(UserVF, Config.getVScaleForTuning())) *
+          UserIC)
     return false;
 
   EpilogueCM.ValuesToIgnore.insert_range(CM.ValuesToIgnore);
   EpilogueCM.VecValuesToIgnore.insert_range(CM.VecValuesToIgnore);
 
   FixedScalableVFPair MaxFactors =
-      EpilogueCM.computeMaxVF(EpilogueUserVF, UserIC);
+      EpilogueCM.computeMaxVF(EpilogueUserVF, /*UserIC*/ 1);
   if (!MaxFactors ||
-      !EpilogueCM.foldTailByMasking()) // Cases that should not to be vectorized
-                                       // nor tail-folded.
+      !EpilogueCM
+           .foldTailByMasking()) { // Cases that should not to be vectorized
+                                   // or tail-folded.
+    reportVectorizationInfo("This case of epilogue loop can't be tail-folded",
+                            "InvalidTailFoldedEpilogue", ORE, OrigLoop);
     return false;
+  }
 
   auto VPlan1 = tryToBuildVPlan1(EpilogueCM);
   if (!VPlan1)
     return false;
 
-  // Invalidate interleave groups if all blocks of loop will be predicated.
-  if (EpilogueCM.blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
-      !useMaskedInterleavedAccesses(TTI)) {
+  if (!useMaskedInterleavedAccesses(TTI)) {
     LLVM_DEBUG(
-        dbgs() << "LV: [EpilogueTF] Invalidate all interleaved groups due to "
-                  "fold-tail "
-                  "by masking which requires masked-interleaved support.\n");
+        dbgs() << "LV: Invalidate all interleaved groups due to fold-tail by "
+                  "masking which requires masked-interleaved support.\n");
     if (EpilogueCM.InterleaveInfo.invalidateGroups())
       // Invalidating interleave groups also requires invalidating all decisions
       // based on them, which includes widening decisions and uniform and scalar
       // values.
       EpilogueCM.invalidateCostModelingDecisions();
   }
-
-  if (EpilogueCM.foldTailByMasking())
-    Legal->prepareToFoldTailByMasking();
+  Legal->prepareToFoldTailByMasking();
 
   // Collect the instructions (and their associated costs) that will be more
   // profitable to scalarize.
   EpilogueCM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
 
+  // Expecting only 2 vplans as epilogue TF is applied only for forced VFs.
   assert(VPlans.size() == 2 &&
          "For tail-folded epilogue, VPlans size is expected to be 2");
-  // remove last vplan which should be the epilogue plan to replace it by our
-  // tail-folded vplan:
+  // Remove the last vplan, which should be the epilogue plan to replace it by
+  // the tail-folded vplan:
   assert(VPlans.back()->getSingleVF() == EpilogueUserVF &&
-         "For tail-folded epilogue, first vplan is expected to have "
+         "For tail-folded epilogue, last vplan is expected to have "
          "EpilogueUserVF");
   VPlans.pop_back();
   buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, EpilogueCM);
@@ -7327,13 +7335,6 @@ getEpilogueTailLowering(const LoopVectorizationCostModel &MainCM, const Loop *L,
     return CM_EpilogueAllowed;
   }
 
-  if (LVL.hasUncountableEarlyExit()) {
-    LLVM_DEBUG(dbgs() << "LV: Epilogue tail-folding can't be applied because "
-                         " of loop has early exit\n"
-                         "LV: Fall back to a normal epilogue\n");
-    return CM_EpilogueAllowed;
-  }
-
   // If having epilogue is NOT allowed, then no epilogue to apply TF for.
   if (!MainCM.isEpilogueAllowed()) {
     LLVM_DEBUG(dbgs() << "LV: No epilogue to apply tail-folding for.\n"
@@ -8010,7 +8011,7 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
 
   if (IsEpilogueTFEnabled) {
     // The epilogue vector loop is tail-folded, so it can safely handle
-    // any remaining trip count, including zero, via masking.
+    // any remaining iterations, including zero, via masking.
     // vec.epilog.iter.check's own min-iters check was therefore built with a
     // compile-time-known-false condition (see
     // addMinimumVectorEpilogueIterationCheck) that never needs to bail out to
@@ -8029,7 +8030,7 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
         {{DominatorTree::Delete, VecEpilogueIterationCountCheck, DeadSucc}});
 
     if (!SCEVCheckBlock && !MemCheckBlock) {
-      // Delete the scalar loop as it's dead right
+      // Delete the scalar loop as it's dead right now.
       assert(pred_empty(ScalarPH) &&
              "scalar preheader should have no predecessors left");
       SmallVector<BasicBlock *> Blocks(L->block_begin(), L->block_end());
@@ -8231,17 +8232,14 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   EpilogueLowering EpilogueTailLoweringStatus =
       getEpilogueTailLowering(CM, L, ORE, LVL, Hints);
-  bool IsEpilogueTFEnabled = false;
   std::optional<InterleavedAccessInfo> TailFoldingCMIAI;
   std::optional<LoopVectorizationCostModel> EpilogueTailFoldingCM;
   if (EpilogueTailLoweringStatus ==
       EpilogueLowering::CM_EpilogueNotNeededFoldTail) {
     LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is enabled\n");
-    IsEpilogueTFEnabled = true;
     TailFoldingCMIAI.emplace(PSE, L, DT, LI, LVL.getLAI(), OptForSize);
-    if (UseInterleaved && useMaskedInterleavedAccesses(*TTI))
-      TailFoldingCMIAI->analyzeInterleaving(
-          /*useMaskedInterleavedAccesses*/ true);
+    if (UseInterleaved)
+      TailFoldingCMIAI->analyzeInterleaving(useMaskedInterleavedAccesses(*TTI));
     EpilogueTailFoldingCM.emplace(CM_EpilogueNotNeededFoldTail, L, PSE, LI,
                                   &LVL, *TTI, TLI, AC, ORE, GetBFI, F, &Hints,
                                   *TailFoldingCMIAI, Config);
@@ -8259,13 +8257,13 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   // Plan how to best vectorize.
   LVP.plan(UserVF, UserIC);
-  if (IsEpilogueTFEnabled)
-    if (!LVP.planForEpilogueTF(UserVF, /*UserIC*/ 1,
-                               EpilogueVectorizationForceVF,
-                               *EpilogueTailFoldingCM)) {
+  if (EpilogueTailFoldingCM.has_value())
+    if (!LVP.planForEpilogueTF(UserVF, UserIC, EpilogueVectorizationForceVF,
+                               EpilogueTailFoldingCM.value())) {
       // we can't apply epilogue TF:
+      LLVM_DEBUG(
+          dbgs() << "LV: Applying epilogue tail-folding failed, disable it.\n");
       EpilogueTailFoldingCM.reset();
-      IsEpilogueTFEnabled = false;
     }
 
   auto [VF, BestPlanPtr] = LVP.computeBestVF();
@@ -8538,7 +8536,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
         EPI.EpilogueVF, EPI.EpilogueUF, BestEpiPlan, EpilogILV, DT,
         LoopVectorizationPlanner::EpilogueVectorizationKind::Epilogue);
     connectEpilogueVectorLoop(BestEpiPlan, L, EPI, DT, LI, Checks, InstsToMove,
-                              ResumeValues, IsEpilogueTFEnabled);
+                              ResumeValues, EpilogueTailFoldingCM.has_value());
     ++LoopsEpilogueVectorized;
   } else {
     InnerLoopVectorizer LB(L, PSE, LI, DT, TTI, AC, VF.Width, IC, Checks,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 30c7f16ac88aa..6576a0256644c 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1579,6 +1579,8 @@ void VPlanTransforms::addMinimumVectorEpilogueIterationCheck(
   VPBuilder Builder(cast<VPBasicBlock>(Plan.getEntry()));
 
   if (Plan.hasTailFolded()) {
+    assert(!RequiresScalarEpilogue &&
+           "Expected no scalar epilogue for tail-folded plan");
     Builder.createNaryOp(VPInstruction::BranchOnCond, Plan.getFalse());
     return;
   }
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index 746e4495e059e..02ce78dcc8e14 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -1,7 +1,11 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug -mtriple=aarch64-linux-gnu -mcpu=neoverse-v1 -S %s | FileCheck %s
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug -mcpu=neoverse-v1 -S %s | FileCheck %s
 
+; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALIDATE-INTERLEAVE
+
+target triple = "aarch64-linux-gnu"
 
 define void @test_epilogue_tf(ptr %A, i64 %n) {
 ; CHECK-LABEL: define void @test_epilogue_tf(
@@ -63,3 +67,30 @@ for.body:
 exit:
   ret void
 }
+
+define i64 @test_no_masked_interleave_support(i64 %y, i32 %n) {
+; CHECK-INVALIDATE-INTERLEAVE-LABEL: Checking a loop in 'test_no_masked_interleave_support'
+; CHECK-INVALIDATE-INTERLEAVE: LV: epilogue tail-folding is enabled
+; CHECK-INVALIDATE-INTERLEAVE: LV: Analyzing interleaved accesses...
+; CHECK-INVALIDATE-INTERLEAVE: LV: Invalidate all interleaved groups due to fold-tail by masking which requires masked-interleaved support
+entry:
+  br label %for.body
+
+for.body:
+  %i = phi i32 [ 0, %entry ], [ %inc, %cond.end ]
+  %cmp = icmp eq i64 %y, 0
+  br i1 %cmp, label %cond.end, label %cond.false
+
+cond.false:
+  %div = xor i64 3, %y
+  br label %cond.end
+
+cond.end:
+  %cond = phi i64 [ %div, %cond.false ], [ 77, %for.body ]
+  %inc = add nuw nsw i32 %i, 1
+  %exitcond = icmp eq i32 %inc, %n
+  br i1 %exitcond, label %for.cond.cleanup, label %for.body
+
+for.cond.cleanup:
+  ret i64 %cond
+}
diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index 2407c5e28c892..c07fb38709f63 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -1,5 +1,5 @@
 ; REQUIRES: asserts
-; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail\
+; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
 ; RUN: -pass-remarks-analysis=loop-vectorize -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s
 
 ; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -force-vector-width=16 \
@@ -12,6 +12,10 @@
 ; RUN: --disable-output -force-vector-width=16 -epilogue-vectorization-force-VF=8 -epilogue-tail-folding-policy=prefer-fold-tail \
 ; RUN: -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s --check-prefix=CHECK-DISABLED-EPILOG
 
+; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -vectorize-scev-check-threshold=0 < %s 2>&1 | FileCheck %s \
+; RUN: --check-prefix=CHECK-NO-VPLANS
+
 define void @test_epilogue_tf(ptr %A, i64 %n) {
 ; CHECK-LABEL: Checking a loop in 'test_epilogue_tf'
 ; CHECK: LV: epilogue tail-folding is enabled
@@ -31,6 +35,29 @@ exit:
   ret void
 }
 
+; This case can't be tail-folded because all the iterations will be executed by
+; main vector loop.
+define void @test_epilogue_tf_reset(ptr %A) {
+; CHECK-LABEL: Checking a loop in 'test_epilogue_tf_reset'
+; CHECK: LV: epilogue tail-folding is enabled
+; CHECK: LV: This case of epilogue loop can't be tail-folded.
+; CHECK: LV: Applying epilogue tail-folding failed, disable it.
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+  %arrayidx = getelementptr inbounds i8, ptr %A, i64 %iv
+  store i8 1, ptr %arrayidx, align 1
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exitcond = icmp ne i64 %iv.next, 64
+  br i1 %exitcond, label %for.body, label %exit
+
+exit:
+  ret void
+}
+
 define void @test_epilogue_tf_no_fv(ptr %A, i64 %n) {
 ; CHECK-NO-FORCED-MAIN-VF-LABEL: Checking a loop in 'test_epilogue_tf_no_fv'
 ; CHECK-NO-FORCED-MAIN-VF: remark: <unknown>:0:0: For now, Epilogue tail-folding can't be applied without forced epilogue/main loop VF
@@ -123,3 +150,30 @@ for.body:
 for.end:
   ret i32 0
 }
+
+; Can't build a valid vplan for this case because too many SCEV checks needed,
+; more than the specfied limit.
+define i64 @test_no_vplan_built(ptr %dst, i64 %n) {
+; CHECK-NO-VPLANS-LABEL: Checking a loop in 'test_no_vplan_built'
+; CHECK-NO-VPLANS: LV: epilogue tail-folding is enabled
+; CHECK-NO-VPLANS: LV: no vplans have been built for main loop VF, bail out of epilogue tail-folding
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %dead.iv = phi i16 [ 0, %entry ], [ %dead.iv.next, %loop ]
+  %prev = phi i64 [ 0, %entry ], [ %ext, %loop ]
+  %iv.next = add nuw nsw i64 %iv, 1
+  %dead.iv.next = add i16 %dead.iv, 1
+  %ext = zext i16 %dead.iv.next to i64
+  %gep = getelementptr inbounds i64, ptr %dst, i64 %prev
+  store i64 %iv, ptr %gep, align 8
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  %result = phi i64 [ %ext, %loop ]
+  ret i64 %result
+}

>From 440ce908d096cd1114367abbe3da4cb7c5be94b0 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Thu, 6 Aug 2026 00:43:59 +0100
Subject: [PATCH 05/11] resolve review comments

---
 .../Vectorize/LoopVectorizationPlanner.h      |  1 -
 .../Transforms/Vectorize/LoopVectorize.cpp    | 40 +++++++------
 .../AArch64/fold-epilogue-tail.ll             |  5 +-
 .../LoopVectorize/fold-epilogue-tail.ll       | 56 +++++++++++++++----
 4 files changed, 68 insertions(+), 34 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 9e72b3da45cd5..25991be146822 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -912,7 +912,6 @@ class LoopVectorizationPlanner {
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
   /// tail-folding should be avoided up-front, no plans are generated.
   bool planForEpilogueTF(ElementCount UserVF, unsigned UserIC,
-                         ElementCount EpilogueUserVF,
                          LoopVectorizationCostModel &EpilogueCM);
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 08e1c568209cd..9151289639975 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5598,27 +5598,19 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
 }
 
 bool LoopVectorizationPlanner::planForEpilogueTF(
-    ElementCount UserVF, unsigned UserIC, ElementCount EpilogueUserVF,
+    ElementCount UserVF, unsigned UserIC,
     LoopVectorizationCostModel &EpilogueCM) {
   if (VPlans.empty()) {
     LLVM_DEBUG(dbgs() << "LV: no vplans have been built for main loop VF, bail "
                          "out of epilogue tail-folding\n");
     return false;
   }
-  if (!OrigLoop->isInnermost())
-    return false;
-
-  if (!EpilogueUserVF.isVector() ||
-      (estimateElementCount(EpilogueUserVF, Config.getVScaleForTuning()) >=
-       estimateElementCount(UserVF, Config.getVScaleForTuning())) *
-          UserIC)
-    return false;
 
   EpilogueCM.ValuesToIgnore.insert_range(CM.ValuesToIgnore);
   EpilogueCM.VecValuesToIgnore.insert_range(CM.VecValuesToIgnore);
 
   FixedScalableVFPair MaxFactors =
-      EpilogueCM.computeMaxVF(EpilogueUserVF, /*UserIC*/ 1);
+      EpilogueCM.computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
   if (!MaxFactors ||
       !EpilogueCM
            .foldTailByMasking()) { // Cases that should not to be vectorized
@@ -5629,8 +5621,6 @@ bool LoopVectorizationPlanner::planForEpilogueTF(
   }
 
   auto VPlan1 = tryToBuildVPlan1(EpilogueCM);
-  if (!VPlan1)
-    return false;
 
   if (!useMaskedInterleavedAccesses(TTI)) {
     LLVM_DEBUG(
@@ -5646,20 +5636,23 @@ bool LoopVectorizationPlanner::planForEpilogueTF(
 
   // Collect the instructions (and their associated costs) that will be more
   // profitable to scalarize.
-  EpilogueCM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
+  EpilogueCM.collectNonVectorizedAndSetWideningDecisions(
+      EpilogueVectorizationForceVF);
 
   // Expecting only 2 vplans as epilogue TF is applied only for forced VFs.
   assert(VPlans.size() == 2 &&
          "For tail-folded epilogue, VPlans size is expected to be 2");
   // Remove the last vplan, which should be the epilogue plan to replace it by
   // the tail-folded vplan:
-  assert(VPlans.back()->getSingleVF() == EpilogueUserVF &&
+  assert(VPlans.back()->getSingleVF() == EpilogueVectorizationForceVF &&
          "For tail-folded epilogue, last vplan is expected to have "
          "EpilogueUserVF");
   VPlans.pop_back();
-  buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, EpilogueCM);
+  buildVPlans(*VPlan1, EpilogueVectorizationForceVF,
+              EpilogueVectorizationForceVF, EpilogueCM);
 
-  cost(*VPlans.back(), EpilogueUserVF, /*RU=*/nullptr, EpilogueCM);
+  cost(*VPlans.back(), EpilogueVectorizationForceVF, /*RU=*/nullptr,
+       EpilogueCM);
   return true;
 }
 
@@ -7312,6 +7305,11 @@ getEpilogueTailLowering(const LoopVectorizationCostModel &MainCM, const Loop *L,
       EpilogueTailFoldingPolicy != TailFoldingPolicyTy::PreferFoldTail)
     return CM_EpilogueAllowed;
 
+  if (!L->isInnermost())
+    reportVectorizationInfo(
+        "Epilgue tail-folding is not supported for outer loop",
+        "InvalidTailFoldedEpilogue", ORE, L);
+
   if (!EnableEpilogueVectorization) {
     reportVectorizationInfo(
         "Options conflict, epilogue vectorization is disallowed while "
@@ -8257,12 +8255,12 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   // Plan how to best vectorize.
   LVP.plan(UserVF, UserIC);
-  if (EpilogueTailFoldingCM.has_value())
-    if (!LVP.planForEpilogueTF(UserVF, UserIC, EpilogueVectorizationForceVF,
-                               EpilogueTailFoldingCM.value())) {
+  if (EpilogueTailFoldingCM)
+    if (!LVP.planForEpilogueTF(UserVF, UserIC, EpilogueTailFoldingCM.value())) {
       // we can't apply epilogue TF:
-      LLVM_DEBUG(
-          dbgs() << "LV: Applying epilogue tail-folding failed, disable it.\n");
+      reportVectorizationInfo(
+          "Applying epilogue tail-folding failed, disable it.",
+          "InvalidTailFoldedEpilogue", ORE, L);
       EpilogueTailFoldingCM.reset();
     }
 
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index 02ce78dcc8e14..e9b248016c32f 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -1,8 +1,9 @@
+; REQUIRES: asserts
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug -mcpu=neoverse-v1 -S %s | FileCheck %s
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug-only=loop-vectorize -mcpu=neoverse-v1 -S %s | FileCheck %s
 
-; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize,vectorutils --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
 ; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALIDATE-INTERLEAVE
 
 target triple = "aarch64-linux-gnu"
diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index c07fb38709f63..9df6674e792f3 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -1,18 +1,23 @@
 ; REQUIRES: asserts
-; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
+
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
 ; RUN: -pass-remarks-analysis=loop-vectorize -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s
 
-; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -force-vector-width=16 \
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize -enable-vplan-native-path --disable-output \
+; RUN: -epilogue-tail-folding-policy=prefer-fold-tail -pass-remarks-analysis=loop-vectorize \
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s --check-prefix=CHECK-OUTER-LOOP
+
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -force-vector-width=16 \
 ; RUN:  -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s --check-prefix=CHECK-NO-FORCED-MAIN-VF
 
-; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -epilogue-vectorization-force-VF=8  \
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -epilogue-vectorization-force-VF=8  \
 ; RUN: -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s --check-prefix=CHECK-NO-FORCED-EPILOGUE-VF
 
-; RUN: opt -S -p loop-vectorize -debug -enable-epilogue-vectorization=false \
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize -enable-epilogue-vectorization=false \
 ; RUN: --disable-output -force-vector-width=16 -epilogue-vectorization-force-VF=8 -epilogue-tail-folding-policy=prefer-fold-tail \
 ; RUN: -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s --check-prefix=CHECK-DISABLED-EPILOG
 
-; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
 ; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -vectorize-scev-check-threshold=0 < %s 2>&1 | FileCheck %s \
 ; RUN: --check-prefix=CHECK-NO-VPLANS
 
@@ -35,10 +40,38 @@ exit:
   ret void
 }
 
+define void @test_outer_loop(ptr %A, i64 %m) {
+; CHECK-OUTER-LOOP-LABEL: Checking a loop in 'test_outer_loop'
+; CHECK-OUTER-LOOP: LV: Epilgue tail-folding is not supported for outer loop
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %iv.outer = phi i64 [ 0, %entry ], [ %iv.outer.next, %outer.latch ]
+  br label %inner
+
+inner:
+  %iv.inner = phi i64 [ 0, %outer.header ], [ %iv.inner.next, %inner ]
+  %gep = getelementptr inbounds i32, ptr %A, i64 %iv.inner
+  store i32 0, ptr %gep, align 4
+  %iv.inner.next = add nuw nsw i64 %iv.inner, 1
+  %inner.ec = icmp eq i64 %iv.inner.next, 8
+  br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+  %iv.outer.next = add nuw nsw i64 %iv.outer, 1
+  %outer.ec = icmp eq i64 %iv.outer.next, %m
+  br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !1
+
+exit:
+  ret void
+}
+
 ; This case can't be tail-folded because all the iterations will be executed by
 ; main vector loop.
-define void @test_epilogue_tf_reset(ptr %A) {
-; CHECK-LABEL: Checking a loop in 'test_epilogue_tf_reset'
+define void @test_no_iterations_left(ptr %A) {
+; CHECK-LABEL: Checking a loop in 'test_no_iterations_left'
 ; CHECK: LV: epilogue tail-folding is enabled
 ; CHECK: LV: This case of epilogue loop can't be tail-folded.
 ; CHECK: LV: Applying epilogue tail-folding failed, disable it.
@@ -58,11 +91,11 @@ exit:
   ret void
 }
 
-define void @test_epilogue_tf_no_fv(ptr %A, i64 %n) {
-; CHECK-NO-FORCED-MAIN-VF-LABEL: Checking a loop in 'test_epilogue_tf_no_fv'
+define void @test_no_fv(ptr %A, i64 %n) {
+; CHECK-NO-FORCED-MAIN-VF-LABEL: Checking a loop in 'test_no_fv'
 ; CHECK-NO-FORCED-MAIN-VF: remark: <unknown>:0:0: For now, Epilogue tail-folding can't be applied without forced epilogue/main loop VF
 
-; CHECK-NO-FORCED-EPILOGUE-VF-LABEL: Checking a loop in 'test_epilogue_tf_no_fv'
+; CHECK-NO-FORCED-EPILOGUE-VF-LABEL: Checking a loop in 'test_no_fv'
 ; CHECK-NO-FORCED-EPILOGUE-VF: remark: <unknown>:0:0: For now, Epilogue tail-folding can't be applied without forced epilogue/main loop VF
 ;
 entry:
@@ -177,3 +210,6 @@ exit:
   %result = phi i64 [ %ext, %loop ]
   ret i64 %result
 }
+
+!1 = distinct !{!1, !2}
+!2 = !{!"llvm.loop.vectorize.enable"}

>From 143a3443fd723fe5fabf7c83be47adae07316ac0 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Sun, 9 Aug 2026 00:45:24 +0100
Subject: [PATCH 06/11] Swap between default CM instance and EPilogueTF one to
 enable costs for tail-folded epilogue

---
 .../Vectorize/LoopVectorizationPlanner.h      |  35 ++--
 .../Transforms/Vectorize/LoopVectorize.cpp    | 189 +++++++++---------
 .../AArch64/fold-epilogue-tail.ll             |   2 +-
 3 files changed, 121 insertions(+), 105 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 25991be146822..43a52c56b2bd8 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -854,7 +854,9 @@ class LoopVectorizationPlanner {
   LoopVectorizationLegality *Legal;
 
   /// The profitability analysis.
-  LoopVectorizationCostModel &CM;
+  LoopVectorizationCostModel *EnabledCM;
+  LoopVectorizationCostModel *DefaultCM;
+  LoopVectorizationCostModel *EpilogueTFCM;
 
   /// VF selection state independent of cost-modeling decisions.
   VFSelectionContext &Config;
@@ -884,8 +886,7 @@ class LoopVectorizationPlanner {
   ///
   /// TODO: Move to VPlan::cost once the use of LoopVectorizationLegality has
   /// been retired.
-  InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU,
-                       LoopVectorizationCostModel &EnabledCM) const;
+  InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU) const;
 
   /// Precompute costs for certain instructions using the legacy cost model. The
   /// function is used to bring up the VPlan-based cost model to initially avoid
@@ -897,11 +898,22 @@ class LoopVectorizationPlanner {
   LoopVectorizationPlanner(
       Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
       const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
-      LoopVectorizationCostModel &CM, VFSelectionContext &Config,
+      LoopVectorizationCostModel *DefaultCM,
+      LoopVectorizationCostModel *EpilogueTFCM, VFSelectionContext &Config,
       InterleavedAccessInfo &IAI, PredicatedScalarEvolution &PSE,
       const LoopVectorizeHints &Hints, OptimizationRemarkEmitter *ORE)
-      : OrigLoop(L), LI(LI), DT(DT), TLI(TLI), TTI(TTI), Legal(Legal), CM(CM),
-        Config(Config), IAI(IAI), PSE(PSE), Hints(Hints), ORE(ORE) {}
+      : OrigLoop(L), LI(LI), DT(DT), TLI(TLI), TTI(TTI), Legal(Legal),
+        DefaultCM(DefaultCM), EpilogueTFCM(EpilogueTFCM), Config(Config),
+        IAI(IAI), PSE(PSE), Hints(Hints), ORE(ORE) {
+    enableDefaultCM();
+  }
+
+  void enableDefaultCM() { EnabledCM = DefaultCM; }
+
+  void enableEpilogueTFCM() {
+    assert(EpilogueTFCM && "No CM for epilogue tail-folding to enable");
+    EnabledCM = EpilogueTFCM;
+  }
 
   /// Build VPlans for the specified \p UserVF and \p UserIC if they are
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
@@ -911,8 +923,7 @@ class LoopVectorizationPlanner {
   /// Build VPlans for the specified \p EpilogueUserVF and \p IC if they are
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
   /// tail-folding should be avoided up-front, no plans are generated.
-  bool planForEpilogueTF(ElementCount UserVF, unsigned UserIC,
-                         LoopVectorizationCostModel &EpilogueCM);
+  bool planForEpilogueTF(ElementCount UserVF, unsigned UserIC);
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
   /// for each VF.
@@ -1011,7 +1022,7 @@ class LoopVectorizationPlanner {
   /// Build an initial VPlan, with HCFG wrapping the original scalar loop and
   /// scalar transformations applied. Returns null if an initial VPlan cannot
   /// be built.
-  VPlanPtr tryToBuildVPlan1(LoopVectorizationCostModel &EnabledCM);
+  VPlanPtr tryToBuildVPlan1();
 
   /// Build a VPlan using VPRecipes according to the information gathered by
   /// Legal and VPlan-based analysis. For outer loops, performs basic recipe
@@ -1021,14 +1032,12 @@ class LoopVectorizationPlanner {
   /// maximum VF for which no plan could be built. Each VPlan is built starting
   /// from a copy of \p InitialPlan, which is a plain CFG VPlan wrapping the
   /// original scalar loop.
-  VPlanPtr tryToBuildVPlan(VPlanPtr InitialPlan, VFRange &Range,
-                           LoopVectorizationCostModel &EnabledCM);
+  VPlanPtr tryToBuildVPlan(VPlanPtr InitialPlan, VFRange &Range);
 
   /// Build VPlans for power-of-2 VF's between \p MinVF and \p MaxVF inclusive,
   /// based on \p VPlan1 and according to the information gathered by Legal
   /// when it checked if it is legal to vectorize the loop.
-  void buildVPlans(VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF,
-                   LoopVectorizationCostModel &EnabledCM);
+  void buildVPlans(VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF);
 
   /// Add ComputeReductionResult recipes to the middle block to compute the
   /// final reduction results. Add Select recipes to the latch block when
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 9151289639975..54fa8ebdaff95 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3159,7 +3159,7 @@ void LoopVectorizationPlanner::emitInvalidCostRemarks(
       if (VF.isScalar())
         continue;
 
-      VPCostContext CostCtx(*TLI, *Plan, CM, Config,
+      VPCostContext CostCtx(*TLI, *Plan, *EnabledCM, Config,
                             /*ReusePrintingSlotTracker=*/true);
       precomputeCosts(*Plan, VF, CostCtx);
       auto Iter = vp_depth_first_deep(Plan->getVectorLoopRegion()->getEntry());
@@ -3459,13 +3459,13 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     return nullptr;
   }
 
-  if (!CM.isEpilogueAllowed()) {
+  if (!EnabledCM->isEpilogueAllowed()) {
     LLVM_DEBUG(dbgs() << "LEV: Unable to vectorize epilogue because no "
                          "epilogue is allowed.\n");
     return nullptr;
   }
 
-  if (CM.maskPartialAliasing()) {
+  if (EnabledCM->maskPartialAliasing()) {
     LLVM_DEBUG(
         dbgs()
         << "LEV: Epilogue vectorization not supported with alias masking.\n");
@@ -3511,7 +3511,7 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     return nullptr;
   }
 
-  if (!CM.isEpilogueVectorizationProfitable(MainLoopVF, IC)) {
+  if (!EnabledCM->isEpilogueVectorizationProfitable(MainLoopVF, IC)) {
     LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization is not profitable for "
                          "this loop\n");
     return nullptr;
@@ -3655,8 +3655,8 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
   // overhead of multiple instructions to calculate the predicate is likely
   // not beneficial. If an epilogue is not allowed for any other reason,
   // do not interleave.
-  if (!CM.isEpilogueAllowed() &&
-      !(CM.preferTailFoldedLoop() && CM.useWideActiveLaneMask()))
+  if (!EnabledCM->isEpilogueAllowed() && !(EnabledCM->preferTailFoldedLoop() &&
+                                           EnabledCM->useWideActiveLaneMask()))
     return 1;
 
   if (any_of(Plan.getVectorLoopRegion()->getEntryBasicBlock()->phis(),
@@ -3684,16 +3684,16 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
   if (hasFindLastReductionPhi(Plan))
     return 1;
 
-  VPRegisterUsage R =
-      calculateRegisterUsageForPlan(Plan, {VF}, TTI, CM.ValuesToIgnore)[0];
+  VPRegisterUsage R = calculateRegisterUsageForPlan(
+      Plan, {VF}, TTI, EnabledCM->ValuesToIgnore)[0];
 
   // If we did not calculate the cost for VF (because the user selected the VF)
   // then we calculate the cost of VF here.
   if (LoopCost == 0) {
     if (VF.isScalar())
-      LoopCost = CM.expectedCost(VF);
+      LoopCost = EnabledCM->expectedCost(VF);
     else
-      LoopCost = cost(Plan, VF, &R, CM);
+      LoopCost = cost(Plan, VF, &R);
     assert(LoopCost.isValid() && "Expected to have chosen a VF with valid cost");
 
     // Loop body is free and there is no need for interleaving.
@@ -3773,10 +3773,10 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
 
   // Try to get the exact trip count, or an estimate based on profiling data or
   // ConstantMax from PSE, failing that.
-  auto BestKnownTC =
-      getSmallBestKnownTC(PSE, OrigLoop,
-                          /*CanUseConstantMax=*/true,
-                          /*CanExcludeZeroTrips=*/CM.isEpilogueAllowed());
+  auto BestKnownTC = getSmallBestKnownTC(
+      PSE, OrigLoop,
+      /*CanUseConstantMax=*/true,
+      /*CanExcludeZeroTrips=*/EnabledCM->isEpilogueAllowed());
 
   // For fixed length VFs treat a scalable trip count as unknown.
   if (BestKnownTC && (BestKnownTC->isFixed() || VF.isScalable())) {
@@ -5488,10 +5488,10 @@ void LoopVectorizationCostModel::collectValuesToIgnore() {
 }
 
 void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
-  CM.collectValuesToIgnore();
-  Config.collectElementTypesForWidening(&CM.ValuesToIgnore);
+  EnabledCM->collectValuesToIgnore();
+  Config.collectElementTypesForWidening(&EnabledCM->ValuesToIgnore);
 
-  FixedScalableVFPair MaxFactors = CM.computeMaxVF(UserVF, UserIC);
+  FixedScalableVFPair MaxFactors = EnabledCM->computeMaxVF(UserVF, UserIC);
   if (!MaxFactors) // Cases that should not to be vectorized nor interleaved.
     return;
 
@@ -5502,7 +5502,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   if (MaxFactors.FixedVF.isVector() || MaxFactors.ScalableVF.isVector())
     Legal->collectUnitStridePredicates();
 
-  auto VPlan1 = tryToBuildVPlan1(CM);
+  auto VPlan1 = tryToBuildVPlan1();
   if (!VPlan1)
     return;
 
@@ -5511,7 +5511,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
     // plan for that VF only.
     ElementCount VF =
         MaxFactors.FixedVF ? MaxFactors.FixedVF : MaxFactors.ScalableVF;
-    buildVPlans(*VPlan1, VF, VF, CM);
+    buildVPlans(*VPlan1, VF, VF);
     LLVM_DEBUG(printPlans(dbgs()));
     return;
   }
@@ -5521,20 +5521,20 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   Config.computeMinimalBitwidths();
 
   // Invalidate interleave groups if all blocks of loop will be predicated.
-  if (CM.blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
+  if (EnabledCM->blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
       !useMaskedInterleavedAccesses(TTI)) {
     LLVM_DEBUG(
         dbgs()
         << "LV: Invalidate all interleaved groups due to fold-tail by masking "
            "which requires masked-interleaved support.\n");
-    if (CM.InterleaveInfo.invalidateGroups())
+    if (EnabledCM->InterleaveInfo.invalidateGroups())
       // Invalidating interleave groups also requires invalidating all decisions
       // based on them, which includes widening decisions and uniform and scalar
       // values.
-      CM.invalidateCostModelingDecisions();
+      EnabledCM->invalidateCostModelingDecisions();
   }
 
-  if (CM.foldTailByMasking())
+  if (EnabledCM->foldTailByMasking())
     Legal->prepareToFoldTailByMasking();
 
   ElementCount MaxUserVF =
@@ -5549,23 +5549,23 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
              "VF needs to be a power of two");
       // Collect the instructions (and their associated costs) that will be more
       // profitable to scalarize.
-      CM.collectNonVectorizedAndSetWideningDecisions(UserVF);
+      EnabledCM->collectNonVectorizedAndSetWideningDecisions(UserVF);
       // Build the main-loop VPlan firstly because if epilogue tail-folding is
       // enabled, it will be built later, so we keep the epilogue vplans at the
       // end.
-      buildVPlans(*VPlan1, UserVF, UserVF, CM);
+      buildVPlans(*VPlan1, UserVF, UserVF);
 
       ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
       if (EpilogueUserVF.isVector() &&
           ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
-        CM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
-        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, CM);
+        EnabledCM->collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
+        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF);
       }
       if (!VPlans.empty() && VPlans.front()->getSingleVF() == UserVF) {
         // For scalar VF, skip VPlan cost check as VPlan cost is designed for
         // vector VFs only.
         if (UserVF.isScalar() ||
-            cost(*VPlans.front(), UserVF, /*RU=*/nullptr, CM).isValid()) {
+            cost(*VPlans.front(), UserVF, /*RU=*/nullptr).isValid()) {
           LLVM_DEBUG(dbgs() << "LV: Using user VF " << UserVF << ".\n");
           LLVM_DEBUG(printPlans(dbgs()));
           return;
@@ -5588,55 +5588,52 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
 
   for (const auto &VF : VFCandidates) {
     // Collect Uniform and Scalar instructions after vectorization with VF.
-    CM.collectNonVectorizedAndSetWideningDecisions(VF);
+    EnabledCM->collectNonVectorizedAndSetWideningDecisions(VF);
   }
 
-  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF, CM);
-  buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF, CM);
+  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF);
+  buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF);
 
   LLVM_DEBUG(printPlans(dbgs()));
 }
 
-bool LoopVectorizationPlanner::planForEpilogueTF(
-    ElementCount UserVF, unsigned UserIC,
-    LoopVectorizationCostModel &EpilogueCM) {
+bool LoopVectorizationPlanner::planForEpilogueTF(ElementCount UserVF,
+                                                 unsigned UserIC) {
   if (VPlans.empty()) {
     LLVM_DEBUG(dbgs() << "LV: no vplans have been built for main loop VF, bail "
                          "out of epilogue tail-folding\n");
     return false;
   }
 
-  EpilogueCM.ValuesToIgnore.insert_range(CM.ValuesToIgnore);
-  EpilogueCM.VecValuesToIgnore.insert_range(CM.VecValuesToIgnore);
+  EnabledCM->ValuesToIgnore.insert_range(DefaultCM->ValuesToIgnore);
+  EnabledCM->VecValuesToIgnore.insert_range(DefaultCM->VecValuesToIgnore);
 
   FixedScalableVFPair MaxFactors =
-      EpilogueCM.computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
-  if (!MaxFactors ||
-      !EpilogueCM
-           .foldTailByMasking()) { // Cases that should not to be vectorized
-                                   // or tail-folded.
+      EnabledCM->computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
+  if (!MaxFactors || !EnabledCM->foldTailByMasking()) {
+    // Cases that should not to be vectorized // or tail-folded.
     reportVectorizationInfo("This case of epilogue loop can't be tail-folded",
                             "InvalidTailFoldedEpilogue", ORE, OrigLoop);
     return false;
   }
 
-  auto VPlan1 = tryToBuildVPlan1(EpilogueCM);
+  auto VPlan1 = tryToBuildVPlan1();
 
   if (!useMaskedInterleavedAccesses(TTI)) {
     LLVM_DEBUG(
         dbgs() << "LV: Invalidate all interleaved groups due to fold-tail by "
                   "masking which requires masked-interleaved support.\n");
-    if (EpilogueCM.InterleaveInfo.invalidateGroups())
+    if (EnabledCM->InterleaveInfo.invalidateGroups())
       // Invalidating interleave groups also requires invalidating all decisions
       // based on them, which includes widening decisions and uniform and scalar
       // values.
-      EpilogueCM.invalidateCostModelingDecisions();
+      EnabledCM->invalidateCostModelingDecisions();
   }
   Legal->prepareToFoldTailByMasking();
 
   // Collect the instructions (and their associated costs) that will be more
   // profitable to scalarize.
-  EpilogueCM.collectNonVectorizedAndSetWideningDecisions(
+  EnabledCM->collectNonVectorizedAndSetWideningDecisions(
       EpilogueVectorizationForceVF);
 
   // Expecting only 2 vplans as epilogue TF is applied only for forced VFs.
@@ -5649,10 +5646,9 @@ bool LoopVectorizationPlanner::planForEpilogueTF(
          "EpilogueUserVF");
   VPlans.pop_back();
   buildVPlans(*VPlan1, EpilogueVectorizationForceVF,
-              EpilogueVectorizationForceVF, EpilogueCM);
+              EpilogueVectorizationForceVF);
 
-  cost(*VPlans.back(), EpilogueVectorizationForceVF, /*RU=*/nullptr,
-       EpilogueCM);
+  cost(*VPlans.back(), EpilogueVectorizationForceVF, /*RU=*/nullptr);
   return true;
 }
 
@@ -5850,8 +5846,8 @@ LoopVectorizationPlanner::precomputeCosts(VPlan &Plan, ElementCount VF,
 }
 
 InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
-                                               VPRegisterUsage *RU, LoopVectorizationCostModel &EnabledCM) const {
-  VPCostContext CostCtx(*TLI, Plan, EnabledCM, Config,
+                                               VPRegisterUsage *RU) const {
+  VPCostContext CostCtx(*TLI, Plan, *EnabledCM, Config,
                         /*ReusePrintingSlotTracker=*/true);
   InstructionCost Cost = precomputeCosts(Plan, VF, CostCtx);
 
@@ -5927,7 +5923,7 @@ LoopVectorizationPlanner::computeBestVF() {
          "More than a single plan/VF w/o any plan having scalar VF");
 
   // TODO: Compute scalar cost using VPlan-based cost model.
-  InstructionCost ScalarCost = CM.expectedCost(ScalarVF);
+  InstructionCost ScalarCost = EnabledCM->expectedCost(ScalarVF);
   LLVM_DEBUG(dbgs() << "LV: Scalar loop costs: " << ScalarCost << ".\n");
   VectorizationFactor ScalarFactor(ScalarVF, ScalarCost, ScalarCost);
   VectorizationFactor BestFactor = ScalarFactor;
@@ -5951,7 +5947,8 @@ LoopVectorizationPlanner::computeBestVF() {
       return Config.shouldConsiderRegPressureForVF(VF);
     });
     if (ConsiderRegPressure)
-      RUs = calculateRegisterUsageForPlan(*P, VFs, TTI, CM.ValuesToIgnore);
+      RUs = calculateRegisterUsageForPlan(*P, VFs, TTI,
+                                          EnabledCM->ValuesToIgnore);
 
     for (unsigned I = 0; I < VFs.size(); I++) {
       ElementCount VF = VFs[I];
@@ -5974,7 +5971,7 @@ LoopVectorizationPlanner::computeBestVF() {
       }
 
       InstructionCost Cost =
-          cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr, CM);
+          cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
       VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
 
       if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
@@ -6010,7 +6007,7 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
 
   RUN_VPLAN_PASS(VPlanTransforms::replaceWideCanonicalIVWithWideIV, BestVPlan,
                  *PSE.getSE(), TTI, Config.CostKind, BestVF, BestUF,
-                 CM.ValuesToIgnore);
+                 EnabledCM->ValuesToIgnore);
   // TODO: Move to VPlan transform stage once the transition to the VPlan-based
   // cost model is complete for better cost estimates.
   RUN_VPLAN_PASS(VPlanTransforms::unrollByUF, BestVPlan, BestUF);
@@ -6025,7 +6022,7 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
                    BestVPlan, BestVF, VScale);
   }
 
-  if (CM.maskPartialAliasing()) {
+  if (EnabledCM->maskPartialAliasing()) {
     assert(BestVPlan.hasTailFolded() && "Expected tail folding to be enabled");
     RUN_VPLAN_PASS(VPlanTransforms::materializeAliasMaskCheckBlock, BestVPlan,
                    *Legal->getRuntimePointerChecking()->getDiffChecks(),
@@ -6600,8 +6597,7 @@ VPRecipeBuilder::tryToCreateWidenNonPhiRecipe(VPSingleDefRecipe *R,
 // optimizations.
 static void printOptimizedVPlan(VPlan &) {}
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(
-    LoopVectorizationCostModel &EnabledCM) {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
   bool IsInnerLoop = OrigLoop->isInnermost();
 
   // Set up loop versioning for inner loops with memory runtime checks.
@@ -6651,8 +6647,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(
   bool ForceVectorization = Hints.getForce() == LoopVectorizeHints::FK_Enabled;
   bool OptForSize =
       !ForceVectorization &&
-      (EnabledCM.EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
-       EnabledCM.EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
+      (EnabledCM->EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
+       EnabledCM->EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
   unsigned SCEVCheckThreshold = ForceVectorization
                                     ? PragmaVectorizeSCEVCheckThreshold
                                     : VectorizeSCEVCheckThreshold;
@@ -6680,24 +6676,23 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(
 
   RUN_VPLAN_PASS(VPlanTransforms::createLoopRegions, *VPlan0,
                  getDebugLocFromInstOrOperands(Legal->getPrimaryInduction()));
-  if (EnabledCM.foldTailByMasking())
+  if (EnabledCM->foldTailByMasking())
     RUN_VPLAN_PASS(VPlanTransforms::foldTailByMasking, *VPlan0);
   RUN_VPLAN_PASS(VPlanTransforms::introduceMasksAndLinearize, *VPlan0);
 
   return VPlan0;
 }
 
-void LoopVectorizationPlanner::buildVPlans(
-    VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF,
-    LoopVectorizationCostModel &EnabledCM) {
+void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
+                                           ElementCount MaxVF) {
   if (ElementCount::isKnownGT(MinVF, MaxVF))
     return;
 
   auto MaxVFTimes2 = MaxVF * 2;
   for (ElementCount VF = MinVF; ElementCount::isKnownLT(VF, MaxVFTimes2);) {
     VFRange SubRange = {VF, MaxVFTimes2};
-    auto Plan = tryToBuildVPlan(std::unique_ptr<VPlan>(VPlan1.duplicate()),
-                                SubRange, EnabledCM);
+    auto Plan =
+        tryToBuildVPlan(std::unique_ptr<VPlan>(VPlan1.duplicate()), SubRange);
     VF = SubRange.End;
 
     if (!Plan)
@@ -6710,7 +6705,7 @@ void LoopVectorizationPlanner::buildVPlans(
                    Config.getMinimalBitwidths());
     RUN_VPLAN_PASS(VPlanTransforms::optimize, *Plan);
     // TODO: try to put addExplicitVectorLength close to addActiveLaneMask
-    if (EnabledCM.foldTailWithEVL()) {
+    if (EnabledCM->foldTailWithEVL()) {
       RUN_VPLAN_PASS(VPlanTransforms::addExplicitVectorLength, *Plan,
                      Config.getMaxSafeElements());
       RUN_VPLAN_PASS(VPlanTransforms::optimizeEVLMasks, *Plan);
@@ -6720,7 +6715,7 @@ void LoopVectorizationPlanner::buildVPlans(
             RUN_VPLAN_PASS(VPlanTransforms::narrowInterleaveGroups, *Plan, TTI))
       VPlans.push_back(std::move(P));
 
-    TailFoldingStyle Style = EnabledCM.getTailFoldingStyle();
+    TailFoldingStyle Style = EnabledCM->getTailFoldingStyle();
     RUN_VPLAN_PASS(VPlanTransforms::materializeHeaderMask, *Plan,
                    useActiveLaneMask(Style),
                    useActiveLaneMaskForControlFlow(Style));
@@ -6731,8 +6726,8 @@ void LoopVectorizationPlanner::buildVPlans(
   }
 }
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
-    VPlanPtr Plan, VFRange &Range, LoopVectorizationCostModel &EnabledCM) {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
+                                                   VFRange &Range) {
 
   // For outer loops, the plan only needs basic recipe conversion and induction
   // live-out optimization; the full inner-loop recipe building below does not
@@ -6758,8 +6753,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
 
   bool RequiresScalarEpilogueCheck =
       LoopVectorizationPlanner::getDecisionAndClampRange(
-          [EnabledCM](ElementCount VF) {
-            return !EnabledCM.requiresScalarEpilogue(VF.isVector());
+          [&](ElementCount VF) {
+            return !EnabledCM->requiresScalarEpilogue(VF.isVector());
           },
           Range);
   // Update the branch in the middle block if a scalar epilogue is required.
@@ -6777,9 +6772,9 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
   // TODO: Consider using getDecisionAndClampRange here to split up VPlans.
   bool IVUpdateMayOverflow = false;
   for (ElementCount VF : Range)
-    IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(&EnabledCM, VF);
+    IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(EnabledCM, VF);
 
-  TailFoldingStyle Style = EnabledCM.getTailFoldingStyle();
+  TailFoldingStyle Style = EnabledCM->getTailFoldingStyle();
   // Use NUW for the induction increment if we proved that it won't overflow in
   // the vector loop or when not folding the tail. In the later case, we know
   // that the canonical induction increment will not overflow as the vector trip
@@ -6806,10 +6801,10 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
   // placeholders for its members' Recipes which we'll be replacing with a
   // single VPInterleaveRecipe.
   for (InterleaveGroup<Instruction> *IG :
-       EnabledCM.InterleaveInfo.getInterleaveGroups()) {
-    auto ApplyIG = [IG, EnabledCM](ElementCount VF) -> bool {
+       EnabledCM->InterleaveInfo.getInterleaveGroups()) {
+    auto ApplyIG = [IG, this](ElementCount VF) -> bool {
       bool Result = (VF.isVector() && // Query is illegal for VF == 1
-                     EnabledCM.getWideningDecision(IG->getInsertPos(), VF) ==
+                     EnabledCM->getWideningDecision(IG->getInsertPos(), VF) ==
                          LoopVectorizationCostModel::CM_Interleave);
       // For scalable vectors, the interleave factors must be <= 8 since we
       // require the (de)interleaveN intrinsics instead of shufflevectors.
@@ -6826,7 +6821,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
   // Construct wide recipes and apply predication for original scalar
   // VPInstructions in the loop.
   // ---------------------------------------------------------------------------
-  VPRecipeBuilder RecipeBuilder(*Plan, Legal, EnabledCM, Builder);
+  VPRecipeBuilder RecipeBuilder(*Plan, Legal, *EnabledCM, Builder);
 
   // Scan the body of the loop in a topological order to visit each basic block
   // after having visited its predecessor basic blocks.
@@ -6837,7 +6832,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
   RUN_VPLAN_PASS(VPlanTransforms::createInLoopReductionRecipes, *Plan,
                  Range.Start);
 
-  VPCostContext CostCtx(*TLI, *Plan, EnabledCM, Config);
+  VPCostContext CostCtx(*TLI, *Plan, *EnabledCM, Config);
 
   RUN_VPLAN_PASS(VPlanTransforms::makeMemOpWideningDecisions, *Plan, Range,
                  RecipeBuilder, CostCtx);
@@ -6937,7 +6932,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
   // range for better cost estimation.
   // TODO: Enable following transform when the EVL-version of extended-reduction
   // and mulacc-reduction are implemented.
-  if (!EnabledCM.foldTailWithEVL()) {
+  if (!EnabledCM->foldTailWithEVL()) {
     RUN_VPLAN_PASS(VPlanTransforms::createPartialReductions, *Plan, CostCtx,
                    Range);
     RUN_VPLAN_PASS(VPlanTransforms::convertToAbstractRecipes, *Plan, CostCtx,
@@ -6948,7 +6943,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
   // for this VPlan, replace the Recipes widening its memory instructions with a
   // single VPInterleaveRecipe at its insertion point.
   RUN_VPLAN_PASS(VPlanTransforms::createInterleaveGroups, *Plan,
-                 InterleaveGroups, EnabledCM.isEpilogueAllowed());
+                 InterleaveGroups, EnabledCM->isEpilogueAllowed());
 
   // Convert memory recipes to strided access recipes if the strided access is
   // legal and profitable.
@@ -6965,7 +6960,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
 
   RUN_VPLAN_PASS(VPlanTransforms::dropPoisonGeneratingRecipes, *Plan);
 
-  if (EnabledCM.maskPartialAliasing())
+  if (EnabledCM->maskPartialAliasing())
     RUN_VPLAN_PASS(VPlanTransforms::attachAliasMaskToHeaderMask, *Plan);
 
   assert(verifyVPlanIsValid(*Plan) && "VPlan is invalid");
@@ -7009,7 +7004,7 @@ void LoopVectorizationPlanner::addReductionResultComputation(
 
     // Remove the predicated select if the target doesn't want it.
     VPValue *V;
-    if (!CM.usePredicatedReductionSelect(RecurrenceKind) &&
+    if (!EnabledCM->usePredicatedReductionSelect(RecurrenceKind) &&
         match(PhiR->getBackedgeValue(),
               m_Select(m_Specific(HeaderMask), m_VPValue(V), m_Specific(PhiR))))
       PhiR->setBackedgeValue(V);
@@ -7188,7 +7183,7 @@ void LoopVectorizationPlanner::attachRuntimeChecks(
   const auto &[SCEVCheckCond, SCEVCheckBlock] = RTChecks.getSCEVChecks();
   if (SCEVCheckBlock && SCEVCheckBlock->hasNPredecessors(0)) {
     assert((!Config.OptForSize ||
-            CM.Hints->getForce() == LoopVectorizeHints::FK_Enabled) &&
+            EnabledCM->Hints->getForce() == LoopVectorizeHints::FK_Enabled) &&
            "Cannot SCEV check stride or overflow when optimizing for size");
     RUN_VPLAN_PASS(VPlanTransforms::attachCheckBlock, Plan, SCEVCheckCond,
                    SCEVCheckBlock, HasBranchWeights);
@@ -7202,7 +7197,7 @@ void LoopVectorizationPlanner::attachRuntimeChecks(
 
     if (Config.OptForSize) {
       assert(
-          CM.Hints->getForce() == LoopVectorizeHints::FK_Enabled &&
+          EnabledCM->Hints->getForce() == LoopVectorizeHints::FK_Enabled &&
           "Cannot emit memory checks when optimizing for size, unless forced "
           "to vectorize.");
       ORE->emit([&]() {
@@ -7226,8 +7221,9 @@ bool LoopVectorizationPlanner::requiresScalarEpilogue(VPlan &Plan,
   // loop. Must be called before removeBranchOnConst.
   VPBasicBlock *MiddleVPBB = Plan.getMiddleBlock();
   bool Result = MiddleVPBB->getSingleSuccessor() == Plan.getScalarPreheader();
-  assert(CM.requiresScalarEpilogue(VF.isVector()) == Result &&
-         "CM.requiresScalarEpilogue and the VPlan-based check must agree");
+  assert(
+      EnabledCM->requiresScalarEpilogue(VF.isVector()) == Result &&
+      "EnabledCM->requiresScalarEpilogue and the VPlan-based check must agree");
   return Result;
 }
 
@@ -8224,9 +8220,6 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                             OptForSize);
   LoopVectorizationCostModel CM(SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE,
                                 GetBFI, F, &Hints, IAI, Config);
-  // Use the planner for vectorization.
-  LoopVectorizationPlanner LVP(L, LI, DT, TLI, *TTI, &LVL, CM, Config, IAI, PSE,
-                               Hints, ORE);
 
   EpilogueLowering EpilogueTailLoweringStatus =
       getEpilogueTailLowering(CM, L, ORE, LVL, Hints);
@@ -8243,6 +8236,12 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                                   *TailFoldingCMIAI, Config);
   }
 
+  // Use the planner for vectorization.
+  LoopVectorizationPlanner LVP(L, LI, DT, TLI, *TTI, &LVL, &CM,
+                               EpilogueTailFoldingCM ? &*EpilogueTailFoldingCM
+                                                     : nullptr,
+                               Config, IAI, PSE, Hints, ORE);
+
   // Get user vectorization factor and interleave count.
   ElementCount UserVF = Hints.getWidth();
   unsigned UserIC = Hints.getInterleave();
@@ -8255,14 +8254,19 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   // Plan how to best vectorize.
   LVP.plan(UserVF, UserIC);
-  if (EpilogueTailFoldingCM)
-    if (!LVP.planForEpilogueTF(UserVF, UserIC, EpilogueTailFoldingCM.value())) {
+  if (EpilogueTailFoldingCM) {
+    // Enable the epilogue tail-folding CM
+    LVP.enableEpilogueTFCM();
+    if (!LVP.planForEpilogueTF(UserVF, UserIC)) {
       // we can't apply epilogue TF:
       reportVectorizationInfo(
           "Applying epilogue tail-folding failed, disable it.",
           "InvalidTailFoldedEpilogue", ORE, L);
       EpilogueTailFoldingCM.reset();
     }
+    // Get back the default CM:
+    LVP.enableDefaultCM();
+  }
 
   auto [VF, BestPlanPtr] = LVP.computeBestVF();
   unsigned IC = 1;
@@ -8504,6 +8508,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
         LoopVectorizationPlanner::EpilogueVectorizationKind::MainLoop);
     ++LoopsVectorized;
 
+    if (EpilogueTailFoldingCM)
+      LVP.enableEpilogueTFCM();
     // Derive EPI fields from VPlan-generated IR.
     BasicBlock *EntryBB =
         cast<VPIRBasicBlock>(BestMainPlan.getEntry())->getIRBasicBlock();
@@ -8536,6 +8542,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     connectEpilogueVectorLoop(BestEpiPlan, L, EPI, DT, LI, Checks, InstsToMove,
                               ResumeValues, EpilogueTailFoldingCM.has_value());
     ++LoopsEpilogueVectorized;
+    LVP.enableDefaultCM();
   } else {
     InnerLoopVectorizer LB(L, PSE, LI, DT, TTI, AC, VF.Width, IC, Checks,
                            BestPlan);
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index e9b248016c32f..ece1fc3a15b11 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -1,5 +1,5 @@
-; REQUIRES: asserts
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; REQUIRES: asserts
 ; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
 ; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug-only=loop-vectorize -mcpu=neoverse-v1 -S %s | FileCheck %s
 

>From 5ddd4659698738c0bd627983265ca9660d434d03 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Mon, 10 Aug 2026 21:03:58 +0100
Subject: [PATCH 07/11] resolve review comments - improve readability

---
 .../Vectorize/LoopVectorizationPlanner.h          | 15 +++++++++++----
 llvm/lib/Transforms/Vectorize/LoopVectorize.cpp   |  7 +++----
 .../LoopVectorize/fold-epilogue-tail.ll           |  2 +-
 3 files changed, 15 insertions(+), 9 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 43a52c56b2bd8..95aeedfab5b0c 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -854,8 +854,16 @@ class LoopVectorizationPlanner {
   LoopVectorizationLegality *Legal;
 
   /// The profitability analysis.
+  /// The CM currently in effect for the VPlan being built or costed; it always
+  /// aliases either \c DefaultCM or \c EpilogueTFCM.
   LoopVectorizationCostModel *EnabledCM;
+  /// The CM used for the main-loop VPlan, and for the epilogue VPlan in all
+  /// cases except tail-folded epilogue vectorization.
   LoopVectorizationCostModel *DefaultCM;
+  /// The CM used only when the epilogue loop is vectorized with tail-folding.
+  /// \c EnabledCM is switched to point here (via enableEpilogueTFCM()) while
+  /// the epilogue VPlan's costs are computed, so that they correctly account
+  /// for the tail-folded epilogue.
   LoopVectorizationCostModel *EpilogueTFCM;
 
   /// VF selection state independent of cost-modeling decisions.
@@ -920,10 +928,9 @@ class LoopVectorizationPlanner {
   /// interleaving should be avoided up-front, no plans are generated.
   void plan(ElementCount UserVF, unsigned UserIC);
 
-  /// Build VPlans for the specified \p EpilogueUserVF and \p IC if they are
-  /// non-zero or all applicable candidate VFs otherwise. If vectorization and
-  /// tail-folding should be avoided up-front, no plans are generated.
-  bool planForEpilogueTF(ElementCount UserVF, unsigned UserIC);
+  /// Build VPlan for the forced epilogue VF. If vectorization and tail-folding
+  /// should be avoided up-front, no tail-folded plans are generated.
+  bool planForEpilogueTF();
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
   /// for each VF.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 54fa8ebdaff95..7468be329e28a 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5597,8 +5597,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   LLVM_DEBUG(printPlans(dbgs()));
 }
 
-bool LoopVectorizationPlanner::planForEpilogueTF(ElementCount UserVF,
-                                                 unsigned UserIC) {
+bool LoopVectorizationPlanner::planForEpilogueTF() {
   if (VPlans.empty()) {
     LLVM_DEBUG(dbgs() << "LV: no vplans have been built for main loop VF, bail "
                          "out of epilogue tail-folding\n");
@@ -7303,7 +7302,7 @@ getEpilogueTailLowering(const LoopVectorizationCostModel &MainCM, const Loop *L,
 
   if (!L->isInnermost())
     reportVectorizationInfo(
-        "Epilgue tail-folding is not supported for outer loop",
+        "Epilogue tail-folding is not supported for outer loop",
         "InvalidTailFoldedEpilogue", ORE, L);
 
   if (!EnableEpilogueVectorization) {
@@ -8257,7 +8256,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   if (EpilogueTailFoldingCM) {
     // Enable the epilogue tail-folding CM
     LVP.enableEpilogueTFCM();
-    if (!LVP.planForEpilogueTF(UserVF, UserIC)) {
+    if (!LVP.planForEpilogueTF()) {
       // we can't apply epilogue TF:
       reportVectorizationInfo(
           "Applying epilogue tail-folding failed, disable it.",
diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index 9df6674e792f3..2c8f994fd3104 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -42,7 +42,7 @@ exit:
 
 define void @test_outer_loop(ptr %A, i64 %m) {
 ; CHECK-OUTER-LOOP-LABEL: Checking a loop in 'test_outer_loop'
-; CHECK-OUTER-LOOP: LV: Epilgue tail-folding is not supported for outer loop
+; CHECK-OUTER-LOOP: LV: Epilogue tail-folding is not supported for outer loop
 ;
 entry:
   br label %outer.header

>From adcf25c0553a68dbeebd7e9d3c74da977a9f21b9 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Tue, 11 Aug 2026 15:59:31 +0100
Subject: [PATCH 08/11] disallow epilogue TF for outer loop

---
 .../Transforms/Vectorize/LoopVectorize.cpp    |   4 +-
 .../LoopVectorize/fold-epilogue-tail.ll       | 108 +++++++++---------
 2 files changed, 60 insertions(+), 52 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 7468be329e28a..1753a0b3e2d0d 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7300,10 +7300,12 @@ getEpilogueTailLowering(const LoopVectorizationCostModel &MainCM, const Loop *L,
       EpilogueTailFoldingPolicy != TailFoldingPolicyTy::PreferFoldTail)
     return CM_EpilogueAllowed;
 
-  if (!L->isInnermost())
+  if (!L->isInnermost()) {
     reportVectorizationInfo(
         "Epilogue tail-folding is not supported for outer loop",
         "InvalidTailFoldedEpilogue", ORE, L);
+    return CM_EpilogueAllowed;
+  }
 
   if (!EnableEpilogueVectorization) {
     reportVectorizationInfo(
diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index 2c8f994fd3104..ea94e709920e8 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -40,34 +40,6 @@ exit:
   ret void
 }
 
-define void @test_outer_loop(ptr %A, i64 %m) {
-; CHECK-OUTER-LOOP-LABEL: Checking a loop in 'test_outer_loop'
-; CHECK-OUTER-LOOP: LV: Epilogue tail-folding is not supported for outer loop
-;
-entry:
-  br label %outer.header
-
-outer.header:
-  %iv.outer = phi i64 [ 0, %entry ], [ %iv.outer.next, %outer.latch ]
-  br label %inner
-
-inner:
-  %iv.inner = phi i64 [ 0, %outer.header ], [ %iv.inner.next, %inner ]
-  %gep = getelementptr inbounds i32, ptr %A, i64 %iv.inner
-  store i32 0, ptr %gep, align 4
-  %iv.inner.next = add nuw nsw i64 %iv.inner, 1
-  %inner.ec = icmp eq i64 %iv.inner.next, 8
-  br i1 %inner.ec, label %outer.latch, label %inner
-
-outer.latch:
-  %iv.outer.next = add nuw nsw i64 %iv.outer, 1
-  %outer.ec = icmp eq i64 %iv.outer.next, %m
-  br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !1
-
-exit:
-  ret void
-}
-
 ; This case can't be tail-folded because all the iterations will be executed by
 ; main vector loop.
 define void @test_no_iterations_left(ptr %A) {
@@ -91,12 +63,41 @@ exit:
   ret void
 }
 
-define void @test_no_fv(ptr %A, i64 %n) {
-; CHECK-NO-FORCED-MAIN-VF-LABEL: Checking a loop in 'test_no_fv'
+; Can't build a valid vplan for this case because too many SCEV checks needed,
+; more than the specfied limit.
+define i64 @test_no_vplan_built(ptr %dst, i64 %n) {
+; CHECK-NO-VPLANS-LABEL: Checking a loop in 'test_no_vplan_built'
+; CHECK-NO-VPLANS: LV: epilogue tail-folding is enabled
+; CHECK-NO-VPLANS: LV: no vplans have been built for main loop VF, bail out of epilogue tail-folding
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %dead.iv = phi i16 [ 0, %entry ], [ %dead.iv.next, %loop ]
+  %prev = phi i64 [ 0, %entry ], [ %ext, %loop ]
+  %iv.next = add nuw nsw i64 %iv, 1
+  %dead.iv.next = add i16 %dead.iv, 1
+  %ext = zext i16 %dead.iv.next to i64
+  %gep = getelementptr inbounds i64, ptr %dst, i64 %prev
+  store i64 %iv, ptr %gep, align 8
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  %result = phi i64 [ %ext, %loop ]
+  ret i64 %result
+}
+
+define void @test_no_vf(ptr %A, i64 %n) {
+; CHECK-NO-FORCED-MAIN-VF-LABEL: Checking a loop in 'test_no_vf'
 ; CHECK-NO-FORCED-MAIN-VF: remark: <unknown>:0:0: For now, Epilogue tail-folding can't be applied without forced epilogue/main loop VF
+; CHECK-NO-FORCED-MAIN-VF-NOT: LV: epilogue tail-folding is enabled
 
-; CHECK-NO-FORCED-EPILOGUE-VF-LABEL: Checking a loop in 'test_no_fv'
+; CHECK-NO-FORCED-EPILOGUE-VF-LABEL: Checking a loop in 'test_no_vf'
 ; CHECK-NO-FORCED-EPILOGUE-VF: remark: <unknown>:0:0: For now, Epilogue tail-folding can't be applied without forced epilogue/main loop VF
+; CHECK-NO-FORCED-EPILOGUE-VF-NOT: LV: epilogue tail-folding is enabled
 ;
 entry:
   br label %for.body
@@ -116,6 +117,7 @@ exit:
 define void @epilogue_is_disabled(ptr %a, i64 %n) {
 ; CHECK-DISABLED-EPILOG-LABEL: Checking a loop in 'epilogue_is_disabled'
 ; CHECK-DISABLED-EPILOG: remark: <unknown>:0:0: Options conflict, epilogue vectorization is disallowed while epilogue tail-folding allowed!
+; CHECK-DISABLED-EPILOG-NOT: LV: epilogue tail-folding is enabled
 ;
 entry:
   br label %for.body
@@ -136,6 +138,7 @@ define i16 @require_scalar_epilogue(ptr %dst, i64 %x) {
 ; CHECK-LABEL: Checking a loop in 'require_scalar_epilogue'
 ; CHECK: LV: Epilogue tail-folding can't be applied because scalar epilogue is required
 ; CHECK-NEXT: LV: Fall back to a normal epilogue
+; CHECK-NOT: LV: epilogue tail-folding is enabled
 ;
 entry:
   br label %loop.header
@@ -165,6 +168,7 @@ define i32 @opt_for_size(ptr %p, i32 %n) optsize {
 ; CHECK-LABEL: Checking a loop in 'opt_for_size'
 ; CHECK: LV: No epilogue to apply tail-folding for.
 ; CHECK-NEXT: LV: Fall back to a normal epilogue
+; CHECK-NOT: LV: epilogue tail-folding is enabled
 ;
 entry:
   br label %for.body
@@ -184,31 +188,33 @@ for.end:
   ret i32 0
 }
 
-; Can't build a valid vplan for this case because too many SCEV checks needed,
-; more than the specfied limit.
-define i64 @test_no_vplan_built(ptr %dst, i64 %n) {
-; CHECK-NO-VPLANS-LABEL: Checking a loop in 'test_no_vplan_built'
-; CHECK-NO-VPLANS: LV: epilogue tail-folding is enabled
-; CHECK-NO-VPLANS: LV: no vplans have been built for main loop VF, bail out of epilogue tail-folding
+define void @test_outer_loop(ptr %A, i64 %m) {
+; CHECK-OUTER-LOOP-LABEL: Checking a loop in 'test_outer_loop'
+; CHECK-OUTER-LOOP: remark: <unknown>:0:0: Epilogue tail-folding is not supported for outer loop
+; CHECK-OUTER-LOOP-NOT: LV: epilogue tail-folding is enabled
 ;
 entry:
-  br label %loop
+  br label %outer.header
 
-loop:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-  %dead.iv = phi i16 [ 0, %entry ], [ %dead.iv.next, %loop ]
-  %prev = phi i64 [ 0, %entry ], [ %ext, %loop ]
-  %iv.next = add nuw nsw i64 %iv, 1
-  %dead.iv.next = add i16 %dead.iv, 1
-  %ext = zext i16 %dead.iv.next to i64
-  %gep = getelementptr inbounds i64, ptr %dst, i64 %prev
-  store i64 %iv, ptr %gep, align 8
-  %cmp = icmp slt i64 %iv.next, %n
-  br i1 %cmp, label %loop, label %exit
+outer.header:
+  %iv.outer = phi i64 [ 0, %entry ], [ %iv.outer.next, %outer.latch ]
+  br label %inner
+
+inner:
+  %iv.inner = phi i64 [ 0, %outer.header ], [ %iv.inner.next, %inner ]
+  %gep = getelementptr inbounds i32, ptr %A, i64 %iv.inner
+  store i32 0, ptr %gep, align 4
+  %iv.inner.next = add nuw nsw i64 %iv.inner, 1
+  %inner.ec = icmp eq i64 %iv.inner.next, 8
+  br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+  %iv.outer.next = add nuw nsw i64 %iv.outer, 1
+  %outer.ec = icmp eq i64 %iv.outer.next, %m
+  br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !1
 
 exit:
-  %result = phi i64 [ %ext, %loop ]
-  ret i64 %result
+  ret void
 }
 
 !1 = distinct !{!1, !2}

>From a060fc7f24f2a8a883cc0368bf3f0e9d37c37e48 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Fri, 14 Aug 2026 16:57:33 +0100
Subject: [PATCH 09/11] revert commit of swapping between different CMs

---
 .../Vectorize/LoopVectorizationPlanner.h      |  42 ++--
 .../Transforms/Vectorize/LoopVectorize.cpp    | 184 +++++++++---------
 2 files changed, 101 insertions(+), 125 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 95aeedfab5b0c..a6a88fb011a18 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -854,17 +854,7 @@ class LoopVectorizationPlanner {
   LoopVectorizationLegality *Legal;
 
   /// The profitability analysis.
-  /// The CM currently in effect for the VPlan being built or costed; it always
-  /// aliases either \c DefaultCM or \c EpilogueTFCM.
-  LoopVectorizationCostModel *EnabledCM;
-  /// The CM used for the main-loop VPlan, and for the epilogue VPlan in all
-  /// cases except tail-folded epilogue vectorization.
-  LoopVectorizationCostModel *DefaultCM;
-  /// The CM used only when the epilogue loop is vectorized with tail-folding.
-  /// \c EnabledCM is switched to point here (via enableEpilogueTFCM()) while
-  /// the epilogue VPlan's costs are computed, so that they correctly account
-  /// for the tail-folded epilogue.
-  LoopVectorizationCostModel *EpilogueTFCM;
+  LoopVectorizationCostModel &CM;
 
   /// VF selection state independent of cost-modeling decisions.
   VFSelectionContext &Config;
@@ -894,7 +884,8 @@ class LoopVectorizationPlanner {
   ///
   /// TODO: Move to VPlan::cost once the use of LoopVectorizationLegality has
   /// been retired.
-  InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU) const;
+  InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU,
+                       LoopVectorizationCostModel &EnabledCM) const;
 
   /// Precompute costs for certain instructions using the legacy cost model. The
   /// function is used to bring up the VPlan-based cost model to initially avoid
@@ -906,22 +897,11 @@ class LoopVectorizationPlanner {
   LoopVectorizationPlanner(
       Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
       const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
-      LoopVectorizationCostModel *DefaultCM,
-      LoopVectorizationCostModel *EpilogueTFCM, VFSelectionContext &Config,
+      LoopVectorizationCostModel &CM, VFSelectionContext &Config,
       InterleavedAccessInfo &IAI, PredicatedScalarEvolution &PSE,
       const LoopVectorizeHints &Hints, OptimizationRemarkEmitter *ORE)
-      : OrigLoop(L), LI(LI), DT(DT), TLI(TLI), TTI(TTI), Legal(Legal),
-        DefaultCM(DefaultCM), EpilogueTFCM(EpilogueTFCM), Config(Config),
-        IAI(IAI), PSE(PSE), Hints(Hints), ORE(ORE) {
-    enableDefaultCM();
-  }
-
-  void enableDefaultCM() { EnabledCM = DefaultCM; }
-
-  void enableEpilogueTFCM() {
-    assert(EpilogueTFCM && "No CM for epilogue tail-folding to enable");
-    EnabledCM = EpilogueTFCM;
-  }
+      : OrigLoop(L), LI(LI), DT(DT), TLI(TLI), TTI(TTI), Legal(Legal), CM(CM),
+        Config(Config), IAI(IAI), PSE(PSE), Hints(Hints), ORE(ORE) {}
 
   /// Build VPlans for the specified \p UserVF and \p UserIC if they are
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
@@ -930,7 +910,7 @@ class LoopVectorizationPlanner {
 
   /// Build VPlan for the forced epilogue VF. If vectorization and tail-folding
   /// should be avoided up-front, no tail-folded plans are generated.
-  bool planForEpilogueTF();
+  bool planForEpilogueTF(LoopVectorizationCostModel &EpilogueCM);
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
   /// for each VF.
@@ -1029,7 +1009,7 @@ class LoopVectorizationPlanner {
   /// Build an initial VPlan, with HCFG wrapping the original scalar loop and
   /// scalar transformations applied. Returns null if an initial VPlan cannot
   /// be built.
-  VPlanPtr tryToBuildVPlan1();
+  VPlanPtr tryToBuildVPlan1(LoopVectorizationCostModel &EnabledCM);
 
   /// Build a VPlan using VPRecipes according to the information gathered by
   /// Legal and VPlan-based analysis. For outer loops, performs basic recipe
@@ -1039,12 +1019,14 @@ class LoopVectorizationPlanner {
   /// maximum VF for which no plan could be built. Each VPlan is built starting
   /// from a copy of \p InitialPlan, which is a plain CFG VPlan wrapping the
   /// original scalar loop.
-  VPlanPtr tryToBuildVPlan(VPlanPtr InitialPlan, VFRange &Range);
+  VPlanPtr tryToBuildVPlan(VPlanPtr InitialPlan, VFRange &Range,
+                           LoopVectorizationCostModel &EnabledCM);
 
   /// Build VPlans for power-of-2 VF's between \p MinVF and \p MaxVF inclusive,
   /// based on \p VPlan1 and according to the information gathered by Legal
   /// when it checked if it is legal to vectorize the loop.
-  void buildVPlans(VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF);
+  void buildVPlans(VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF,
+                   LoopVectorizationCostModel &EnabledCM);
 
   /// Add ComputeReductionResult recipes to the middle block to compute the
   /// final reduction results. Add Select recipes to the latch block when
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 1753a0b3e2d0d..f70bc573fabc3 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3159,7 +3159,7 @@ void LoopVectorizationPlanner::emitInvalidCostRemarks(
       if (VF.isScalar())
         continue;
 
-      VPCostContext CostCtx(*TLI, *Plan, *EnabledCM, Config,
+      VPCostContext CostCtx(*TLI, *Plan, CM, Config,
                             /*ReusePrintingSlotTracker=*/true);
       precomputeCosts(*Plan, VF, CostCtx);
       auto Iter = vp_depth_first_deep(Plan->getVectorLoopRegion()->getEntry());
@@ -3459,13 +3459,13 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     return nullptr;
   }
 
-  if (!EnabledCM->isEpilogueAllowed()) {
+  if (!CM.isEpilogueAllowed()) {
     LLVM_DEBUG(dbgs() << "LEV: Unable to vectorize epilogue because no "
                          "epilogue is allowed.\n");
     return nullptr;
   }
 
-  if (EnabledCM->maskPartialAliasing()) {
+  if (CM.maskPartialAliasing()) {
     LLVM_DEBUG(
         dbgs()
         << "LEV: Epilogue vectorization not supported with alias masking.\n");
@@ -3511,7 +3511,7 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     return nullptr;
   }
 
-  if (!EnabledCM->isEpilogueVectorizationProfitable(MainLoopVF, IC)) {
+  if (!CM.isEpilogueVectorizationProfitable(MainLoopVF, IC)) {
     LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization is not profitable for "
                          "this loop\n");
     return nullptr;
@@ -3655,8 +3655,8 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
   // overhead of multiple instructions to calculate the predicate is likely
   // not beneficial. If an epilogue is not allowed for any other reason,
   // do not interleave.
-  if (!EnabledCM->isEpilogueAllowed() && !(EnabledCM->preferTailFoldedLoop() &&
-                                           EnabledCM->useWideActiveLaneMask()))
+  if (!CM.isEpilogueAllowed() &&
+      !(CM.preferTailFoldedLoop() && CM.useWideActiveLaneMask()))
     return 1;
 
   if (any_of(Plan.getVectorLoopRegion()->getEntryBasicBlock()->phis(),
@@ -3684,16 +3684,16 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
   if (hasFindLastReductionPhi(Plan))
     return 1;
 
-  VPRegisterUsage R = calculateRegisterUsageForPlan(
-      Plan, {VF}, TTI, EnabledCM->ValuesToIgnore)[0];
+  VPRegisterUsage R =
+      calculateRegisterUsageForPlan(Plan, {VF}, TTI, CM.ValuesToIgnore)[0];
 
   // If we did not calculate the cost for VF (because the user selected the VF)
   // then we calculate the cost of VF here.
   if (LoopCost == 0) {
     if (VF.isScalar())
-      LoopCost = EnabledCM->expectedCost(VF);
+      LoopCost = CM.expectedCost(VF);
     else
-      LoopCost = cost(Plan, VF, &R);
+      LoopCost = cost(Plan, VF, &R, CM);
     assert(LoopCost.isValid() && "Expected to have chosen a VF with valid cost");
 
     // Loop body is free and there is no need for interleaving.
@@ -3773,10 +3773,10 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
 
   // Try to get the exact trip count, or an estimate based on profiling data or
   // ConstantMax from PSE, failing that.
-  auto BestKnownTC = getSmallBestKnownTC(
-      PSE, OrigLoop,
-      /*CanUseConstantMax=*/true,
-      /*CanExcludeZeroTrips=*/EnabledCM->isEpilogueAllowed());
+  auto BestKnownTC =
+      getSmallBestKnownTC(PSE, OrigLoop,
+                          /*CanUseConstantMax=*/true,
+                          /*CanExcludeZeroTrips=*/CM.isEpilogueAllowed());
 
   // For fixed length VFs treat a scalable trip count as unknown.
   if (BestKnownTC && (BestKnownTC->isFixed() || VF.isScalable())) {
@@ -5488,10 +5488,10 @@ void LoopVectorizationCostModel::collectValuesToIgnore() {
 }
 
 void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
-  EnabledCM->collectValuesToIgnore();
-  Config.collectElementTypesForWidening(&EnabledCM->ValuesToIgnore);
+  CM.collectValuesToIgnore();
+  Config.collectElementTypesForWidening(&CM.ValuesToIgnore);
 
-  FixedScalableVFPair MaxFactors = EnabledCM->computeMaxVF(UserVF, UserIC);
+  FixedScalableVFPair MaxFactors = CM.computeMaxVF(UserVF, UserIC);
   if (!MaxFactors) // Cases that should not to be vectorized nor interleaved.
     return;
 
@@ -5502,7 +5502,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   if (MaxFactors.FixedVF.isVector() || MaxFactors.ScalableVF.isVector())
     Legal->collectUnitStridePredicates();
 
-  auto VPlan1 = tryToBuildVPlan1();
+  auto VPlan1 = tryToBuildVPlan1(CM);
   if (!VPlan1)
     return;
 
@@ -5511,7 +5511,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
     // plan for that VF only.
     ElementCount VF =
         MaxFactors.FixedVF ? MaxFactors.FixedVF : MaxFactors.ScalableVF;
-    buildVPlans(*VPlan1, VF, VF);
+    buildVPlans(*VPlan1, VF, VF, CM);
     LLVM_DEBUG(printPlans(dbgs()));
     return;
   }
@@ -5521,20 +5521,20 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   Config.computeMinimalBitwidths();
 
   // Invalidate interleave groups if all blocks of loop will be predicated.
-  if (EnabledCM->blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
+  if (CM.blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
       !useMaskedInterleavedAccesses(TTI)) {
     LLVM_DEBUG(
         dbgs()
         << "LV: Invalidate all interleaved groups due to fold-tail by masking "
            "which requires masked-interleaved support.\n");
-    if (EnabledCM->InterleaveInfo.invalidateGroups())
+    if (CM.InterleaveInfo.invalidateGroups())
       // Invalidating interleave groups also requires invalidating all decisions
       // based on them, which includes widening decisions and uniform and scalar
       // values.
-      EnabledCM->invalidateCostModelingDecisions();
+      CM.invalidateCostModelingDecisions();
   }
 
-  if (EnabledCM->foldTailByMasking())
+  if (CM.foldTailByMasking())
     Legal->prepareToFoldTailByMasking();
 
   ElementCount MaxUserVF =
@@ -5549,23 +5549,23 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
              "VF needs to be a power of two");
       // Collect the instructions (and their associated costs) that will be more
       // profitable to scalarize.
-      EnabledCM->collectNonVectorizedAndSetWideningDecisions(UserVF);
+      CM.collectNonVectorizedAndSetWideningDecisions(UserVF);
       // Build the main-loop VPlan firstly because if epilogue tail-folding is
       // enabled, it will be built later, so we keep the epilogue vplans at the
       // end.
-      buildVPlans(*VPlan1, UserVF, UserVF);
+      buildVPlans(*VPlan1, UserVF, UserVF, CM);
 
       ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
       if (EpilogueUserVF.isVector() &&
           ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
-        EnabledCM->collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
-        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF);
+        CM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
+        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, CM);
       }
       if (!VPlans.empty() && VPlans.front()->getSingleVF() == UserVF) {
         // For scalar VF, skip VPlan cost check as VPlan cost is designed for
         // vector VFs only.
         if (UserVF.isScalar() ||
-            cost(*VPlans.front(), UserVF, /*RU=*/nullptr).isValid()) {
+            cost(*VPlans.front(), UserVF, /*RU=*/nullptr, CM).isValid()) {
           LLVM_DEBUG(dbgs() << "LV: Using user VF " << UserVF << ".\n");
           LLVM_DEBUG(printPlans(dbgs()));
           return;
@@ -5588,51 +5588,52 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
 
   for (const auto &VF : VFCandidates) {
     // Collect Uniform and Scalar instructions after vectorization with VF.
-    EnabledCM->collectNonVectorizedAndSetWideningDecisions(VF);
+    CM.collectNonVectorizedAndSetWideningDecisions(VF);
   }
 
-  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF);
-  buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF);
+  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF, CM);
+  buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF, CM);
 
   LLVM_DEBUG(printPlans(dbgs()));
 }
 
-bool LoopVectorizationPlanner::planForEpilogueTF() {
+bool LoopVectorizationPlanner::planForEpilogueTF(
+    LoopVectorizationCostModel &EpilogueCM) {
   if (VPlans.empty()) {
     LLVM_DEBUG(dbgs() << "LV: no vplans have been built for main loop VF, bail "
                          "out of epilogue tail-folding\n");
     return false;
   }
 
-  EnabledCM->ValuesToIgnore.insert_range(DefaultCM->ValuesToIgnore);
-  EnabledCM->VecValuesToIgnore.insert_range(DefaultCM->VecValuesToIgnore);
+  EpilogueCM.ValuesToIgnore.insert_range(CM.ValuesToIgnore);
+  EpilogueCM.VecValuesToIgnore.insert_range(CM.VecValuesToIgnore);
 
   FixedScalableVFPair MaxFactors =
-      EnabledCM->computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
-  if (!MaxFactors || !EnabledCM->foldTailByMasking()) {
+      EpilogueCM.computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
+  if (!MaxFactors || !EpilogueCM.foldTailByMasking()) {
     // Cases that should not to be vectorized // or tail-folded.
     reportVectorizationInfo("This case of epilogue loop can't be tail-folded",
                             "InvalidTailFoldedEpilogue", ORE, OrigLoop);
     return false;
   }
 
-  auto VPlan1 = tryToBuildVPlan1();
+  auto VPlan1 = tryToBuildVPlan1(EpilogueCM);
 
   if (!useMaskedInterleavedAccesses(TTI)) {
     LLVM_DEBUG(
         dbgs() << "LV: Invalidate all interleaved groups due to fold-tail by "
                   "masking which requires masked-interleaved support.\n");
-    if (EnabledCM->InterleaveInfo.invalidateGroups())
+    if (EpilogueCM.InterleaveInfo.invalidateGroups())
       // Invalidating interleave groups also requires invalidating all decisions
       // based on them, which includes widening decisions and uniform and scalar
       // values.
-      EnabledCM->invalidateCostModelingDecisions();
+      EpilogueCM.invalidateCostModelingDecisions();
   }
   Legal->prepareToFoldTailByMasking();
 
   // Collect the instructions (and their associated costs) that will be more
   // profitable to scalarize.
-  EnabledCM->collectNonVectorizedAndSetWideningDecisions(
+  EpilogueCM.collectNonVectorizedAndSetWideningDecisions(
       EpilogueVectorizationForceVF);
 
   // Expecting only 2 vplans as epilogue TF is applied only for forced VFs.
@@ -5645,9 +5646,10 @@ bool LoopVectorizationPlanner::planForEpilogueTF() {
          "EpilogueUserVF");
   VPlans.pop_back();
   buildVPlans(*VPlan1, EpilogueVectorizationForceVF,
-              EpilogueVectorizationForceVF);
+              EpilogueVectorizationForceVF, EpilogueCM);
 
-  cost(*VPlans.back(), EpilogueVectorizationForceVF, /*RU=*/nullptr);
+  cost(*VPlans.back(), EpilogueVectorizationForceVF, /*RU=*/nullptr,
+       EpilogueCM);
   return true;
 }
 
@@ -5844,9 +5846,11 @@ LoopVectorizationPlanner::precomputeCosts(VPlan &Plan, ElementCount VF,
   return Cost;
 }
 
-InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
-                                               VPRegisterUsage *RU) const {
-  VPCostContext CostCtx(*TLI, Plan, *EnabledCM, Config,
+InstructionCost
+LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
+                               VPRegisterUsage *RU,
+                               LoopVectorizationCostModel &EnabledCM) const {
+  VPCostContext CostCtx(*TLI, Plan, EnabledCM, Config,
                         /*ReusePrintingSlotTracker=*/true);
   InstructionCost Cost = precomputeCosts(Plan, VF, CostCtx);
 
@@ -5922,7 +5926,7 @@ LoopVectorizationPlanner::computeBestVF() {
          "More than a single plan/VF w/o any plan having scalar VF");
 
   // TODO: Compute scalar cost using VPlan-based cost model.
-  InstructionCost ScalarCost = EnabledCM->expectedCost(ScalarVF);
+  InstructionCost ScalarCost = CM.expectedCost(ScalarVF);
   LLVM_DEBUG(dbgs() << "LV: Scalar loop costs: " << ScalarCost << ".\n");
   VectorizationFactor ScalarFactor(ScalarVF, ScalarCost, ScalarCost);
   VectorizationFactor BestFactor = ScalarFactor;
@@ -5946,8 +5950,7 @@ LoopVectorizationPlanner::computeBestVF() {
       return Config.shouldConsiderRegPressureForVF(VF);
     });
     if (ConsiderRegPressure)
-      RUs = calculateRegisterUsageForPlan(*P, VFs, TTI,
-                                          EnabledCM->ValuesToIgnore);
+      RUs = calculateRegisterUsageForPlan(*P, VFs, TTI, CM.ValuesToIgnore);
 
     for (unsigned I = 0; I < VFs.size(); I++) {
       ElementCount VF = VFs[I];
@@ -5970,7 +5973,7 @@ LoopVectorizationPlanner::computeBestVF() {
       }
 
       InstructionCost Cost =
-          cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
+          cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr, CM);
       VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
 
       if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
@@ -6006,7 +6009,7 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
 
   RUN_VPLAN_PASS(VPlanTransforms::replaceWideCanonicalIVWithWideIV, BestVPlan,
                  *PSE.getSE(), TTI, Config.CostKind, BestVF, BestUF,
-                 EnabledCM->ValuesToIgnore);
+                 CM.ValuesToIgnore);
   // TODO: Move to VPlan transform stage once the transition to the VPlan-based
   // cost model is complete for better cost estimates.
   RUN_VPLAN_PASS(VPlanTransforms::unrollByUF, BestVPlan, BestUF);
@@ -6021,7 +6024,7 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
                    BestVPlan, BestVF, VScale);
   }
 
-  if (EnabledCM->maskPartialAliasing()) {
+  if (CM.maskPartialAliasing()) {
     assert(BestVPlan.hasTailFolded() && "Expected tail folding to be enabled");
     RUN_VPLAN_PASS(VPlanTransforms::materializeAliasMaskCheckBlock, BestVPlan,
                    *Legal->getRuntimePointerChecking()->getDiffChecks(),
@@ -6596,7 +6599,8 @@ VPRecipeBuilder::tryToCreateWidenNonPhiRecipe(VPSingleDefRecipe *R,
 // optimizations.
 static void printOptimizedVPlan(VPlan &) {}
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(
+    LoopVectorizationCostModel &EnabledCM) {
   bool IsInnerLoop = OrigLoop->isInnermost();
 
   // Set up loop versioning for inner loops with memory runtime checks.
@@ -6646,8 +6650,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
   bool ForceVectorization = Hints.getForce() == LoopVectorizeHints::FK_Enabled;
   bool OptForSize =
       !ForceVectorization &&
-      (EnabledCM->EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
-       EnabledCM->EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
+      (EnabledCM.EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
+       EnabledCM.EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
   unsigned SCEVCheckThreshold = ForceVectorization
                                     ? PragmaVectorizeSCEVCheckThreshold
                                     : VectorizeSCEVCheckThreshold;
@@ -6675,23 +6679,24 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
 
   RUN_VPLAN_PASS(VPlanTransforms::createLoopRegions, *VPlan0,
                  getDebugLocFromInstOrOperands(Legal->getPrimaryInduction()));
-  if (EnabledCM->foldTailByMasking())
+  if (EnabledCM.foldTailByMasking())
     RUN_VPLAN_PASS(VPlanTransforms::foldTailByMasking, *VPlan0);
   RUN_VPLAN_PASS(VPlanTransforms::introduceMasksAndLinearize, *VPlan0);
 
   return VPlan0;
 }
 
-void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
-                                           ElementCount MaxVF) {
+void LoopVectorizationPlanner::buildVPlans(
+    VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF,
+    LoopVectorizationCostModel &EnabledCM) {
   if (ElementCount::isKnownGT(MinVF, MaxVF))
     return;
 
   auto MaxVFTimes2 = MaxVF * 2;
   for (ElementCount VF = MinVF; ElementCount::isKnownLT(VF, MaxVFTimes2);) {
     VFRange SubRange = {VF, MaxVFTimes2};
-    auto Plan =
-        tryToBuildVPlan(std::unique_ptr<VPlan>(VPlan1.duplicate()), SubRange);
+    auto Plan = tryToBuildVPlan(std::unique_ptr<VPlan>(VPlan1.duplicate()),
+                                SubRange, EnabledCM);
     VF = SubRange.End;
 
     if (!Plan)
@@ -6704,7 +6709,7 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
                    Config.getMinimalBitwidths());
     RUN_VPLAN_PASS(VPlanTransforms::optimize, *Plan);
     // TODO: try to put addExplicitVectorLength close to addActiveLaneMask
-    if (EnabledCM->foldTailWithEVL()) {
+    if (EnabledCM.foldTailWithEVL()) {
       RUN_VPLAN_PASS(VPlanTransforms::addExplicitVectorLength, *Plan,
                      Config.getMaxSafeElements());
       RUN_VPLAN_PASS(VPlanTransforms::optimizeEVLMasks, *Plan);
@@ -6714,7 +6719,7 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
             RUN_VPLAN_PASS(VPlanTransforms::narrowInterleaveGroups, *Plan, TTI))
       VPlans.push_back(std::move(P));
 
-    TailFoldingStyle Style = EnabledCM->getTailFoldingStyle();
+    TailFoldingStyle Style = EnabledCM.getTailFoldingStyle();
     RUN_VPLAN_PASS(VPlanTransforms::materializeHeaderMask, *Plan,
                    useActiveLaneMask(Style),
                    useActiveLaneMaskForControlFlow(Style));
@@ -6725,8 +6730,8 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
   }
 }
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
-                                                   VFRange &Range) {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
+    VPlanPtr Plan, VFRange &Range, LoopVectorizationCostModel &EnabledCM) {
 
   // For outer loops, the plan only needs basic recipe conversion and induction
   // live-out optimization; the full inner-loop recipe building below does not
@@ -6752,8 +6757,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
 
   bool RequiresScalarEpilogueCheck =
       LoopVectorizationPlanner::getDecisionAndClampRange(
-          [&](ElementCount VF) {
-            return !EnabledCM->requiresScalarEpilogue(VF.isVector());
+          [EnabledCM](ElementCount VF) {
+            return !EnabledCM.requiresScalarEpilogue(VF.isVector());
           },
           Range);
   // Update the branch in the middle block if a scalar epilogue is required.
@@ -6771,9 +6776,9 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // TODO: Consider using getDecisionAndClampRange here to split up VPlans.
   bool IVUpdateMayOverflow = false;
   for (ElementCount VF : Range)
-    IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(EnabledCM, VF);
+    IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(&EnabledCM, VF);
 
-  TailFoldingStyle Style = EnabledCM->getTailFoldingStyle();
+  TailFoldingStyle Style = EnabledCM.getTailFoldingStyle();
   // Use NUW for the induction increment if we proved that it won't overflow in
   // the vector loop or when not folding the tail. In the later case, we know
   // that the canonical induction increment will not overflow as the vector trip
@@ -6800,10 +6805,10 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // placeholders for its members' Recipes which we'll be replacing with a
   // single VPInterleaveRecipe.
   for (InterleaveGroup<Instruction> *IG :
-       EnabledCM->InterleaveInfo.getInterleaveGroups()) {
-    auto ApplyIG = [IG, this](ElementCount VF) -> bool {
+       EnabledCM.InterleaveInfo.getInterleaveGroups()) {
+    auto ApplyIG = [IG, EnabledCM](ElementCount VF) -> bool {
       bool Result = (VF.isVector() && // Query is illegal for VF == 1
-                     EnabledCM->getWideningDecision(IG->getInsertPos(), VF) ==
+                     EnabledCM.getWideningDecision(IG->getInsertPos(), VF) ==
                          LoopVectorizationCostModel::CM_Interleave);
       // For scalable vectors, the interleave factors must be <= 8 since we
       // require the (de)interleaveN intrinsics instead of shufflevectors.
@@ -6820,7 +6825,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // Construct wide recipes and apply predication for original scalar
   // VPInstructions in the loop.
   // ---------------------------------------------------------------------------
-  VPRecipeBuilder RecipeBuilder(*Plan, Legal, *EnabledCM, Builder);
+  VPRecipeBuilder RecipeBuilder(*Plan, Legal, EnabledCM, Builder);
 
   // Scan the body of the loop in a topological order to visit each basic block
   // after having visited its predecessor basic blocks.
@@ -6831,7 +6836,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   RUN_VPLAN_PASS(VPlanTransforms::createInLoopReductionRecipes, *Plan,
                  Range.Start);
 
-  VPCostContext CostCtx(*TLI, *Plan, *EnabledCM, Config);
+  VPCostContext CostCtx(*TLI, *Plan, EnabledCM, Config);
 
   RUN_VPLAN_PASS(VPlanTransforms::makeMemOpWideningDecisions, *Plan, Range,
                  RecipeBuilder, CostCtx);
@@ -6931,7 +6936,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // range for better cost estimation.
   // TODO: Enable following transform when the EVL-version of extended-reduction
   // and mulacc-reduction are implemented.
-  if (!EnabledCM->foldTailWithEVL()) {
+  if (!EnabledCM.foldTailWithEVL()) {
     RUN_VPLAN_PASS(VPlanTransforms::createPartialReductions, *Plan, CostCtx,
                    Range);
     RUN_VPLAN_PASS(VPlanTransforms::convertToAbstractRecipes, *Plan, CostCtx,
@@ -6942,7 +6947,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // for this VPlan, replace the Recipes widening its memory instructions with a
   // single VPInterleaveRecipe at its insertion point.
   RUN_VPLAN_PASS(VPlanTransforms::createInterleaveGroups, *Plan,
-                 InterleaveGroups, EnabledCM->isEpilogueAllowed());
+                 InterleaveGroups, EnabledCM.isEpilogueAllowed());
 
   // Convert memory recipes to strided access recipes if the strided access is
   // legal and profitable.
@@ -6959,7 +6964,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
 
   RUN_VPLAN_PASS(VPlanTransforms::dropPoisonGeneratingRecipes, *Plan);
 
-  if (EnabledCM->maskPartialAliasing())
+  if (EnabledCM.maskPartialAliasing())
     RUN_VPLAN_PASS(VPlanTransforms::attachAliasMaskToHeaderMask, *Plan);
 
   assert(verifyVPlanIsValid(*Plan) && "VPlan is invalid");
@@ -7003,7 +7008,7 @@ void LoopVectorizationPlanner::addReductionResultComputation(
 
     // Remove the predicated select if the target doesn't want it.
     VPValue *V;
-    if (!EnabledCM->usePredicatedReductionSelect(RecurrenceKind) &&
+    if (!CM.usePredicatedReductionSelect(RecurrenceKind) &&
         match(PhiR->getBackedgeValue(),
               m_Select(m_Specific(HeaderMask), m_VPValue(V), m_Specific(PhiR))))
       PhiR->setBackedgeValue(V);
@@ -7182,7 +7187,7 @@ void LoopVectorizationPlanner::attachRuntimeChecks(
   const auto &[SCEVCheckCond, SCEVCheckBlock] = RTChecks.getSCEVChecks();
   if (SCEVCheckBlock && SCEVCheckBlock->hasNPredecessors(0)) {
     assert((!Config.OptForSize ||
-            EnabledCM->Hints->getForce() == LoopVectorizeHints::FK_Enabled) &&
+            CM.Hints->getForce() == LoopVectorizeHints::FK_Enabled) &&
            "Cannot SCEV check stride or overflow when optimizing for size");
     RUN_VPLAN_PASS(VPlanTransforms::attachCheckBlock, Plan, SCEVCheckCond,
                    SCEVCheckBlock, HasBranchWeights);
@@ -7196,7 +7201,7 @@ void LoopVectorizationPlanner::attachRuntimeChecks(
 
     if (Config.OptForSize) {
       assert(
-          EnabledCM->Hints->getForce() == LoopVectorizeHints::FK_Enabled &&
+          CM.Hints->getForce() == LoopVectorizeHints::FK_Enabled &&
           "Cannot emit memory checks when optimizing for size, unless forced "
           "to vectorize.");
       ORE->emit([&]() {
@@ -7220,9 +7225,8 @@ bool LoopVectorizationPlanner::requiresScalarEpilogue(VPlan &Plan,
   // loop. Must be called before removeBranchOnConst.
   VPBasicBlock *MiddleVPBB = Plan.getMiddleBlock();
   bool Result = MiddleVPBB->getSingleSuccessor() == Plan.getScalarPreheader();
-  assert(
-      EnabledCM->requiresScalarEpilogue(VF.isVector()) == Result &&
-      "EnabledCM->requiresScalarEpilogue and the VPlan-based check must agree");
+  assert(CM.requiresScalarEpilogue(VF.isVector()) == Result &&
+         "CM.requiresScalarEpilogue and the VPlan-based check must agree");
   return Result;
 }
 
@@ -8221,6 +8225,9 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                             OptForSize);
   LoopVectorizationCostModel CM(SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE,
                                 GetBFI, F, &Hints, IAI, Config);
+  // Use the planner for vectorization.
+  LoopVectorizationPlanner LVP(L, LI, DT, TLI, *TTI, &LVL, CM, Config, IAI, PSE,
+                               Hints, ORE);
 
   EpilogueLowering EpilogueTailLoweringStatus =
       getEpilogueTailLowering(CM, L, ORE, LVL, Hints);
@@ -8237,12 +8244,6 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                                   *TailFoldingCMIAI, Config);
   }
 
-  // Use the planner for vectorization.
-  LoopVectorizationPlanner LVP(L, LI, DT, TLI, *TTI, &LVL, &CM,
-                               EpilogueTailFoldingCM ? &*EpilogueTailFoldingCM
-                                                     : nullptr,
-                               Config, IAI, PSE, Hints, ORE);
-
   // Get user vectorization factor and interleave count.
   ElementCount UserVF = Hints.getWidth();
   unsigned UserIC = Hints.getInterleave();
@@ -8256,17 +8257,13 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   // Plan how to best vectorize.
   LVP.plan(UserVF, UserIC);
   if (EpilogueTailFoldingCM) {
-    // Enable the epilogue tail-folding CM
-    LVP.enableEpilogueTFCM();
-    if (!LVP.planForEpilogueTF()) {
+    if (!LVP.planForEpilogueTF(EpilogueTailFoldingCM.value())) {
       // we can't apply epilogue TF:
       reportVectorizationInfo(
           "Applying epilogue tail-folding failed, disable it.",
           "InvalidTailFoldedEpilogue", ORE, L);
       EpilogueTailFoldingCM.reset();
     }
-    // Get back the default CM:
-    LVP.enableDefaultCM();
   }
 
   auto [VF, BestPlanPtr] = LVP.computeBestVF();
@@ -8509,8 +8506,6 @@ bool LoopVectorizePass::processLoop(Loop *L) {
         LoopVectorizationPlanner::EpilogueVectorizationKind::MainLoop);
     ++LoopsVectorized;
 
-    if (EpilogueTailFoldingCM)
-      LVP.enableEpilogueTFCM();
     // Derive EPI fields from VPlan-generated IR.
     BasicBlock *EntryBB =
         cast<VPIRBasicBlock>(BestMainPlan.getEntry())->getIRBasicBlock();
@@ -8543,7 +8538,6 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     connectEpilogueVectorLoop(BestEpiPlan, L, EPI, DT, LI, Checks, InstsToMove,
                               ResumeValues, EpilogueTailFoldingCM.has_value());
     ++LoopsEpilogueVectorized;
-    LVP.enableDefaultCM();
   } else {
     InnerLoopVectorizer LB(L, PSE, LI, DT, TTI, AC, VF.Width, IC, Checks,
                            BestPlan);

>From 4469b6fa408c34b31465b76ee34b92304480b798 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Wed, 19 Aug 2026 16:06:42 +0100
Subject: [PATCH 10/11] Add test cases for different scenarios that should be
 supported by the feature

---
 .../AArch64/fold-epilogue-tail.ll             | 412 +++++++++++++++++-
 .../LoopVectorize/fold-epilogue-tail.ll       | 147 +------
 2 files changed, 411 insertions(+), 148 deletions(-)

diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index ece1fc3a15b11..79495c4ad9c3e 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -8,9 +8,9 @@
 
 target triple = "aarch64-linux-gnu"
 
-define void @test_epilogue_tf(ptr %A, i64 %n) {
+define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-LABEL: define void @test_epilogue_tf(
-; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i32 [[VAL:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ITER_CHECK:.*]]:
 ; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
@@ -20,13 +20,15 @@ define void @test_epilogue_tf(ptr %A, i64 %n) {
 ; CHECK:       [[VECTOR_PH]]:
 ; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <16 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <16 x i32> [[BROADCAST_SPLATINSERT]], <16 x i32> poison, <16 x i32> zeroinitializer
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
-; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP1]], i64 16
-; CHECK-NEXT:    store <16 x i8> splat (i8 1), ptr [[TMP1]], align 1
-; CHECK-NEXT:    store <16 x i8> splat (i8 1), ptr [[TMP2]], align 1
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 16
+; CHECK-NEXT:    store <16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP1]], align 4
+; CHECK-NEXT:    store <16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP2]], align 4
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
@@ -38,14 +40,16 @@ define void @test_epilogue_tf(ptr %A, i64 %n) {
 ; CHECK:       [[VEC_EPILOG_PH]]:
 ; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <8 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT3:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT2]], <8 x i32> poison, <8 x i32> zeroinitializer
 ; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
 ; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT5:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX2]]
-; CHECK-NEXT:    call void @llvm.masked.store.v8i8.p0(<8 x i8> splat (i8 1), ptr align 1 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], 8
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX4]]
+; CHECK-NEXT:    call void @llvm.masked.store.v8i32.p0(<8 x i32> [[BROADCAST_SPLAT3]], ptr align 4 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-NEXT:    [[INDEX_NEXT5]] = add i64 [[INDEX4]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT5]], i64 [[N]])
 ; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP6:%.*]] = xor i1 [[TMP5]], true
 ; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
@@ -59,8 +63,8 @@ entry:
 
 for.body:
   %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
-  %arrayidx = getelementptr inbounds i8, ptr %A, i64 %iv
-  store i8 1, ptr %arrayidx, align 1
+  %arrayidx = getelementptr inbounds i32, ptr %A, i64 %iv
+  store i32 %val, ptr %arrayidx, align 4
   %iv.next = add nuw nsw i64 %iv, 1
   %exitcond = icmp ne i64 %iv.next, %n
   br i1 %exitcond, label %for.body, label %exit
@@ -69,6 +73,388 @@ exit:
   ret void
 }
 
+define i32 @add_redc(ptr %src, i64 %n) {
+; CHECK-LABEL: define i32 @add_redc(
+; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI2:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 16
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 1
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i32>, ptr [[TMP3]], align 1
+; CHECK-NEXT:    [[TMP4]] = add <16 x i32> [[WIDE_LOAD]], [[VEC_PHI]]
+; CHECK-NEXT:    [[TMP5]] = add <16 x i32> [[WIDE_LOAD3]], [[VEC_PHI2]]
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[BIN_RDX:%.*]] = add <16 x i32> [[TMP5]], [[TMP4]]
+; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.add.v16i32(<16 x i32> [[BIN_RDX]])
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <8 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI5:%.*]] = phi <8 x i32> [ [[TMP8]], %[[VEC_EPILOG_PH]] ], [ [[TMP11:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
+; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 [[TMP9]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
+; CHECK-NEXT:    [[TMP10:%.*]] = add <8 x i32> [[WIDE_MASKED_LOAD]], [[VEC_PHI5]]
+; CHECK-NEXT:    [[TMP11]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> [[TMP10]], <8 x i32> [[VEC_PHI5]]
+; CHECK-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
+; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP13:%.*]] = xor i1 [[TMP12]], true
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP14:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP11]])
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[TMP14]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i32 [[ADD_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %red = phi i32 [ 0, %entry ], [ %add, %loop ]
+  %gep = getelementptr inbounds i32, ptr %src, i64 %iv
+  %load = load i32, ptr %gep, align 1
+  %add = add i32 %load, %red
+  %iv.next = add i64 %iv, 1
+  %icmp3 = icmp eq i64 %iv, %n
+  br i1 %icmp3, label %exit, label %loop
+
+exit:
+  ret i32 %add
+}
+
+define i32 @max_redc(ptr %src, i64 %n) {
+; CHECK-LABEL: define i32 @max_redc(
+; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI2:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 16
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 1
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i32>, ptr [[TMP3]], align 1
+; CHECK-NEXT:    [[TMP4]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[WIDE_LOAD]], <16 x i32> [[VEC_PHI]])
+; CHECK-NEXT:    [[TMP5]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[WIDE_LOAD3]], <16 x i32> [[VEC_PHI2]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[RDX_MINMAX:%.*]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[TMP4]], <16 x i32> [[TMP5]])
+; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.umax.v16i32(<16 x i32> [[RDX_MINMAX]])
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i32> poison, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT]], <8 x i32> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI5:%.*]] = phi <8 x i32> [ [[BROADCAST_SPLAT]], %[[VEC_EPILOG_PH]] ], [ [[TMP10:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
+; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 [[TMP8]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
+; CHECK-NEXT:    [[TMP9:%.*]] = call <8 x i32> @llvm.umax.v8i32(<8 x i32> [[WIDE_MASKED_LOAD]], <8 x i32> [[VEC_PHI5]])
+; CHECK-NEXT:    [[TMP10]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> [[TMP9]], <8 x i32> [[VEC_PHI5]]
+; CHECK-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
+; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP12:%.*]] = xor i1 [[TMP11]], true
+; CHECK-NEXT:    br i1 [[TMP12]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP13:%.*]] = call i32 @llvm.vector.reduce.umax.v8i32(<8 x i32> [[TMP10]])
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[TMP13]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i32 [[MAX_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %red = phi i32 [ 0, %entry ], [ %max, %loop ]
+  %gep = getelementptr inbounds i32, ptr %src, i64 %iv
+  %load = load i32, ptr %gep, align 1
+  %max = call i32 @llvm.umax(i32 %load, i32 %red)
+  %iv.next = add i64 %iv, 1
+  %icmp3 = icmp eq i64 %iv, %n
+  br i1 %icmp3, label %exit, label %loop
+
+exit:
+  ret i32 %max
+}
+
+define i32 @live-out(ptr %src, i64 %n) {
+; CHECK-LABEL: define i32 @live-out(
+; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP1]], i64 16
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <16 x i32> [[WIDE_LOAD]], i64 15
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX2]]
+; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 4 [[TMP5]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
+; CHECK-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP7:%.*]] = xor i1 [[TMP6]], true
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP8:%.*]] = xor <8 x i1> [[ACTIVE_LANE_MASK]], splat (i1 true)
+; CHECK-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v8i1(<8 x i1> [[TMP8]], i1 false)
+; CHECK-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
+; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <8 x i32> [[WIDE_MASKED_LOAD]], i64 [[LAST_ACTIVE_LANE]]
+; CHECK-NEXT:    br label %[[FOR_END]]
+; CHECK:       [[FOR_END]]:
+; CHECK-NEXT:    [[LOAD_LCSSA:%.*]] = phi i32 [ [[TMP9]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP4]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i32 [[LOAD_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %gep = getelementptr inbounds nuw i32, ptr %src, i64 %iv
+  %load = load i32, ptr %gep, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, %n
+  br i1 %ec, label %for.end, label %loop
+
+for.end:
+  ret i32 %load
+}
+
+define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
+; CHECK-LABEL: define void @reversed-loop(
+; CHECK-SAME: ptr [[A:%.*]], i32 [[N:%.*]], i32 [[VAL:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[ST:%.*]] = sub i32 [[N]], 1
+; CHECK-NEXT:    [[TMP0:%.*]] = add i32 [[N]], -1
+; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[N]], -2
+; CHECK-NEXT:    [[SMIN1:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP1]], i32 -1)
+; CHECK-NEXT:    [[TMP2:%.*]] = sub i32 [[TMP0]], [[SMIN1]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK:       [[VECTOR_SCEVCHECK]]:
+; CHECK-NEXT:    [[TMP3:%.*]] = add i32 [[N]], -2
+; CHECK-NEXT:    [[SMIN:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP3]], i32 -1)
+; CHECK-NEXT:    [[TMP4:%.*]] = sub i32 [[TMP3]], [[SMIN]]
+; CHECK-NEXT:    [[TMP5:%.*]] = sub i32 [[ST]], [[TMP4]]
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp sgt i32 [[TMP5]], [[ST]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK2:%.*]] = icmp ult i32 [[TMP2]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK2]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = and i32 [[TMP2]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[TMP2]], [[TMP7]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <16 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <16 x i32> [[BROADCAST_SPLATINSERT]], <16 x i32> poison, <16 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP8:%.*]] = sub i32 [[ST]], [[N_VEC]]
+; CHECK-NEXT:    [[REVERSE:%.*]] = shufflevector <16 x i32> [[BROADCAST_SPLAT]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP9:%.*]] = sub i32 [[ST]], [[INDEX]]
+; CHECK-NEXT:    [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP9]]
+; CHECK-NEXT:    [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[TMP10]], i64 -15
+; CHECK-NEXT:    [[TMP12:%.*]] = getelementptr inbounds i32, ptr [[TMP10]], i64 -31
+; CHECK-NEXT:    store <16 x i32> [[REVERSE]], ptr [[TMP11]], align 4
+; CHECK-NEXT:    store <16 x i32> [[REVERSE]], ptr [[TMP12]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 32
+; CHECK-NEXT:    [[TMP13:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[TMP2]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <8 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT4:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT3]], <8 x i32> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    [[REVERSE5:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLAT4]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[VEC_EPILOG_RESUME_VAL]], i32 [[TMP2]])
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX6:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT8:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP14:%.*]] = sub i32 [[ST]], [[INDEX6]]
+; CHECK-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP14]]
+; CHECK-NEXT:    [[TMP16:%.*]] = getelementptr i32, ptr [[TMP15]], i64 -7
+; CHECK-NEXT:    [[REVERSE7:%.*]] = shufflevector <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    call void @llvm.masked.store.v8i32.p0(<8 x i32> [[REVERSE5]], ptr align 4 [[TMP16]], <8 x i1> [[REVERSE7]])
+; CHECK-NEXT:    [[INDEX_NEXT8]] = add i32 [[INDEX6]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[INDEX_NEXT8]], i32 [[TMP2]])
+; CHECK-NEXT:    [[TMP17:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP18:%.*]] = xor i1 [[TMP17]], true
+; CHECK-NEXT:    br i1 [[TMP18]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[ST]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[IV]]
+; CHECK-NEXT:    store i32 [[VAL]], ptr [[ARRAYIDX]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = sub nuw nsw i32 [[IV]], 1
+; CHECK-NEXT:    [[EXITCOND:%.*]] = icmp sge i32 [[IV_NEXT]], 0
+; CHECK-NEXT:    br i1 [[EXITCOND]], label %[[FOR_BODY]], label %[[EXIT]], !llvm.loop [[LOOP14:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %st = sub i32 %n, 1
+  br label %for.body
+
+for.body:
+  %iv = phi i32 [ %st, %entry ], [ %iv.next, %for.body ]
+  %arrayidx = getelementptr inbounds i32, ptr %A, i32 %iv
+  store i32 %val, ptr %arrayidx, align 4
+  %iv.next = sub nuw nsw i32 %iv, 1
+  %exitcond = icmp sge i32 %iv.next, 0
+  br i1 %exitcond, label %for.body, label %exit
+
+exit:
+  ret void
+}
+
+define void @math_func(ptr %A, i32 %n) {
+; CHECK-LABEL: define void @math_func(
+; CHECK-SAME: ptr [[A:%.*]], i32 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[UMAX:%.*]] = call i32 @llvm.umax.i32(i32 [[N]], i32 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[UMAX]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i32 [[UMAX]], 16
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i32 [[UMAX]], 15
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[UMAX]], [[TMP0]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds float, ptr [[A]], i32 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x float>, ptr [[TMP1]], align 4
+; CHECK-NEXT:    [[TMP2:%.*]] = call <16 x float> @llvm.pow.v16f32(<16 x float> [[WIDE_LOAD]], <16 x float> splat (float 2.000000e+00))
+; CHECK-NEXT:    store <16 x float> [[TMP2]], ptr [[TMP1]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 16
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[UMAX]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[VEC_EPILOG_RESUME_VAL]], i32 [[UMAX]])
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX2:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds float, ptr [[A]], i32 [[INDEX2]]
+; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x float> @llvm.masked.load.v8f32.p0(ptr align 4 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x float> poison)
+; CHECK-NEXT:    [[TMP5:%.*]] = call <8 x float> @llvm.pow.v8f32(<8 x float> [[WIDE_MASKED_LOAD]], <8 x float> splat (float 2.000000e+00))
+; CHECK-NEXT:    call void @llvm.masked.store.v8f32.p0(<8 x float> [[TMP5]], ptr align 4 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-NEXT:    [[INDEX_NEXT3]] = add i32 [[INDEX2]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[INDEX_NEXT3]], i32 [[UMAX]])
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP7:%.*]] = xor i1 [[TMP6]], true
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP16:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %for.body ]
+  %arrayidx = getelementptr inbounds float, ptr %A, i32 %iv
+  %load = load float, ptr %arrayidx, align 4
+  %val = call float @llvm.pow.f32(float %load, float 2.0)
+  store float %val, ptr %arrayidx, align 4
+  %iv.next = add nuw nsw i32 %iv, 1
+  %exitcond = icmp ult i32 %iv.next, %n
+  br i1 %exitcond, label %for.body, label %exit
+
+exit:
+  ret void
+}
+
 define i64 @test_no_masked_interleave_support(i64 %y, i32 %n) {
 ; CHECK-INVALIDATE-INTERLEAVE-LABEL: Checking a loop in 'test_no_masked_interleave_support'
 ; CHECK-INVALIDATE-INTERLEAVE: LV: epilogue tail-folding is enabled
diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index ea94e709920e8..2d02f4006fcc1 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -1,103 +1,15 @@
 ; REQUIRES: asserts
+; RUN: opt -S < %s -p loop-vectorize -debug-only=loop-vectorize --disable-output \
+; RUN: -epilogue-tail-folding-policy=prefer-fold-tail -pass-remarks-analysis=loop-vectorize 2>&1 | FileCheck %s
 
-; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -pass-remarks-analysis=loop-vectorize -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s
-
-; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize -enable-vplan-native-path --disable-output \
-; RUN: -epilogue-tail-folding-policy=prefer-fold-tail -pass-remarks-analysis=loop-vectorize \
-; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s --check-prefix=CHECK-OUTER-LOOP
-
-; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -force-vector-width=16 \
-; RUN:  -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s --check-prefix=CHECK-NO-FORCED-MAIN-VF
-
-; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -epilogue-vectorization-force-VF=8  \
-; RUN: -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s --check-prefix=CHECK-NO-FORCED-EPILOGUE-VF
-
-; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize -enable-epilogue-vectorization=false \
-; RUN: --disable-output -force-vector-width=16 -epilogue-vectorization-force-VF=8 -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s --check-prefix=CHECK-DISABLED-EPILOG
-
-; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -vectorize-scev-check-threshold=0 < %s 2>&1 | FileCheck %s \
-; RUN: --check-prefix=CHECK-NO-VPLANS
+; RUN: opt -S < %s -p loop-vectorize -debug-only=loop-vectorize -enable-epilogue-vectorization=false \
+; RUN: --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -pass-remarks-analysis=loop-vectorize 2>&1 \
+; RUN: | FileCheck %s --check-prefix=CHECK-DISABLED-EPILOG
 
 define void @test_epilogue_tf(ptr %A, i64 %n) {
-; CHECK-LABEL: Checking a loop in 'test_epilogue_tf'
-; CHECK: LV: epilogue tail-folding is enabled
-;
-entry:
-  br label %for.body
-
-for.body:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
-  %arrayidx = getelementptr inbounds i8, ptr %A, i64 %iv
-  store i8 1, ptr %arrayidx, align 1
-  %iv.next = add nuw nsw i64 %iv, 1
-  %exitcond = icmp ne i64 %iv.next, %n
-  br i1 %exitcond, label %for.body, label %exit
-
-exit:
-  ret void
-}
-
-; This case can't be tail-folded because all the iterations will be executed by
-; main vector loop.
-define void @test_no_iterations_left(ptr %A) {
-; CHECK-LABEL: Checking a loop in 'test_no_iterations_left'
-; CHECK: LV: epilogue tail-folding is enabled
-; CHECK: LV: This case of epilogue loop can't be tail-folded.
-; CHECK: LV: Applying epilogue tail-folding failed, disable it.
-;
-entry:
-  br label %for.body
-
-for.body:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
-  %arrayidx = getelementptr inbounds i8, ptr %A, i64 %iv
-  store i8 1, ptr %arrayidx, align 1
-  %iv.next = add nuw nsw i64 %iv, 1
-  %exitcond = icmp ne i64 %iv.next, 64
-  br i1 %exitcond, label %for.body, label %exit
-
-exit:
-  ret void
-}
-
-; Can't build a valid vplan for this case because too many SCEV checks needed,
-; more than the specfied limit.
-define i64 @test_no_vplan_built(ptr %dst, i64 %n) {
-; CHECK-NO-VPLANS-LABEL: Checking a loop in 'test_no_vplan_built'
-; CHECK-NO-VPLANS: LV: epilogue tail-folding is enabled
-; CHECK-NO-VPLANS: LV: no vplans have been built for main loop VF, bail out of epilogue tail-folding
-;
-entry:
-  br label %loop
-
-loop:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-  %dead.iv = phi i16 [ 0, %entry ], [ %dead.iv.next, %loop ]
-  %prev = phi i64 [ 0, %entry ], [ %ext, %loop ]
-  %iv.next = add nuw nsw i64 %iv, 1
-  %dead.iv.next = add i16 %dead.iv, 1
-  %ext = zext i16 %dead.iv.next to i64
-  %gep = getelementptr inbounds i64, ptr %dst, i64 %prev
-  store i64 %iv, ptr %gep, align 8
-  %cmp = icmp slt i64 %iv.next, %n
-  br i1 %cmp, label %loop, label %exit
-
-exit:
-  %result = phi i64 [ %ext, %loop ]
-  ret i64 %result
-}
-
-define void @test_no_vf(ptr %A, i64 %n) {
-; CHECK-NO-FORCED-MAIN-VF-LABEL: Checking a loop in 'test_no_vf'
-; CHECK-NO-FORCED-MAIN-VF: remark: <unknown>:0:0: For now, Epilogue tail-folding can't be applied without forced epilogue/main loop VF
-; CHECK-NO-FORCED-MAIN-VF-NOT: LV: epilogue tail-folding is enabled
-
-; CHECK-NO-FORCED-EPILOGUE-VF-LABEL: Checking a loop in 'test_no_vf'
-; CHECK-NO-FORCED-EPILOGUE-VF: remark: <unknown>:0:0: For now, Epilogue tail-folding can't be applied without forced epilogue/main loop VF
-; CHECK-NO-FORCED-EPILOGUE-VF-NOT: LV: epilogue tail-folding is enabled
+; CHECK-LABEL: LV: Checking a loop in 'test_epilogue_tf'
+; CHECK: LV: epilogue tail-folding is not supported yet
+; CHECK: remark: <unknown>:0:0: The epilogue-tail-folding policy prefer-fold-tail is not supported yet, fall back to a normal epilogue
 ;
 entry:
   br label %for.body
@@ -115,9 +27,8 @@ exit:
 }
 
 define void @epilogue_is_disabled(ptr %a, i64 %n) {
-; CHECK-DISABLED-EPILOG-LABEL: Checking a loop in 'epilogue_is_disabled'
+; CHECK-DISABLED-EPILOG-LABEL: LV: Checking a loop in 'epilogue_is_disabled'
 ; CHECK-DISABLED-EPILOG: remark: <unknown>:0:0: Options conflict, epilogue vectorization is disallowed while epilogue tail-folding allowed!
-; CHECK-DISABLED-EPILOG-NOT: LV: epilogue tail-folding is enabled
 ;
 entry:
   br label %for.body
@@ -135,10 +46,9 @@ for.end:
 }
 
 define i16 @require_scalar_epilogue(ptr %dst, i64 %x) {
-; CHECK-LABEL: Checking a loop in 'require_scalar_epilogue'
+; CHECK-LABEL: LV: Checking a loop in 'require_scalar_epilogue'
 ; CHECK: LV: Epilogue tail-folding can't be applied because scalar epilogue is required
 ; CHECK-NEXT: LV: Fall back to a normal epilogue
-; CHECK-NOT: LV: epilogue tail-folding is enabled
 ;
 entry:
   br label %loop.header
@@ -165,10 +75,9 @@ exit.2:
 }
 
 define i32 @opt_for_size(ptr %p, i32 %n) optsize {
-; CHECK-LABEL: Checking a loop in 'opt_for_size'
+; CHECK-LABEL: LV: Checking a loop in 'opt_for_size'
 ; CHECK: LV: No epilogue to apply tail-folding for.
 ; CHECK-NEXT: LV: Fall back to a normal epilogue
-; CHECK-NOT: LV: epilogue tail-folding is enabled
 ;
 entry:
   br label %for.body
@@ -186,36 +95,4 @@ for.body:
 
 for.end:
   ret i32 0
-}
-
-define void @test_outer_loop(ptr %A, i64 %m) {
-; CHECK-OUTER-LOOP-LABEL: Checking a loop in 'test_outer_loop'
-; CHECK-OUTER-LOOP: remark: <unknown>:0:0: Epilogue tail-folding is not supported for outer loop
-; CHECK-OUTER-LOOP-NOT: LV: epilogue tail-folding is enabled
-;
-entry:
-  br label %outer.header
-
-outer.header:
-  %iv.outer = phi i64 [ 0, %entry ], [ %iv.outer.next, %outer.latch ]
-  br label %inner
-
-inner:
-  %iv.inner = phi i64 [ 0, %outer.header ], [ %iv.inner.next, %inner ]
-  %gep = getelementptr inbounds i32, ptr %A, i64 %iv.inner
-  store i32 0, ptr %gep, align 4
-  %iv.inner.next = add nuw nsw i64 %iv.inner, 1
-  %inner.ec = icmp eq i64 %iv.inner.next, 8
-  br i1 %inner.ec, label %outer.latch, label %inner
-
-outer.latch:
-  %iv.outer.next = add nuw nsw i64 %iv.outer, 1
-  %outer.ec = icmp eq i64 %iv.outer.next, %m
-  br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !1
-
-exit:
-  ret void
-}
-
-!1 = distinct !{!1, !2}
-!2 = !{!"llvm.loop.vectorize.enable"}
+}
\ No newline at end of file

>From b9c2c5029489883777fb3e6ec54f4525bcca9c2f Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Mon, 24 Aug 2026 10:31:47 +0100
Subject: [PATCH 11/11] Give Planner its own epilogue tail-folding CM and make
 plan() responsible for handling epilogueTF planning

---
 .../Vectorize/LoopVectorizationPlanner.h      |  21 +-
 .../Transforms/Vectorize/LoopVectorize.cpp    | 304 +++++++-----
 .../AArch64/fold-epilogue-tail.ll             | 435 ++++++++++++++----
 .../LoopVectorize/fold-epilogue-tail.ll       |  80 +++-
 4 files changed, 619 insertions(+), 221 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index a6a88fb011a18..0888c32e16393 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -856,6 +856,10 @@ class LoopVectorizationPlanner {
   /// The profitability analysis.
   LoopVectorizationCostModel &CM;
 
+  /// The profitability analysis for epilogue tail-folding.
+  /// Cleared after making cost based decisions.
+  std::unique_ptr<LoopVectorizationCostModel> EpilogueTfCM;
+
   /// VF selection state independent of cost-modeling decisions.
   VFSelectionContext &Config;
 
@@ -897,11 +901,16 @@ class LoopVectorizationPlanner {
   LoopVectorizationPlanner(
       Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
       const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
-      LoopVectorizationCostModel &CM, VFSelectionContext &Config,
-      InterleavedAccessInfo &IAI, PredicatedScalarEvolution &PSE,
-      const LoopVectorizeHints &Hints, OptimizationRemarkEmitter *ORE)
-      : OrigLoop(L), LI(LI), DT(DT), TLI(TLI), TTI(TTI), Legal(Legal), CM(CM),
-        Config(Config), IAI(IAI), PSE(PSE), Hints(Hints), ORE(ORE) {}
+      LoopVectorizationCostModel &CM,
+      std::unique_ptr<LoopVectorizationCostModel> EpilogueTfCM,
+      VFSelectionContext &Config, InterleavedAccessInfo &IAI,
+      PredicatedScalarEvolution &PSE, const LoopVectorizeHints &Hints,
+      OptimizationRemarkEmitter *ORE);
+
+  ~LoopVectorizationPlanner();
+
+  /// Destroy the cost model.
+  void clearEpilogueTfCM();
 
   /// Build VPlans for the specified \p UserVF and \p UserIC if they are
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
@@ -910,7 +919,7 @@ class LoopVectorizationPlanner {
 
   /// Build VPlan for the forced epilogue VF. If vectorization and tail-folding
   /// should be avoided up-front, no tail-folded plans are generated.
-  bool planForEpilogueTF(LoopVectorizationCostModel &EpilogueCM);
+  bool planForEpilogueTF();
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
   /// for each VF.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index f70bc573fabc3..a19891db2ffe3 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3034,6 +3034,9 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
       MaxPowerOf2RuntimeVF = std::nullopt; // Stick with tail-folding for now.
   }
 
+  // TODO: Make NoScalarEpilogueNeeded lambda a separate function to be used
+  // only for main loop VF not also epilogueVF. Using it for epilogueVF against
+  // full TC is inaccurate.
   auto NoScalarEpilogueNeeded = [this, &UserIC](unsigned MaxVF) {
     // Return false if the loop is neither a single-latch-exit loop nor an
     // early-exit loop as tail-folding is not supported in that case.
@@ -3151,9 +3154,13 @@ void LoopVectorizationPlanner::emitInvalidCostRemarks(
   using RecipeVFPair = std::pair<VPRecipeBase *, ElementCount>;
   SmallVector<RecipeVFPair> InvalidCosts;
   for (const auto &Plan : VPlans) {
+    // Skip cost remarks when Plan is not compatible with the CM.
+    // Specifically for the case of epilogue tail-folded Plans.
+    if (Plan->hasTailFolded() ^ CM.preferTailFoldedLoop())
+      continue;
     for (ElementCount VF : Plan->vectorFactors()) {
       // The VPlan-based cost model is designed for computing vector cost.
-      // Querying VPlan-based cost model with a scarlar VF will cause some
+      // Querying VPlan-based cost model with a scalar VF will cause some
       // errors because we expect the VF is vector for most of the widen
       // recipes.
       if (VF.isScalar())
@@ -3417,6 +3424,61 @@ static bool hasUnsupportedHeaderPhiRecipe(VPlan &Plan) {
       });
 }
 
+/// Determine how to lower the epilogue for the vector epilogue loop.
+/// Check if there are any conflicts that prevent tail-folding the epilogue.
+/// \return CM_EpilogueNotNeededFoldTail if epilogue tail-folding is possible,
+/// otherwise CM_EpilogueAllowed.
+static EpilogueLowering
+getEpilogueTailLowering(const LoopVectorizationCostModel &MainCM, const Loop *L,
+                        OptimizationRemarkEmitter *ORE,
+                        const LoopVectorizationLegality &LVL,
+                        const LoopVectorizeHints &Hints) {
+  // Epilogue TF is only enabled when explicitly requested via command line.
+  if (!EpilogueTailFoldingPolicy.getNumOccurrences() ||
+      EpilogueTailFoldingPolicy != TailFoldingPolicyTy::PreferFoldTail)
+    return CM_EpilogueAllowed;
+
+  if (!L->isInnermost()) {
+    reportVectorizationInfo(
+        "Epilogue tail-folding is not supported for outer loop",
+        "InvalidTailFoldedEpilogue", ORE, L);
+    return CM_EpilogueAllowed;
+  }
+
+  if (!EnableEpilogueVectorization) {
+    reportVectorizationInfo(
+        "Options conflict, epilogue vectorization is disallowed while "
+        "epilogue tail-folding allowed!\n",
+        "UnsupportedEpilogueTailFoldingPolicy", ORE, L);
+    return CM_EpilogueAllowed;
+  }
+
+  if (!hasForcedEpilogueVF() || !Hints.getWidth()) {
+    reportVectorizationInfo("For now, Epilogue tail-folding can't be "
+                            "applied without forced epilogue/main loop VF\n",
+                            "UnsupportedEpilogueTailFoldingPolicy", ORE, L);
+    return CM_EpilogueAllowed;
+  }
+
+  // If scalar epilogue is explicitly required, we can't apply TF.
+  if (MainCM.requiresScalarEpilogue(/*IsVectorizing*/ true)) {
+    LLVM_DEBUG(dbgs() << "LV: Epilogue tail-folding can't be applied because "
+                         "scalar epilogue is required\n"
+                         "LV: Fall back to a normal epilogue\n");
+    return CM_EpilogueAllowed;
+  }
+
+  // If having epilogue is NOT allowed, then no epilogue to apply TF for.
+  if (!MainCM.isEpilogueAllowed()) {
+    LLVM_DEBUG(dbgs() << "LV: No epilogue to apply tail-folding for.\n"
+                         "LV: Fall back to a normal epilogue\n");
+    return CM_EpilogueAllowed;
+  }
+
+  // We can apply tail-folding on the vectorized epilogue loop.
+  return CM_EpilogueNotNeededFoldTail;
+}
+
 bool LoopVectorizationPlanner::isCandidateForEpilogueVectorization(
     VPlan &MainPlan) const {
   // Bail out if the plan contains header phi recipes not yet supported
@@ -5555,21 +5617,27 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
       // end.
       buildVPlans(*VPlan1, UserVF, UserVF, CM);
 
-      ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
-      if (EpilogueUserVF.isVector() &&
-          ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
-        CM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
-        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, CM);
-      }
-      if (!VPlans.empty() && VPlans.front()->getSingleVF() == UserVF) {
-        // For scalar VF, skip VPlan cost check as VPlan cost is designed for
-        // vector VFs only.
-        if (UserVF.isScalar() ||
-            cost(*VPlans.front(), UserVF, /*RU=*/nullptr, CM).isValid()) {
-          LLVM_DEBUG(dbgs() << "LV: Using user VF " << UserVF << ".\n");
-          LLVM_DEBUG(printPlans(dbgs()));
-          return;
+      // For scalar VF, skip VPlan cost check as VPlan cost is designed for
+      // vector VFs only.
+      if (!VPlans.empty() &&
+          (UserVF.isScalar() ||
+           cost(*VPlans.front(), UserVF, /*RU=*/nullptr, CM).isValid())) {
+        // Plan for epilogue only if we succeeded in building main loop vplan.
+
+        // Try to plan for tail-folded epilogue if it's enabled/doable,
+        // otherwise plan for unpredicated epilogue:
+        bool EpilogueTfPlanCreated = planForEpilogueTF();
+        if (!EpilogueTfPlanCreated) {
+          ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
+          if (EpilogueUserVF.isVector() &&
+              ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
+            CM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
+            buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, CM);
+          }
         }
+        LLVM_DEBUG(dbgs() << "LV: Using user VF " << UserVF << ".\n");
+        LLVM_DEBUG(printPlans(dbgs()));
+        return;
       }
       VPlans.clear();
       reportVectorizationInfo("UserVF ignored because of invalid costs.",
@@ -5597,59 +5665,81 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   LLVM_DEBUG(printPlans(dbgs()));
 }
 
-bool LoopVectorizationPlanner::planForEpilogueTF(
-    LoopVectorizationCostModel &EpilogueCM) {
-  if (VPlans.empty()) {
-    LLVM_DEBUG(dbgs() << "LV: no vplans have been built for main loop VF, bail "
-                         "out of epilogue tail-folding\n");
+bool LoopVectorizationPlanner::planForEpilogueTF() {
+  if (!EpilogueTfCM)
     return false;
-  }
+  assert(EpilogueTfCM->preferTailFoldedLoop() &&
+         "Epilogue tail-folding is expected to be enabled");
 
-  EpilogueCM.ValuesToIgnore.insert_range(CM.ValuesToIgnore);
-  EpilogueCM.VecValuesToIgnore.insert_range(CM.VecValuesToIgnore);
+  LLVM_DEBUG(dbgs() << "LV: plan for tail-folded epilogue\n");
+
+  EpilogueTfCM->ValuesToIgnore.insert_range(CM.ValuesToIgnore);
+  EpilogueTfCM->VecValuesToIgnore.insert_range(CM.VecValuesToIgnore);
 
   FixedScalableVFPair MaxFactors =
-      EpilogueCM.computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
-  if (!MaxFactors || !EpilogueCM.foldTailByMasking()) {
-    // Cases that should not to be vectorized // or tail-folded.
+      EpilogueTfCM->computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
+  if (!MaxFactors || !EpilogueTfCM->preferTailFoldedLoop() ||
+      !EpilogueTfCM->foldTailByMasking()) {
+    // Cases that should not to be vectorized or tail-folded.
     reportVectorizationInfo("This case of epilogue loop can't be tail-folded",
                             "InvalidTailFoldedEpilogue", ORE, OrigLoop);
     return false;
   }
 
-  auto VPlan1 = tryToBuildVPlan1(EpilogueCM);
+  auto VPlan1 = tryToBuildVPlan1(*EpilogueTfCM);
+
+  // If we're here, the main loop's initial VPlan was built successfully.
+  // Building one for the tail-folded loop should therefore also succeed, since
+  // nothing tail-folding-specific happens yet at this point. Still check below
+  // to catch any unexpected failure.
+  if (!VPlan1) {
+    reportVectorizationInfo(
+        "Failed to build initial tail-folded epilogue VPlan",
+        "InvalidTailFoldedEpilogue", ORE, OrigLoop);
+    return false;
+  }
 
   if (!useMaskedInterleavedAccesses(TTI)) {
     LLVM_DEBUG(
         dbgs() << "LV: Invalidate all interleaved groups due to fold-tail by "
                   "masking which requires masked-interleaved support.\n");
-    if (EpilogueCM.InterleaveInfo.invalidateGroups())
+    if (EpilogueTfCM->InterleaveInfo.invalidateGroups())
       // Invalidating interleave groups also requires invalidating all decisions
       // based on them, which includes widening decisions and uniform and scalar
       // values.
-      EpilogueCM.invalidateCostModelingDecisions();
+      EpilogueTfCM->invalidateCostModelingDecisions();
   }
   Legal->prepareToFoldTailByMasking();
 
   // Collect the instructions (and their associated costs) that will be more
   // profitable to scalarize.
-  EpilogueCM.collectNonVectorizedAndSetWideningDecisions(
+  EpilogueTfCM->collectNonVectorizedAndSetWideningDecisions(
       EpilogueVectorizationForceVF);
 
-  // Expecting only 2 vplans as epilogue TF is applied only for forced VFs.
-  assert(VPlans.size() == 2 &&
-         "For tail-folded epilogue, VPlans size is expected to be 2");
-  // Remove the last vplan, which should be the epilogue plan to replace it by
-  // the tail-folded vplan:
-  assert(VPlans.back()->getSingleVF() == EpilogueVectorizationForceVF &&
-         "For tail-folded epilogue, last vplan is expected to have "
-         "EpilogueUserVF");
-  VPlans.pop_back();
+  size_t NumPlansBefore = VPlans.size();
   buildVPlans(*VPlan1, EpilogueVectorizationForceVF,
-              EpilogueVectorizationForceVF, EpilogueCM);
+              EpilogueVectorizationForceVF, *EpilogueTfCM);
+
+  // Check that a vplan is successfully built:
+  if (VPlans.size() == NumPlansBefore ||
+      VPlans.back()->getSingleVF() != EpilogueVectorizationForceVF ||
+      !VPlans.back()->hasTailFolded()) {
+    reportVectorizationInfo(
+        "Failed to build a valid tail-folded epilogue VPlan",
+        "InvalidTailFoldedEpilogue", ORE, OrigLoop);
+    return false;
+  }
 
-  cost(*VPlans.back(), EpilogueVectorizationForceVF, /*RU=*/nullptr,
-       EpilogueCM);
+  if (!cost(*VPlans.back(), EpilogueVectorizationForceVF, /*RU=*/nullptr,
+            *EpilogueTfCM)
+           .isValid()) {
+    VPlans.pop_back();
+    reportVectorizationInfo("This case of epilogue loop can't be tail-folded "
+                            "- Invalid costs",
+                            "InvalidTailFoldedEpilogue", ORE, OrigLoop);
+    return false;
+  }
+  LLVM_DEBUG(dbgs() << "LV: Tail-folded epilogue VPlan is created\n");
   return true;
 }
 
@@ -5996,6 +6086,22 @@ LoopVectorizationPlanner::computeBestVF() {
   return {BestFactor, &BestPlan};
 }
 
+LoopVectorizationPlanner::LoopVectorizationPlanner(
+    Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
+    const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
+    LoopVectorizationCostModel &CM,
+    std::unique_ptr<LoopVectorizationCostModel> EpilogueTfCM,
+    VFSelectionContext &Config, InterleavedAccessInfo &IAI,
+    PredicatedScalarEvolution &PSE, const LoopVectorizeHints &Hints,
+    OptimizationRemarkEmitter *ORE)
+    : OrigLoop(L), LI(LI), DT(DT), TLI(TLI), TTI(TTI), Legal(Legal), CM(CM),
+      EpilogueTfCM(std::move(EpilogueTfCM)), Config(Config), IAI(IAI), PSE(PSE),
+      Hints(Hints), ORE(ORE) {}
+
+LoopVectorizationPlanner::~LoopVectorizationPlanner() = default;
+
+void LoopVectorizationPlanner::clearEpilogueTfCM() { EpilogueTfCM.reset(); }
+
 DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
     ElementCount BestVF, unsigned BestUF, VPlan &BestVPlan,
     InnerLoopVectorizer &ILV, DominatorTree *DT,
@@ -6024,8 +6130,9 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
                    BestVPlan, BestVF, VScale);
   }
 
+  const bool IsTailFolded = BestVPlan.hasTailFolded();
   if (CM.maskPartialAliasing()) {
-    assert(BestVPlan.hasTailFolded() && "Expected tail folding to be enabled");
+    assert(IsTailFolded && "Expected tail folding to be enabled");
     RUN_VPLAN_PASS(VPlanTransforms::materializeAliasMaskCheckBlock, BestVPlan,
                    *Legal->getRuntimePointerChecking()->getDiffChecks(),
                    HasBranchWeights);
@@ -6065,7 +6172,6 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
   RUN_VPLAN_PASS(VPlanTransforms::convertEVLExitCond, BestVPlan);
   // Regions are dissolved after optimizing for VF and UF, which completely
   // removes unneeded loop regions first.
-  const bool HasTailFolded = BestVPlan.hasTailFolded();
   RUN_VPLAN_PASS(VPlanTransforms::dissolveLoopRegions, BestVPlan);
   // Expand BranchOnTwoConds after dissolution, when latch has direct access to
   // its successors.
@@ -6083,7 +6189,7 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
   assert((LI->getUniqueLatchExitBlock(*OrigLoop) || RequiresScalarEpilogue) &&
          "loops not exiting via the latch without required epilogue?");
   VPlanTransforms::materializeVectorTripCount(
-      BestVPlan, VectorPH, HasTailFolded, RequiresScalarEpilogue,
+      BestVPlan, VectorPH, IsTailFolded, RequiresScalarEpilogue,
       &BestVPlan.getVFxUF(), MaxRuntimeStep);
   VPlanTransforms::materializeFactors(BestVPlan, VectorPH, BestVF);
   // Limit expansions to VPInstruction to when not vectorizing the epilogue.
@@ -7290,61 +7396,6 @@ getEpilogueLowering(Function *F, Loop *L, LoopVectorizeHints &Hints,
   return CM_EpilogueAllowed;
 }
 
-/// Determine how to lower the epilogue for the vector epilogue loop.
-/// Check if there are any conflicts that prevent tail-folding the epilogue.
-/// \return CM_EpilogueNotNeededFoldTail if epilogue tail-folding is possible,
-/// otherwise CM_EpilogueAllowed.
-static EpilogueLowering
-getEpilogueTailLowering(const LoopVectorizationCostModel &MainCM, const Loop *L,
-                        OptimizationRemarkEmitter *ORE,
-                        LoopVectorizationLegality &LVL,
-                        LoopVectorizeHints &Hints) {
-  // Epilogue TF is only enabled when explicitly requested via command line.
-  if (!EpilogueTailFoldingPolicy.getNumOccurrences() ||
-      EpilogueTailFoldingPolicy != TailFoldingPolicyTy::PreferFoldTail)
-    return CM_EpilogueAllowed;
-
-  if (!L->isInnermost()) {
-    reportVectorizationInfo(
-        "Epilogue tail-folding is not supported for outer loop",
-        "InvalidTailFoldedEpilogue", ORE, L);
-    return CM_EpilogueAllowed;
-  }
-
-  if (!EnableEpilogueVectorization) {
-    reportVectorizationInfo(
-        "Options conflict, epilogue vectorization is disallowed while "
-        "epilogue tail-folding allowed!\n",
-        "UnsupportedEpilogueTailFoldingPolicy", ORE, L);
-    return CM_EpilogueAllowed;
-  }
-
-  if (!hasForcedEpilogueVF() || !Hints.getWidth()) {
-    reportVectorizationInfo("For now, Epilogue tail-folding can't be "
-                            "applied without forced epilogue/main loop VF\n",
-                            "UnsupportedEpilogueTailFoldingPolicy", ORE, L);
-    return CM_EpilogueAllowed;
-  }
-
-  // If scalar epilogue is explicitly required, we can't apply TF.
-  if (MainCM.requiresScalarEpilogue(/*IsVectorizing*/ true)) {
-    LLVM_DEBUG(dbgs() << "LV: Epilogue tail-folding can't be applied because "
-                         "scalar epilogue is required\n"
-                         "LV: Fall back to a normal epilogue\n");
-    return CM_EpilogueAllowed;
-  }
-
-  // If having epilogue is NOT allowed, then no epilogue to apply TF for.
-  if (!MainCM.isEpilogueAllowed()) {
-    LLVM_DEBUG(dbgs() << "LV: No epilogue to apply tail-folding for.\n"
-                         "LV: Fall back to a normal epilogue\n");
-    return CM_EpilogueAllowed;
-  }
-
-  // We can apply tail-folding on the vectorized epilogue loop.
-  return CM_EpilogueNotNeededFoldTail;
-}
-
 // Emit a remark if there are stores to floats that required a floating point
 // extension. If the vectorized loop was generated with floating point there
 // will be a performance penalty from the conversion overhead and the change in
@@ -7916,7 +7967,7 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
                                       GeneratedRTChecks &Checks,
                                       ArrayRef<Instruction *> InstsToMove,
                                       ArrayRef<VPInstruction *> ResumeValues,
-                                      bool IsEpilogueTFEnabled) {
+                                      bool IsEpilogueTfEnabled) {
   BasicBlock *VecEpilogueIterationCountCheck =
       cast<VPIRBasicBlock>(EpiPlan.getEntry())->getIRBasicBlock();
   BasicBlock *VecEpiloguePreHeader =
@@ -7946,7 +7997,7 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
   // to, even a trip count too small for the epilogue VF is handled safely by
   // the masked epilogue vector loop, so skip straight to its preheader.
   RedirectEdge(EPI.EpilogueIterationCountCheck,
-               IsEpilogueTFEnabled ? VecEpiloguePreHeader : ScalarPH);
+               IsEpilogueTfEnabled ? VecEpiloguePreHeader : ScalarPH);
 
   // Adjust the terminators of runtime check blocks and phis using them.
   BasicBlock *SCEVCheckBlock = Checks.getSCEVChecks().second;
@@ -7983,12 +8034,12 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
     }
     // When the epilogue is tail-folded, EpilogueIterationCountCheck
     // (iter.check) is redirected to branch straight into the vector epilogue
-    // preheader (see the IsEpilogueTFEnabled redirect above), so it is now a
+    // preheader (see the IsEpilogueTfEnabled redirect above), so it is now a
     // predecessor and its incoming value must be kept rather than stripped.
     // TODO: revisit for reduction phis, whose resume value on this bypass
     // edge may need dedicated handling rather than reusing the value already
     // present here.
-    if (!IsEpilogueTFEnabled)
+    if (!IsEpilogueTfEnabled)
       Phi->removeIncomingValue(EPI.EpilogueIterationCountCheck);
   }
 
@@ -8008,7 +8059,7 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
     if (Phi.use_empty())
       Phi.eraseFromParent();
 
-  if (IsEpilogueTFEnabled) {
+  if (IsEpilogueTfEnabled) {
     // The epilogue vector loop is tail-folded, so it can safely handle
     // any remaining iterations, including zero, via masking.
     // vec.epilog.iter.check's own min-iters check was therefore built with a
@@ -8225,25 +8276,32 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                             OptForSize);
   LoopVectorizationCostModel CM(SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE,
                                 GetBFI, F, &Hints, IAI, Config);
-  // Use the planner for vectorization.
-  LoopVectorizationPlanner LVP(L, LI, DT, TLI, *TTI, &LVL, CM, Config, IAI, PSE,
-                               Hints, ORE);
 
+  // Setup the epilogue tail-folding CM. Only built when tail-folding the
+  // epilogue is actually a candidate, to avoid the cost of an extra
+  // InterleavedAccessInfo scan and LoopVectorizationCostModel construction
+  // for the common case where this (experimental, off-by-default) feature
+  // isn't in use.
   EpilogueLowering EpilogueTailLoweringStatus =
       getEpilogueTailLowering(CM, L, ORE, LVL, Hints);
-  std::optional<InterleavedAccessInfo> TailFoldingCMIAI;
-  std::optional<LoopVectorizationCostModel> EpilogueTailFoldingCM;
+  std::optional<InterleavedAccessInfo> EpilogueTfCMIAI;
+  std::unique_ptr<LoopVectorizationCostModel> EpilogueTfCM;
   if (EpilogueTailLoweringStatus ==
       EpilogueLowering::CM_EpilogueNotNeededFoldTail) {
     LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is enabled\n");
-    TailFoldingCMIAI.emplace(PSE, L, DT, LI, LVL.getLAI(), OptForSize);
+    EpilogueTfCMIAI.emplace(PSE, L, DT, LI, LVL.getLAI(), OptForSize);
     if (UseInterleaved)
-      TailFoldingCMIAI->analyzeInterleaving(useMaskedInterleavedAccesses(*TTI));
-    EpilogueTailFoldingCM.emplace(CM_EpilogueNotNeededFoldTail, L, PSE, LI,
-                                  &LVL, *TTI, TLI, AC, ORE, GetBFI, F, &Hints,
-                                  *TailFoldingCMIAI, Config);
+      EpilogueTfCMIAI->analyzeInterleaving(useMaskedInterleavedAccesses(*TTI));
+    EpilogueTfCM = std::make_unique<LoopVectorizationCostModel>(
+        EpilogueTailLoweringStatus, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE,
+        GetBFI, F, &Hints, *EpilogueTfCMIAI, Config);
   }
 
+  // Use the planner for vectorization.
+  LoopVectorizationPlanner LVP(L, LI, DT, TLI, *TTI, &LVL, CM,
+                               std::move(EpilogueTfCM), Config, IAI, PSE, Hints,
+                               ORE);
+
   // Get user vectorization factor and interleave count.
   ElementCount UserVF = Hints.getWidth();
   unsigned UserIC = Hints.getInterleave();
@@ -8256,15 +8314,9 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   // Plan how to best vectorize.
   LVP.plan(UserVF, UserIC);
-  if (EpilogueTailFoldingCM) {
-    if (!LVP.planForEpilogueTF(EpilogueTailFoldingCM.value())) {
-      // we can't apply epilogue TF:
-      reportVectorizationInfo(
-          "Applying epilogue tail-folding failed, disable it.",
-          "InvalidTailFoldedEpilogue", ORE, L);
-      EpilogueTailFoldingCM.reset();
-    }
-  }
+  // Right now, after planning, the epilogue tail-folding CM is not needed
+  // anymore. Clear it.
+  LVP.clearEpilogueTfCM();
 
   auto [VF, BestPlanPtr] = LVP.computeBestVF();
   unsigned IC = 1;
@@ -8532,11 +8584,13 @@ bool LoopVectorizePass::processLoop(Loop *L) {
         BestMainPlan, BestEpiPlan, L, ExpandedSCEVs, EPI, LVP, Config,
         *PSE.getSE(), ResumeValues);
     LVP.attachRuntimeChecks(BestEpiPlan, Checks, HasBranchWeights);
+    // Save the status of epilogue tail-folding:
+    const bool IsTailFolded = BestEpiPlan.hasTailFolded();
     LVP.executePlan(
         EPI.EpilogueVF, EPI.EpilogueUF, BestEpiPlan, EpilogILV, DT,
         LoopVectorizationPlanner::EpilogueVectorizationKind::Epilogue);
     connectEpilogueVectorLoop(BestEpiPlan, L, EPI, DT, LI, Checks, InstsToMove,
-                              ResumeValues, EpilogueTailFoldingCM.has_value());
+                              ResumeValues, IsTailFolded);
     ++LoopsEpilogueVectorized;
   } else {
     InnerLoopVectorizer LB(L, PSE, LI, DT, TTI, AC, VF.Width, IC, Checks,
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index 79495c4ad9c3e..ffde0d0084933 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -1,7 +1,10 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; REQUIRES: asserts
 ; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug-only=loop-vectorize -mcpu=neoverse-v1 -S %s | FileCheck %s
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug-only=loop-vectorize -mattr=+sve -S %s | FileCheck %s
+
+; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -force-vector-width="vscale x 16" -epilogue-vectorization-force-VF="vscale x 8" -debug-only=loop-vectorize -mattr=+sve -S %s | FileCheck %s --check-prefix=CHECK-VS
 
 ; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize,vectorutils --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
 ; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALIDATE-INTERLEAVE
@@ -31,7 +34,7 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-NEXT:    store <16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP2]], align 4
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
@@ -52,12 +55,68 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT5]], i64 [[N]])
 ; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP6:%.*]] = xor i1 [[TMP5]], true
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    br label %[[EXIT]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
+; CHECK-VS-LABEL: define void @test_epilogue_tf(
+; CHECK-VS-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i32 [[VAL:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 4
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP3]], 1
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP4]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[TMP5]], i64 [[TMP3]]
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP5]], align 4
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP6]], align 4
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = shl nuw i64 [[TMP8]], 3
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT3:%.*]] = shufflevector <vscale x 8 x i32> [[BROADCAST_SPLATINSERT2]], <vscale x 8 x i32> poison, <vscale x 8 x i32> zeroinitializer
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT5:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX4]]
+; CHECK-VS-NEXT:    call void @llvm.masked.store.nxv8i32.p0(<vscale x 8 x i32> [[BROADCAST_SPLAT3]], ptr align 4 [[TMP10]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-VS-NEXT:    [[INDEX_NEXT5]] = add i64 [[INDEX4]], [[TMP9]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT5]], i64 [[N]])
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = xor i1 [[TMP11]], true
+; CHECK-VS-NEXT:    br i1 [[TMP12]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[EXIT]]:
+; CHECK-VS-NEXT:    ret void
+;
 entry:
   br label %for.body
 
@@ -99,7 +158,7 @@ define i32 @add_redc(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[TMP5]] = add <16 x i32> [[WIDE_LOAD3]], [[VEC_PHI2]]
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[BIN_RDX:%.*]] = add <16 x i32> [[TMP5]], [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.add.v16i32(<16 x i32> [[BIN_RDX]])
@@ -125,7 +184,7 @@ define i32 @add_redc(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
 ; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP13:%.*]] = xor i1 [[TMP12]], true
-; CHECK-NEXT:    br i1 [[TMP13]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[TMP14:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP11]])
 ; CHECK-NEXT:    br label %[[EXIT]]
@@ -133,6 +192,72 @@ define i32 @add_redc(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[TMP14]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
 ; CHECK-NEXT:    ret i32 [[ADD_LCSSA]]
 ;
+; CHECK-VS-LABEL: define i32 @add_redc(
+; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], [[TMP3]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP1]], 4
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP5]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 [[TMP4]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP6]], align 1
+; CHECK-VS-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP7]], align 1
+; CHECK-VS-NEXT:    [[TMP8]] = add <vscale x 16 x i32> [[WIDE_LOAD]], [[VEC_PHI]]
+; CHECK-VS-NEXT:    [[TMP9]] = add <vscale x 16 x i32> [[WIDE_LOAD3]], [[VEC_PHI2]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP5]]
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[BIN_RDX:%.*]] = add <vscale x 16 x i32> [[TMP9]], [[TMP8]]
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i32 @llvm.vector.reduce.add.nxv16i32(<vscale x 16 x i32> [[BIN_RDX]])
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = shl nuw i64 [[TMP12]], 3
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = insertelement <vscale x 8 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI5:%.*]] = phi <vscale x 8 x i32> [ [[TMP14]], %[[VEC_EPILOG_PH]] ], [ [[TMP17:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
+; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 1 [[TMP15]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
+; CHECK-VS-NEXT:    [[TMP16:%.*]] = add <vscale x 8 x i32> [[WIDE_MASKED_LOAD]], [[VEC_PHI5]]
+; CHECK-VS-NEXT:    [[TMP17]] = select <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> [[TMP16]], <vscale x 8 x i32> [[VEC_PHI5]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], [[TMP13]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
+; CHECK-VS-NEXT:    [[TMP18:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP19:%.*]] = xor i1 [[TMP18]], true
+; CHECK-VS-NEXT:    br i1 [[TMP19]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP20:%.*]] = call i32 @llvm.vector.reduce.add.nxv8i32(<vscale x 8 x i32> [[TMP17]])
+; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[EXIT]]:
+; CHECK-VS-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[TMP20]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    ret i32 [[ADD_LCSSA]]
+;
 entry:
   br label %loop
 
@@ -176,7 +301,7 @@ define i32 @max_redc(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[TMP5]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[WIDE_LOAD3]], <16 x i32> [[VEC_PHI2]])
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[RDX_MINMAX:%.*]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[TMP4]], <16 x i32> [[TMP5]])
 ; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.umax.v16i32(<16 x i32> [[RDX_MINMAX]])
@@ -203,7 +328,7 @@ define i32 @max_redc(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
 ; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP12:%.*]] = xor i1 [[TMP11]], true
-; CHECK-NEXT:    br i1 [[TMP12]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP12]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[TMP13:%.*]] = call i32 @llvm.vector.reduce.umax.v8i32(<8 x i32> [[TMP10]])
 ; CHECK-NEXT:    br label %[[EXIT]]
@@ -211,6 +336,73 @@ define i32 @max_redc(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[TMP13]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
 ; CHECK-NEXT:    ret i32 [[MAX_LCSSA]]
 ;
+; CHECK-VS-LABEL: define i32 @max_redc(
+; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], [[TMP3]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP1]], 4
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP5]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 [[TMP4]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP6]], align 1
+; CHECK-VS-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP7]], align 1
+; CHECK-VS-NEXT:    [[TMP8]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD]], <vscale x 16 x i32> [[VEC_PHI]])
+; CHECK-VS-NEXT:    [[TMP9]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], <vscale x 16 x i32> [[VEC_PHI2]])
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP5]]
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[RDX_MINMAX:%.*]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[TMP8]], <vscale x 16 x i32> [[TMP9]])
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i32 @llvm.vector.reduce.umax.nxv16i32(<vscale x 16 x i32> [[RDX_MINMAX]])
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = shl nuw i64 [[TMP12]], 3
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 8 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 8 x i32> poison, <vscale x 8 x i32> zeroinitializer
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI5:%.*]] = phi <vscale x 8 x i32> [ [[BROADCAST_SPLAT]], %[[VEC_EPILOG_PH]] ], [ [[TMP16:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
+; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 1 [[TMP14]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = call <vscale x 8 x i32> @llvm.umax.nxv8i32(<vscale x 8 x i32> [[WIDE_MASKED_LOAD]], <vscale x 8 x i32> [[VEC_PHI5]])
+; CHECK-VS-NEXT:    [[TMP16]] = select <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> [[TMP15]], <vscale x 8 x i32> [[VEC_PHI5]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], [[TMP13]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
+; CHECK-VS-NEXT:    [[TMP17:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP18:%.*]] = xor i1 [[TMP17]], true
+; CHECK-VS-NEXT:    br i1 [[TMP18]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP19:%.*]] = call i32 @llvm.vector.reduce.umax.nxv8i32(<vscale x 8 x i32> [[TMP16]])
+; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[EXIT]]:
+; CHECK-VS-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[TMP19]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    ret i32 [[MAX_LCSSA]]
+;
 entry:
   br label %loop
 
@@ -248,7 +440,7 @@ define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 4
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <16 x i32> [[WIDE_LOAD]], i64 15
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
@@ -268,7 +460,7 @@ define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
 ; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP7:%.*]] = xor i1 [[TMP6]], true
-; CHECK-NEXT:    br i1 [[TMP7]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[TMP8:%.*]] = xor <8 x i1> [[ACTIVE_LANE_MASK]], splat (i1 true)
 ; CHECK-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v8i1(<8 x i1> [[TMP8]], i1 false)
@@ -279,6 +471,66 @@ define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[LOAD_LCSSA:%.*]] = phi i32 [ [[TMP9]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP4]], %[[MIDDLE_BLOCK]] ]
 ; CHECK-NEXT:    ret i32 [[LOAD_LCSSA]]
 ;
+; CHECK-VS-LABEL: define i32 @live-out(
+; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 4
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP3]], 1
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP4]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP5]], i64 [[TMP3]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP6]], align 4
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = call i32 @llvm.vscale.i32()
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = mul nuw i32 [[TMP8]], 16
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = sub i32 [[TMP9]], 1
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = extractelement <vscale x 16 x i32> [[WIDE_LOAD]], i32 [[TMP10]]
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = shl nuw i64 [[TMP12]], 3
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX2]]
+; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 4 [[TMP14]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
+; CHECK-VS-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], [[TMP13]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP16:%.*]] = xor i1 [[TMP15]], true
+; CHECK-VS-NEXT:    br i1 [[TMP16]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP17:%.*]] = xor <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], splat (i1 true)
+; CHECK-VS-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP17]], i1 false)
+; CHECK-VS-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
+; CHECK-VS-NEXT:    [[TMP18:%.*]] = extractelement <vscale x 8 x i32> [[WIDE_MASKED_LOAD]], i64 [[LAST_ACTIVE_LANE]]
+; CHECK-VS-NEXT:    br label %[[FOR_END]]
+; CHECK-VS:       [[FOR_END]]:
+; CHECK-VS-NEXT:    [[LOAD_LCSSA:%.*]] = phi i32 [ [[TMP18]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    ret i32 [[LOAD_LCSSA]]
+;
 entry:
   br label %loop
 
@@ -333,7 +585,7 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-NEXT:    store <16 x i32> [[REVERSE]], ptr [[TMP12]], align 4
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 32
 ; CHECK-NEXT:    [[TMP13:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[TMP2]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
@@ -358,7 +610,7 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[INDEX_NEXT8]], i32 [[TMP2]])
 ; CHECK-NEXT:    [[TMP17:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP18:%.*]] = xor i1 [[TMP17]], true
-; CHECK-NEXT:    br i1 [[TMP18]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP18]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    br label %[[EXIT]]
 ; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
@@ -369,10 +621,102 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-NEXT:    store i32 [[VAL]], ptr [[ARRAYIDX]], align 4
 ; CHECK-NEXT:    [[IV_NEXT]] = sub nuw nsw i32 [[IV]], 1
 ; CHECK-NEXT:    [[EXITCOND:%.*]] = icmp sge i32 [[IV_NEXT]], 0
-; CHECK-NEXT:    br i1 [[EXITCOND]], label %[[FOR_BODY]], label %[[EXIT]], !llvm.loop [[LOOP14:![0-9]+]]
+; CHECK-NEXT:    br i1 [[EXITCOND]], label %[[FOR_BODY]], label %[[EXIT]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
+; CHECK-VS-LABEL: define void @reversed-loop(
+; CHECK-VS-SAME: ptr [[A:%.*]], i32 [[N:%.*]], i32 [[VAL:%.*]]) #[[ATTR0]] {
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:    [[ST:%.*]] = sub i32 [[N]], 1
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i32 [[N]], -1
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = add i32 [[N]], -2
+; CHECK-VS-NEXT:    [[SMIN1:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP1]], i32 -1)
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = sub i32 [[TMP0]], [[SMIN1]]
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = call i32 @llvm.vscale.i32()
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i32 [[TMP3]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], [[TMP4]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK-VS:       [[VECTOR_SCEVCHECK]]:
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = add i32 [[N]], -2
+; CHECK-VS-NEXT:    [[SMIN:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP5]], i32 -1)
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = sub i32 [[TMP5]], [[SMIN]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = sub i32 [[ST]], [[TMP6]]
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = icmp sgt i32 [[TMP7]], [[ST]]
+; CHECK-VS-NEXT:    br i1 [[TMP8]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = shl nuw i32 [[TMP3]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK2:%.*]] = icmp ult i32 [[TMP2]], [[TMP9]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK2]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = shl nuw i32 [[TMP3]], 4
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = shl nuw i32 [[TMP10]], 1
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i32 [[TMP2]], [[TMP11]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i32 [[TMP2]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = sub i32 [[ST]], [[N_VEC]]
+; CHECK-VS-NEXT:    [[REVERSE:%.*]] = call <vscale x 16 x i32> @llvm.vector.reverse.nxv16i32(<vscale x 16 x i32> [[BROADCAST_SPLAT]])
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = sub i32 [[ST]], [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP13]]
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = zext i32 [[TMP10]] to i64
+; CHECK-VS-NEXT:    [[TMP16:%.*]] = sub nuw nsw i64 [[TMP15]], 1
+; CHECK-VS-NEXT:    [[TMP17:%.*]] = sub i64 0, [[TMP16]]
+; CHECK-VS-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i32, ptr [[TMP14]], i64 [[TMP17]]
+; CHECK-VS-NEXT:    [[TMP19:%.*]] = sub i64 [[TMP17]], [[TMP15]]
+; CHECK-VS-NEXT:    [[TMP20:%.*]] = getelementptr inbounds i32, ptr [[TMP14]], i64 [[TMP19]]
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[REVERSE]], ptr [[TMP18]], align 4
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[REVERSE]], ptr [[TMP20]], align 4
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], [[TMP11]]
+; CHECK-VS-NEXT:    [[TMP21:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP21]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[TMP2]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP22:%.*]] = call i32 @llvm.vscale.i32()
+; CHECK-VS-NEXT:    [[TMP23:%.*]] = shl nuw i32 [[TMP22]], 3
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT4:%.*]] = shufflevector <vscale x 8 x i32> [[BROADCAST_SPLATINSERT3]], <vscale x 8 x i32> poison, <vscale x 8 x i32> zeroinitializer
+; CHECK-VS-NEXT:    [[REVERSE5:%.*]] = call <vscale x 8 x i32> @llvm.vector.reverse.nxv8i32(<vscale x 8 x i32> [[BROADCAST_SPLAT4]])
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i32(i32 [[VEC_EPILOG_RESUME_VAL]], i32 [[TMP2]])
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX6:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT8:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP24:%.*]] = sub i32 [[ST]], [[INDEX6]]
+; CHECK-VS-NEXT:    [[TMP25:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP24]]
+; CHECK-VS-NEXT:    [[TMP26:%.*]] = zext i32 [[TMP23]] to i64
+; CHECK-VS-NEXT:    [[TMP27:%.*]] = sub nuw nsw i64 [[TMP26]], 1
+; CHECK-VS-NEXT:    [[TMP28:%.*]] = sub i64 0, [[TMP27]]
+; CHECK-VS-NEXT:    [[TMP29:%.*]] = getelementptr i32, ptr [[TMP25]], i64 [[TMP28]]
+; CHECK-VS-NEXT:    [[REVERSE7:%.*]] = call <vscale x 8 x i1> @llvm.vector.reverse.nxv8i1(<vscale x 8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-VS-NEXT:    call void @llvm.masked.store.nxv8i32.p0(<vscale x 8 x i32> [[REVERSE5]], ptr align 4 [[TMP29]], <vscale x 8 x i1> [[REVERSE7]])
+; CHECK-VS-NEXT:    [[INDEX_NEXT8]] = add i32 [[INDEX6]], [[TMP23]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i32(i32 [[INDEX_NEXT8]], i32 [[TMP2]])
+; CHECK-VS-NEXT:    [[TMP30:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP31:%.*]] = xor i1 [[TMP30]], true
+; CHECK-VS-NEXT:    br i1 [[TMP31]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-VS-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK-VS:       [[FOR_BODY]]:
+; CHECK-VS-NEXT:    [[IV:%.*]] = phi i32 [ [[ST]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[IV]]
+; CHECK-VS-NEXT:    store i32 [[VAL]], ptr [[ARRAYIDX]], align 4
+; CHECK-VS-NEXT:    [[IV_NEXT]] = sub nuw nsw i32 [[IV]], 1
+; CHECK-VS-NEXT:    [[EXITCOND:%.*]] = icmp sge i32 [[IV_NEXT]], 0
+; CHECK-VS-NEXT:    br i1 [[EXITCOND]], label %[[FOR_BODY]], label %[[EXIT]]
+; CHECK-VS:       [[EXIT]]:
+; CHECK-VS-NEXT:    ret void
+;
 entry:
   %st = sub i32 %n, 1
   br label %for.body
@@ -389,72 +733,6 @@ exit:
   ret void
 }
 
-define void @math_func(ptr %A, i32 %n) {
-; CHECK-LABEL: define void @math_func(
-; CHECK-SAME: ptr [[A:%.*]], i32 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-NEXT:    [[UMAX:%.*]] = call i32 @llvm.umax.i32(i32 [[N]], i32 1)
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[UMAX]], 8
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i32 [[UMAX]], 16
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP0:%.*]] = and i32 [[UMAX]], 15
-; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[UMAX]], [[TMP0]]
-; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds float, ptr [[A]], i32 [[INDEX]]
-; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x float>, ptr [[TMP1]], align 4
-; CHECK-NEXT:    [[TMP2:%.*]] = call <16 x float> @llvm.pow.v16f32(<16 x float> [[WIDE_LOAD]], <16 x float> splat (float 2.000000e+00))
-; CHECK-NEXT:    store <16 x float> [[TMP2]], ptr [[TMP1]], align 4
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 16
-; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
-; CHECK:       [[MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[UMAX]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
-; CHECK:       [[VEC_EPILOG_PH]]:
-; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[VEC_EPILOG_RESUME_VAL]], i32 [[UMAX]])
-; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX2:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds float, ptr [[A]], i32 [[INDEX2]]
-; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x float> @llvm.masked.load.v8f32.p0(ptr align 4 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x float> poison)
-; CHECK-NEXT:    [[TMP5:%.*]] = call <8 x float> @llvm.pow.v8f32(<8 x float> [[WIDE_MASKED_LOAD]], <8 x float> splat (float 2.000000e+00))
-; CHECK-NEXT:    call void @llvm.masked.store.v8f32.p0(<8 x float> [[TMP5]], ptr align 4 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-NEXT:    [[INDEX_NEXT3]] = add i32 [[INDEX2]], 8
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[INDEX_NEXT3]], i32 [[UMAX]])
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-NEXT:    [[TMP7:%.*]] = xor i1 [[TMP6]], true
-; CHECK-NEXT:    br i1 [[TMP7]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP16:![0-9]+]]
-; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-NEXT:    br label %[[EXIT]]
-; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    ret void
-;
-entry:
-  br label %for.body
-
-for.body:
-  %iv = phi i32 [ 0, %entry ], [ %iv.next, %for.body ]
-  %arrayidx = getelementptr inbounds float, ptr %A, i32 %iv
-  %load = load float, ptr %arrayidx, align 4
-  %val = call float @llvm.pow.f32(float %load, float 2.0)
-  store float %val, ptr %arrayidx, align 4
-  %iv.next = add nuw nsw i32 %iv, 1
-  %exitcond = icmp ult i32 %iv.next, %n
-  br i1 %exitcond, label %for.body, label %exit
-
-exit:
-  ret void
-}
-
 define i64 @test_no_masked_interleave_support(i64 %y, i32 %n) {
 ; CHECK-INVALIDATE-INTERLEAVE-LABEL: Checking a loop in 'test_no_masked_interleave_support'
 ; CHECK-INVALIDATE-INTERLEAVE: LV: epilogue tail-folding is enabled
@@ -481,3 +759,4 @@ cond.end:
 for.cond.cleanup:
   ret i64 %cond
 }
+
diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index 2d02f4006fcc1..42c48d4b8db8b 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -1,15 +1,68 @@
 ; REQUIRES: asserts
-; RUN: opt -S < %s -p loop-vectorize -debug-only=loop-vectorize --disable-output \
-; RUN: -epilogue-tail-folding-policy=prefer-fold-tail -pass-remarks-analysis=loop-vectorize 2>&1 | FileCheck %s
 
-; RUN: opt -S < %s -p loop-vectorize -debug-only=loop-vectorize -enable-epilogue-vectorization=false \
-; RUN: --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -pass-remarks-analysis=loop-vectorize 2>&1 \
-; RUN: | FileCheck %s --check-prefix=CHECK-DISABLED-EPILOG
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -pass-remarks-analysis=loop-vectorize -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s
+
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -force-vector-width=16 \
+; RUN:  -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s --check-prefix=CHECK-NO-FORCED-MAIN-VF
+
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize --disable-output -epilogue-tail-folding-policy=prefer-fold-tail -epilogue-vectorization-force-VF=8  \
+; RUN: -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s --check-prefix=CHECK-NO-FORCED-EPILOGUE-VF
+
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize -enable-epilogue-vectorization=false \
+; RUN: --disable-output -force-vector-width=16 -epilogue-vectorization-force-VF=8 -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s --check-prefix=CHECK-DISABLED-EPILOG
+
 
 define void @test_epilogue_tf(ptr %A, i64 %n) {
-; CHECK-LABEL: LV: Checking a loop in 'test_epilogue_tf'
-; CHECK: LV: epilogue tail-folding is not supported yet
-; CHECK: remark: <unknown>:0:0: The epilogue-tail-folding policy prefer-fold-tail is not supported yet, fall back to a normal epilogue
+; CHECK-LABEL: Checking a loop in 'test_epilogue_tf'
+; CHECK: LV: epilogue tail-folding is enabled
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+  %arrayidx = getelementptr inbounds i8, ptr %A, i64 %iv
+  store i8 1, ptr %arrayidx, align 1
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exitcond = icmp ne i64 %iv.next, %n
+  br i1 %exitcond, label %for.body, label %exit
+
+exit:
+  ret void
+}
+
+; This case can't be tail-folded because all the iterations will be executed by
+; main vector loop.
+define void @test_no_iterations_left(ptr %A) {
+; CHECK-LABEL: Checking a loop in 'test_no_iterations_left'
+; CHECK: LV: epilogue tail-folding is enabled
+; CHECK: remark: <unknown>:0:0: This case of epilogue loop can't be tail-folded
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+  %arrayidx = getelementptr inbounds i8, ptr %A, i64 %iv
+  store i8 1, ptr %arrayidx, align 1
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exitcond = icmp ne i64 %iv.next, 64
+  br i1 %exitcond, label %for.body, label %exit
+
+exit:
+  ret void
+}
+
+define void @test_no_vf(ptr %A, i64 %n) {
+; CHECK-NO-FORCED-MAIN-VF-LABEL: Checking a loop in 'test_no_vf'
+; CHECK-NO-FORCED-MAIN-VF: remark: <unknown>:0:0: For now, Epilogue tail-folding can't be applied without forced epilogue/main loop VF
+; CHECK-NO-FORCED-MAIN-VF-NOT: LV: epilogue tail-folding is enabled
+
+; CHECK-NO-FORCED-EPILOGUE-VF-LABEL: Checking a loop in 'test_no_vf'
+; CHECK-NO-FORCED-EPILOGUE-VF: remark: <unknown>:0:0: For now, Epilogue tail-folding can't be applied without forced epilogue/main loop VF
+; CHECK-NO-FORCED-EPILOGUE-VF-NOT: LV: epilogue tail-folding is enabled
 ;
 entry:
   br label %for.body
@@ -27,8 +80,9 @@ exit:
 }
 
 define void @epilogue_is_disabled(ptr %a, i64 %n) {
-; CHECK-DISABLED-EPILOG-LABEL: LV: Checking a loop in 'epilogue_is_disabled'
+; CHECK-DISABLED-EPILOG-LABEL: Checking a loop in 'epilogue_is_disabled'
 ; CHECK-DISABLED-EPILOG: remark: <unknown>:0:0: Options conflict, epilogue vectorization is disallowed while epilogue tail-folding allowed!
+; CHECK-DISABLED-EPILOG-NOT: LV: epilogue tail-folding is enabled
 ;
 entry:
   br label %for.body
@@ -46,9 +100,10 @@ for.end:
 }
 
 define i16 @require_scalar_epilogue(ptr %dst, i64 %x) {
-; CHECK-LABEL: LV: Checking a loop in 'require_scalar_epilogue'
+; CHECK-LABEL: Checking a loop in 'require_scalar_epilogue'
 ; CHECK: LV: Epilogue tail-folding can't be applied because scalar epilogue is required
 ; CHECK-NEXT: LV: Fall back to a normal epilogue
+; CHECK-NOT: LV: epilogue tail-folding is enabled
 ;
 entry:
   br label %loop.header
@@ -75,9 +130,10 @@ exit.2:
 }
 
 define i32 @opt_for_size(ptr %p, i32 %n) optsize {
-; CHECK-LABEL: LV: Checking a loop in 'opt_for_size'
+; CHECK-LABEL: Checking a loop in 'opt_for_size'
 ; CHECK: LV: No epilogue to apply tail-folding for.
 ; CHECK-NEXT: LV: Fall back to a normal epilogue
+; CHECK-NOT: LV: epilogue tail-folding is enabled
 ;
 entry:
   br label %for.body
@@ -95,4 +151,4 @@ for.body:
 
 for.end:
   ret i32 0
-}
\ No newline at end of file
+}



More information about the llvm-commits mailing list